-
Notifications
You must be signed in to change notification settings - Fork 11
Update sklearn models for feature probabilities #365
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
Changes from all commits
78b1e2f
2ac3c6b
6ac5c0b
8fab188
b3e5e92
815dc96
82473f0
File filter
Filter by extension
Conversations
Jump to
Diff view
Diff view
There are no files selected for viewing
| Original file line number | Diff line number | Diff line change |
|---|---|---|
|
|
@@ -81,13 +81,15 @@ def to_sympy(self): | |
| class GeneticEngineEstimator(GEBaseEstimator): | ||
| max_time: float | int | ||
|
|
||
| def __init__(self, max_time: float | int = 1, seed: int = 0): | ||
| def __init__(self, max_time: float | int = 1, seed: int = 0, weight_features_by_correlation: bool = False): | ||
| self.max_time = max_time | ||
| self.seed = 0 | ||
| self.seed = seed | ||
| self.weight_features_by_correlation = weight_features_by_correlation | ||
|
Contributor
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. Bug: Seed Parameter Not Applied: Always Zeroed SeedThe seed parameter is hardcoded to 0 instead of using the seed parameter passed to init. Line 86 should be |
||
|
|
||
| _parameter_constraints = { | ||
| "max_time": [float, int], | ||
| "seed": [int], | ||
| "weight_features_by_correlation": [bool], | ||
| } | ||
|
|
||
| def get_population(self) -> list[BaseEstimator]: | ||
|
|
@@ -177,3 +179,28 @@ def search( | |
| budget: SearchBudget, | ||
| population_recorder: PopulationRecorder, | ||
| ) -> list[Individual] | None: ... | ||
|
|
||
|
|
||
| def correlation_weights(self, feature_names: list[str], data, target) -> list[float]: | ||
|
|
||
| def safe_corrcoef(xv, yv) -> float: | ||
| with np.errstate(all="ignore"): | ||
| x = np.asarray(xv, dtype=float) | ||
| y = np.asarray(yv, dtype=float) | ||
| if len(x) < 2: | ||
| return 0.0 | ||
| c = np.corrcoef(x, y) | ||
| # For 2x2 corr matrix, off-diagonal holds the correlation | ||
| try: | ||
| corr = float(c[0, 1]) | ||
| except Exception: | ||
| corr = 0.0 | ||
| if not np.isfinite(corr): | ||
| return 0.0 | ||
| return corr | ||
|
|
||
| def wrapper(corr_value: float) -> float: | ||
| # Higher absolute correlation -> smaller weight (bias search), add epsilon | ||
| return 1 - abs(corr_value) + 0.00001 | ||
|
Contributor
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. Bug: Inverted Correlation Weighting Misaligns Sampling LikelihoodThe correlation weighting logic is inverted. The documentation states features should be "sampled with probabilities proportional to their absolute Pearson correlation", but the implementation returns |
||
|
|
||
| return [wrapper(safe_corrcoef(data[:, i], target)) for i in range(len(feature_names))] | ||
| Original file line number | Diff line number | Diff line change |
|---|---|---|
|
|
@@ -35,10 +35,12 @@ class GeneticEngineRegressor( | |
| ): | ||
|
|
||
| def get_grammar(self, feature_names: list[str], data, target) -> Grammar: | ||
| Var = make_var(feature_names, relative_weight=10) | ||
| weights = self.correlation_weights(feature_names, data, target) if self.weight_features_by_correlation else None | ||
| Var = make_var(feature_names, weights=weights, relative_weight=10) | ||
|
Contributor
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. Bug: Correlation weighting ignored when disabled.The |
||
|
|
||
| Var.feature_names = feature_names | ||
| index_of = {n: i for i, n in enumerate(feature_names)} | ||
| Var.to_numpy = lambda s: f"dataset[:,{index_of[s.name]}]" | ||
| Var.to_numpy = lambda s: f"dataset[:,{index_of[s.name]}]" # pyright:ignore | ||
| complete_components = components + [Var] | ||
| return extract_grammar(complete_components, Expression) | ||
|
|
||
|
|
@@ -72,14 +74,15 @@ def __str__(self): | |
|
|
||
| class HillClimbingRegressor(GeneticEngineRegressor): | ||
|
|
||
| def __init__(self, max_time: float | int = 1, seed: int = 0, number_of_mutations: int = 5): | ||
| super().__init__(max_time, seed) | ||
| def __init__(self, max_time: float | int = 1, seed: int = 0, number_of_mutations: int = 5, weight_features_by_correlation: bool = False): | ||
| super().__init__(max_time, seed, weight_features_by_correlation) | ||
| self.number_of_mutations = number_of_mutations | ||
|
|
||
| _parameter_constraints = { | ||
| "max_time": [float, int], | ||
| "seed": [int], | ||
| "number_of_mutations": [int], | ||
| "weight_features_by_correlation": [bool], | ||
| } | ||
|
|
||
| def search( | ||
|
|
||
There was a problem hiding this comment.
Choose a reason for hiding this comment
The reason will be displayed to describe this comment to others. Learn more.
Bug: Conditional weighting bypasses flag in get_grammar
The
get_grammarmethod unconditionally callsself.correlation_weights(feature_names, data, target)and applies the weights, ignoring theweight_features_by_correlationparameter. The correlation-based weighting should only be applied whenself.weight_features_by_correlationis True, otherwise uniform weights should be used.