From 78b1e2f550df2a760f2911a46193145e303138b8 Mon Sep 17 00:00:00 2001 From: Alcides Fonseca Date: Tue, 4 Nov 2025 13:13:43 +0000 Subject: [PATCH 1/6] Update sklearn models for feature probabilities --- .github/workflows/run_examples.yml | 55 +++++++++++++ docs/source/index.md | 19 +++++ docs/source/sklearn.md | 4 + docs/source/synthetic_problems.md | 93 ++++++++++++++++++++++ geml/classifiers.py | 29 ++++++- geml/common.py | 4 +- geml/regressors.py | 30 ++++++- geneticengine/grammar/metahandlers/vars.py | 54 ++++++++++++- run_examples.sh | 27 ++++++- 9 files changed, 307 insertions(+), 8 deletions(-) create mode 100644 .github/workflows/run_examples.yml create mode 100644 docs/source/synthetic_problems.md diff --git a/.github/workflows/run_examples.yml b/.github/workflows/run_examples.yml new file mode 100644 index 00000000..892def29 --- /dev/null +++ b/.github/workflows/run_examples.yml @@ -0,0 +1,55 @@ +name: Run Examples (uv + optional Codon) + +on: + push: + branches: [ main ] + pull_request: + branches: [ main ] + +jobs: + run-examples: + runs-on: ubuntu-latest + strategy: + fail-fast: false + matrix: + codon: [false, true] + + steps: + - name: Checkout + uses: actions/checkout@v4 + + - name: Set up Python + uses: actions/setup-python@v5 + with: + python-version: '3.11' + + - name: Install uv + shell: bash + run: | + curl -LsSf https://astral.sh/uv/install.sh | sh + echo "$HOME/.local/bin" >> $GITHUB_PATH + + - name: Install Codon (only when matrix.codon=true) + if: matrix.codon == true + shell: bash + run: | + curl -LsSf https://exaloop.io/install.sh | sh + echo "$HOME/.codon/bin" >> $GITHUB_PATH + + - name: Show tool versions + shell: bash + run: | + which uv || true + uv --version || true + which codon || true + codon --version || true + + - name: Run examples + shell: bash + env: + CODON: ${{ matrix.codon && '1' || '0' }} + run: | + chmod +x ./run_examples.sh + ./run_examples.sh + + diff --git a/docs/source/index.md b/docs/source/index.md index a177e87b..57c0396f 100644 --- a/docs/source/index.md +++ b/docs/source/index.md @@ -74,10 +74,29 @@ representations genetic_operators optimizations autoapi/index +synthetic_problems ``` +## Running examples with uv and Codon (optional) + +- Install uv (`https://docs.astral.sh/uv/getting-started/installation/`). The script will automatically use uv to run Python and resolve dependencies from the project. +- Run all examples normally: + +```bash +./run_examples.sh +``` + +- To enable Codon, install Codon (`https://exaloop.io/docs/codon/latest/guide/installation`) and run: + +```bash +CODON=1 ./run_examples.sh +``` + +When `CODON=1`, the script will try to run each example with `codon run -release` when feasible and automatically fall back to `uv run python` for examples that require third-party packages unsupported by Codon. + + ## Indices and tables * {ref}`genindex` diff --git a/docs/source/sklearn.md b/docs/source/sklearn.md index 22491232..85cd32e1 100644 --- a/docs/source/sklearn.md +++ b/docs/source/sklearn.md @@ -30,3 +30,7 @@ model.fit(X, y) For complete examples, check the `sklearn-type-examples.py` in the examples folder. + +## Feature probability weighting + +All sklearn-compatible estimators accept an optional parameter `weight_features_by_correlation` (default: `False`). When enabled, terminals corresponding to input features are sampled with probabilities proportional to their absolute Pearson correlation with the output, biasing the grammar towards more predictive variables. diff --git a/docs/source/synthetic_problems.md b/docs/source/synthetic_problems.md new file mode 100644 index 00000000..0d20d6c4 --- /dev/null +++ b/docs/source/synthetic_problems.md @@ -0,0 +1,93 @@ +# Synthetic problem generation + +Synthetic problem generation lets you quickly produce random grammars and corresponding target individuals to benchmark search behavior without hand-crafting domain grammars. + +This is useful for stress-testing operators, budgets, and configuration choices under controlled complexity. + +## API + +```{eval-rst} +.. autofunction:: geneticengine.grammar.synthetic_grammar.create_arbitrary_grammar +``` + +### Parameters (summary) +- **seed**: Controls reproducibility of the generated grammar. +- **non_terminals_count**: Number of abstract non-terminals to create. +- **recursive_non_terminals_count**: How many of the last non-terminals are allowed to be recursive. +- **productions_per_non_terminal(rd)**: Callable returning how many productions each non-terminal has (called per non-terminal with a `random.Random`). +- **non_terminals_per_production(rd)**: Callable returning the arity (number of fields) per production (called per production with a `random.Random`). +- **base_types**: Set of terminal/base field types to allow as leaves (defaults to `{int, bool}`). + +The function returns `(nodes, root)` where `nodes` are the dynamically created classes (non-terminals and productions) and `root` is the designated root non-terminal. + +## Minimal example + +```python +from geneticengine.grammar.grammar import extract_grammar +from geneticengine.grammar.synthetic_grammar import create_arbitrary_grammar +from geneticengine.random.sources import NativeRandomSource +from geneticengine.representations.tree.initializations import MaxDepthDecider +from geneticengine.representations.tree.treebased import TreeBasedRepresentation + +# 1) Generate a random grammar +nodes, root = create_arbitrary_grammar( + seed=0, + non_terminals_count=3, + recursive_non_terminals_count=2, + productions_per_non_terminal=lambda rd: 2, + non_terminals_per_production=lambda rd: 1, +) + +# 2) Build a Grammar object +G = extract_grammar(nodes, root) + +# 3) Create a random target individual (phenotype) from this grammar +r = NativeRandomSource(0) +rep = TreeBasedRepresentation(G, decider=MaxDepthDecider(r, G, G.get_min_tree_depth())) +G_target = rep.create_genotype(r, depth=8) +P_target = rep.genotype_to_phenotype(G_target) +print(P_target) +``` + +## Using with Genetic Programming + +A common pattern is to generate a target individual and then define a distance-based fitness to measure how close candidates are to that target. For example, using string distance over the stringified phenotypes: + +```python +from polyleven import levenshtein +from geneticengine.algorithms.gp.gp import GeneticProgramming +from geneticengine.evaluation.budget import EvaluationBudget +from geneticengine.problems import SingleObjectiveProblem + +# Assume G and P_target are created as above + +def fitness_function(p): + return levenshtein(str(p), str(P_target)) + +problem = SingleObjectiveProblem( + fitness_function=fitness_function, + minimize=True, + target=0, +) + +r = NativeRandomSource(0) +alg = GeneticProgramming( + problem=problem, + budget=EvaluationBudget(100), + representation=TreeBasedRepresentation(G, decider=MaxDepthDecider(r, G, G.get_min_tree_depth() + 10)), + population_size=10, + random=r, +) +ind = alg.search()[0] +print(ind, ind.get_fitness(problem)) +``` + +## Complexity control tips +- **Grammar size**: Increase `non_terminals_count` and/or `productions_per_non_terminal` to grow the search space. +- **Depth/recursion**: Control recursion potential with `recursive_non_terminals_count`, and control tree growth with `MaxDepthDecider` and the depth you pass to `create_genotype`. +- **Arity**: Use `non_terminals_per_production` to increase/decrease branching factor per production. +- **Leaf variety**: Adjust `base_types` to change the terminal set (e.g., `{int, bool, float}`). + +## End-to-end runnable example +See `examples/synthetic_grammar_example.py` for a complete script with CLI flags to vary grammar size, recursion, and arity. + diff --git a/geml/classifiers.py b/geml/classifiers.py index 0dba886d..b20642cf 100644 --- a/geml/classifiers.py +++ b/geml/classifiers.py @@ -1,4 +1,5 @@ import numpy as np +from typing import Annotated from geml.common import GeneticEngineEstimator, PopulationRecorder from geml.grammars.ruleset_classification import make_grammar from geneticengine.algorithms.gp.gp import GeneticProgramming @@ -8,6 +9,8 @@ from geneticengine.evaluation.budget import SearchBudget from geneticengine.evaluation.tracker import ProgressTracker from geneticengine.grammar.grammar import Grammar, extract_grammar +from geneticengine.grammar.metahandlers.vars import VarRangeWithProbabilities +from geneticengine.grammar.metahandlers.vars import VarRange from geneticengine.problems import Problem from geneticengine.random.sources import RandomSource from geneticengine.representations.tree.initializations import ProgressivelyTerminalDecider @@ -17,10 +20,31 @@ class GeneticEngineClassifier(GeneticEngineEstimator): + def _maybe_weight_features(self, Var, feature_names: list[str], data, target) -> None: + if self.weight_features_by_correlation: + try: + y = target.reshape(-1) if hasattr(target, "reshape") else target + # For classification, use absolute Pearson correlation as a simple heuristic + corrs: list[float] = [] + for i in range(len(feature_names)): + xi = data[:, i] + with np.errstate(all="ignore"): + c = np.corrcoef(xi, y)[0, 1] + if np.isnan(c): + c = 0.0 + corrs.append(abs(float(c))) + s = float(sum(corrs)) + if s > 0: + weights = [c / s for c in corrs] + Var.__init__.__annotations__["name"] = Annotated[str, VarRangeWithProbabilities(feature_names, weights)] # type: ignore + except Exception: + pass + def get_grammar(self, feature_names: list[str], data, target) -> Grammar: classes = np.unique(target).tolist() components, RuleSet = make_grammar(feature_names, classes) Var = components[-1] + self._maybe_weight_features(Var, feature_names, data, target) Var.feature_names = feature_names # type:ignore index_of = {n: i for i, n in enumerate(feature_names)} Var.to_numpy = lambda s: f"dataset[:,{index_of[s.name]}]" # type:ignore @@ -56,14 +80,15 @@ def __str__(self): class HillClimbingClassifier(GeneticEngineClassifier): - def __init__(self, max_time: float | int = 1, seed: int = 0, number_of_mutations: int = 5): - super().__init__(max_time, seed) + def __init__(self, max_time: float | int = 1, seed: int = 0, number_of_mutations: int = 5, weight_features_by_correlation: bool = False): + super().__init__(max_time, seed, weight_features_by_correlation) self.number_of_mutations = number_of_mutations _parameter_constraints = { "max_time": [float, int], "seed": [int], "number_of_mutations": [int], + "weight_features_by_correlation": [bool], } def search( diff --git a/geml/common.py b/geml/common.py index 87ae9080..88295071 100644 --- a/geml/common.py +++ b/geml/common.py @@ -81,13 +81,15 @@ def to_sympy(self): class GeneticEngineEstimator(GEBaseEstimator): max_time: float | int - def __init__(self, max_time: float | int = 1, seed: int = 0): + def __init__(self, max_time: float | int = 1, seed: int = 0, weight_features_by_correlation: bool = False): self.max_time = max_time self.seed = 0 + self.weight_features_by_correlation = weight_features_by_correlation _parameter_constraints = { "max_time": [float, int], "seed": [int], + "weight_features_by_correlation": [bool], } def get_population(self) -> list[BaseEstimator]: diff --git a/geml/regressors.py b/geml/regressors.py index 66e27a7a..c9220e66 100644 --- a/geml/regressors.py +++ b/geml/regressors.py @@ -1,8 +1,10 @@ from abc import ABC, abstractmethod +from typing import Annotated, Any from sklearn.base import BaseEstimator from geml.common import GeneticEngineEstimator, PopulationRecorder from geml.grammars.symbolic_regression import make_var, components, Expression +from geneticengine.grammar.metahandlers.vars import VarRangeWithProbabilities from geneticengine.algorithms.gp.gp import GeneticProgramming from geneticengine.algorithms.hill_climbing import HC from geneticengine.algorithms.one_plus_one import OnePlusOne @@ -34,8 +36,31 @@ class GeneticEngineRegressor( GeneticEngineEstimator, ): + def _maybe_weight_features(self, Var, feature_names: list[str], data, target) -> None: + if self.weight_features_by_correlation: + try: + import numpy as np # local import to avoid polluting module namespace + y = target.reshape(-1) if hasattr(target, "reshape") else target + corrs: list[float] = [] + for i in range(len(feature_names)): + xi = data[:, i] + with np.errstate(all="ignore"): + c = np.corrcoef(xi, y)[0, 1] + if np.isnan(c): + c = 0.0 + corrs.append(abs(float(c))) + s = sum(corrs) + if s > 0: + weights = [c / s for c in corrs] + Var.__init__.__annotations__["name"] = Annotated[str, VarRangeWithProbabilities(feature_names, weights)] # type: ignore + except Exception: + # If anything goes wrong, fall back to uniform selection + pass + def get_grammar(self, feature_names: list[str], data, target) -> Grammar: Var = make_var(feature_names, relative_weight=10) + # Optionally weight features by their absolute Pearson correlation with target + self._maybe_weight_features(Var, feature_names, data, target) Var.feature_names = feature_names index_of = {n: i for i, n in enumerate(feature_names)} Var.to_numpy = lambda s: f"dataset[:,{index_of[s.name]}]" @@ -72,14 +97,15 @@ def __str__(self): class HillClimbingRegressor(GeneticEngineRegressor): - def __init__(self, max_time: float | int = 1, seed: int = 0, number_of_mutations: int = 5): - super().__init__(max_time, seed) + def __init__(self, max_time: float | int = 1, seed: int = 0, number_of_mutations: int = 5, weight_features_by_correlation: bool = False): + super().__init__(max_time, seed, weight_features_by_correlation) self.number_of_mutations = number_of_mutations _parameter_constraints = { "max_time": [float, int], "seed": [int], "number_of_mutations": [int], + "weight_features_by_correlation": [bool], } def search( diff --git a/geneticengine/grammar/metahandlers/vars.py b/geneticengine/grammar/metahandlers/vars.py index 29d8c0a4..7cc802ad 100644 --- a/geneticengine/grammar/metahandlers/vars.py +++ b/geneticengine/grammar/metahandlers/vars.py @@ -1,6 +1,6 @@ from __future__ import annotations -from typing import Any, Callable, Generator, TypeVar +from typing import Any, Callable, Generator, TypeVar, Sequence from geneticengine.grammar.grammar import Grammar from geneticengine.random.sources import RandomSource @@ -53,3 +53,55 @@ def iterate( dependent_values: dict[str, Any], ): yield from self.options + + +class VarRangeWithProbabilities(MetaHandlerGenerator): + """Like VarRange but allows specifying per-option selection probabilities. + + Usage: Annotated[str, VarRangeWithProbabilities(options, weights)] + """ + + def __init__(self, options: list[T], weights: Sequence[float]): + if not options: + raise SynthesisException( + f"The VarRangeWithProbabilities metahandler requires a non-empty set of options. Options found: {options}", + ) + if not weights or len(weights) != len(options): + raise SynthesisException( + "The weights list must be the same length as options and non-empty.", + ) + self.options = options + self.weights = list(weights) + + def validate(self, v) -> bool: + return v in self.options + + def generate( + self, + random: RandomSource, + grammar: Grammar, + base_type: type, + rec: Callable[[type[T]], T], + dependent_values: dict[str, Any], + parent_values: list[dict[str, Any]], + ): + total = sum(self.weights) + if total <= 0: + return random.choice(self.options) + normalized = [w / total if w >= 0 else 0.0 for w in self.weights] + return random.choice_weighted(self.options, normalized) + + def __repr__(self): + return f"{self.options} (weighted)" + + def __class_getitem__(cls, args): + return VarRangeWithProbabilities(*args) + + def iterate( + self, + base_type: type, + combine_lists: Callable[[list[type]], Generator[Any, Any, Any]], + rec: Any, + dependent_values: dict[str, Any], + ): + yield from self.options diff --git a/run_examples.sh b/run_examples.sh index e3090a65..9347c1e8 100755 --- a/run_examples.sh +++ b/run_examples.sh @@ -1,7 +1,24 @@ #!/bin/bash export PYTHONPATH="${PYTHONPATH:+${PYTHONPATH}:}." -PYTHON_BINARY=python3 +# Preferred Python runner: uv (falls back to system python3) +if command -v uv >/dev/null 2>&1; then + PYTHON_CMD=(uv run -q -- python) +else + PYTHON_CMD=(python3) +fi + +# Optional: attempt to use Codon when requested, but fall back for packages Codon doesn't support +# Enable with: CODON=1 ./run_examples.sh +function can_run_with_codon { + # Return 0 (true) if the example likely works with Codon; 1 otherwise + local file="$1" + # Heuristic: Codon generally doesn't support heavy third-party modules + if grep -E '^(from|import)\s+(pandas|numpy|sklearn|seaborn|z3|pathos|matplotlib|sympy)\b' "$file" >/dev/null 2>&1; then + return 1 + fi + return 0 +} set -o errexit set -o nounset @@ -14,7 +31,13 @@ cd "$(dirname "$0")" function run_example { printf "Running $1..." - $PYTHON_BINARY $1 > /dev/null + if [[ "${CODON-0}" == "1" ]] && command -v codon >/dev/null 2>&1 && can_run_with_codon "$1"; then + codon run -release "$1" > /dev/null || { echo "(failed)"; exit 111; } + echo "(success)" + return + fi + + "${PYTHON_CMD[@]}" "$1" > /dev/null RESULT=$? if [ $RESULT -eq 0 ]; then echo "(success)" From 2ac3c6b03dc93e970f462e0f6a6a3d3dd8afe074 Mon Sep 17 00:00:00 2001 From: Alcides Fonseca Date: Wed, 5 Nov 2025 08:22:24 +0000 Subject: [PATCH 2/6] Fix ruff linting issues: remove unused imports and type ignore comments --- geml/classifiers.py | 32 ++++++++++++++------------------ geml/common.py | 2 +- geml/regressors.py | 34 +++++++++++++++------------------- 3 files changed, 30 insertions(+), 38 deletions(-) diff --git a/geml/classifiers.py b/geml/classifiers.py index b20642cf..295c6528 100644 --- a/geml/classifiers.py +++ b/geml/classifiers.py @@ -10,7 +10,6 @@ from geneticengine.evaluation.tracker import ProgressTracker from geneticengine.grammar.grammar import Grammar, extract_grammar from geneticengine.grammar.metahandlers.vars import VarRangeWithProbabilities -from geneticengine.grammar.metahandlers.vars import VarRange from geneticengine.problems import Problem from geneticengine.random.sources import RandomSource from geneticengine.representations.tree.initializations import ProgressivelyTerminalDecider @@ -22,23 +21,20 @@ class GeneticEngineClassifier(GeneticEngineEstimator): def _maybe_weight_features(self, Var, feature_names: list[str], data, target) -> None: if self.weight_features_by_correlation: - try: - y = target.reshape(-1) if hasattr(target, "reshape") else target - # For classification, use absolute Pearson correlation as a simple heuristic - corrs: list[float] = [] - for i in range(len(feature_names)): - xi = data[:, i] - with np.errstate(all="ignore"): - c = np.corrcoef(xi, y)[0, 1] - if np.isnan(c): - c = 0.0 - corrs.append(abs(float(c))) - s = float(sum(corrs)) - if s > 0: - weights = [c / s for c in corrs] - Var.__init__.__annotations__["name"] = Annotated[str, VarRangeWithProbabilities(feature_names, weights)] # type: ignore - except Exception: - pass + y = target.reshape(-1) if hasattr(target, "reshape") else target + # For classification, use absolute Pearson correlation as a simple heuristic + corrs: list[float] = [] + for i in range(len(feature_names)): + xi = data[:, i] + with np.errstate(all="ignore"): + c = np.corrcoef(xi, y)[0, 1] + if np.isnan(c): + c = 0.0 + corrs.append(abs(float(c))) + s = float(sum(corrs)) + if s > 0: + weights = [c / s for c in corrs] + Var.__init__.__annotations__["name"] = Annotated[str, VarRangeWithProbabilities(feature_names, weights)] def get_grammar(self, feature_names: list[str], data, target) -> Grammar: classes = np.unique(target).tolist() diff --git a/geml/common.py b/geml/common.py index 88295071..76225fea 100644 --- a/geml/common.py +++ b/geml/common.py @@ -83,7 +83,7 @@ class GeneticEngineEstimator(GEBaseEstimator): def __init__(self, max_time: float | int = 1, seed: int = 0, weight_features_by_correlation: bool = False): self.max_time = max_time - self.seed = 0 + self.seed = seed self.weight_features_by_correlation = weight_features_by_correlation _parameter_constraints = { diff --git a/geml/regressors.py b/geml/regressors.py index c9220e66..31c137f4 100644 --- a/geml/regressors.py +++ b/geml/regressors.py @@ -1,5 +1,5 @@ from abc import ABC, abstractmethod -from typing import Annotated, Any +from typing import Annotated from sklearn.base import BaseEstimator from geml.common import GeneticEngineEstimator, PopulationRecorder @@ -38,24 +38,20 @@ class GeneticEngineRegressor( def _maybe_weight_features(self, Var, feature_names: list[str], data, target) -> None: if self.weight_features_by_correlation: - try: - import numpy as np # local import to avoid polluting module namespace - y = target.reshape(-1) if hasattr(target, "reshape") else target - corrs: list[float] = [] - for i in range(len(feature_names)): - xi = data[:, i] - with np.errstate(all="ignore"): - c = np.corrcoef(xi, y)[0, 1] - if np.isnan(c): - c = 0.0 - corrs.append(abs(float(c))) - s = sum(corrs) - if s > 0: - weights = [c / s for c in corrs] - Var.__init__.__annotations__["name"] = Annotated[str, VarRangeWithProbabilities(feature_names, weights)] # type: ignore - except Exception: - # If anything goes wrong, fall back to uniform selection - pass + import numpy as np # local import to avoid polluting module namespace + y = target.reshape(-1) if hasattr(target, "reshape") else target + corrs: list[float] = [] + for i in range(len(feature_names)): + xi = data[:, i] + with np.errstate(all="ignore"): + c = np.corrcoef(xi, y)[0, 1] + if np.isnan(c): + c = 0.0 + corrs.append(abs(float(c))) + s = sum(corrs) + if s > 0: + weights = [c / s for c in corrs] + Var.__init__.__annotations__["name"] = Annotated[str, VarRangeWithProbabilities(feature_names, weights)] def get_grammar(self, feature_names: list[str], data, target) -> Grammar: Var = make_var(feature_names, relative_weight=10) From 8fab1889dd8ab33a46828542735f425ba889616d Mon Sep 17 00:00:00 2001 From: Alcides Fonseca Date: Wed, 5 Nov 2025 09:22:45 +0000 Subject: [PATCH 3/6] Removed unused import --- geneticengine/grammar/metahandlers/vars.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/geneticengine/grammar/metahandlers/vars.py b/geneticengine/grammar/metahandlers/vars.py index 5beadb57..30703df2 100644 --- a/geneticengine/grammar/metahandlers/vars.py +++ b/geneticengine/grammar/metahandlers/vars.py @@ -1,6 +1,6 @@ from __future__ import annotations -from typing import Any, Callable, Generator, TypeVar, Sequence +from typing import Any, Callable, Generator, TypeVar from geneticengine.grammar.grammar import Grammar from geneticengine.random.sources import RandomSource From b3e5e92b83c9d67111fcfb8055ac1fd7b69a82d3 Mon Sep 17 00:00:00 2001 From: Alcides Fonseca Date: Wed, 5 Nov 2025 09:55:12 +0000 Subject: [PATCH 4/6] Fixed weights --- .github/workflows/run_examples.yml | 55 ---------------------------- examples/geml/classifier_example.py | 2 +- geml/classifiers.py | 21 ++--------- geml/common.py | 8 ++++ geml/grammars/symbolic_regression.py | 11 +++--- geml/regressors.py | 27 +------------- 6 files changed, 21 insertions(+), 103 deletions(-) delete mode 100644 .github/workflows/run_examples.yml diff --git a/.github/workflows/run_examples.yml b/.github/workflows/run_examples.yml deleted file mode 100644 index 892def29..00000000 --- a/.github/workflows/run_examples.yml +++ /dev/null @@ -1,55 +0,0 @@ -name: Run Examples (uv + optional Codon) - -on: - push: - branches: [ main ] - pull_request: - branches: [ main ] - -jobs: - run-examples: - runs-on: ubuntu-latest - strategy: - fail-fast: false - matrix: - codon: [false, true] - - steps: - - name: Checkout - uses: actions/checkout@v4 - - - name: Set up Python - uses: actions/setup-python@v5 - with: - python-version: '3.11' - - - name: Install uv - shell: bash - run: | - curl -LsSf https://astral.sh/uv/install.sh | sh - echo "$HOME/.local/bin" >> $GITHUB_PATH - - - name: Install Codon (only when matrix.codon=true) - if: matrix.codon == true - shell: bash - run: | - curl -LsSf https://exaloop.io/install.sh | sh - echo "$HOME/.codon/bin" >> $GITHUB_PATH - - - name: Show tool versions - shell: bash - run: | - which uv || true - uv --version || true - which codon || true - codon --version || true - - - name: Run examples - shell: bash - env: - CODON: ${{ matrix.codon && '1' || '0' }} - run: | - chmod +x ./run_examples.sh - ./run_examples.sh - - diff --git a/examples/geml/classifier_example.py b/examples/geml/classifier_example.py index 8316a88d..4f81d2cf 100644 --- a/examples/geml/classifier_example.py +++ b/examples/geml/classifier_example.py @@ -27,5 +27,5 @@ model = model_class(max_time=20.0, seed=seed) model.fit(data, target) y_pred = model.predict(test_data) - r2 = f1_score(test_target, y_pred) + r2 = f1_score(test_target, y_pred, average="weighted") print(f"{model}: {r2}") diff --git a/geml/classifiers.py b/geml/classifiers.py index 295c6528..6ca69ae0 100644 --- a/geml/classifiers.py +++ b/geml/classifiers.py @@ -1,3 +1,4 @@ +from geneticengine.grammar.decorators import weight import numpy as np from typing import Annotated from geml.common import GeneticEngineEstimator, PopulationRecorder @@ -19,31 +20,17 @@ class GeneticEngineClassifier(GeneticEngineEstimator): - def _maybe_weight_features(self, Var, feature_names: list[str], data, target) -> None: - if self.weight_features_by_correlation: - y = target.reshape(-1) if hasattr(target, "reshape") else target - # For classification, use absolute Pearson correlation as a simple heuristic - corrs: list[float] = [] - for i in range(len(feature_names)): - xi = data[:, i] - with np.errstate(all="ignore"): - c = np.corrcoef(xi, y)[0, 1] - if np.isnan(c): - c = 0.0 - corrs.append(abs(float(c))) - s = float(sum(corrs)) - if s > 0: - weights = [c / s for c in corrs] - Var.__init__.__annotations__["name"] = Annotated[str, VarRangeWithProbabilities(feature_names, weights)] def get_grammar(self, feature_names: list[str], data, target) -> Grammar: classes = np.unique(target).tolist() components, RuleSet = make_grammar(feature_names, classes) Var = components[-1] - self._maybe_weight_features(Var, feature_names, data, target) + weights = self.correlation_weights(feature_names, data, target) + Var.__init__.__annotations__["name"] = Annotated[str, VarRangeWithProbabilities(feature_names, weights)] # type:ignore Var.feature_names = feature_names # type:ignore index_of = {n: i for i, n in enumerate(feature_names)} Var.to_numpy = lambda s: f"dataset[:,{index_of[s.name]}]" # type:ignore + Var = weight(10)(Var) return extract_grammar(components, RuleSet) def get_goal(self) -> tuple[bool, float]: diff --git a/geml/common.py b/geml/common.py index 76225fea..05759e8b 100644 --- a/geml/common.py +++ b/geml/common.py @@ -179,3 +179,11 @@ def search( budget: SearchBudget, population_recorder: PopulationRecorder, ) -> list[Individual] | None: ... + + + def correlation_weights(self, feature_names: list[str], data, target) -> list[float]: + + def wrapper(v:float) -> float: + return 1 - abs(float(np.average(v))) + 0.00001 + + return [wrapper(np.corrcoef(data[:, i], target)) for i in range(len(feature_names))] diff --git a/geml/grammars/symbolic_regression.py b/geml/grammars/symbolic_regression.py index d6c9ca3f..9ecc7888 100644 --- a/geml/grammars/symbolic_regression.py +++ b/geml/grammars/symbolic_regression.py @@ -3,7 +3,7 @@ from typing import Annotated from geneticengine.grammar.decorators import weight -from geneticengine.grammar.metahandlers.vars import VarRange +from geneticengine.grammar.metahandlers.vars import VarRangeWithProbabilities class Expression(ABC): @@ -190,18 +190,19 @@ def to_numpy(self) -> str: return f"{self.value}" -def make_var(options: list[str], relative_weight: float = 1): +def make_var(options: list[str], weights: list[float], relative_weight: float = 1): @weight(relative_weight) @dataclass class Var(Expression): - name: Annotated[str, VarRange(options)] - # The list of vars should always be filled in dynamically + name: Annotated[str, VarRangeWithProbabilities(options, weights)] + feature_names : list[str] def to_sympy(self) -> str: return f"{self.name}" def to_numpy(self) -> str: - return f"{self.name}" + index_of = {n: i for i, n in enumerate(options)} + return f"dataset[:,{index_of[self.name]}]" return Var diff --git a/geml/regressors.py b/geml/regressors.py index 31c137f4..87620b58 100644 --- a/geml/regressors.py +++ b/geml/regressors.py @@ -1,10 +1,8 @@ from abc import ABC, abstractmethod -from typing import Annotated from sklearn.base import BaseEstimator from geml.common import GeneticEngineEstimator, PopulationRecorder from geml.grammars.symbolic_regression import make_var, components, Expression -from geneticengine.grammar.metahandlers.vars import VarRangeWithProbabilities from geneticengine.algorithms.gp.gp import GeneticProgramming from geneticengine.algorithms.hill_climbing import HC from geneticengine.algorithms.one_plus_one import OnePlusOne @@ -36,30 +34,9 @@ class GeneticEngineRegressor( GeneticEngineEstimator, ): - def _maybe_weight_features(self, Var, feature_names: list[str], data, target) -> None: - if self.weight_features_by_correlation: - import numpy as np # local import to avoid polluting module namespace - y = target.reshape(-1) if hasattr(target, "reshape") else target - corrs: list[float] = [] - for i in range(len(feature_names)): - xi = data[:, i] - with np.errstate(all="ignore"): - c = np.corrcoef(xi, y)[0, 1] - if np.isnan(c): - c = 0.0 - corrs.append(abs(float(c))) - s = sum(corrs) - if s > 0: - weights = [c / s for c in corrs] - Var.__init__.__annotations__["name"] = Annotated[str, VarRangeWithProbabilities(feature_names, weights)] - def get_grammar(self, feature_names: list[str], data, target) -> Grammar: - Var = make_var(feature_names, relative_weight=10) - # Optionally weight features by their absolute Pearson correlation with target - self._maybe_weight_features(Var, feature_names, data, target) - Var.feature_names = feature_names - index_of = {n: i for i, n in enumerate(feature_names)} - Var.to_numpy = lambda s: f"dataset[:,{index_of[s.name]}]" + weights = self.correlation_weights(feature_names, data, target) + Var = make_var(feature_names, weights=weights, relative_weight=10) complete_components = components + [Var] return extract_grammar(complete_components, Expression) From 815dc96df69cf6df12ad0a1fa2eafca361a6c8a3 Mon Sep 17 00:00:00 2001 From: Alcides Fonseca Date: Wed, 5 Nov 2025 11:15:49 +0000 Subject: [PATCH 5/6] Handled numpy weirdness. --- geml/classifiers.py | 2 +- geml/common.py | 25 +++++++++++++++++++++---- geml/grammars/symbolic_regression.py | 24 ++++++++++++++++++------ geml/regressors.py | 7 +++++++ geneticengine/random/sources.py | 7 ++++++- 5 files changed, 53 insertions(+), 12 deletions(-) diff --git a/geml/classifiers.py b/geml/classifiers.py index 6ca69ae0..9e29469b 100644 --- a/geml/classifiers.py +++ b/geml/classifiers.py @@ -1,6 +1,6 @@ -from geneticengine.grammar.decorators import weight import numpy as np from typing import Annotated +from geneticengine.grammar.decorators import weight from geml.common import GeneticEngineEstimator, PopulationRecorder from geml.grammars.ruleset_classification import make_grammar from geneticengine.algorithms.gp.gp import GeneticProgramming diff --git a/geml/common.py b/geml/common.py index 05759e8b..e5013abb 100644 --- a/geml/common.py +++ b/geml/common.py @@ -183,7 +183,24 @@ def search( def correlation_weights(self, feature_names: list[str], data, target) -> list[float]: - def wrapper(v:float) -> float: - return 1 - abs(float(np.average(v))) + 0.00001 - - return [wrapper(np.corrcoef(data[:, i], target)) for i in range(len(feature_names))] + def safe_corrcoef(xv, yv) -> float: + with np.errstate(all="ignore"): + x = np.asarray(xv, dtype=float) + y = np.asarray(yv, dtype=float) + if len(x) < 2: + return 0.0 + c = np.corrcoef(x, y) + # For 2x2 corr matrix, off-diagonal holds the correlation + try: + corr = float(c[0, 1]) + except Exception: + corr = 0.0 + if not np.isfinite(corr): + return 0.0 + return corr + + def wrapper(corr_value: float) -> float: + # Higher absolute correlation -> smaller weight (bias search), add epsilon + return 1 - abs(corr_value) + 0.00001 + + return [wrapper(safe_corrcoef(data[:, i], target)) for i in range(len(feature_names))] diff --git a/geml/grammars/symbolic_regression.py b/geml/grammars/symbolic_regression.py index 9ecc7888..57feb7ca 100644 --- a/geml/grammars/symbolic_regression.py +++ b/geml/grammars/symbolic_regression.py @@ -1,6 +1,7 @@ from abc import ABC, abstractmethod -from dataclasses import dataclass +from dataclasses import dataclass, field from typing import Annotated +from typing import cast from geneticengine.grammar.decorators import weight from geneticengine.grammar.metahandlers.vars import VarRangeWithProbabilities @@ -190,19 +191,30 @@ def to_numpy(self) -> str: return f"{self.value}" -def make_var(options: list[str], weights: list[float], relative_weight: float = 1): +def make_var(options: list[str], weights: list[float] | int | float | None = None, relative_weight: float = 1): + # Backward/lenient compatibility: if the second positional argument is a number, + # interpret it as relative_weight and default to uniform feature weights. + if isinstance(weights, (int, float)) and relative_weight == 1: + relative_weight = int(weights) + weights = None + # If no explicit weights are provided, use uniform weights. + if weights is None: + weights_list: list[float] = [1.0 for _ in options] + else: + weights_list = [float(w) for w in cast(list[float], weights)] + # Ensure static type correctness for annotation usage below. + probabilities: list[float] = weights_list @weight(relative_weight) @dataclass class Var(Expression): - name: Annotated[str, VarRangeWithProbabilities(options, weights)] - feature_names : list[str] + name: Annotated[str, VarRangeWithProbabilities(options, probabilities)] + feature_names: list[str] = field(default_factory=list) def to_sympy(self) -> str: return f"{self.name}" def to_numpy(self) -> str: - index_of = {n: i for i, n in enumerate(options)} - return f"dataset[:,{index_of[self.name]}]" + return f"{self.name}" return Var diff --git a/geml/regressors.py b/geml/regressors.py index 87620b58..4c19a191 100644 --- a/geml/regressors.py +++ b/geml/regressors.py @@ -37,6 +37,13 @@ class GeneticEngineRegressor( def get_grammar(self, feature_names: list[str], data, target) -> Grammar: weights = self.correlation_weights(feature_names, data, target) Var = make_var(feature_names, weights=weights, relative_weight=10) + # Align Var with dataset-based evaluation used during training + from typing import Annotated + from geneticengine.grammar.metahandlers.vars import VarRangeWithProbabilities + Var.__init__.__annotations__["name"] = Annotated[str, VarRangeWithProbabilities(feature_names, weights)] + Var.feature_names = feature_names + index_of = {n: i for i, n in enumerate(feature_names)} + Var.to_numpy = lambda s: f"dataset[:,{index_of[s.name]}]" # pyright:ignore complete_components = components + [Var] return extract_grammar(complete_components, Expression) diff --git a/geneticengine/random/sources.py b/geneticengine/random/sources.py index 9f1d8dda..d9ebe6d6 100644 --- a/geneticengine/random/sources.py +++ b/geneticengine/random/sources.py @@ -26,7 +26,12 @@ def choice_weighted( choices: list[T], weights: list[float], ) -> T: - acc_weights: list[int] = [int(x * 100000) for x in accumulate(weights)] + # Sanitize weights: replace non-finite values with 0 and ensure non-negative totals. + sanitized: list[float] = [w if isinstance(w, (int, float)) and math.isfinite(w) and w > 0 else 0.0 for w in weights] + # If all weights are zero or negative, fall back to uniform selection. + if not any(sanitized): + return self.choice(choices) + acc_weights: list[int] = [int(x * 100000) for x in accumulate(sanitized)] total = acc_weights[-1] rand_value: float = self.randint(0, total) From 82473f0d5b3ca14d593d702864bd7b9044b4aae5 Mon Sep 17 00:00:00 2001 From: Alcides Fonseca Date: Wed, 5 Nov 2025 17:52:44 +0000 Subject: [PATCH 6/6] Refactored Var in regression --- geml/grammars/symbolic_regression.py | 21 ++++++++++++--------- geml/regressors.py | 7 ++----- 2 files changed, 14 insertions(+), 14 deletions(-) diff --git a/geml/grammars/symbolic_regression.py b/geml/grammars/symbolic_regression.py index 57feb7ca..38f2f197 100644 --- a/geml/grammars/symbolic_regression.py +++ b/geml/grammars/symbolic_regression.py @@ -4,7 +4,7 @@ from typing import cast from geneticengine.grammar.decorators import weight -from geneticengine.grammar.metahandlers.vars import VarRangeWithProbabilities +from geneticengine.grammar.metahandlers.vars import VarRange, VarRangeWithProbabilities class Expression(ABC): @@ -197,17 +197,11 @@ def make_var(options: list[str], weights: list[float] | int | float | None = Non if isinstance(weights, (int, float)) and relative_weight == 1: relative_weight = int(weights) weights = None - # If no explicit weights are provided, use uniform weights. - if weights is None: - weights_list: list[float] = [1.0 for _ in options] - else: - weights_list = [float(w) for w in cast(list[float], weights)] - # Ensure static type correctness for annotation usage below. - probabilities: list[float] = weights_list + @weight(relative_weight) @dataclass class Var(Expression): - name: Annotated[str, VarRangeWithProbabilities(options, probabilities)] + name: str # Annotation will be set dynamically below feature_names: list[str] = field(default_factory=list) def to_sympy(self) -> str: @@ -216,6 +210,15 @@ def to_sympy(self) -> str: def to_numpy(self) -> str: return f"{self.name}" + # Choose metahandler based on whether weights are provided + if weights is None: + # Use uniform selection (no probabilities) + Var.__init__.__annotations__["name"] = Annotated[str, VarRange(options)] + else: + # Use weighted selection with probabilities + weights_list = [float(w) for w in cast(list[float], weights)] + Var.__init__.__annotations__["name"] = Annotated[str, VarRangeWithProbabilities(options, weights_list)] + return Var diff --git a/geml/regressors.py b/geml/regressors.py index 4c19a191..44df024f 100644 --- a/geml/regressors.py +++ b/geml/regressors.py @@ -35,12 +35,9 @@ class GeneticEngineRegressor( ): def get_grammar(self, feature_names: list[str], data, target) -> Grammar: - weights = self.correlation_weights(feature_names, data, target) + weights = self.correlation_weights(feature_names, data, target) if self.weight_features_by_correlation else None Var = make_var(feature_names, weights=weights, relative_weight=10) - # Align Var with dataset-based evaluation used during training - from typing import Annotated - from geneticengine.grammar.metahandlers.vars import VarRangeWithProbabilities - Var.__init__.__annotations__["name"] = Annotated[str, VarRangeWithProbabilities(feature_names, weights)] + Var.feature_names = feature_names index_of = {n: i for i, n in enumerate(feature_names)} Var.to_numpy = lambda s: f"dataset[:,{index_of[s.name]}]" # pyright:ignore