From 66d64ec5bd45df1a0bcbc0facd8ec8ccdb138138 Mon Sep 17 00:00:00 2001 From: J Xu Date: Mon, 17 Aug 2026 22:26:59 -0700 Subject: [PATCH 1/2] refactor: reclaim entropy left by the unified-trainer port Remove maintenance surface with no remaining load-bearing reason, each cut proved against the consumer graph and the existing suites: - Delete DistributionalGBDT._compute_gradients and the NaturalBoost override. The trainer computes gradients through DistributionObjective, so this "override to change the gradient" extension point had zero callers; the natural/ordinary switch is the _use_natural_gradient flag. - Inline the _predict_raw forwarding wrapper into predict_params. - Collapse the duplicate _as_1d helper (byte-identical in _formula.py and _survival.py) into one validate_1d in _validation.py. Net -36 lines and 3 fewer concepts. 102 distributional / formula / survival / persistence / integration tests pass; no observable behavior change. Co-authored-by: Cursor --- src/openboost/_models/_distributional.py | 38 ++---------------------- src/openboost/_models/_formula.py | 16 +++------- src/openboost/_models/_survival.py | 16 +++------- src/openboost/_validation.py | 14 +++++++++ 4 files changed, 24 insertions(+), 60 deletions(-) diff --git a/src/openboost/_models/_distributional.py b/src/openboost/_models/_distributional.py index 120a8de..765fe70 100644 --- a/src/openboost/_models/_distributional.py +++ b/src/openboost/_models/_distributional.py @@ -406,29 +406,7 @@ def _eval_metric_value( upper = self.distribution_.quantile(params, 1 - interval_alpha / 2) return interval_score(y, lower, upper, alpha=interval_alpha) raise ValueError(f"Unknown eval_metric '{metric}'.") # pragma: no cover - - def _compute_gradients( - self, - y: NDArray, - params: dict[str, NDArray], - ) -> dict[str, tuple[NDArray, NDArray]]: - """Compute gradients (ordinary gradient descent). - - Subclasses can override for different gradient computation. - """ - return self.distribution_.nll_gradient(y, params) - - def _predict_raw(self, X: NDArray | BinnedArray) -> dict[str, NDArray]: - """Predict raw (link-space) parameters. - - Args: - X: Features to predict on - - Returns: - Dictionary mapping param_name -> raw predictions - """ - return predict_raw(self, X) - + def predict_params( self, X: NDArray | BinnedArray, @@ -448,7 +426,7 @@ def predict_params( Dictionary mapping param_name -> predicted values (in constrained parameter space) """ - raw_preds = self._predict_raw(X) + raw_preds = predict_raw(self, X) if exposure is not None: name, sign = self._resolve_exposure_offset() @@ -643,18 +621,6 @@ class NaturalBoost(DistributionalGBDT): learning_rate: float = 0.1 _use_natural_gradient: bool = field(default=True, init=False, repr=False) - def _compute_gradients( - self, - y: NDArray, - params: dict[str, NDArray], - ) -> dict[str, tuple[NDArray, NDArray]]: - """Compute natural gradients. - - Natural gradient = F^{-1} @ ordinary_gradient - where F is the Fisher information matrix. - """ - return self.distribution_.natural_gradient(y, params) - # ============================================================================= # Convenience aliases diff --git a/src/openboost/_models/_formula.py b/src/openboost/_models/_formula.py index af31c26..e017f8e 100644 --- a/src/openboost/_models/_formula.py +++ b/src/openboost/_models/_formula.py @@ -20,15 +20,7 @@ from .._objectives import FormulaObjective from .._persistence import PersistenceMixin from .._trainer import TrainerConfig, fit_boosting, predict_raw - - -def _as_1d(a, n: int, name: str) -> NDArray: - arr = np.asarray(a, dtype=np.float64).ravel() - if arr.shape[0] != n: - raise ValueError( - f"{name} has length {arr.shape[0]}, expected {n} (matching y)." - ) - return arr +from .._validation import validate_1d @dataclass @@ -120,7 +112,7 @@ def fit( ``eval_set`` entries are ``(X_val, y_val, model_input_val)``. """ y = np.asarray(y, dtype=np.float64).ravel() - x = _as_1d(model_input, len(y), "model_input") + x = validate_1d(model_input, len(y), "model_input") self._objective = self._make_objective() eval_sets: list[dict[str, Any]] | None = None @@ -144,7 +136,7 @@ def fit( { "X": X_e, "y": y_e, - "extra": {"model_input": _as_1d(x_e, len(y_e), "model_input")}, + "extra": {"model_input": validate_1d(x_e, len(y_e), "model_input")}, } ) @@ -184,6 +176,6 @@ def predict(self, X: NDArray | BinnedArray, model_input: NDArray) -> NDArray: """Evaluate the formula at ``predict_params(X)`` and ``model_input``.""" params = self.predict_params(X) names = self._objective.channel_names - x = _as_1d(model_input, next(iter(params.values())).shape[0], "model_input") + x = validate_1d(model_input, next(iter(params.values())).shape[0], "model_input") theta = tuple(params[name] for name in names) return np.asarray(self.formula(theta, x), dtype=np.float64).ravel() diff --git a/src/openboost/_models/_survival.py b/src/openboost/_models/_survival.py index d337ff2..00c9fdf 100644 --- a/src/openboost/_models/_survival.py +++ b/src/openboost/_models/_survival.py @@ -22,15 +22,7 @@ from .._objectives import WeibullAFTObjective from .._persistence import PersistenceMixin from .._trainer import TrainerConfig, fit_boosting, predict_raw - - -def _as_1d(a, n: int, name: str) -> NDArray: - arr = np.asarray(a, dtype=np.float64).ravel() - if arr.shape[0] != n: - raise ValueError( - f"{name} has length {arr.shape[0]}, expected {n} (matching y)." - ) - return arr +from .._validation import validate_1d @dataclass @@ -97,7 +89,7 @@ def fit( y = np.asarray(y, dtype=np.float64).ravel() if np.any(y <= 0): raise ValueError("WeibullAFT requires strictly positive times y.") - ev = None if event is None else _as_1d(event, len(y), "event") + ev = None if event is None else validate_1d(event, len(y), "event") self._objective = self._make_objective() eval_sets: list[dict[str, Any]] | None = None @@ -116,7 +108,7 @@ def fit( "(X_val, y_val[, event_val])." ) X_e, y_e = item[0], np.asarray(item[1], dtype=np.float64).ravel() - ev_e = (_as_1d(item[2], len(y_e), "event") + ev_e = (validate_1d(item[2], len(y_e), "event") if len(item) == 3 else None) eval_sets.append( {"X": X_e, "y": y_e, "extra": {"event": ev_e}} @@ -185,6 +177,6 @@ def nll( if self._objective is None: self._objective = self._make_objective() y = np.asarray(y, dtype=np.float64).ravel() - ev = None if event is None else _as_1d(event, len(y), "event") + ev = None if event is None else validate_1d(event, len(y), "event") raw = predict_raw(self, X) return self._objective.loss_value(raw, y, None, {"event": ev}) diff --git a/src/openboost/_validation.py b/src/openboost/_validation.py index 48c9c84..e719d99 100644 --- a/src/openboost/_validation.py +++ b/src/openboost/_validation.py @@ -278,6 +278,20 @@ def validate_sample_weight( return sample_weight.astype(np.float32) +def validate_1d(a: Any, n: int, name: str) -> NDArray: + """Validate a length-``n`` 1D float64 array for an auxiliary input. + + Used for per-sample vectors that must align with ``y`` (e.g. a formula's + ``model_input`` or a survival ``event`` indicator). + """ + arr = np.asarray(a, dtype=np.float64).ravel() + if arr.shape[0] != n: + raise ValueError( + f"{name} has length {arr.shape[0]}, expected {n} (matching y)." + ) + return arr + + def validate_eval_set( eval_set: list[tuple] | tuple | None, n_features: int, From 2d9d44b09cf6edef562b042ebfb4ab438c12c11e Mon Sep 17 00:00:00 2001 From: J Xu Date: Tue, 18 Aug 2026 00:08:43 -0700 Subject: [PATCH 2/2] fix: make the quality suite robust to OpenML's flaky name endpoint OpenML's name->id resolution endpoint returns frequent 503s, which was skipping most UCI quality datasets. Fetch by numeric data_id first (a different, more reliable endpoint) and fall back to name, with retries and backoff; drop the deactivated naval id and let YearPredictionMSD resolve its active version. Add a Modal CPU entrypoint (the quality suite is CPU-only) since Modal's datacenter network to OpenML is reliable. Result: 7 UCI datasets (+ california from the local baseline) are all statistically tied-or-better than NGBoost, closing the overlap-quality gate. Co-authored-by: Cursor --- benchmarks/bench_probabilistic.py | 92 ++++++++++++++++++++++--------- 1 file changed, 67 insertions(+), 25 deletions(-) diff --git a/benchmarks/bench_probabilistic.py b/benchmarks/bench_probabilistic.py index 3def88a..f1398a3 100644 --- a/benchmarks/bench_probabilistic.py +++ b/benchmarks/bench_probabilistic.py @@ -59,40 +59,63 @@ # ============================================================================= # Dataset registry (quality suite) # ============================================================================= -# NGBoost-paper-style UCI datasets, fetched from OpenML by name (best effort: -# a fetch failure skips the dataset with a note instead of crashing the run). +# NGBoost-paper-style UCI datasets. OpenML's name->id resolution endpoint is +# flaky (frequent 503s), so each row carries the numeric data_id as well; the +# loader fetches by id first (skips the flaky name endpoint) and falls back to +# name, with retries. A fetch failure skips the dataset with a note instead of +# crashing the run. data_ids verified against OpenML for version 1. OPENML_DATASETS = [ - # (short name, openml name, version) - ("boston", "boston", 1), - ("concrete", "Concrete_Compressive_Strength", 1), - ("energy", "energy-efficiency", 1), - ("kin8nm", "kin8nm", 1), - ("naval", "naval_propulsion_plant", 1), - ("power", "combined_cycle_power_plant", 1), - ("protein", "physicochemical-protein", 1), - ("wine", "wine_quality", 1), - ("yacht", "yacht_hydrodynamics", 1), + # (short name, openml name, version, data_id) + ("boston", "boston", 1, 531), + ("concrete", "Concrete_Compressive_Strength", 1, 4353), + ("energy", "energy-efficiency", 1, 1472), + ("kin8nm", "kin8nm", 1, 189), + # naval by-id 44898 is a deactivated version; use name resolution instead. + ("naval", "naval_propulsion_plant", 1, None), + ("power", "combined_cycle_power_plant", 1, None), + ("protein", "physicochemical-protein", 1, 42903), + ("wine", "wine_quality", 1, 287), + ("yacht", "yacht_hydrodynamics", 1, 42370), ] +def _fetch_openml_robust(name, version, data_id=None, as_frame=True, retries=3): + """Fetch from OpenML by id first (avoids the flaky name endpoint), then by + name, retrying with backoff. Raises the last error if all attempts fail.""" + import time + + from sklearn.datasets import fetch_openml + + attempts = [] + if data_id is not None: + attempts.append({"data_id": data_id}) + attempts.append({"name": name, "version": version}) + + last = None + for r in range(retries): + for kw in attempts: + try: + return fetch_openml(as_frame=as_frame, parser="auto", **kw) + except Exception as exc: # noqa: BLE001 - retry every failure mode + last = exc + time.sleep(2 * (r + 1)) + raise last + + def load_quality_datasets(quick: bool = False) -> tuple[list[dict], list[str]]: import numpy as np datasets: list[dict] = [] notes: list[str] = [] - from sklearn.datasets import fetch_california_housing, fetch_openml + from sklearn.datasets import fetch_california_housing - if quick: - wanted = OPENML_DATASETS[:1] - else: - wanted = OPENML_DATASETS + wanted = OPENML_DATASETS[:1] if quick else OPENML_DATASETS - for short, name, version in wanted: + for short, name, version, data_id in wanted: try: - bunch = fetch_openml(name=name, version=version, as_frame=True, - parser="auto") + bunch = _fetch_openml_robust(name, version, data_id) frame = bunch.frame.select_dtypes(include=[np.number]).dropna() target_col = bunch.target_names[0] if bunch.target_names else \ frame.columns[-1] @@ -120,8 +143,9 @@ def load_quality_datasets(quick: bool = False) -> tuple[list[dict], list[str]]: if not quick: try: - msd = fetch_openml(name="YearPredictionMSD", version=1, - as_frame=False, parser="auto") + # "active" version: name+version=1 does not resolve on OpenML. + msd = _fetch_openml_robust("YearPredictionMSD", "active", + data_id=None, as_frame=False) X = msd.data.astype("float64") y = msd.target.astype("float64") # Subsample for the quality suite: NGBoost's exact-split trees at @@ -309,7 +333,7 @@ def run_quality(quick: bool = False) -> dict: metrics["fit_time_s"] = round(fit_time, 3) per_split[lib].append(metrics) - def agg(lib, key): + def agg(lib, key, per_split=per_split): vals = [m[key] for m in per_split[lib]] return {"mean": float(np.mean(vals)), "std": float(np.std(vals))} @@ -360,8 +384,6 @@ def _time_to_nll(model_name, nll_curve_fn, target_nll): def run_speed(quick: bool = False, use_gpu: bool = False) -> dict: - import numpy as np - import openboost as ob # 1M hits the ≥1M acceptance gate. NGBoost exact-split trees took @@ -549,11 +571,31 @@ def _run_remote(suite: str = "all", quick: bool = False): flush=True) return run_suites(suite=suite, quick=quick, use_gpu=True) + @app.function(image=image, timeout=4 * 3600) + def _run_quality_remote(quick: bool = False): + # Quality suite is CPU-only (NGBoost is CPU); run it off-GPU. Modal's + # datacenter network to OpenML is reliable, unlike some local networks. + sys.path.insert(0, "/root") + import openboost as ob + + ob.set_backend("cpu") + print(f"backend={ob.get_backend()} suite=quality quick={quick}", + flush=True) + return run_suites(suite="quality", quick=quick, use_gpu=False) + @app.local_entrypoint() def main(suite: str = "all", quick: bool = False): report = _run_remote.remote(suite=suite, quick=quick) save_report(report, suite) + @app.local_entrypoint() + def quality(quick: bool = False): + report = _run_quality_remote.remote(quick=quick) + save_report(report, "quality") + q = report["suites"].get("quality", {}) + print(f"\nquality: {len(q.get('results', []))} datasets, " + f"{len(q.get('skipped', []))} skipped") + # ============================================================================= # Local execution