diff --git a/.gitignore b/.gitignore index be0e7d1..dcd7bb7 100644 --- a/.gitignore +++ b/.gitignore @@ -31,3 +31,8 @@ docs/architecture-performance-map.json docs/disparity-report.json docs/disparity-history.json docs/scaled-ci-history.json +docs/estimator-comparison.json + +# Raw ESTIMATOR| capture from the generated wide benchmark; the joined result +# is benchmarks/estimator_comparison.json. +benchmarks/estimator_flow_raw.txt diff --git a/benchmarks/README.md b/benchmarks/README.md index 19ba316..a51b355 100644 --- a/benchmarks/README.md +++ b/benchmarks/README.md @@ -225,6 +225,64 @@ python benchmarks/summarize_scaled_ci.py --baseline benchmarks/scaled_flow_basel Each run directory needs `scaled_flow.json` beside `scaled_comparison.json`. Build it from CI artifacts; a developer machine's numbers would make a fast laptop the standard CI has to meet. +## Every estimator + +The canonical benchmark races 12 estimators. `lib/scikit` exports 203, and a +claim about Flow against scikit-learn says little while the other 191 are +unmeasured. Four pieces cover the rest, all driven from one registry so the two +sides cannot drift apart. + +[`estimator_coverage.py`](estimator_coverage.py) parses every exported `*_fit` +out of `lib/scikit`, resolves its arguments from a table keyed by parameter +name, finds the scikit-learn class by name, and sorts each estimator into a +bucket. Nothing is dropped silently; every entry that is not raced carries its +reason. + +| bucket | meaning | +| --- | --- | +| `runnable` | arguments resolved and a scikit-learn counterpart exists | +| `different_shape` | takes a pipeline, a vectorizer input or a list of fitted models first, so it is not an estimator over a feature matrix | +| `flow_only` | Flow implements it and scikit-learn has no equivalent | +| `simplified` | the implementation's own comments call it a simplified stand-in | +| `blocked` | the signature is not resolved yet, with the missing parameter named | + +The `simplified` bucket is detected from the source rather than listed, so it +stays true as the implementations are filled in. It matters: `spectral_biclustering` +thresholds row and column means where scikit-learn does an SVD and k-means, and +came out at 25106x. Three of the four it catches would otherwise have been the +widest wins in the whole matrix. + +[`generate_estimator_bench.py`](generate_estimator_bench.py) emits the Flow +timing blocks from that registry, split across several files because Flow issue +#469 miscompiles some functions once a program grows past a certain size. Each +block prints one line and flushes it. Without the flush, one estimator trapping +takes the whole file's buffered output with it and the run looks empty rather +than partial, which is how a RANSAC crash first presented. + +[`bench_estimators_sklearn.py`](bench_estimators_sklearn.py) times the +scikit-learn side of the same registry, and +[`compare_estimators.py`](compare_estimators.py) joins them. + +``` +python benchmarks/estimator_coverage.py +python benchmarks/generate_estimator_bench.py +for f in benchmarks/generated/bench_estimators_*.flow; do flow run "$f"; done | tee /tmp/flow.txt +python benchmarks/bench_estimators_sklearn.py +python benchmarks/compare_estimators.py /tmp/flow.txt +``` + +Read the result for what it is. These rows have no parity contract, no declared +tolerances and no disparity report: each library runs its own defaults over the +same data. That answers whether a Flow implementation is in the same +performance league and says nothing about whether it computes the same thing. +The per-estimator numerical contract remains the canonical benchmark's job, and +`estimator_comparison.json` says so in its own `contract` field. + +The breadth is worth having for correctness as much as for speed. The first run +of it found a heap-buffer-overflow in `_solve_lstsq_qr`, which assumed a design +has at least as many rows as columns; RANSAC reaches it by fitting 5 sampled +rows against the 10 features of diabetes. + ## Pages publication [`publish_headline_v2.py`](publish_headline_v2.py) validates the committed canonical benchmark and architecture map, then copies the JSON artifacts into `docs/` for the static site. The public benchmark and architecture pages render those artifacts directly instead of embedding hand-maintained timing claims. diff --git a/benchmarks/bench_estimators_sklearn.py b/benchmarks/bench_estimators_sklearn.py new file mode 100644 index 0000000..a99c9cc --- /dev/null +++ b/benchmarks/bench_estimators_sklearn.py @@ -0,0 +1,164 @@ +#!/usr/bin/env python3 +"""Time the scikit-learn counterpart of every runnable registry entry. + +Reads `estimator_coverage.json` so the two sides cannot drift apart: an +estimator is raced here exactly when the Flow generator emitted a block for it. + +Each estimator is constructed with its own defaults rather than a translation +of the Flow hyperparameters. The Flow side passes a fixed set of small values, +and matching them one by one across 172 estimators would be a second source of +error without making the comparison more honest: this measures each library +doing its own default thing on the same data, and the per-estimator parameter +contract stays the canonical benchmark's job. +""" +from __future__ import annotations + +import argparse +import json +import time +import warnings +from pathlib import Path + +import numpy as np +from sklearn.datasets import load_diabetes, load_iris +from sklearn.utils import all_estimators + +# IterativeImputer is behind an experimental flag. +from sklearn.experimental import enable_iterative_imputer # noqa: F401 + +ROOT = Path(__file__).resolve().parents[1] +REGISTRY = ROOT / "benchmarks" / "estimator_coverage.json" + +# Meta-estimators need a base estimator, and a few refuse their own defaults on +# this data. A shallow tree keeps the wrapper's own overhead visible rather than +# burying it under the base estimator's work. +def _constructors() -> dict: + from sklearn.linear_model import Ridge + from sklearn.tree import DecisionTreeClassifier, DecisionTreeRegressor + import numpy as np + + tree_c = lambda: DecisionTreeClassifier(max_depth=3) + tree_r = lambda: DecisionTreeRegressor(max_depth=3) + return { + "ClassifierChain": lambda c: c(tree_c()), + "MultiOutputClassifier": lambda c: c(tree_c()), + "MultiOutputRegressor": lambda c: c(tree_r()), + "OneVsOneClassifier": lambda c: c(tree_c()), + "OneVsRestClassifier": lambda c: c(tree_c()), + "OutputCodeClassifier": lambda c: c(tree_c()), + "RegressorChain": lambda c: c(tree_r()), + "RFE": lambda c: c(tree_c()), + "RFECV": lambda c: c(tree_c()), + "SequentialFeatureSelector": lambda c: c(tree_c(), n_features_to_select=2), + "StackingClassifier": lambda c: c(estimators=[("t", tree_c())]), + "StackingRegressor": lambda c: c(estimators=[("r", Ridge())]), + "VotingClassifier": lambda c: c(estimators=[("t", tree_c())]), + "VotingRegressor": lambda c: c(estimators=[("r", Ridge())]), + "SparseCoder": lambda c: c(dictionary=np.eye(4)), + # nu=0.5 is infeasible for this class balance. + "NuSVC": lambda c: c(nu=0.1), + } + + +CLASSIFICATION = {"classification"} + + +def dataset_kind(entry: dict) -> str: + names = [p["name"] for p in entry["fit"]["parameters"]] + types = {p["name"]: p["type"] for p in entry["fit"]["parameters"]} + if "y" not in names and "Y" not in names: + return "unsupervised" + if "Y" in names: + return "multioutput_class" if "classifier" in entry["flow_estimator"] or "chain" in entry["flow_estimator"] else "multioutput" + if "n_classes" in names or types.get("y") == "ptr": + return "classification" + return "regression" + + +def timed(fn, repeats: int) -> float: + best = float("inf") + for _ in range(repeats): + t0 = time.perf_counter() + fn() + best = min(best, (time.perf_counter() - t0) * 1000.0) + return best + + +def main() -> int: + ap = argparse.ArgumentParser() + ap.add_argument("--repeats", type=int, default=5) + ap.add_argument("--output", type=Path, default=ROOT / "benchmarks" / "estimators_sklearn.json") + args = ap.parse_args() + + registry = json.loads(REGISTRY.read_text()) + runnable = [e for e in registry["entries"] if e["bucket"] == "runnable"] + classes = dict(all_estimators()) + constructors = _constructors() + + iris = load_iris() + diabetes = load_diabetes() + data = { + "classification": (iris.data.astype(np.float64), iris.target.astype(np.float64)), + "unsupervised": (iris.data.astype(np.float64), None), + "regression": (diabetes.data.astype(np.float64), diabetes.target.astype(np.float64)), + } + y_multi = np.column_stack([diabetes.target, diabetes.target * 0.5]) + data["multioutput"] = (diabetes.data.astype(np.float64), y_multi) + y_labels = np.column_stack([iris.target, (iris.target + 1) % 3]) + data["multioutput_class"] = (iris.data.astype(np.float64), y_labels) + + rows = [] + for entry in runnable: + name = entry["sklearn_estimator"] + cls = classes.get(name) + if cls is None: + rows.append({"flow_estimator": entry["flow_estimator"], "sklearn_estimator": name, + "status": "unavailable", "reason": "not in sklearn.utils.all_estimators()"}) + continue + kind = dataset_kind(entry) + X, y = data[kind] + try: + with warnings.catch_warnings(): + warnings.simplefilter("ignore") + build = constructors.get(name) + model = build(cls) if build else cls() + fit_ms = timed((lambda: model.fit(X)) if y is None else (lambda: model.fit(X, y)), args.repeats) + pred_ms = 0.0 + for method in ("predict", "transform"): + if hasattr(model, method): + try: + pred_ms = timed(lambda m=method: getattr(model, m)(X), args.repeats) + except Exception: + pred_ms = 0.0 + break + rows.append({ + "flow_estimator": entry["flow_estimator"], + "sklearn_estimator": name, + "dataset": kind, + "fit_ms": fit_ms, + "pred_ms": pred_ms, + "timing_unit": "ms", + "status": "ok", + }) + except Exception as exc: # the estimator refused this data or these defaults + rows.append({ + "flow_estimator": entry["flow_estimator"], + "sklearn_estimator": name, + "dataset": kind, + "status": "failed", + "reason": f"{type(exc).__name__}: {exc}"[:200], + }) + + ok = sum(1 for r in rows if r["status"] == "ok") + args.output.write_text(json.dumps({"schema_version": 1, "repeats": args.repeats, + "counts": {"rows": len(rows), "ok": ok}, + "rows": rows}, indent=2) + "\n") + print(f"timed {ok} of {len(rows)} scikit-learn estimators -> {args.output.name}") + for r in rows: + if r["status"] != "ok": + print(f" {r['status']}: {r['flow_estimator']} -> {r['sklearn_estimator']}: {r.get('reason','')[:110]}") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/benchmarks/compare_estimators.py b/benchmarks/compare_estimators.py new file mode 100644 index 0000000..3973b57 --- /dev/null +++ b/benchmarks/compare_estimators.py @@ -0,0 +1,151 @@ +#!/usr/bin/env python3 +"""Join the wide Flow and scikit-learn estimator timings into one artifact. + +This is a breadth measurement, and it says so in its own output. The canonical +19 rows carry a parity contract, declared tolerances and a disparity report; +these rows carry none of that. An estimator here is timed on the same data with +each library's own defaults, which answers "is the Flow implementation in the +same performance league" and does not answer "does it compute the same thing". + +`speedup` is `sklearn_ms / flow_ms` over fit plus predict, as everywhere else. +""" +from __future__ import annotations + +import argparse +import json +import re +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] +LINE = re.compile(r"^ESTIMATOR\|([a-z_0-9]+)\|([0-9.eE+-]+)\|([0-9.eE+-]+)\|(\d+)\|ok$") + +# flow_now_ns resolves to about a microsecond through this harness, so a total +# at or below this is a rounding artifact rather than a measurement. +RESOLUTION_MS = 0.00002 + + +def parse_flow(path: Path) -> dict[str, dict]: + rows: dict[str, dict] = {} + for line in path.read_text().splitlines(): + m = LINE.match(line.strip()) + if not m: + continue + name, fit, pred, reps = m.group(1), float(m.group(2)), float(m.group(3)), int(m.group(4)) + prior = rows.get(name) + # Repeated runs of the same file: keep the fastest, as every other + # harness here does. + if prior is None or fit + pred < prior["fit_ms"] + prior["pred_ms"]: + rows[name] = {"fit_ms": fit, "pred_ms": pred, "repeats": reps} + return rows + + +def main() -> int: + ap = argparse.ArgumentParser() + ap.add_argument("flow", type=Path, help="captured ESTIMATOR| lines from the generated benchmarks") + ap.add_argument("--sklearn", type=Path, default=ROOT / "benchmarks" / "estimators_sklearn.json") + ap.add_argument("--registry", type=Path, default=ROOT / "benchmarks" / "estimator_coverage.json") + ap.add_argument("--output", type=Path, default=ROOT / "benchmarks" / "estimator_comparison.json") + ap.add_argument("--note", default="", help="recorded verbatim in the artifact, for measurement conditions") + args = ap.parse_args() + + flow = parse_flow(args.flow) + sk = {r["flow_estimator"]: r for r in json.loads(args.sklearn.read_text())["rows"]} + registry = json.loads(args.registry.read_text()) + + rows = [] + for entry in registry["entries"]: + name = entry["flow_estimator"] + if entry["bucket"] != "runnable": + row = {"flow_estimator": name, "status": entry["bucket"], "reason": entry.get("reason", "")} + if entry.get("sklearn_estimator"): + row["sklearn_estimator"] = entry["sklearn_estimator"] + # A simplified implementation still gets its times shown, because + # hiding them reads as concealment. They carry no ratio, because a + # ratio against a different algorithm is not a speedup. + f, k = flow.get(name), sk.get(name) + if entry["bucket"] == "simplified" and f and k and k.get("status") == "ok": + row["flow_ms"] = f["fit_ms"] + f["pred_ms"] + row["sklearn_ms"] = k["fit_ms"] + k["pred_ms"] + row["timing_unit"] = "ms" + rows.append(row) + continue + f, s = flow.get(name), sk.get(name) + if f is None: + rows.append({"flow_estimator": name, "sklearn_estimator": entry["sklearn_estimator"], + "status": "flow_missing", "reason": "no ESTIMATOR line in the Flow output"}) + continue + if s is None or s.get("status") != "ok": + rows.append({"flow_estimator": name, "sklearn_estimator": entry["sklearn_estimator"], + "status": "sklearn_missing", "reason": (s or {}).get("reason", "absent")}) + continue + flow_ms = f["fit_ms"] + f["pred_ms"] + sk_ms = s["fit_ms"] + s["pred_ms"] + # The harness repeats a fast estimator until the clock can see it, so + # this should not trigger any more. It stays as the backstop it always + # was: a Flow side at the floor has been rounded rather than measured. + if flow_ms <= RESOLUTION_MS: + rows.append({ + "flow_estimator": name, + "sklearn_estimator": entry["sklearn_estimator"], + "dataset": s["dataset"], + "flow_ms": flow_ms, + "sklearn_ms": sk_ms, + "speedup": None, + "timing_unit": "ms", + "status": "below_resolution", + "reason": f"Flow side at or under {RESOLUTION_MS} ms, which is the clock's floor here", + }) + continue + rows.append({ + "flow_estimator": name, + "sklearn_estimator": entry["sklearn_estimator"], + "dataset": s["dataset"], + "flow_ms": flow_ms, + "sklearn_ms": sk_ms, + "speedup": (sk_ms / flow_ms) if flow_ms > 0 else None, + "flow_repeats": f["repeats"], + "timing_unit": "ms", + "status": "ok", + }) + + compared = [r for r in rows if r["status"] == "ok" and r["speedup"] is not None] + unmeasured = [r for r in rows if r["status"] == "below_resolution"] + wins = sum(1 for r in compared if r["speedup"] >= 1.0) + payload = { + "schema_version": 1, + "contract": "breadth timing only; no parity contract, no declared tolerances, " + "each library on its own defaults over the same data", + "measurement": "fastest of several rounds per estimator on both sides, which is " + "what survives a machine that is not idle", + "note": args.note, + "counts": { + "registry_estimators": len(registry["entries"]), + "compared": len(compared), + "flow_wins": wins, + "sklearn_wins": len(compared) - wins, + "below_resolution": len(unmeasured), + "simplified": sum(1 for r in rows if r["status"] == "simplified"), + }, + "rows": rows, + } + args.output.write_text(json.dumps(payload, indent=2) + "\n") + print(f"compared {len(compared)} estimators: {wins} Flow wins, {len(compared) - wins} scikit-learn wins") + simplified = [r for r in rows if r["status"] == "simplified"] + if simplified: + print(f"{len(simplified)} rows excluded, implementation declared simplified in its own comments:") + for r in sorted(simplified, key=lambda r: r["flow_estimator"]): + times = (f" flow={r['flow_ms']:.3f} sklearn={r['sklearn_ms']:.3f}" + if "flow_ms" in r else "") + print(f" {r['flow_estimator']:32s}{times}") + if unmeasured: + print(f"{len(unmeasured)} rows left unranked, Flow side under the clock's resolution:") + for r in sorted(unmeasured, key=lambda r: r["flow_estimator"]): + print(f" {r['flow_estimator']:32s} flow={r['flow_ms']:.4f} sklearn={r['sklearn_ms']:9.3f}") + losers = sorted((r for r in compared if r["speedup"] < 1.0), key=lambda r: r["speedup"]) + for r in losers: + print(f" {r['speedup']:6.2f}x {r['flow_estimator']:32s} flow={r['flow_ms']:9.3f} sklearn={r['sklearn_ms']:9.3f}") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/benchmarks/estimator_comparison.json b/benchmarks/estimator_comparison.json new file mode 100644 index 0000000..8b3338c --- /dev/null +++ b/benchmarks/estimator_comparison.json @@ -0,0 +1,2080 @@ +{ + "schema_version": 1, + "contract": "breadth timing only; no parity contract, no declared tolerances, each library on its own defaults over the same data", + "measurement": "fastest of several rounds per estimator on both sides, which is what survives a machine that is not idle", + "note": "Apple M4 Max; Flow at -O3 with adaptive repeats, fastest of 3 rounds; scikit-learn 1.9.0 wheel, fastest of 3; developer machine under load. CI is the authority for anything published.", + "counts": { + "registry_estimators": 203, + "compared": 166, + "flow_wins": 145, + "sklearn_wins": 21, + "below_resolution": 2, + "simplified": 4 + }, + "rows": [ + { + "flow_estimator": "adaboost_classifier", + "sklearn_estimator": "AdaBoostClassifier", + "dataset": "classification", + "flow_ms": 0.293650003, + "sklearn_ms": 23.29525000823196, + "speedup": 79.32998389321304, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "adaboost_regressor", + "sklearn_estimator": "AdaBoostRegressor", + "dataset": "regression", + "flow_ms": 8.074000001, + "sklearn_ms": 15.209208999294788, + "speedup": 1.8837266531348849, + "flow_repeats": 1, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "additive_chi2_sampler", + "status": "different_shape", + "reason": "takes i32 first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "AdditiveChi2Sampler" + }, + { + "flow_estimator": "affinity_propagation", + "sklearn_estimator": "AffinityPropagation", + "dataset": "unsupervised", + "flow_ms": 139.29699707, + "sklearn_ms": 4.076251003425568, + "speedup": 0.029263021379973875, + "flow_repeats": 1, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "agglomerative_clustering", + "sklearn_estimator": "AgglomerativeClustering", + "dataset": "unsupervised", + "flow_ms": 0.128594995, + "sklearn_ms": 0.3488329966785386, + "speedup": 2.7126483163558475, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "ard_regression", + "sklearn_estimator": "ARDRegression", + "dataset": "regression", + "flow_ms": 0.63499999, + "sklearn_ms": 3.3072089863708243, + "speedup": 5.208203210161979, + "flow_repeats": 1, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "bagging_classifier", + "sklearn_estimator": "BaggingClassifier", + "dataset": "classification", + "flow_ms": 0.14565, + "sklearn_ms": 6.647166999755427, + "speedup": 45.63794713186012, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "bagging_regressor", + "sklearn_estimator": "BaggingRegressor", + "dataset": "regression", + "flow_ms": 7.729000173, + "sklearn_ms": 12.465458989026956, + "speedup": 1.6128164976076727, + "flow_repeats": 1, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "bayesian_gaussian_mixture", + "sklearn_estimator": "BayesianGaussianMixture", + "dataset": "unsupervised", + "flow_ms": 0.115610001, + "sklearn_ms": 1.1429580044932663, + "speedup": 9.886324665746402, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "bayesian_ridge", + "sklearn_estimator": "BayesianRidge", + "dataset": "regression", + "flow_ms": 0.096429996, + "sklearn_ms": 0.45233400305733085, + "speedup": 4.690801844037521, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "bernoulli_nb", + "sklearn_estimator": "BernoulliNB", + "dataset": "classification", + "flow_ms": 0.002665, + "sklearn_ms": 0.5652910040225834, + "speedup": 212.1166994456223, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "bernoulli_rbm", + "sklearn_estimator": "BernoulliRBM", + "dataset": "unsupervised", + "flow_ms": 0.456850013, + "sklearn_ms": 8.775667010922916, + "speedup": 19.20907685499599, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "birch", + "sklearn_estimator": "Birch", + "dataset": "unsupervised", + "flow_ms": 0.012905001, + "sklearn_ms": 1.0970829898724332, + "speedup": 85.01223594422297, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "bisecting_kmeans", + "sklearn_estimator": "BisectingKMeans", + "dataset": "unsupervised", + "flow_ms": 0.019985, + "sklearn_ms": 4.4658750121016055, + "speedup": 223.46134661504158, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "calibrated_classifier_cv", + "sklearn_estimator": "CalibratedClassifierCV", + "dataset": "classification", + "flow_ms": 0.34106499, + "sklearn_ms": 10.508707986446097, + "speedup": 30.811453226102444, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "categorical_nb", + "sklearn_estimator": "CategoricalNB", + "dataset": "classification", + "flow_ms": 0.003265, + "sklearn_ms": 0.68374999682419, + "speedup": 209.41806947142112, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "cca", + "sklearn_estimator": "CCA", + "dataset": "multioutput", + "flow_ms": 0.122035002, + "sklearn_ms": 0.4619169922079891, + "speedup": 3.785118897347083, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "classifier_chain", + "sklearn_estimator": "ClassifierChain", + "dataset": "multioutput_class", + "flow_ms": 2.099999905, + "sklearn_ms": 0.9616669995011762, + "speedup": 0.45793668714531505, + "flow_repeats": 1, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "column_transformer", + "status": "different_shape", + "reason": "takes ColumnTransformer first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "ColumnTransformer" + }, + { + "flow_estimator": "complement_nb", + "sklearn_estimator": "ComplementNB", + "dataset": "classification", + "flow_ms": 0.00224, + "sklearn_ms": 0.422583005274646, + "speedup": 188.65312735475268, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "count_vectorizer", + "status": "different_shape", + "reason": "takes ptr first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "CountVectorizer" + }, + { + "flow_estimator": "dbscan", + "sklearn_estimator": "DBSCAN", + "dataset": "unsupervised", + "flow_ms": 0.074029997, + "sklearn_ms": 0.4175000067334622, + "speedup": 5.639605884807239, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "decision_tree_classifier", + "sklearn_estimator": "DecisionTreeClassifier", + "dataset": "classification", + "flow_ms": 0.012285, + "sklearn_ms": 0.30191600671969354, + "speedup": 24.575987522970575, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "decision_tree_regressor", + "sklearn_estimator": "DecisionTreeRegressor", + "dataset": "regression", + "flow_ms": 0.661349993, + "sklearn_ms": 1.2142920022597536, + "speedup": 1.8360807667835777, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "dict_vectorizer", + "status": "different_shape", + "reason": "takes ptr > first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "DictVectorizer" + }, + { + "flow_estimator": "dictionary_learning", + "sklearn_estimator": "DictionaryLearning", + "dataset": "unsupervised", + "flow_ms": 0.37209999699999996, + "sklearn_ms": 199.1827499878127, + "speedup": 535.2936081528986, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "discriminant_lda", + "sklearn_estimator": "LinearDiscriminantAnalysis", + "dataset": "classification", + "flow_ms": 0.00833, + "sklearn_ms": 0.3475420089671388, + "speedup": 41.72172976796384, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "dummy_classifier", + "status": "different_shape", + "reason": "takes ptr first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "DummyClassifier" + }, + { + "flow_estimator": "dummy_regressor", + "status": "different_shape", + "reason": "takes ptr first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "DummyRegressor" + }, + { + "flow_estimator": "elastic_net_cv", + "sklearn_estimator": "ElasticNetCV", + "dataset": "regression", + "flow_ms": 167.832992554, + "sklearn_ms": 13.482125010341406, + "speedup": 0.08033060011131932, + "flow_repeats": 1, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "elastic_net", + "sklearn_estimator": "ElasticNet", + "dataset": "regression", + "flow_ms": 0.206999997, + "sklearn_ms": 0.21425001614261419, + "speedup": 1.0350242475733669, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "elliptic_envelope", + "status": "simplified", + "reason": "the implementation's own comments call it simplified, so timing it against scikit-learn's algorithm compares two different things", + "sklearn_estimator": "EllipticEnvelope", + "flow_ms": 0.020725001, + "sklearn_ms": 8.94445800804533, + "timing_unit": "ms" + }, + { + "flow_estimator": "empirical_covariance", + "sklearn_estimator": "EmpiricalCovariance", + "dataset": "unsupervised", + "flow_ms": 0.002105, + "sklearn_ms": 0.12658300693146884, + "speedup": 60.13444509808496, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "extra_tree_classifier", + "sklearn_estimator": "ExtraTreeClassifier", + "dataset": "classification", + "flow_ms": 0.00545, + "sklearn_ms": 0.25049901159945875, + "speedup": 45.963121394396104, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "extra_tree_regressor", + "sklearn_estimator": "ExtraTreeRegressor", + "dataset": "regression", + "flow_ms": 0.020909999000000002, + "sklearn_ms": 0.6777079979656264, + "speedup": 32.410714030432345, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "extra_trees_classifier", + "sklearn_estimator": "ExtraTreesClassifier", + "dataset": "classification", + "flow_ms": 0.159500001, + "sklearn_ms": 30.058165997616015, + "speedup": 188.45245021419163, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "extra_trees_regressor", + "sklearn_estimator": "ExtraTreesRegressor", + "dataset": "regression", + "flow_ms": 7.530000068, + "sklearn_ms": 68.69250000454485, + "speedup": 9.122509878381697, + "flow_repeats": 1, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "factor_analysis", + "sklearn_estimator": "FactorAnalysis", + "dataset": "unsupervised", + "flow_ms": 0.015670001, + "sklearn_ms": 0.48095900274347514, + "speedup": 30.69297843334376, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "fast_ica", + "sklearn_estimator": "FastICA", + "dataset": "unsupervised", + "flow_ms": 0.014385, + "sklearn_ms": 0.6858330016257241, + "speedup": 47.67695527464193, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "feature_agglomeration", + "sklearn_estimator": "FeatureAgglomeration", + "dataset": "unsupervised", + "flow_ms": 0.001525, + "sklearn_ms": 0.24379098613280803, + "speedup": 159.86294172643147, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "feature_union", + "status": "different_shape", + "reason": "takes FeatureUnion first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "FeatureUnion" + }, + { + "flow_estimator": "gamma_regressor", + "sklearn_estimator": "GammaRegressor", + "dataset": "regression", + "flow_ms": 0.5390499790000001, + "sklearn_ms": 0.7119990041246638, + "speedup": 1.3208404264211346, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "gaussian_mixture", + "sklearn_estimator": "GaussianMixture", + "dataset": "unsupervised", + "flow_ms": 0.022855001, + "sklearn_ms": 0.8785410027485341, + "speedup": 38.43977091703186, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "gaussian_nb", + "sklearn_estimator": "GaussianNB", + "dataset": "classification", + "flow_ms": 0.002875, + "sklearn_ms": 0.33529099891893566, + "speedup": 116.62295614571676, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "gaussian_process_classifier", + "sklearn_estimator": "GaussianProcessClassifier", + "dataset": "regression", + "flow_ms": 59.547999621, + "sklearn_ms": 2805.2895829896443, + "speedup": 47.10971990401404, + "flow_repeats": 1, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "gaussian_process_regressor", + "sklearn_estimator": "GaussianProcessRegressor", + "dataset": "regression", + "flow_ms": 15.13599968, + "sklearn_ms": 72.50166700396221, + "speedup": 4.790015098887886, + "flow_repeats": 1, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "gaussian_random_projection", + "status": "different_shape", + "reason": "takes i32 first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "GaussianRandomProjection" + }, + { + "flow_estimator": "gradient_boosting_classifier", + "sklearn_estimator": "GradientBoostingClassifier", + "dataset": "classification", + "flow_ms": 0.250799992, + "sklearn_ms": 62.244832995929755, + "speedup": 248.1851474537916, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "gradient_boosting_regressor", + "sklearn_estimator": "GradientBoostingRegressor", + "dataset": "regression", + "flow_ms": 6.671000181999999, + "sklearn_ms": 47.3774999845773, + "speedup": 7.102008498277883, + "flow_repeats": 1, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "graphical_lasso_cv", + "status": "flow_only", + "reason": "scikit-learn spells this GraphicalLassoCV; covered by that entry" + }, + { + "flow_estimator": "graphical_lasso", + "sklearn_estimator": "GraphicalLasso", + "dataset": "unsupervised", + "flow_ms": 0.00982, + "sklearn_ms": 0.3913340042345226, + "speedup": 39.85071326217134, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "hdbscan", + "sklearn_estimator": "HDBSCAN", + "dataset": "unsupervised", + "flow_ms": 5.703999996, + "sklearn_ms": 0.7184580026660115, + "speedup": 0.1259568729259886, + "flow_repeats": 1, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "hist_gradient_boosting_classifier", + "sklearn_estimator": "HistGradientBoostingClassifier", + "dataset": "classification", + "flow_ms": 0.317299995, + "sklearn_ms": 722.8900420013815, + "speedup": 2278.2541865510634, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "hist_gradient_boosting_regressor", + "sklearn_estimator": "HistGradientBoostingRegressor", + "dataset": "regression", + "flow_ms": 1.983999979, + "sklearn_ms": 839.5987920084735, + "speedup": 423.18487948354635, + "flow_repeats": 1, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "huber_regressor", + "sklearn_estimator": "HuberRegressor", + "dataset": "regression", + "flow_ms": 2.4989999789999997, + "sklearn_ms": 6.450457018218003, + "speedup": 2.581215315095448, + "flow_repeats": 1, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "incremental_pca_partial", + "status": "different_shape", + "reason": "takes IncrementalPCA first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "IncrementalPCA" + }, + { + "flow_estimator": "isolation_forest", + "sklearn_estimator": "IsolationForest", + "dataset": "unsupervised", + "flow_ms": 0.090750001, + "sklearn_ms": 48.3180420123972, + "speedup": 532.4302091456418, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "isomap", + "sklearn_estimator": "Isomap", + "dataset": "unsupervised", + "flow_ms": 5.301000118, + "sklearn_ms": 5.399625006248243, + "speedup": 1.0186049586970114, + "flow_repeats": 1, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "isotonic", + "status": "different_shape", + "reason": "takes ptr first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "IsotonicRegression" + }, + { + "flow_estimator": "iterative_imputer", + "sklearn_estimator": "IterativeImputer", + "dataset": "unsupervised", + "flow_ms": 0.001765, + "sklearn_ms": 2.774332999251783, + "speedup": 1571.8600562333047, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "kbins_discretizer", + "sklearn_estimator": "KBinsDiscretizer", + "dataset": "unsupervised", + "flow_ms": 0.00756, + "sklearn_ms": 0.9220420179190114, + "speedup": 121.96322988346712, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "kernel_density", + "sklearn_estimator": "KernelDensity", + "dataset": "unsupervised", + "flow_ms": 0.0, + "sklearn_ms": 0.10016700252890587, + "speedup": null, + "timing_unit": "ms", + "status": "below_resolution", + "reason": "Flow side at or under 2e-05 ms, which is the clock's floor here" + }, + { + "flow_estimator": "kernel_pca", + "sklearn_estimator": "KernelPCA", + "dataset": "unsupervised", + "flow_ms": 1.0669000149999999, + "sklearn_ms": 1.8091249949065968, + "speedup": 1.6956837280638684, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "kernel_ridge", + "sklearn_estimator": "KernelRidge", + "dataset": "regression", + "flow_ms": 2.2022999519999997, + "sklearn_ms": 1.3769590004812926, + "speedup": 0.6252368117389364, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "kernel_svc", + "sklearn_estimator": "SVC", + "dataset": "classification", + "flow_ms": 0.11696500000000001, + "sklearn_ms": 0.6641249929089099, + "speedup": 5.677980531859188, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "kernel_svc_multi", + "sklearn_estimator": "SVC", + "dataset": "classification", + "flow_ms": 0.086784996, + "sklearn_ms": 0.6252509920159355, + "speedup": 7.204597808772561, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "kmeans", + "sklearn_estimator": "KMeans", + "dataset": "unsupervised", + "flow_ms": 0.011829999, + "sklearn_ms": 0.7295419927686453, + "speedup": 61.66881271660676, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "kneighbors_transformer", + "sklearn_estimator": "KNeighborsTransformer", + "dataset": "unsupervised", + "flow_ms": 0.436674982, + "sklearn_ms": 0.36108397762291133, + "speedup": 0.8268941260823395, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "knn_classifier", + "sklearn_estimator": "KNeighborsClassifier", + "dataset": "classification", + "flow_ms": 0.112904995, + "sklearn_ms": 1.0927499970421195, + "speedup": 9.67849116898787, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "knn_imputer", + "sklearn_estimator": "KNNImputer", + "dataset": "unsupervised", + "flow_ms": 0.0011749999999999998, + "sklearn_ms": 0.1957500062417239, + "speedup": 166.59574999295654, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "knn_regressor", + "sklearn_estimator": "KNeighborsRegressor", + "dataset": "regression", + "flow_ms": 1.247499936, + "sklearn_ms": 1.2899580033263192, + "speedup": 1.0340345246529288, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "label_binarizer", + "status": "different_shape", + "reason": "takes ptr first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "LabelBinarizer" + }, + { + "flow_estimator": "label_encoder", + "status": "different_shape", + "reason": "takes ptr first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "LabelEncoder" + }, + { + "flow_estimator": "label_propagation", + "sklearn_estimator": "LabelPropagation", + "dataset": "classification", + "flow_ms": 0.112865001, + "sklearn_ms": 0.9385419834870845, + "speedup": 8.315615781433294, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "label_spreading", + "sklearn_estimator": "LabelSpreading", + "dataset": "classification", + "flow_ms": 0.841149986, + "sklearn_ms": 0.8673739939695224, + "speedup": 1.0311763756832808, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "lars_cv", + "sklearn_estimator": "LarsCV", + "dataset": "regression", + "flow_ms": 0.136189997, + "sklearn_ms": 2.6954999921144918, + "speedup": 19.79220244871943, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "lars", + "sklearn_estimator": "Lars", + "dataset": "regression", + "flow_ms": 0.016220000000000002, + "sklearn_ms": 0.5449580057756975, + "speedup": 33.597904178526356, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "lasso_cv", + "sklearn_estimator": "LassoCV", + "dataset": "regression", + "flow_ms": 1.776449919, + "sklearn_ms": 19.727291000890546, + "speedup": 11.10489566291598, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "lasso", + "sklearn_estimator": "Lasso", + "dataset": "regression", + "flow_ms": 0.032230001, + "sklearn_ms": 0.2537920081522316, + "speedup": 7.87440273899562, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "lasso_lars_cv", + "sklearn_estimator": "LassoLarsCV", + "dataset": "regression", + "flow_ms": 3.488999916, + "sklearn_ms": 3.7985839881002903, + "speedup": 1.0887314644751314, + "flow_repeats": 1, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "lasso_lars", + "sklearn_estimator": "LassoLars", + "dataset": "regression", + "flow_ms": 0.046024998, + "sklearn_ms": 0.4093340103281662, + "speedup": 8.893732278449338, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "lasso_lars_ic", + "sklearn_estimator": "LassoLarsIC", + "dataset": "regression", + "flow_ms": 0.827899959, + "sklearn_ms": 1.0350410011596978, + "speedup": 1.2502005706219605, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "lda", + "sklearn_estimator": "LatentDirichletAllocation", + "dataset": "unsupervised", + "flow_ms": 6.449000026999999, + "sklearn_ms": 103.45783299999312, + "speedup": 16.042461244665322, + "flow_repeats": 1, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "ledoit_wolf_cv", + "status": "flow_only", + "reason": "cross-validated variant scikit-learn does not expose" + }, + { + "flow_estimator": "ledoit_wolf_estimator", + "sklearn_estimator": "LedoitWolf", + "dataset": "unsupervised", + "flow_ms": 0.006625, + "sklearn_ms": 0.17004100664053112, + "speedup": 25.66656704008017, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "linear_regression", + "sklearn_estimator": "LinearRegression", + "dataset": "regression", + "flow_ms": 0.01975, + "sklearn_ms": 0.26516699290368706, + "speedup": 13.42617685588289, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "linear_svc", + "sklearn_estimator": "LinearSVC", + "dataset": "classification", + "flow_ms": 0.019065, + "sklearn_ms": 0.3953749983338639, + "speedup": 20.73826374685885, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "linear_svc_multi", + "sklearn_estimator": "LinearSVC", + "dataset": "classification", + "flow_ms": 0.06523, + "sklearn_ms": 0.37720799446105957, + "speedup": 5.782737919071893, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "linear_svr", + "sklearn_estimator": "LinearSVR", + "dataset": "regression", + "flow_ms": 0.278600007, + "sklearn_ms": 0.23625099856872112, + "speedup": 0.8479935126804254, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "lle", + "sklearn_estimator": "LocallyLinearEmbedding", + "dataset": "unsupervised", + "flow_ms": 10.61400032, + "sklearn_ms": 7.336875001783483, + "speedup": 0.6912450330304383, + "flow_repeats": 1, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "local_outlier_factor", + "sklearn_estimator": "LocalOutlierFactor", + "dataset": "unsupervised", + "flow_ms": 5.397999763, + "sklearn_ms": 0.395416995161213, + "speedup": 0.07325250324602747, + "flow_repeats": 1, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "logistic_inference", + "status": "flow_only", + "reason": "inference summary (coefficient standard errors, Wald tests); statsmodels territory" + }, + { + "flow_estimator": "logistic_regression_cv", + "sklearn_estimator": "LogisticRegressionCV", + "dataset": "classification", + "flow_ms": 59.974998474, + "sklearn_ms": 92.93033300491516, + "speedup": 1.5494845413827187, + "flow_repeats": 1, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "logistic_regression", + "sklearn_estimator": "LogisticRegression", + "dataset": "classification", + "flow_ms": 0.27125001, + "sklearn_ms": 4.276374995242804, + "speedup": 15.76543718926611, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "lssvm_classifier", + "status": "flow_only", + "reason": "least-squares SVM; not in the scikit-learn public surface" + }, + { + "flow_estimator": "maxabs_scaler", + "sklearn_estimator": "MaxAbsScaler", + "dataset": "unsupervised", + "flow_ms": 0.0004, + "sklearn_ms": 0.10537500202190131, + "speedup": 263.4375050547533, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "mds", + "sklearn_estimator": "MDS", + "dataset": "unsupervised", + "flow_ms": 39.657001495, + "sklearn_ms": 19.73887500935234, + "speedup": 0.4977399769329771, + "flow_repeats": 1, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "mean_shift", + "sklearn_estimator": "MeanShift", + "dataset": "unsupervised", + "flow_ms": 1.259599924, + "sklearn_ms": 124.62391699955333, + "speedup": 98.93928589944352, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "min_cov_det", + "sklearn_estimator": "MinCovDet", + "dataset": "unsupervised", + "flow_ms": 0.20085001, + "sklearn_ms": 8.830791994114406, + "speedup": 43.96709760738576, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "minibatch_dictionary_learning", + "sklearn_estimator": "MiniBatchDictionaryLearning", + "dataset": "unsupervised", + "flow_ms": 0.121835001, + "sklearn_ms": 191.07929299934767, + "speedup": 1568.3448223499229, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "minibatch_kmeans", + "sklearn_estimator": "MiniBatchKMeans", + "dataset": "unsupervised", + "flow_ms": 0.080454998, + "sklearn_ms": 6.989082990912721, + "speedup": 86.8694694506452, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "minibatch_nmf", + "sklearn_estimator": "MiniBatchNMF", + "dataset": "unsupervised", + "flow_ms": 0.079229996, + "sklearn_ms": 7.578208009363152, + "speedup": 95.64821900739655, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "minibatch_sparse_pca", + "sklearn_estimator": "MiniBatchSparsePCA", + "dataset": "unsupervised", + "flow_ms": 0.00238, + "sklearn_ms": 5.608374995063059, + "speedup": 2356.4600819592683, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "minmax_scaler", + "sklearn_estimator": "MinMaxScaler", + "dataset": "unsupervised", + "flow_ms": 0.00096, + "sklearn_ms": 0.09683400276117027, + "speedup": 100.86875287621902, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "missing_indicator", + "sklearn_estimator": "MissingIndicator", + "dataset": "unsupervised", + "flow_ms": 0.00037, + "sklearn_ms": 0.29858300695195794, + "speedup": 806.9810998701566, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "mlp_classifier", + "sklearn_estimator": "MLPClassifier", + "dataset": "classification", + "flow_ms": 0.9091500579999999, + "sklearn_ms": 32.084249003673904, + "speedup": 35.29037777795963, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "mlp_regressor", + "sklearn_estimator": "MLPRegressor", + "dataset": "regression", + "flow_ms": 3.5689998860000003, + "sklearn_ms": 79.0593749989057, + "speedup": 22.151688855197037, + "flow_repeats": 1, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "multi_output_classifier", + "sklearn_estimator": "MultiOutputClassifier", + "dataset": "multioutput_class", + "flow_ms": 0.5503500100000001, + "sklearn_ms": 0.8460840035695583, + "speedup": 1.537356206406825, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "multi_output_regressor", + "sklearn_estimator": "MultiOutputRegressor", + "dataset": "multioutput", + "flow_ms": 0.041465000999999994, + "sklearn_ms": 1.3581249804701656, + "speedup": 32.753525810120344, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "multiclass_logistic", + "sklearn_estimator": "LogisticRegression", + "dataset": "classification", + "flow_ms": 0.31905, + "sklearn_ms": 4.690916000981815, + "speedup": 14.702761325754004, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "multilabel_binarizer", + "status": "different_shape", + "reason": "takes ptr > first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "MultiLabelBinarizer" + }, + { + "flow_estimator": "multinomial_nb", + "sklearn_estimator": "MultinomialNB", + "dataset": "classification", + "flow_ms": 0.00238, + "sklearn_ms": 0.43749899487011135, + "speedup": 183.82310708828206, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "multitask_elastic_net_cv", + "sklearn_estimator": "MultiTaskElasticNetCV", + "dataset": "multioutput", + "flow_ms": 171.37600708, + "sklearn_ms": 17.658540993579663, + "speedup": 0.10303975039712814, + "flow_repeats": 1, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "multitask_elastic_net", + "sklearn_estimator": "MultiTaskElasticNet", + "dataset": "multioutput", + "flow_ms": 0.45095000500000004, + "sklearn_ms": 0.2088330074911937, + "speedup": 0.4630956983606058, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "multitask_lasso_cv", + "sklearn_estimator": "MultiTaskLassoCV", + "dataset": "multioutput", + "flow_ms": 187.596993042, + "sklearn_ms": 111.14133299270179, + "speedup": 0.5924473052071736, + "flow_repeats": 1, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "multitask_lasso", + "sklearn_estimator": "MultiTaskLasso", + "dataset": "multioutput", + "flow_ms": 0.449449994, + "sklearn_ms": 0.2170000079786405, + "speedup": 0.48281235037382264, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "nca", + "status": "simplified", + "reason": "the implementation's own comments call it simplified, so timing it against scikit-learn's algorithm compares two different things", + "sklearn_estimator": "NeighborhoodComponentsAnalysis", + "flow_ms": 169.713004028, + "sklearn_ms": 36.182417010422796, + "timing_unit": "ms" + }, + { + "flow_estimator": "nearest_centroid", + "sklearn_estimator": "NearestCentroid", + "dataset": "classification", + "flow_ms": 0.001915, + "sklearn_ms": 0.6288340082392097, + "speedup": 328.37285025546197, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "nearest_neighbors", + "sklearn_estimator": "NearestNeighbors", + "dataset": "unsupervised", + "flow_ms": 0.0, + "sklearn_ms": 0.1144580019172281, + "speedup": null, + "timing_unit": "ms", + "status": "below_resolution", + "reason": "Flow side at or under 2e-05 ms, which is the clock's floor here" + }, + { + "flow_estimator": "nmf", + "sklearn_estimator": "NMF", + "dataset": "unsupervised", + "flow_ms": 0.477299983, + "sklearn_ms": 2.7645420050248504, + "speedup": 5.792042957237755, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "nu_svc", + "sklearn_estimator": "NuSVC", + "dataset": "regression", + "flow_ms": 2.119000114, + "sklearn_ms": 63.21224999555852, + "speedup": 29.83116875639702, + "flow_repeats": 1, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "nu_svr", + "sklearn_estimator": "NuSVR", + "dataset": "regression", + "flow_ms": 128.55500328600002, + "sklearn_ms": 4.312457997002639, + "speedup": 0.03354562550481672, + "flow_repeats": 1, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "nystroem", + "sklearn_estimator": "Nystroem", + "dataset": "unsupervised", + "flow_ms": 0.018185, + "sklearn_ms": 1.2665829999605194, + "speedup": 69.64987626948141, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "oas_cv", + "status": "flow_only", + "reason": "cross-validated variant scikit-learn does not expose" + }, + { + "flow_estimator": "oas_estimator", + "sklearn_estimator": "OAS", + "dataset": "unsupervised", + "flow_ms": 0.006135, + "sklearn_ms": 0.11737500608433038, + "speedup": 19.132030331594194, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "ols_inference", + "status": "flow_only", + "reason": "inference summary for ordinary least squares; statsmodels territory" + }, + { + "flow_estimator": "omp_cv", + "sklearn_estimator": "OrthogonalMatchingPursuitCV", + "dataset": "regression", + "flow_ms": 0.037935, + "sklearn_ms": 1.2267079873709008, + "speedup": 32.337102606323995, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "one_class_svm", + "sklearn_estimator": "OneClassSVM", + "dataset": "unsupervised", + "flow_ms": 8.137999535, + "sklearn_ms": 0.5150840006535873, + "speedup": 0.06329368764870386, + "flow_repeats": 1, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "one_vs_one", + "sklearn_estimator": "OneVsOneClassifier", + "dataset": "classification", + "flow_ms": 0.116269994, + "sklearn_ms": 1.3165410055080429, + "speedup": 11.323136436285038, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "one_vs_rest", + "sklearn_estimator": "OneVsRestClassifier", + "dataset": "classification", + "flow_ms": 0.2055, + "sklearn_ms": 1.3594179908977821, + "speedup": 6.615172705098697, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "onehot_encoder", + "sklearn_estimator": "OneHotEncoder", + "dataset": "unsupervised", + "flow_ms": 0.029215002, + "sklearn_ms": 0.5092510109534487, + "speedup": 17.431147564304418, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "optics", + "sklearn_estimator": "OPTICS", + "dataset": "unsupervised", + "flow_ms": 2.779999971, + "sklearn_ms": 33.755458003724925, + "speedup": 12.142251207140363, + "flow_repeats": 1, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "ordinal_encoder", + "sklearn_estimator": "OrdinalEncoder", + "dataset": "unsupervised", + "flow_ms": 0.0181, + "sklearn_ms": 0.47520900261588395, + "speedup": 26.254641028501872, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "orthogonal_matching_pursuit", + "sklearn_estimator": "OrthogonalMatchingPursuit", + "dataset": "regression", + "flow_ms": 0.027035, + "sklearn_ms": 0.1994580088648945, + "speedup": 7.377769885884761, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "output_code", + "sklearn_estimator": "OutputCodeClassifier", + "dataset": "classification", + "flow_ms": 0.264249997, + "sklearn_ms": 1.7742080090101808, + "speedup": 6.714126884210261, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "passive_aggressive_classifier", + "sklearn_estimator": "PassiveAggressiveClassifier", + "dataset": "regression", + "flow_ms": 0.116094999, + "sklearn_ms": 29.98487501463387, + "speedup": 258.2787826599996, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "passive_aggressive_regressor", + "sklearn_estimator": "PassiveAggressiveRegressor", + "dataset": "regression", + "flow_ms": 0.822950006, + "sklearn_ms": 0.758833994041197, + "speedup": 0.9220900279587543, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "pca", + "sklearn_estimator": "PCA", + "dataset": "unsupervised", + "flow_ms": 0.00726, + "sklearn_ms": 0.1589170133229345, + "speedup": 21.889395774508884, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "perceptron", + "sklearn_estimator": "Perceptron", + "dataset": "regression", + "flow_ms": 0.21455000700000001, + "sklearn_ms": 26.87045800848864, + "speedup": 125.24100271172976, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "pipeline", + "status": "different_shape", + "reason": "takes Pipeline first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "Pipeline" + }, + { + "flow_estimator": "pls_canonical", + "sklearn_estimator": "PLSCanonical", + "dataset": "multioutput", + "flow_ms": 0.043340000000000004, + "sklearn_ms": 0.3192919975845143, + "speedup": 7.367143460648691, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "pls", + "sklearn_estimator": "PLSRegression", + "dataset": "multioutput", + "flow_ms": 0.067739999, + "sklearn_ms": 0.351291018887423, + "speedup": 5.185872808876526, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "pls_svd", + "sklearn_estimator": "PLSSVD", + "dataset": "multioutput", + "flow_ms": 0.019864999, + "sklearn_ms": 0.2077509998343885, + "speedup": 10.458142979739817, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "poisson_regressor", + "sklearn_estimator": "PoissonRegressor", + "dataset": "regression", + "flow_ms": 0.472000023, + "sklearn_ms": 3.4686659928411245, + "speedup": 7.348868270799056, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "polynomial_count_sketch", + "status": "different_shape", + "reason": "takes i32 first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "PolynomialCountSketch" + }, + { + "flow_estimator": "polynomial_features", + "status": "different_shape", + "reason": "takes i32 first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "PolynomialFeatures" + }, + { + "flow_estimator": "power_transformer", + "sklearn_estimator": "PowerTransformer", + "dataset": "unsupervised", + "flow_ms": 0.063644999, + "sklearn_ms": 7.3907919868361205, + "speedup": 116.12525890425611, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "qda", + "sklearn_estimator": "QuadraticDiscriminantAnalysis", + "dataset": "classification", + "flow_ms": 0.00932, + "sklearn_ms": 0.29850001737941056, + "speedup": 32.02789886045178, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "quantile_regressor", + "sklearn_estimator": "QuantileRegressor", + "dataset": "regression", + "flow_ms": 0.207649991, + "sklearn_ms": 7.948208003654145, + "speedup": 38.27694846205962, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "quantile_transformer", + "sklearn_estimator": "QuantileTransformer", + "dataset": "unsupervised", + "flow_ms": 0.008825, + "sklearn_ms": 0.33608300145715475, + "speedup": 38.08305965520167, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "radius_neighbors_classifier", + "sklearn_estimator": "RadiusNeighborsClassifier", + "dataset": "classification", + "flow_ms": 0.097194999, + "sklearn_ms": 0.6490420055342838, + "speedup": 6.6777304615671, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "radius_neighbors_regressor", + "sklearn_estimator": "RadiusNeighborsRegressor", + "dataset": "regression", + "flow_ms": 0.975350022, + "sklearn_ms": 2.382626000326127, + "speedup": 2.442842001931207, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "radius_neighbors_transformer", + "sklearn_estimator": "RadiusNeighborsTransformer", + "dataset": "unsupervised", + "flow_ms": 0.061634999, + "sklearn_ms": 0.4621660045813769, + "speedup": 7.498434527132496, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "random_forest_classifier", + "sklearn_estimator": "RandomForestClassifier", + "dataset": "classification", + "flow_ms": 0.10339999400000001, + "sklearn_ms": 42.424207989824936, + "speedup": 410.29217071158564, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "random_forest_regressor", + "sklearn_estimator": "RandomForestRegressor", + "dataset": "regression", + "flow_ms": 7.646000094000001, + "sklearn_ms": 106.43045901088044, + "speedup": 13.919756435054058, + "flow_repeats": 1, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "random_trees_embedding", + "sklearn_estimator": "RandomTreesEmbedding", + "dataset": "unsupervised", + "flow_ms": 0.72344997, + "sklearn_ms": 36.00624999671709, + "speedup": 49.77020041443514, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "ransac_regressor", + "sklearn_estimator": "RANSACRegressor", + "dataset": "regression", + "flow_ms": 0.025629999, + "sklearn_ms": 24.211207986809313, + "speedup": 944.6433449649886, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "rbf_sampler", + "status": "different_shape", + "reason": "takes i32 first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "RBFSampler" + }, + { + "flow_estimator": "regressor_chain", + "sklearn_estimator": "RegressorChain", + "dataset": "multioutput_class", + "flow_ms": 0.00694, + "sklearn_ms": 0.6237920024432242, + "speedup": 89.8835738390813, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "rfe", + "sklearn_estimator": "RFE", + "dataset": "regression", + "flow_ms": 0.016205, + "sklearn_ms": 8.93216700933408, + "speedup": 551.1982110048799, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "rfecv", + "sklearn_estimator": "RFECV", + "dataset": "regression", + "flow_ms": 4.782000065, + "sklearn_ms": 74.68254200648516, + "speedup": 15.61742806176335, + "flow_repeats": 1, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "ridge_classifier_cv", + "sklearn_estimator": "RidgeClassifierCV", + "dataset": "classification", + "flow_ms": 25.680000305, + "sklearn_ms": 0.9509170049568638, + "speedup": 0.03702947794637356, + "flow_repeats": 1, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "ridge_classifier", + "sklearn_estimator": "RidgeClassifier", + "dataset": "regression", + "flow_ms": 0.206200003, + "sklearn_ms": 1.3081669894745573, + "speedup": 6.344165715043939, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "ridge_cv", + "sklearn_estimator": "RidgeCV", + "dataset": "regression", + "flow_ms": 0.723699987, + "sklearn_ms": 0.3534170100465417, + "speedup": 0.48834740416617095, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "ridge", + "sklearn_estimator": "Ridge", + "dataset": "regression", + "flow_ms": 0.017119999, + "sklearn_ms": 0.24566700449213386, + "speedup": 14.34970904450017, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "robust_scaler", + "sklearn_estimator": "RobustScaler", + "dataset": "unsupervised", + "flow_ms": 0.00709, + "sklearn_ms": 0.2803750103339553, + "speedup": 39.54513544907691, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "select_fdr", + "sklearn_estimator": "SelectFdr", + "dataset": "regression", + "flow_ms": 1.146900088, + "sklearn_ms": 4.7365840000566095, + "speedup": 4.1299011567053, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "select_fpr", + "sklearn_estimator": "SelectFpr", + "dataset": "regression", + "flow_ms": 1.137300007, + "sklearn_ms": 4.5432499900925905, + "speedup": 3.9947682776129545, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "select_from_model", + "status": "different_shape", + "reason": "takes ptr first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "SelectFromModel" + }, + { + "flow_estimator": "select_fwe", + "sklearn_estimator": "SelectFwe", + "dataset": "regression", + "flow_ms": 1.1540500530000002, + "sklearn_ms": 4.57766700128559, + "speedup": 3.966610451068182, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "select_k_best", + "sklearn_estimator": "SelectKBest", + "dataset": "regression", + "flow_ms": 1.096850038, + "sklearn_ms": 4.83825099945534, + "speedup": 4.411041465866586, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "select_percentile", + "sklearn_estimator": "SelectPercentile", + "dataset": "regression", + "flow_ms": 1.097350095, + "sklearn_ms": 4.657249999581836, + "speedup": 4.244087662453646, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "self_training_classifier", + "status": "different_shape", + "reason": "takes ptr first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "SelfTrainingClassifier" + }, + { + "flow_estimator": "sequential_feature_selector", + "sklearn_estimator": "SequentialFeatureSelector", + "dataset": "regression", + "flow_ms": 3.413000107, + "sklearn_ms": 78.29795898578595, + "speedup": 22.94109479375588, + "flow_repeats": 1, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "sgd_classifier", + "sklearn_estimator": "SGDClassifier", + "dataset": "classification", + "flow_ms": 0.231749991, + "sklearn_ms": 0.8100010018097237, + "speedup": 3.4951500896054997, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "sgd_one_class_svm", + "sklearn_estimator": "SGDOneClassSVM", + "dataset": "unsupervised", + "flow_ms": 0.046094998, + "sklearn_ms": 0.7897070026956499, + "speedup": 17.13216264150071, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "sgd_regressor", + "sklearn_estimator": "SGDRegressor", + "dataset": "regression", + "flow_ms": 0.390450001, + "sklearn_ms": 11.209875010536052, + "speedup": 28.710142097133847, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "shrunk_covariance", + "sklearn_estimator": "ShrunkCovariance", + "dataset": "unsupervised", + "flow_ms": 0.002245, + "sklearn_ms": 0.16125000547617674, + "speedup": 71.82628306288495, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "simple_imputer", + "sklearn_estimator": "SimpleImputer", + "dataset": "unsupervised", + "flow_ms": 0.00126, + "sklearn_ms": 0.417499992181547, + "speedup": 331.3492001440849, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "skewed_chi2_sampler", + "status": "different_shape", + "reason": "takes i32 first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "SkewedChi2Sampler" + }, + { + "flow_estimator": "sparse_coder", + "sklearn_estimator": "SparseCoder", + "dataset": "unsupervised", + "flow_ms": 0.206245005, + "sklearn_ms": 1.0421250044601038, + "speedup": 5.05284966518391, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "sparse_pca", + "sklearn_estimator": "SparsePCA", + "dataset": "unsupervised", + "flow_ms": 0.004145, + "sklearn_ms": 7.9594590060878545, + "speedup": 1920.2554900091325, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "sparse_random_projection", + "status": "different_shape", + "reason": "takes i32 first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "SparseRandomProjection" + }, + { + "flow_estimator": "spectral_biclustering", + "status": "simplified", + "reason": "the implementation's own comments call it simplified, so timing it against scikit-learn's algorithm compares two different things", + "sklearn_estimator": "SpectralBiclustering", + "flow_ms": 0.00094, + "sklearn_ms": 23.599958993145265, + "timing_unit": "ms" + }, + { + "flow_estimator": "spectral_clustering", + "sklearn_estimator": "SpectralClustering", + "dataset": "unsupervised", + "flow_ms": 4.743000031, + "sklearn_ms": 5.601083001238294, + "speedup": 1.1809156577334827, + "flow_repeats": 1, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "spectral_coclustering", + "status": "simplified", + "reason": "the implementation's own comments call it simplified, so timing it against scikit-learn's algorithm compares two different things", + "sklearn_estimator": "SpectralCoclustering", + "flow_ms": 0.000915, + "sklearn_ms": 3.2601659913780168, + "timing_unit": "ms" + }, + { + "flow_estimator": "spectral_embedding", + "sklearn_estimator": "SpectralEmbedding", + "dataset": "unsupervised", + "flow_ms": 6.156000137, + "sklearn_ms": 1.5892500086920336, + "speedup": 0.25816276369781266, + "flow_repeats": 1, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "spline_transformer", + "sklearn_estimator": "SplineTransformer", + "dataset": "unsupervised", + "flow_ms": 0.024345, + "sklearn_ms": 0.4049989947816357, + "speedup": 16.635818228861602, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "stacking_classifier", + "status": "different_shape", + "reason": "takes ptr > first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "StackingClassifier" + }, + { + "flow_estimator": "stacking_regressor", + "sklearn_estimator": "StackingRegressor", + "dataset": "regression", + "flow_ms": 2.340000026, + "sklearn_ms": 3.0410410108743235, + "speedup": 1.2995901611474272, + "flow_repeats": 1, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "standard_scaler", + "sklearn_estimator": "StandardScaler", + "dataset": "unsupervised", + "flow_ms": 0.001025, + "sklearn_ms": 0.13524999667424709, + "speedup": 131.95121626755812, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "svc", + "sklearn_estimator": "SVC", + "dataset": "classification", + "flow_ms": 0.20099499799999998, + "sklearn_ms": 0.6359160033753142, + "speedup": 3.1638399447896424, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "svr", + "sklearn_estimator": "SVR", + "dataset": "regression", + "flow_ms": 66.409997702, + "sklearn_ms": 6.55341699894052, + "speedup": 0.09868118093223723, + "flow_repeats": 1, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "target_encoder", + "sklearn_estimator": "TargetEncoder", + "dataset": "regression", + "flow_ms": 0.783399999, + "sklearn_ms": 13.760207992163487, + "speedup": 17.564728120664046, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "tfidf_vectorizer", + "status": "different_shape", + "reason": "takes ptr first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "TfidfVectorizer" + }, + { + "flow_estimator": "theil_sen_regressor", + "sklearn_estimator": "TheilSenRegressor", + "dataset": "regression", + "flow_ms": 0.156299993, + "sklearn_ms": 185.37908300640993, + "speedup": 1186.04664944809, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "transformed_target_regressor", + "sklearn_estimator": "TransformedTargetRegressor", + "dataset": "regression", + "flow_ms": 0.021449998999999997, + "sklearn_ms": 0.4650419868994504, + "speedup": 21.68028012026716, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "truncated_svd", + "sklearn_estimator": "TruncatedSVD", + "dataset": "unsupervised", + "flow_ms": 0.00762, + "sklearn_ms": 0.25675000506453216, + "speedup": 33.69422638642154, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "tsne", + "sklearn_estimator": "TSNE", + "dataset": "unsupervised", + "flow_ms": 6.71999979, + "sklearn_ms": 415.7947920029983, + "speedup": 61.874226934015766, + "flow_repeats": 1, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "tweedie_regressor", + "sklearn_estimator": "TweedieRegressor", + "dataset": "regression", + "flow_ms": 0.44105001499999996, + "sklearn_ms": 0.7525420078309253, + "speedup": 1.7062509516770459, + "flow_repeats": 20, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "variance_threshold", + "sklearn_estimator": "VarianceThreshold", + "dataset": "unsupervised", + "flow_ms": 0.0014, + "sklearn_ms": 0.13425000361166894, + "speedup": 95.89285972262067, + "flow_repeats": 200, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "voting_classifier", + "status": "different_shape", + "reason": "takes ptr first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "VotingClassifier" + }, + { + "flow_estimator": "voting_regressor", + "status": "different_shape", + "reason": "takes ptr first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "VotingRegressor" + } + ] +} diff --git a/benchmarks/estimator_coverage.json b/benchmarks/estimator_coverage.json new file mode 100644 index 0000000..cc1fbea --- /dev/null +++ b/benchmarks/estimator_coverage.json @@ -0,0 +1,12283 @@ +{ + "schema_version": 1, + "counts": { + "estimators": 203, + "runnable": 168, + "different_shape": 25, + "simplified": 4, + "flow_only": 6 + }, + "sklearn_surface": 208, + "entries": [ + { + "flow_estimator": "adaboost_classifier", + "module": "ensemble.flow", + "fit": { + "name": "adaboost_classifier_fit", + "returns": "AdaBoostClassifier", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_classes", + "type": "i32" + }, + { + "name": "n_trees", + "type": "i32" + }, + { + "name": "max_depth", + "type": "i32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "adaboost_classifier_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "AdaBoostClassifier" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "adaboost_classifier_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "AdaBoostClassifier" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_classes", + "n_trees", + "max_depth", + "seed" + ], + "bucket": "runnable", + "sklearn_estimator": "AdaBoostClassifier" + }, + { + "flow_estimator": "adaboost_regressor", + "module": "ensemble.flow", + "fit": { + "name": "adaboost_regressor_fit", + "returns": "AdaBoostRegressor", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_trees", + "type": "i32" + }, + { + "name": "max_depth", + "type": "i32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "adaboost_regressor_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "AdaBoostRegressor" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "adaboost_regressor_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "AdaBoostRegressor" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_trees", + "max_depth", + "seed" + ], + "bucket": "runnable", + "sklearn_estimator": "AdaBoostRegressor" + }, + { + "flow_estimator": "additive_chi2_sampler", + "module": "kernel_approximation.flow", + "fit": { + "name": "additive_chi2_sampler_fit", + "returns": "AdditiveChi2Sampler", + "parameters": [ + { + "name": "n_samples", + "type": "i32" + }, + { + "name": "n_features", + "type": "i32" + }, + { + "name": "sample_steps", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "additive_chi2_sampler_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "AdditiveChi2Sampler" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "additive_chi2_sampler_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "AdditiveChi2Sampler" + } + ] + } + }, + "parameters": [ + "n_samples", + "n_features", + "sample_steps" + ], + "bucket": "different_shape", + "reason": "takes i32 first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "AdditiveChi2Sampler" + }, + { + "flow_estimator": "affinity_propagation", + "module": "cluster.flow", + "fit": { + "name": "affinity_propagation_fit", + "returns": "AffinityPropagation", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "damping", + "type": "f32" + }, + { + "name": "max_iter", + "type": "i32" + }, + { + "name": "convergence_iter", + "type": "i32" + } + ] + }, + "companions": { + "free": { + "name": "affinity_propagation_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "AffinityPropagation" + } + ] + } + }, + "parameters": [ + "X", + "damping", + "max_iter", + "convergence_iter" + ], + "bucket": "runnable", + "sklearn_estimator": "AffinityPropagation" + }, + { + "flow_estimator": "agglomerative_clustering", + "module": "cluster.flow", + "fit": { + "name": "agglomerative_clustering_fit", + "returns": "AgglomerativeClustering", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "n_clusters", + "type": "i32" + } + ] + }, + "companions": { + "free": { + "name": "agglomerative_clustering_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "AgglomerativeClustering" + } + ] + } + }, + "parameters": [ + "X", + "n_clusters" + ], + "bucket": "runnable", + "sklearn_estimator": "AgglomerativeClustering" + }, + { + "flow_estimator": "ard_regression", + "module": "linear.flow", + "fit": { + "name": "ard_regression_fit", + "returns": "ARDRegression", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "max_iter", + "type": "i32" + }, + { + "name": "tol", + "type": "f32" + } + ] + }, + "companions": { + "predict": { + "name": "ard_regression_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "ARDRegression" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "ard_regression_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "ARDRegression" + } + ] + } + }, + "parameters": [ + "X", + "y", + "max_iter", + "tol" + ], + "bucket": "runnable", + "sklearn_estimator": "ARDRegression" + }, + { + "flow_estimator": "bagging_classifier", + "module": "ensemble.flow", + "fit": { + "name": "bagging_classifier_fit", + "returns": "BaggingClassifier", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_classes", + "type": "i32" + }, + { + "name": "n_estimators", + "type": "i32" + }, + { + "name": "max_depth", + "type": "i32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "bagging_classifier_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "BaggingClassifier" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "bagging_classifier_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "BaggingClassifier" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_classes", + "n_estimators", + "max_depth", + "seed" + ], + "bucket": "runnable", + "sklearn_estimator": "BaggingClassifier" + }, + { + "flow_estimator": "bagging_regressor", + "module": "ensemble.flow", + "fit": { + "name": "bagging_regressor_fit", + "returns": "BaggingRegressor", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_estimators", + "type": "i32" + }, + { + "name": "max_depth", + "type": "i32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "bagging_regressor_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "BaggingRegressor" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "bagging_regressor_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "BaggingRegressor" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_estimators", + "max_depth", + "seed" + ], + "bucket": "runnable", + "sklearn_estimator": "BaggingRegressor" + }, + { + "flow_estimator": "bayesian_gaussian_mixture", + "module": "mixture.flow", + "fit": { + "name": "bayesian_gaussian_mixture_fit", + "returns": "BayesianGaussianMixture", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "n_components", + "type": "i32" + }, + { + "name": "max_iter", + "type": "i32" + }, + { + "name": "tol", + "type": "f32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "bayesian_gaussian_mixture_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "BayesianGaussianMixture" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "bayesian_gaussian_mixture_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "BayesianGaussianMixture" + } + ] + } + }, + "parameters": [ + "X", + "n_components", + "max_iter", + "tol", + "seed" + ], + "bucket": "runnable", + "sklearn_estimator": "BayesianGaussianMixture" + }, + { + "flow_estimator": "bayesian_ridge", + "module": "linear.flow", + "fit": { + "name": "bayesian_ridge_fit", + "returns": "BayesianRidge", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "max_iter", + "type": "i32" + }, + { + "name": "tol", + "type": "f32" + } + ] + }, + "companions": { + "predict": { + "name": "bayesian_ridge_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "BayesianRidge" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "bayesian_ridge_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "BayesianRidge" + } + ] + } + }, + "parameters": [ + "X", + "y", + "max_iter", + "tol" + ], + "bucket": "runnable", + "sklearn_estimator": "BayesianRidge" + }, + { + "flow_estimator": "bernoulli_nb", + "module": "naive_bayes.flow", + "fit": { + "name": "bernoulli_nb_fit", + "returns": "BernoulliNB", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_classes", + "type": "i32" + }, + { + "name": "alpha", + "type": "f32" + } + ] + }, + "companions": { + "predict": { + "name": "bernoulli_nb_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "BernoulliNB" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "bernoulli_nb_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "BernoulliNB" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_classes", + "alpha" + ], + "bucket": "runnable", + "sklearn_estimator": "BernoulliNB" + }, + { + "flow_estimator": "bernoulli_rbm", + "module": "neural_network.flow", + "fit": { + "name": "bernoulli_rbm_fit", + "returns": "BernoulliRBM", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "n_components", + "type": "i32" + }, + { + "name": "learning_rate", + "type": "f32" + }, + { + "name": "n_iter", + "type": "i32" + }, + { + "name": "batch_size", + "type": "i32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "bernoulli_rbm_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "BernoulliRBM" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "bernoulli_rbm_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "BernoulliRBM" + } + ] + } + }, + "parameters": [ + "X", + "n_components", + "learning_rate", + "n_iter", + "batch_size", + "seed" + ], + "bucket": "runnable", + "sklearn_estimator": "BernoulliRBM" + }, + { + "flow_estimator": "birch", + "module": "cluster.flow", + "fit": { + "name": "birch_fit", + "returns": "Birch", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "threshold", + "type": "f32" + }, + { + "name": "branching_factor", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "birch_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "Birch" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "birch_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "Birch" + } + ] + } + }, + "parameters": [ + "X", + "threshold", + "branching_factor" + ], + "bucket": "runnable", + "sklearn_estimator": "Birch" + }, + { + "flow_estimator": "bisecting_kmeans", + "module": "cluster.flow", + "fit": { + "name": "bisecting_kmeans_fit", + "returns": "BisectingKMeans", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "n_clusters", + "type": "i32" + }, + { + "name": "max_iter", + "type": "i32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "free": { + "name": "bisecting_kmeans_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "BisectingKMeans" + } + ] + } + }, + "parameters": [ + "X", + "n_clusters", + "max_iter", + "seed" + ], + "bucket": "runnable", + "sklearn_estimator": "BisectingKMeans" + }, + { + "flow_estimator": "calibrated_classifier_cv", + "module": "calibration.flow", + "fit": { + "name": "calibrated_classifier_cv_fit", + "returns": "CalibratedClassifierCV", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_classes", + "type": "i32" + }, + { + "name": "n_folds", + "type": "i32" + }, + { + "name": "method", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "calibrated_classifier_cv_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "CalibratedClassifierCV" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "calibrated_classifier_cv_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "CalibratedClassifierCV" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_classes", + "n_folds", + "method" + ], + "bucket": "runnable", + "sklearn_estimator": "CalibratedClassifierCV" + }, + { + "flow_estimator": "categorical_nb", + "module": "naive_bayes.flow", + "fit": { + "name": "categorical_nb_fit", + "returns": "CategoricalNB", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_classes", + "type": "i32" + }, + { + "name": "n_categories", + "type": "ptr" + }, + { + "name": "alpha", + "type": "f32" + } + ] + }, + "companions": { + "predict": { + "name": "categorical_nb_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "CategoricalNB" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "categorical_nb_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "CategoricalNB" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_classes", + "n_categories", + "alpha" + ], + "bucket": "runnable", + "sklearn_estimator": "CategoricalNB" + }, + { + "flow_estimator": "cca", + "module": "cross_decomposition.flow", + "fit": { + "name": "cca_fit", + "returns": "CCA", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "Y", + "type": "Matrix" + }, + { + "name": "n_components", + "type": "i32" + }, + { + "name": "max_iter", + "type": "i32" + }, + { + "name": "tol", + "type": "f32" + } + ] + }, + "companions": { + "transform": { + "name": "cca_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "CCA" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "cca_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "CCA" + } + ] + } + }, + "parameters": [ + "X", + "Y", + "n_components", + "max_iter", + "tol" + ], + "bucket": "runnable", + "sklearn_estimator": "CCA" + }, + { + "flow_estimator": "classifier_chain", + "module": "multioutput.flow", + "fit": { + "name": "classifier_chain_fit", + "returns": "ClassifierChain", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "Y", + "type": "ptr >" + }, + { + "name": "n_samples", + "type": "i32" + }, + { + "name": "n_features", + "type": "i32" + }, + { + "name": "n_outputs", + "type": "i32" + }, + { + "name": "lr", + "type": "f32" + }, + { + "name": "n_iter", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "classifier_chain_predict", + "returns": "ptr >", + "parameters": [ + { + "name": "model", + "type": "ClassifierChain" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "classifier_chain_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "ClassifierChain" + } + ] + } + }, + "parameters": [ + "X", + "Y", + "n_samples", + "n_features", + "n_outputs", + "lr", + "n_iter" + ], + "bucket": "runnable", + "sklearn_estimator": "ClassifierChain" + }, + { + "flow_estimator": "column_transformer", + "module": "compose.flow", + "fit": { + "name": "column_transformer_fit", + "returns": "ColumnTransformer", + "parameters": [ + { + "name": "ct", + "type": "ColumnTransformer" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "companions": { + "transform": { + "name": "column_transformer_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "ct", + "type": "ColumnTransformer" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "column_transformer_free", + "returns": "void", + "parameters": [ + { + "name": "ct", + "type": "ColumnTransformer" + } + ] + } + }, + "parameters": [ + "ct", + "X" + ], + "bucket": "different_shape", + "reason": "takes ColumnTransformer first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "ColumnTransformer" + }, + { + "flow_estimator": "complement_nb", + "module": "naive_bayes.flow", + "fit": { + "name": "complement_nb_fit", + "returns": "ComplementNB", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_classes", + "type": "i32" + }, + { + "name": "alpha", + "type": "f32" + } + ] + }, + "companions": { + "predict": { + "name": "complement_nb_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "ComplementNB" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "complement_nb_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "ComplementNB" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_classes", + "alpha" + ], + "bucket": "runnable", + "sklearn_estimator": "ComplementNB" + }, + { + "flow_estimator": "count_vectorizer", + "module": "feature_extraction.flow", + "fit": { + "name": "count_vectorizer_fit", + "returns": "CountVectorizer", + "parameters": [ + { + "name": "documents", + "type": "ptr" + }, + { + "name": "n_docs", + "type": "i32" + }, + { + "name": "max_features", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "count_vectorizer_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "CountVectorizer" + }, + { + "name": "documents", + "type": "ptr" + }, + { + "name": "n_docs", + "type": "i32" + } + ] + }, + "free": { + "name": "count_vectorizer_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "CountVectorizer" + } + ] + } + }, + "parameters": [ + "documents", + "n_docs", + "max_features" + ], + "bucket": "different_shape", + "reason": "takes ptr first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "CountVectorizer" + }, + { + "flow_estimator": "dbscan", + "module": "cluster.flow", + "fit": { + "name": "dbscan_fit", + "returns": "DBSCAN", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "eps", + "type": "f32" + }, + { + "name": "min_samples", + "type": "i32" + } + ] + }, + "companions": { + "free": { + "name": "dbscan_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "DBSCAN" + } + ] + } + }, + "parameters": [ + "X", + "eps", + "min_samples" + ], + "bucket": "runnable", + "sklearn_estimator": "DBSCAN" + }, + { + "flow_estimator": "decision_tree_classifier", + "module": "tree.flow", + "fit": { + "name": "decision_tree_classifier_fit", + "returns": "DecisionTreeClassifier", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_classes", + "type": "i32" + }, + { + "name": "max_depth", + "type": "i32" + }, + { + "name": "criterion", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "decision_tree_classifier_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "DecisionTreeClassifier" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "decision_tree_classifier_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "DecisionTreeClassifier" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_classes", + "max_depth", + "criterion" + ], + "bucket": "runnable", + "sklearn_estimator": "DecisionTreeClassifier" + }, + { + "flow_estimator": "decision_tree_regressor", + "module": "tree.flow", + "fit": { + "name": "decision_tree_regressor_fit", + "returns": "DecisionTreeRegressor", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "max_depth", + "type": "i32" + }, + { + "name": "criterion", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "decision_tree_regressor_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "DecisionTreeRegressor" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "decision_tree_regressor_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "DecisionTreeRegressor" + } + ] + } + }, + "parameters": [ + "X", + "y", + "max_depth", + "criterion" + ], + "bucket": "runnable", + "sklearn_estimator": "DecisionTreeRegressor" + }, + { + "flow_estimator": "dict_vectorizer", + "module": "feature_extraction_extra.flow", + "fit": { + "name": "dict_vectorizer_fit", + "returns": "DictVectorizer", + "parameters": [ + { + "name": "X_keys", + "type": "ptr >" + }, + { + "name": "X_vals", + "type": "ptr >" + }, + { + "name": "n_samples", + "type": "i32" + }, + { + "name": "n_keys_per_sample", + "type": "ptr" + } + ] + }, + "companions": { + "transform": { + "name": "dict_vectorizer_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "DictVectorizer" + }, + { + "name": "X_keys", + "type": "ptr >" + }, + { + "name": "X_vals", + "type": "ptr >" + }, + { + "name": "n_samples", + "type": "i32" + }, + { + "name": "n_keys_per_sample", + "type": "ptr" + } + ] + }, + "free": { + "name": "dict_vectorizer_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "DictVectorizer" + } + ] + } + }, + "parameters": [ + "X_keys", + "X_vals", + "n_samples", + "n_keys_per_sample" + ], + "bucket": "different_shape", + "reason": "takes ptr > first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "DictVectorizer" + }, + { + "flow_estimator": "dictionary_learning", + "module": "decomposition.flow", + "fit": { + "name": "dictionary_learning_fit", + "returns": "DictionaryLearning", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "n_components", + "type": "i32" + }, + { + "name": "alpha", + "type": "f32" + }, + { + "name": "max_iter", + "type": "i32" + }, + { + "name": "tol", + "type": "f32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "dictionary_learning_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "DictionaryLearning" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "dictionary_learning_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "DictionaryLearning" + } + ] + } + }, + "parameters": [ + "X", + "n_components", + "alpha", + "max_iter", + "tol", + "seed" + ], + "bucket": "runnable", + "sklearn_estimator": "DictionaryLearning" + }, + { + "flow_estimator": "discriminant_lda", + "module": "discriminant_analysis.flow", + "fit": { + "name": "discriminant_lda_fit", + "returns": "LinearDiscriminantAnalysis", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_classes", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "discriminant_lda_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "LinearDiscriminantAnalysis" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "discriminant_lda_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "LinearDiscriminantAnalysis" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_classes" + ], + "bucket": "runnable", + "sklearn_estimator": "LinearDiscriminantAnalysis" + }, + { + "flow_estimator": "dummy_classifier", + "module": "dummy.flow", + "fit": { + "name": "dummy_classifier_fit", + "returns": "DummyClassifier", + "parameters": [ + { + "name": "y", + "type": "ptr" + }, + { + "name": "n", + "type": "i32" + }, + { + "name": "n_classes", + "type": "i32" + }, + { + "name": "strategy", + "type": "i32" + }, + { + "name": "constant_label", + "type": "f32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "dummy_classifier_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "DummyClassifier" + }, + { + "name": "n_samples", + "type": "i32" + } + ] + }, + "predict_proba": { + "name": "dummy_classifier_predict_proba", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "DummyClassifier" + }, + { + "name": "n_samples", + "type": "i32" + } + ] + }, + "free": { + "name": "dummy_classifier_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "DummyClassifier" + } + ] + } + }, + "parameters": [ + "y", + "n", + "n_classes", + "strategy", + "constant_label", + "seed" + ], + "bucket": "different_shape", + "reason": "takes ptr first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "DummyClassifier" + }, + { + "flow_estimator": "dummy_regressor", + "module": "dummy.flow", + "fit": { + "name": "dummy_regressor_fit", + "returns": "DummyRegressor", + "parameters": [ + { + "name": "y", + "type": "ptr" + }, + { + "name": "n", + "type": "i32" + }, + { + "name": "strategy", + "type": "i32" + }, + { + "name": "constant_value", + "type": "f32" + } + ] + }, + "companions": { + "predict": { + "name": "dummy_regressor_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "DummyRegressor" + }, + { + "name": "n_samples", + "type": "i32" + } + ] + }, + "free": { + "name": "dummy_regressor_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "DummyRegressor" + } + ] + } + }, + "parameters": [ + "y", + "n", + "strategy", + "constant_value" + ], + "bucket": "different_shape", + "reason": "takes ptr first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "DummyRegressor" + }, + { + "flow_estimator": "elastic_net_cv", + "module": "linear.flow", + "fit": { + "name": "elastic_net_cv_fit", + "returns": "ElasticNetCV", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_alphas", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "elastic_net_cv_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "ElasticNetCV" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "elastic_net_cv_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "ElasticNetCV" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_alphas" + ], + "bucket": "runnable", + "sklearn_estimator": "ElasticNetCV" + }, + { + "flow_estimator": "elastic_net", + "module": "linear.flow", + "fit": { + "name": "elastic_net_fit", + "returns": "ElasticNet", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "alpha", + "type": "f32" + }, + { + "name": "l1_ratio", + "type": "f32" + }, + { + "name": "epochs", + "type": "i32" + }, + { + "name": "lr", + "type": "f32" + } + ] + }, + "companions": { + "predict": { + "name": "elastic_net_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "ElasticNet" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "elastic_net_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "ElasticNet" + } + ] + } + }, + "parameters": [ + "X", + "y", + "alpha", + "l1_ratio", + "epochs", + "lr" + ], + "bucket": "runnable", + "sklearn_estimator": "ElasticNet" + }, + { + "flow_estimator": "elliptic_envelope", + "module": "covariance.flow", + "fit": { + "name": "elliptic_envelope_fit", + "returns": "EllipticEnvelope", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "contamination", + "type": "f32" + }, + { + "name": "n_trials", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "elliptic_envelope_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "EllipticEnvelope" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "elliptic_envelope_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "EllipticEnvelope" + } + ] + } + }, + "parameters": [ + "X", + "contamination", + "n_trials" + ], + "bucket": "simplified", + "reason": "the implementation's own comments call it simplified, so timing it against scikit-learn's algorithm compares two different things", + "sklearn_estimator": "EllipticEnvelope" + }, + { + "flow_estimator": "empirical_covariance", + "module": "covariance.flow", + "fit": { + "name": "empirical_covariance_fit", + "returns": "EmpiricalCovariance", + "parameters": [ + { + "name": "X", + "type": "Matrix" + } + ] + }, + "companions": { + "free": { + "name": "empirical_covariance_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "EmpiricalCovariance" + } + ] + } + }, + "parameters": [ + "X" + ], + "bucket": "runnable", + "sklearn_estimator": "EmpiricalCovariance" + }, + { + "flow_estimator": "extra_tree_classifier", + "module": "tree.flow", + "fit": { + "name": "extra_tree_classifier_fit", + "returns": "ExtraTreeClassifier", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_samples", + "type": "i32" + }, + { + "name": "n_features", + "type": "i32" + }, + { + "name": "n_classes", + "type": "i32" + }, + { + "name": "max_depth", + "type": "i32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "extra_tree_classifier_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "ExtraTreeClassifier" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "extra_tree_classifier_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "ExtraTreeClassifier" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_samples", + "n_features", + "n_classes", + "max_depth", + "seed" + ], + "bucket": "runnable", + "sklearn_estimator": "ExtraTreeClassifier" + }, + { + "flow_estimator": "extra_tree_regressor", + "module": "tree.flow", + "fit": { + "name": "extra_tree_regressor_fit", + "returns": "ExtraTreeRegressor", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_samples", + "type": "i32" + }, + { + "name": "n_features", + "type": "i32" + }, + { + "name": "max_depth", + "type": "i32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "extra_tree_regressor_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "ExtraTreeRegressor" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "extra_tree_regressor_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "ExtraTreeRegressor" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_samples", + "n_features", + "max_depth", + "seed" + ], + "bucket": "runnable", + "sklearn_estimator": "ExtraTreeRegressor" + }, + { + "flow_estimator": "extra_trees_classifier", + "module": "ensemble.flow", + "fit": { + "name": "extra_trees_classifier_fit", + "returns": "ExtraTreesClassifier", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_classes", + "type": "i32" + }, + { + "name": "n_trees", + "type": "i32" + }, + { + "name": "max_depth", + "type": "i32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "extra_trees_classifier_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "ExtraTreesClassifier" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "extra_trees_classifier_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "ExtraTreesClassifier" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_classes", + "n_trees", + "max_depth", + "seed" + ], + "bucket": "runnable", + "sklearn_estimator": "ExtraTreesClassifier" + }, + { + "flow_estimator": "extra_trees_regressor", + "module": "ensemble.flow", + "fit": { + "name": "extra_trees_regressor_fit", + "returns": "ExtraTreesRegressor", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_trees", + "type": "i32" + }, + { + "name": "max_depth", + "type": "i32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "extra_trees_regressor_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "ExtraTreesRegressor" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "extra_trees_regressor_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "ExtraTreesRegressor" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_trees", + "max_depth", + "seed" + ], + "bucket": "runnable", + "sklearn_estimator": "ExtraTreesRegressor" + }, + { + "flow_estimator": "factor_analysis", + "module": "decomposition.flow", + "fit": { + "name": "factor_analysis_fit", + "returns": "FactorAnalysis", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "n_components", + "type": "i32" + }, + { + "name": "max_iter", + "type": "i32" + }, + { + "name": "tol", + "type": "f32" + } + ] + }, + "companions": { + "transform": { + "name": "factor_analysis_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "FactorAnalysis" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "factor_analysis_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "FactorAnalysis" + } + ] + } + }, + "parameters": [ + "X", + "n_components", + "max_iter", + "tol" + ], + "bucket": "runnable", + "sklearn_estimator": "FactorAnalysis" + }, + { + "flow_estimator": "fast_ica", + "module": "decomposition.flow", + "fit": { + "name": "fast_ica_fit", + "returns": "FastICA", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "n_components", + "type": "i32" + }, + { + "name": "max_iter", + "type": "i32" + }, + { + "name": "tol", + "type": "f32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "fast_ica_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "FastICA" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "fast_ica_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "FastICA" + } + ] + } + }, + "parameters": [ + "X", + "n_components", + "max_iter", + "tol", + "seed" + ], + "bucket": "runnable", + "sklearn_estimator": "FastICA" + }, + { + "flow_estimator": "feature_agglomeration", + "module": "cluster.flow", + "fit": { + "name": "feature_agglomeration_fit", + "returns": "FeatureAgglomeration", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "n_clusters", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "feature_agglomeration_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "FeatureAgglomeration" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "feature_agglomeration_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "FeatureAgglomeration" + } + ] + } + }, + "parameters": [ + "X", + "n_clusters" + ], + "bucket": "runnable", + "sklearn_estimator": "FeatureAgglomeration" + }, + { + "flow_estimator": "feature_union", + "module": "compose.flow", + "fit": { + "name": "feature_union_fit", + "returns": "FeatureUnion", + "parameters": [ + { + "name": "fu", + "type": "FeatureUnion" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "companions": { + "transform": { + "name": "feature_union_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "fu", + "type": "FeatureUnion" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "feature_union_free", + "returns": "void", + "parameters": [ + { + "name": "fu", + "type": "FeatureUnion" + } + ] + } + }, + "parameters": [ + "fu", + "X" + ], + "bucket": "different_shape", + "reason": "takes FeatureUnion first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "FeatureUnion" + }, + { + "flow_estimator": "gamma_regressor", + "module": "linear.flow", + "fit": { + "name": "gamma_regressor_fit", + "returns": "GammaRegressor", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "alpha", + "type": "f32" + }, + { + "name": "max_iter", + "type": "i32" + }, + { + "name": "lr", + "type": "f32" + } + ] + }, + "companions": { + "predict": { + "name": "gamma_regressor_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "GammaRegressor" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "gamma_regressor_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "GammaRegressor" + } + ] + } + }, + "parameters": [ + "X", + "y", + "alpha", + "max_iter", + "lr" + ], + "bucket": "runnable", + "sklearn_estimator": "GammaRegressor" + }, + { + "flow_estimator": "gaussian_mixture", + "module": "mixture.flow", + "fit": { + "name": "gaussian_mixture_fit", + "returns": "GaussianMixture", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "n_components", + "type": "i32" + }, + { + "name": "max_iter", + "type": "i32" + }, + { + "name": "tol", + "type": "f32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "gaussian_mixture_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "GaussianMixture" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "predict_proba": { + "name": "gaussian_mixture_predict_proba", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "GaussianMixture" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "gaussian_mixture_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "GaussianMixture" + } + ] + } + }, + "parameters": [ + "X", + "n_components", + "max_iter", + "tol", + "seed" + ], + "bucket": "runnable", + "sklearn_estimator": "GaussianMixture" + }, + { + "flow_estimator": "gaussian_nb", + "module": "naive_bayes.flow", + "fit": { + "name": "gaussian_nb_fit", + "returns": "GaussianNB", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_classes", + "type": "i32" + }, + { + "name": "var_smoothing", + "type": "f32" + } + ] + }, + "companions": { + "predict": { + "name": "gaussian_nb_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "GaussianNB" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "predict_proba": { + "name": "gaussian_nb_predict_proba", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "GaussianNB" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "gaussian_nb_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "GaussianNB" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_classes", + "var_smoothing" + ], + "bucket": "runnable", + "sklearn_estimator": "GaussianNB" + }, + { + "flow_estimator": "gaussian_process_classifier", + "module": "gaussian_process.flow", + "fit": { + "name": "gaussian_process_classifier_fit", + "returns": "GaussianProcessClassifier", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "gamma", + "type": "f32" + }, + { + "name": "max_iter", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "gaussian_process_classifier_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "GaussianProcessClassifier" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "gaussian_process_classifier_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "GaussianProcessClassifier" + } + ] + } + }, + "parameters": [ + "X", + "y", + "gamma", + "max_iter" + ], + "bucket": "runnable", + "sklearn_estimator": "GaussianProcessClassifier" + }, + { + "flow_estimator": "gaussian_process_regressor", + "module": "gaussian_process.flow", + "fit": { + "name": "gaussian_process_regressor_fit", + "returns": "GaussianProcessRegressor", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "alpha", + "type": "f32" + }, + { + "name": "length_scale", + "type": "f32" + }, + { + "name": "kernel_type", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "gaussian_process_regressor_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "GaussianProcessRegressor" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "gaussian_process_regressor_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "GaussianProcessRegressor" + } + ] + } + }, + "parameters": [ + "X", + "y", + "alpha", + "length_scale", + "kernel_type" + ], + "bucket": "runnable", + "sklearn_estimator": "GaussianProcessRegressor" + }, + { + "flow_estimator": "gaussian_random_projection", + "module": "random_projection.flow", + "fit": { + "name": "gaussian_random_projection_fit", + "returns": "GaussianRandomProjection", + "parameters": [ + { + "name": "n_features", + "type": "i32" + }, + { + "name": "n_components", + "type": "i32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "gaussian_random_projection_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "GaussianRandomProjection" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "gaussian_random_projection_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "GaussianRandomProjection" + } + ] + } + }, + "parameters": [ + "n_features", + "n_components", + "seed" + ], + "bucket": "different_shape", + "reason": "takes i32 first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "GaussianRandomProjection" + }, + { + "flow_estimator": "gradient_boosting_classifier", + "module": "ensemble.flow", + "fit": { + "name": "gradient_boosting_classifier_fit", + "returns": "GradientBoostingClassifier", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_classes", + "type": "i32" + }, + { + "name": "n_trees", + "type": "i32" + }, + { + "name": "learning_rate", + "type": "f32" + }, + { + "name": "max_depth", + "type": "i32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "gradient_boosting_classifier_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "GradientBoostingClassifier" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "predict_proba": { + "name": "gradient_boosting_classifier_predict_proba", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "GradientBoostingClassifier" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "gradient_boosting_classifier_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "GradientBoostingClassifier" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_classes", + "n_trees", + "learning_rate", + "max_depth", + "seed" + ], + "bucket": "runnable", + "sklearn_estimator": "GradientBoostingClassifier" + }, + { + "flow_estimator": "gradient_boosting_regressor", + "module": "ensemble.flow", + "fit": { + "name": "gradient_boosting_regressor_fit", + "returns": "GradientBoostingRegressor", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_trees", + "type": "i32" + }, + { + "name": "learning_rate", + "type": "f32" + }, + { + "name": "max_depth", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "gradient_boosting_regressor_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "GradientBoostingRegressor" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "gradient_boosting_regressor_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "GradientBoostingRegressor" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_trees", + "learning_rate", + "max_depth" + ], + "bucket": "runnable", + "sklearn_estimator": "GradientBoostingRegressor" + }, + { + "flow_estimator": "graphical_lasso_cv", + "module": "covariance.flow", + "fit": { + "name": "graphical_lasso_cv_fit", + "returns": "GraphicalLassoCV", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "n_alphas", + "type": "i32" + }, + { + "name": "n_folds", + "type": "i32" + }, + { + "name": "max_iter", + "type": "i32" + } + ] + }, + "companions": { + "free": { + "name": "graphical_lasso_cv_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "GraphicalLassoCV" + } + ] + } + }, + "parameters": [ + "X", + "n_alphas", + "n_folds", + "max_iter" + ], + "bucket": "flow_only", + "reason": "scikit-learn spells this GraphicalLassoCV; covered by that entry" + }, + { + "flow_estimator": "graphical_lasso", + "module": "covariance.flow", + "fit": { + "name": "graphical_lasso_fit", + "returns": "GraphicalLasso", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "alpha", + "type": "f32" + }, + { + "name": "max_iter", + "type": "i32" + }, + { + "name": "tol", + "type": "f32" + } + ] + }, + "companions": { + "free": { + "name": "graphical_lasso_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "GraphicalLasso" + } + ] + } + }, + "parameters": [ + "X", + "alpha", + "max_iter", + "tol" + ], + "bucket": "runnable", + "sklearn_estimator": "GraphicalLasso" + }, + { + "flow_estimator": "hdbscan", + "module": "cluster.flow", + "fit": { + "name": "hdbscan_fit", + "returns": "HDBSCAN", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "min_cluster_size", + "type": "i32" + }, + { + "name": "min_samples", + "type": "i32" + } + ] + }, + "companions": { + "free": { + "name": "hdbscan_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "HDBSCAN" + } + ] + } + }, + "parameters": [ + "X", + "min_cluster_size", + "min_samples" + ], + "bucket": "runnable", + "sklearn_estimator": "HDBSCAN" + }, + { + "flow_estimator": "hist_gradient_boosting_classifier", + "module": "ensemble.flow", + "fit": { + "name": "hist_gradient_boosting_classifier_fit", + "returns": "HistGradientBoostingClassifier", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_samples", + "type": "i32" + }, + { + "name": "n_classes", + "type": "i32" + }, + { + "name": "n_trees", + "type": "i32" + }, + { + "name": "learning_rate", + "type": "f32" + }, + { + "name": "max_depth", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "hist_gradient_boosting_classifier_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "HistGradientBoostingClassifier" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "hist_gradient_boosting_classifier_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "HistGradientBoostingClassifier" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_samples", + "n_classes", + "n_trees", + "learning_rate", + "max_depth" + ], + "bucket": "runnable", + "sklearn_estimator": "HistGradientBoostingClassifier" + }, + { + "flow_estimator": "hist_gradient_boosting_regressor", + "module": "ensemble.flow", + "fit": { + "name": "hist_gradient_boosting_regressor_fit", + "returns": "HistGradientBoostingRegressor", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_samples", + "type": "i32" + }, + { + "name": "n_trees", + "type": "i32" + }, + { + "name": "learning_rate", + "type": "f32" + }, + { + "name": "max_depth", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "hist_gradient_boosting_regressor_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "HistGradientBoostingRegressor" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "hist_gradient_boosting_regressor_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "HistGradientBoostingRegressor" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_samples", + "n_trees", + "learning_rate", + "max_depth" + ], + "bucket": "runnable", + "sklearn_estimator": "HistGradientBoostingRegressor" + }, + { + "flow_estimator": "huber_regressor", + "module": "linear.flow", + "fit": { + "name": "huber_regressor_fit", + "returns": "HuberRegressor", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "epsilon", + "type": "f32" + }, + { + "name": "max_iter", + "type": "i32" + }, + { + "name": "lr", + "type": "f32" + } + ] + }, + "companions": { + "predict": { + "name": "huber_regressor_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "HuberRegressor" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "huber_regressor_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "HuberRegressor" + } + ] + } + }, + "parameters": [ + "X", + "y", + "epsilon", + "max_iter", + "lr" + ], + "bucket": "runnable", + "sklearn_estimator": "HuberRegressor" + }, + { + "flow_estimator": "incremental_pca_partial", + "module": "decomposition.flow", + "fit": { + "name": "incremental_pca_partial_fit", + "returns": "IncrementalPCA", + "parameters": [ + { + "name": "model", + "type": "IncrementalPCA" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "companions": {}, + "parameters": [ + "model", + "X" + ], + "bucket": "different_shape", + "reason": "takes IncrementalPCA first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "IncrementalPCA" + }, + { + "flow_estimator": "isolation_forest", + "module": "ensemble.flow", + "fit": { + "name": "isolation_forest_fit", + "returns": "IsolationForest", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "n_trees", + "type": "i32" + }, + { + "name": "max_depth", + "type": "i32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "isolation_forest_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "IsolationForest" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "isolation_forest_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "IsolationForest" + } + ] + } + }, + "parameters": [ + "X", + "n_trees", + "max_depth", + "seed" + ], + "bucket": "runnable", + "sklearn_estimator": "IsolationForest" + }, + { + "flow_estimator": "isomap", + "module": "manifold.flow", + "fit": { + "name": "isomap_fit", + "returns": "Isomap", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "n_components", + "type": "i32" + }, + { + "name": "n_neighbors", + "type": "i32" + } + ] + }, + "companions": { + "free": { + "name": "isomap_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "Isomap" + } + ] + } + }, + "parameters": [ + "X", + "n_components", + "n_neighbors" + ], + "bucket": "runnable", + "sklearn_estimator": "Isomap" + }, + { + "flow_estimator": "isotonic", + "module": "isotonic.flow", + "fit": { + "name": "isotonic_fit", + "returns": "IsotonicRegression", + "parameters": [ + { + "name": "X", + "type": "ptr" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n", + "type": "i32" + }, + { + "name": "increasing", + "type": "bool" + } + ] + }, + "companions": { + "predict": { + "name": "isotonic_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "IsotonicRegression" + }, + { + "name": "X", + "type": "ptr" + }, + { + "name": "n", + "type": "i32" + } + ] + }, + "transform": { + "name": "isotonic_transform", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "IsotonicRegression" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n", + "type": "i32" + } + ] + }, + "free": { + "name": "isotonic_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "IsotonicRegression" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n", + "increasing" + ], + "bucket": "different_shape", + "reason": "takes ptr first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "IsotonicRegression" + }, + { + "flow_estimator": "iterative_imputer", + "module": "impute.flow", + "fit": { + "name": "iterative_imputer_fit", + "returns": "IterativeImputer", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "max_iter", + "type": "i32" + }, + { + "name": "tol", + "type": "f32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "iterative_imputer_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "IterativeImputer" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "iterative_imputer_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "IterativeImputer" + } + ] + } + }, + "parameters": [ + "X", + "max_iter", + "tol", + "seed" + ], + "bucket": "runnable", + "sklearn_estimator": "IterativeImputer" + }, + { + "flow_estimator": "kbins_discretizer", + "module": "preprocessing.flow", + "fit": { + "name": "kbins_discretizer_fit", + "returns": "KBinsDiscretizer", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "n_bins", + "type": "i32" + }, + { + "name": "strategy", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "kbins_discretizer_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "discretizer", + "type": "KBinsDiscretizer" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "kbins_discretizer_free", + "returns": "void", + "parameters": [ + { + "name": "discretizer", + "type": "KBinsDiscretizer" + } + ] + } + }, + "parameters": [ + "X", + "n_bins", + "strategy" + ], + "bucket": "runnable", + "sklearn_estimator": "KBinsDiscretizer" + }, + { + "flow_estimator": "kernel_density", + "module": "neighbors.flow", + "fit": { + "name": "kernel_density_fit", + "returns": "KernelDensity", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "bandwidth", + "type": "f32" + }, + { + "name": "kernel", + "type": "i32" + } + ] + }, + "companions": { + "free": { + "name": "kernel_density_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "KernelDensity" + } + ] + } + }, + "parameters": [ + "X", + "bandwidth", + "kernel" + ], + "bucket": "runnable", + "sklearn_estimator": "KernelDensity" + }, + { + "flow_estimator": "kernel_pca", + "module": "decomposition.flow", + "fit": { + "name": "kernel_pca_fit", + "returns": "KernelPCA", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "n_components", + "type": "i32" + }, + { + "name": "kernel", + "type": "i32" + }, + { + "name": "gamma", + "type": "f32" + }, + { + "name": "degree", + "type": "i32" + }, + { + "name": "coef0", + "type": "f32" + } + ] + }, + "companions": { + "transform": { + "name": "kernel_pca_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "KernelPCA" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "kernel_pca_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "KernelPCA" + } + ] + } + }, + "parameters": [ + "X", + "n_components", + "kernel", + "gamma", + "degree", + "coef0" + ], + "bucket": "runnable", + "sklearn_estimator": "KernelPCA" + }, + { + "flow_estimator": "kernel_ridge", + "module": "kernel_ridge.flow", + "fit": { + "name": "kernel_ridge_fit", + "returns": "KernelRidge", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "alpha", + "type": "f32" + }, + { + "name": "kernel_type", + "type": "i32" + }, + { + "name": "gamma", + "type": "f32" + }, + { + "name": "degree", + "type": "i32" + }, + { + "name": "coef0", + "type": "f32" + } + ] + }, + "companions": { + "predict": { + "name": "kernel_ridge_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "KernelRidge" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "kernel_ridge_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "KernelRidge" + } + ] + } + }, + "parameters": [ + "X", + "y", + "alpha", + "kernel_type", + "gamma", + "degree", + "coef0" + ], + "bucket": "runnable", + "sklearn_estimator": "KernelRidge" + }, + { + "flow_estimator": "kernel_svc", + "module": "svm.flow", + "fit": { + "name": "kernel_svc_fit", + "returns": "KernelSVC", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_classes", + "type": "i32" + }, + { + "name": "C", + "type": "f32" + }, + { + "name": "gamma", + "type": "f32" + }, + { + "name": "max_iter", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "kernel_svc_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "KernelSVC" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "decision_function": { + "name": "kernel_svc_decision_function", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "KernelSVC" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "kernel_svc_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "KernelSVC" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_classes", + "C", + "gamma", + "max_iter" + ], + "bucket": "runnable", + "sklearn_estimator": "SVC" + }, + { + "flow_estimator": "kernel_svc_multi", + "module": "svm.flow", + "fit": { + "name": "kernel_svc_multi_fit", + "returns": "KernelSVCMulti", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_classes", + "type": "i32" + }, + { + "name": "gamma", + "type": "f32" + }, + { + "name": "C", + "type": "f32" + }, + { + "name": "max_iter", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "kernel_svc_multi_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "KernelSVCMulti" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "decision_function": { + "name": "kernel_svc_multi_decision_function", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "KernelSVCMulti" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "kernel_svc_multi_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "KernelSVCMulti" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_classes", + "gamma", + "C", + "max_iter" + ], + "bucket": "runnable", + "sklearn_estimator": "SVC" + }, + { + "flow_estimator": "kmeans", + "module": "cluster.flow", + "fit": { + "name": "kmeans_fit", + "returns": "KMeans", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "n_clusters", + "type": "i32" + }, + { + "name": "max_iter", + "type": "i32" + }, + { + "name": "tol", + "type": "f32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "kmeans_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "KMeans" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "transform": { + "name": "kmeans_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "KMeans" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "kmeans_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "KMeans" + } + ] + } + }, + "parameters": [ + "X", + "n_clusters", + "max_iter", + "tol", + "seed" + ], + "bucket": "runnable", + "sklearn_estimator": "KMeans" + }, + { + "flow_estimator": "kneighbors_transformer", + "module": "neighbors.flow", + "fit": { + "name": "kneighbors_transformer_fit", + "returns": "KNeighborsTransformer", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "n_neighbors", + "type": "i32" + }, + { + "name": "mode", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "kneighbors_transformer_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "KNeighborsTransformer" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "kneighbors_transformer_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "KNeighborsTransformer" + } + ] + } + }, + "parameters": [ + "X", + "n_neighbors", + "mode" + ], + "bucket": "runnable", + "sklearn_estimator": "KNeighborsTransformer" + }, + { + "flow_estimator": "knn_classifier", + "module": "neighbors.flow", + "fit": { + "name": "knn_classifier_fit", + "returns": "KNNClassifier", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_classes", + "type": "i32" + }, + { + "name": "k", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "knn_classifier_predict", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "KNNClassifier" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "predict_proba": { + "name": "knn_classifier_predict_proba", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "KNNClassifier" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "knn_classifier_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "KNNClassifier" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_classes", + "k" + ], + "bucket": "runnable", + "sklearn_estimator": "KNeighborsClassifier" + }, + { + "flow_estimator": "knn_imputer", + "module": "impute.flow", + "fit": { + "name": "knn_imputer_fit", + "returns": "KNNImputer", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "n_neighbors", + "type": "i32" + }, + { + "name": "weights", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "knn_imputer_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "KNNImputer" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "knn_imputer_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "KNNImputer" + } + ] + } + }, + "parameters": [ + "X", + "n_neighbors", + "weights" + ], + "bucket": "runnable", + "sklearn_estimator": "KNNImputer" + }, + { + "flow_estimator": "knn_regressor", + "module": "neighbors.flow", + "fit": { + "name": "knn_regressor_fit", + "returns": "KNNRegressor", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "k", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "knn_regressor_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "KNNRegressor" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "knn_regressor_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "KNNRegressor" + } + ] + } + }, + "parameters": [ + "X", + "y", + "k" + ], + "bucket": "runnable", + "sklearn_estimator": "KNeighborsRegressor" + }, + { + "flow_estimator": "label_binarizer", + "module": "preprocessing.flow", + "fit": { + "name": "label_binarizer_fit", + "returns": "LabelBinarizer", + "parameters": [ + { + "name": "y", + "type": "ptr" + }, + { + "name": "n", + "type": "i32" + }, + { + "name": "neg_label", + "type": "f32" + }, + { + "name": "pos_label", + "type": "f32" + } + ] + }, + "companions": { + "transform": { + "name": "label_binarizer_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "LabelBinarizer" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n", + "type": "i32" + } + ] + }, + "free": { + "name": "label_binarizer_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "LabelBinarizer" + } + ] + } + }, + "parameters": [ + "y", + "n", + "neg_label", + "pos_label" + ], + "bucket": "different_shape", + "reason": "takes ptr first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "LabelBinarizer" + }, + { + "flow_estimator": "label_encoder", + "module": "preprocessing.flow", + "fit": { + "name": "label_encoder_fit", + "returns": "LabelEncoder", + "parameters": [ + { + "name": "y", + "type": "ptr" + }, + { + "name": "n", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "label_encoder_transform", + "returns": "ptr", + "parameters": [ + { + "name": "encoder", + "type": "LabelEncoder" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n", + "type": "i32" + } + ] + }, + "free": { + "name": "label_encoder_free", + "returns": "void", + "parameters": [ + { + "name": "encoder", + "type": "LabelEncoder" + } + ] + } + }, + "parameters": [ + "y", + "n" + ], + "bucket": "different_shape", + "reason": "takes ptr first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "LabelEncoder" + }, + { + "flow_estimator": "label_propagation", + "module": "semi_supervised.flow", + "fit": { + "name": "label_propagation_fit", + "returns": "LabelPropagation", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_classes", + "type": "i32" + }, + { + "name": "gamma", + "type": "f32" + }, + { + "name": "max_iter", + "type": "i32" + }, + { + "name": "tol", + "type": "f32" + } + ] + }, + "companions": { + "predict": { + "name": "label_propagation_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "LabelPropagation" + } + ] + }, + "predict_proba": { + "name": "label_propagation_predict_proba", + "returns": "ptr >", + "parameters": [ + { + "name": "model", + "type": "LabelPropagation" + } + ] + }, + "free": { + "name": "label_propagation_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "LabelPropagation" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_classes", + "gamma", + "max_iter", + "tol" + ], + "bucket": "runnable", + "sklearn_estimator": "LabelPropagation" + }, + { + "flow_estimator": "label_spreading", + "module": "semi_supervised.flow", + "fit": { + "name": "label_spreading_fit", + "returns": "LabelSpreading", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_classes", + "type": "i32" + }, + { + "name": "gamma", + "type": "f32" + }, + { + "name": "alpha", + "type": "f32" + }, + { + "name": "max_iter", + "type": "i32" + }, + { + "name": "tol", + "type": "f32" + } + ] + }, + "companions": { + "predict": { + "name": "label_spreading_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "LabelSpreading" + } + ] + }, + "free": { + "name": "label_spreading_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "LabelSpreading" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_classes", + "gamma", + "alpha", + "max_iter", + "tol" + ], + "bucket": "runnable", + "sklearn_estimator": "LabelSpreading" + }, + { + "flow_estimator": "lars_cv", + "module": "linear.flow", + "fit": { + "name": "lars_cv_fit", + "returns": "LarsCV", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "max_features", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "lars_cv_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "LarsCV" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "lars_cv_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "LarsCV" + } + ] + } + }, + "parameters": [ + "X", + "y", + "max_features" + ], + "bucket": "runnable", + "sklearn_estimator": "LarsCV" + }, + { + "flow_estimator": "lars", + "module": "linear.flow", + "fit": { + "name": "lars_fit", + "returns": "Lars", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_features_to_select", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "lars_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "Lars" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "lars_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "Lars" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_features_to_select" + ], + "bucket": "runnable", + "sklearn_estimator": "Lars" + }, + { + "flow_estimator": "lasso_cv", + "module": "linear.flow", + "fit": { + "name": "lasso_cv_fit", + "returns": "LassoCV", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_alphas", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "lasso_cv_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "LassoCV" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "lasso_cv_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "LassoCV" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_alphas" + ], + "bucket": "runnable", + "sklearn_estimator": "LassoCV" + }, + { + "flow_estimator": "lasso", + "module": "linear.flow", + "fit": { + "name": "lasso_fit", + "returns": "Lasso", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "alpha", + "type": "f32" + }, + { + "name": "epochs", + "type": "i32" + }, + { + "name": "lr", + "type": "f32" + } + ] + }, + "companions": { + "predict": { + "name": "lasso_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "Lasso" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "lasso_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "Lasso" + } + ] + } + }, + "parameters": [ + "X", + "y", + "alpha", + "epochs", + "lr" + ], + "bucket": "runnable", + "sklearn_estimator": "Lasso" + }, + { + "flow_estimator": "lasso_lars_cv", + "module": "linear.flow", + "fit": { + "name": "lasso_lars_cv_fit", + "returns": "LassoLarsCV", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_alphas", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "lasso_lars_cv_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "LassoLarsCV" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "lasso_lars_cv_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "LassoLarsCV" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_alphas" + ], + "bucket": "runnable", + "sklearn_estimator": "LassoLarsCV" + }, + { + "flow_estimator": "lasso_lars", + "module": "linear.flow", + "fit": { + "name": "lasso_lars_fit", + "returns": "LassoLars", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "alpha", + "type": "f32" + }, + { + "name": "max_iter", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "lasso_lars_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "LassoLars" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "lasso_lars_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "LassoLars" + } + ] + } + }, + "parameters": [ + "X", + "y", + "alpha", + "max_iter" + ], + "bucket": "runnable", + "sklearn_estimator": "LassoLars" + }, + { + "flow_estimator": "lasso_lars_ic", + "module": "extra_estimators.flow", + "fit": { + "name": "lasso_lars_ic_fit", + "returns": "LassoLarsIC", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "criterion", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "lasso_lars_ic_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "LassoLarsIC" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "lasso_lars_ic_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "LassoLarsIC" + } + ] + } + }, + "parameters": [ + "X", + "y", + "criterion" + ], + "bucket": "runnable", + "sklearn_estimator": "LassoLarsIC" + }, + { + "flow_estimator": "lda", + "module": "decomposition.flow", + "fit": { + "name": "lda_fit", + "returns": "LatentDirichletAllocation", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "n_topics", + "type": "i32" + }, + { + "name": "n_iter", + "type": "i32" + }, + { + "name": "alpha", + "type": "f32" + }, + { + "name": "eta", + "type": "f32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "lda_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "LatentDirichletAllocation" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "lda_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "LatentDirichletAllocation" + } + ] + } + }, + "parameters": [ + "X", + "n_topics", + "n_iter", + "alpha", + "eta", + "seed" + ], + "bucket": "runnable", + "sklearn_estimator": "LatentDirichletAllocation" + }, + { + "flow_estimator": "ledoit_wolf_cv", + "module": "covariance.flow", + "fit": { + "name": "ledoit_wolf_cv_fit", + "returns": "LedoitWolf", + "parameters": [ + { + "name": "X", + "type": "Matrix" + } + ] + }, + "companions": { + "free": { + "name": "ledoit_wolf_cv_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "LedoitWolf" + } + ] + } + }, + "parameters": [ + "X" + ], + "bucket": "flow_only", + "reason": "cross-validated variant scikit-learn does not expose" + }, + { + "flow_estimator": "ledoit_wolf_estimator", + "module": "covariance.flow", + "fit": { + "name": "ledoit_wolf_estimator_fit", + "returns": "ShrunkCovariance", + "parameters": [ + { + "name": "X", + "type": "Matrix" + } + ] + }, + "companions": {}, + "parameters": [ + "X" + ], + "bucket": "runnable", + "sklearn_estimator": "LedoitWolf" + }, + { + "flow_estimator": "linear_regression", + "module": "linear.flow", + "fit": { + "name": "linear_regression_fit", + "returns": "LinearRegression", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "penalty", + "type": "Penalty" + } + ] + }, + "companions": { + "predict": { + "name": "linear_regression_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "LinearRegression" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "linear_regression_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "LinearRegression" + } + ] + } + }, + "parameters": [ + "X", + "y", + "penalty" + ], + "bucket": "runnable", + "sklearn_estimator": "LinearRegression" + }, + { + "flow_estimator": "linear_svc", + "module": "svm.flow", + "fit": { + "name": "linear_svc_fit", + "returns": "LinearSVC", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_classes", + "type": "i32" + }, + { + "name": "C", + "type": "f32" + }, + { + "name": "epochs", + "type": "i32" + }, + { + "name": "lr", + "type": "f32" + } + ] + }, + "companions": { + "predict": { + "name": "linear_svc_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "LinearSVC" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "decision_function": { + "name": "linear_svc_decision_function", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "LinearSVC" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "linear_svc_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "LinearSVC" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_classes", + "C", + "epochs", + "lr" + ], + "bucket": "runnable", + "sklearn_estimator": "LinearSVC" + }, + { + "flow_estimator": "linear_svc_multi", + "module": "svm.flow", + "fit": { + "name": "linear_svc_multi_fit", + "returns": "LinearSVCMulti", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_classes", + "type": "i32" + }, + { + "name": "C", + "type": "f32" + }, + { + "name": "epochs", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "linear_svc_multi_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "LinearSVCMulti" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "decision_function": { + "name": "linear_svc_multi_decision_function", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "LinearSVCMulti" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "linear_svc_multi_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "LinearSVCMulti" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_classes", + "C", + "epochs" + ], + "bucket": "runnable", + "sklearn_estimator": "LinearSVC" + }, + { + "flow_estimator": "linear_svr", + "module": "svm.flow", + "fit": { + "name": "linear_svr_fit", + "returns": "LinearSVR", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "C", + "type": "f32" + }, + { + "name": "epsilon", + "type": "f32" + }, + { + "name": "epochs", + "type": "i32" + }, + { + "name": "lr", + "type": "f32" + } + ] + }, + "companions": { + "predict": { + "name": "linear_svr_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "LinearSVR" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "linear_svr_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "LinearSVR" + } + ] + } + }, + "parameters": [ + "X", + "y", + "C", + "epsilon", + "epochs", + "lr" + ], + "bucket": "runnable", + "sklearn_estimator": "LinearSVR" + }, + { + "flow_estimator": "lle", + "module": "manifold.flow", + "fit": { + "name": "lle_fit", + "returns": "LLE", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "n_components", + "type": "i32" + }, + { + "name": "n_neighbors", + "type": "i32" + } + ] + }, + "companions": { + "free": { + "name": "lle_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "LLE" + } + ] + } + }, + "parameters": [ + "X", + "n_components", + "n_neighbors" + ], + "bucket": "runnable", + "sklearn_estimator": "LocallyLinearEmbedding" + }, + { + "flow_estimator": "local_outlier_factor", + "module": "neighbors.flow", + "fit": { + "name": "local_outlier_factor_fit", + "returns": "LocalOutlierFactor", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "n_neighbors", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "local_outlier_factor_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "LocalOutlierFactor" + } + ] + }, + "free": { + "name": "local_outlier_factor_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "LocalOutlierFactor" + } + ] + } + }, + "parameters": [ + "X", + "n_neighbors" + ], + "bucket": "runnable", + "sklearn_estimator": "LocalOutlierFactor" + }, + { + "flow_estimator": "logistic_inference", + "module": "linear.flow", + "fit": { + "name": "logistic_inference_fit", + "returns": "LogisticInference", + "parameters": [ + { + "name": "model", + "type": "LogisticRegression" + }, + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + } + ] + }, + "companions": { + "free": { + "name": "logistic_inference_free", + "returns": "void", + "parameters": [ + { + "name": "inf", + "type": "LogisticInference" + } + ] + } + }, + "parameters": [ + "model", + "X", + "y" + ], + "bucket": "flow_only", + "reason": "inference summary (coefficient standard errors, Wald tests); statsmodels territory" + }, + { + "flow_estimator": "logistic_regression_cv", + "module": "linear.flow", + "fit": { + "name": "logistic_regression_cv_fit", + "returns": "LogisticRegressionCV", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_classes", + "type": "i32" + }, + { + "name": "n_Cs", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "logistic_regression_cv_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "LogisticRegressionCV" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "logistic_regression_cv_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "LogisticRegressionCV" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_classes", + "n_Cs" + ], + "bucket": "runnable", + "sklearn_estimator": "LogisticRegressionCV" + }, + { + "flow_estimator": "logistic_regression", + "module": "linear.flow", + "fit": { + "name": "logistic_regression_fit", + "returns": "LogisticRegression", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_classes", + "type": "i32" + }, + { + "name": "epochs", + "type": "i32" + }, + { + "name": "lr", + "type": "f32" + }, + { + "name": "penalty", + "type": "Penalty" + } + ] + }, + "companions": { + "free": { + "name": "logistic_regression_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "LogisticRegression" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_classes", + "epochs", + "lr", + "penalty" + ], + "bucket": "runnable", + "sklearn_estimator": "LogisticRegression" + }, + { + "flow_estimator": "lssvm_classifier", + "module": "svm.flow", + "fit": { + "name": "lssvm_classifier_fit", + "returns": "LSSVMClassifier", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_classes", + "type": "i32" + }, + { + "name": "gamma", + "type": "f32" + }, + { + "name": "alpha", + "type": "f32" + } + ] + }, + "companions": { + "predict": { + "name": "lssvm_classifier_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "LSSVMClassifier" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "lssvm_classifier_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "LSSVMClassifier" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_classes", + "gamma", + "alpha" + ], + "bucket": "flow_only", + "reason": "least-squares SVM; not in the scikit-learn public surface" + }, + { + "flow_estimator": "maxabs_scaler", + "module": "preprocessing.flow", + "fit": { + "name": "maxabs_scaler_fit", + "returns": "MaxAbsScaler", + "parameters": [ + { + "name": "X", + "type": "Matrix" + } + ] + }, + "companions": { + "transform": { + "name": "maxabs_scaler_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "scaler", + "type": "MaxAbsScaler" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "maxabs_scaler_free", + "returns": "void", + "parameters": [ + { + "name": "scaler", + "type": "MaxAbsScaler" + } + ] + } + }, + "parameters": [ + "X" + ], + "bucket": "runnable", + "sklearn_estimator": "MaxAbsScaler" + }, + { + "flow_estimator": "mds", + "module": "manifold.flow", + "fit": { + "name": "mds_fit", + "returns": "MDS", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "n_components", + "type": "i32" + }, + { + "name": "max_iter", + "type": "i32" + }, + { + "name": "tol", + "type": "f32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "free": { + "name": "mds_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "MDS" + } + ] + } + }, + "parameters": [ + "X", + "n_components", + "max_iter", + "tol", + "seed" + ], + "bucket": "runnable", + "sklearn_estimator": "MDS" + }, + { + "flow_estimator": "mean_shift", + "module": "cluster.flow", + "fit": { + "name": "mean_shift_fit", + "returns": "MeanShift", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "bandwidth", + "type": "f32" + }, + { + "name": "max_iter", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "mean_shift_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "MeanShift" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "mean_shift_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "MeanShift" + } + ] + } + }, + "parameters": [ + "X", + "bandwidth", + "max_iter" + ], + "bucket": "runnable", + "sklearn_estimator": "MeanShift" + }, + { + "flow_estimator": "min_cov_det", + "module": "covariance.flow", + "fit": { + "name": "min_cov_det_fit", + "returns": "MinCovDet", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "h_fraction", + "type": "f32" + }, + { + "name": "max_iter", + "type": "i32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "free": { + "name": "min_cov_det_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "MinCovDet" + } + ] + } + }, + "parameters": [ + "X", + "h_fraction", + "max_iter", + "seed" + ], + "bucket": "runnable", + "sklearn_estimator": "MinCovDet" + }, + { + "flow_estimator": "minibatch_dictionary_learning", + "module": "decomposition.flow", + "fit": { + "name": "minibatch_dictionary_learning_fit", + "returns": "MiniBatchDictionaryLearning", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "n_components", + "type": "i32" + }, + { + "name": "n_iter", + "type": "i32" + }, + { + "name": "batch_size", + "type": "i32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "minibatch_dictionary_learning_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "MiniBatchDictionaryLearning" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "minibatch_dictionary_learning_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "MiniBatchDictionaryLearning" + } + ] + } + }, + "parameters": [ + "X", + "n_components", + "n_iter", + "batch_size", + "seed" + ], + "bucket": "runnable", + "sklearn_estimator": "MiniBatchDictionaryLearning" + }, + { + "flow_estimator": "minibatch_kmeans", + "module": "cluster.flow", + "fit": { + "name": "minibatch_kmeans_fit", + "returns": "MiniBatchKMeans", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "n_clusters", + "type": "i32" + }, + { + "name": "batch_size", + "type": "i32" + }, + { + "name": "max_iter", + "type": "i32" + }, + { + "name": "tol", + "type": "f32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "minibatch_kmeans_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "MiniBatchKMeans" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "minibatch_kmeans_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "MiniBatchKMeans" + } + ] + } + }, + "parameters": [ + "X", + "n_clusters", + "batch_size", + "max_iter", + "tol", + "seed" + ], + "bucket": "runnable", + "sklearn_estimator": "MiniBatchKMeans" + }, + { + "flow_estimator": "minibatch_nmf", + "module": "decomposition.flow", + "fit": { + "name": "minibatch_nmf_fit", + "returns": "MiniBatchNMF", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "n_components", + "type": "i32" + }, + { + "name": "max_iter", + "type": "i32" + }, + { + "name": "batch_size", + "type": "i32" + }, + { + "name": "tol", + "type": "f32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "minibatch_nmf_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "MiniBatchNMF" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "minibatch_nmf_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "MiniBatchNMF" + } + ] + } + }, + "parameters": [ + "X", + "n_components", + "max_iter", + "batch_size", + "tol", + "seed" + ], + "bucket": "runnable", + "sklearn_estimator": "MiniBatchNMF" + }, + { + "flow_estimator": "minibatch_sparse_pca", + "module": "decomposition.flow", + "fit": { + "name": "minibatch_sparse_pca_fit", + "returns": "MiniBatchSparsePCA", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "n_components", + "type": "i32" + }, + { + "name": "alpha", + "type": "f32" + }, + { + "name": "max_iter", + "type": "i32" + }, + { + "name": "batch_size", + "type": "i32" + }, + { + "name": "tol", + "type": "f32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "minibatch_sparse_pca_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "MiniBatchSparsePCA" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "minibatch_sparse_pca_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "MiniBatchSparsePCA" + } + ] + } + }, + "parameters": [ + "X", + "n_components", + "alpha", + "max_iter", + "batch_size", + "tol", + "seed" + ], + "bucket": "runnable", + "sklearn_estimator": "MiniBatchSparsePCA" + }, + { + "flow_estimator": "minmax_scaler", + "module": "preprocessing.flow", + "fit": { + "name": "minmax_scaler_fit", + "returns": "MinMaxScaler", + "parameters": [ + { + "name": "X", + "type": "Matrix" + } + ] + }, + "companions": { + "transform": { + "name": "minmax_scaler_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "scaler", + "type": "MinMaxScaler" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "minmax_scaler_free", + "returns": "void", + "parameters": [ + { + "name": "scaler", + "type": "MinMaxScaler" + } + ] + } + }, + "parameters": [ + "X" + ], + "bucket": "runnable", + "sklearn_estimator": "MinMaxScaler" + }, + { + "flow_estimator": "missing_indicator", + "module": "impute.flow", + "fit": { + "name": "missing_indicator_fit", + "returns": "MissingIndicator", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "missing_values", + "type": "f32" + }, + { + "name": "features", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "missing_indicator_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "MissingIndicator" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "missing_indicator_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "MissingIndicator" + } + ] + } + }, + "parameters": [ + "X", + "missing_values", + "features" + ], + "bucket": "runnable", + "sklearn_estimator": "MissingIndicator" + }, + { + "flow_estimator": "mlp_classifier", + "module": "neural_network.flow", + "fit": { + "name": "mlp_classifier_fit", + "returns": "MLPClassifier", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_classes", + "type": "i32" + }, + { + "name": "hidden_sizes", + "type": "ptr" + }, + { + "name": "n_hidden", + "type": "i32" + }, + { + "name": "activation", + "type": "i32" + }, + { + "name": "epochs", + "type": "i32" + }, + { + "name": "lr", + "type": "f32" + }, + { + "name": "momentum", + "type": "f32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "mlp_classifier_predict", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "MLPClassifier" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "mlp_classifier_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "MLPClassifier" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_classes", + "hidden_sizes", + "n_hidden", + "activation", + "epochs", + "lr", + "momentum", + "seed" + ], + "bucket": "runnable", + "sklearn_estimator": "MLPClassifier" + }, + { + "flow_estimator": "mlp_regressor", + "module": "neural_network.flow", + "fit": { + "name": "mlp_regressor_fit", + "returns": "MLPRegressor", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "hidden_sizes", + "type": "ptr" + }, + { + "name": "n_hidden", + "type": "i32" + }, + { + "name": "activation", + "type": "i32" + }, + { + "name": "epochs", + "type": "i32" + }, + { + "name": "lr", + "type": "f32" + }, + { + "name": "momentum", + "type": "f32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "mlp_regressor_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "MLPRegressor" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "mlp_regressor_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "MLPRegressor" + } + ] + } + }, + "parameters": [ + "X", + "y", + "hidden_sizes", + "n_hidden", + "activation", + "epochs", + "lr", + "momentum", + "seed" + ], + "bucket": "runnable", + "sklearn_estimator": "MLPRegressor" + }, + { + "flow_estimator": "multi_output_classifier", + "module": "multioutput.flow", + "fit": { + "name": "multi_output_classifier_fit", + "returns": "MultiOutputClassifier", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "Y", + "type": "Matrix" + }, + { + "name": "n_outputs", + "type": "i32" + }, + { + "name": "n_classes", + "type": "i32" + }, + { + "name": "epochs", + "type": "i32" + }, + { + "name": "lr", + "type": "f32" + } + ] + }, + "companions": { + "predict": { + "name": "multi_output_classifier_predict", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "MultiOutputClassifier" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "multi_output_classifier_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "MultiOutputClassifier" + } + ] + } + }, + "parameters": [ + "X", + "Y", + "n_outputs", + "n_classes", + "epochs", + "lr" + ], + "bucket": "runnable", + "sklearn_estimator": "MultiOutputClassifier" + }, + { + "flow_estimator": "multi_output_regressor", + "module": "multioutput.flow", + "fit": { + "name": "multi_output_regressor_fit", + "returns": "MultiOutputRegressor", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "Y", + "type": "Matrix" + }, + { + "name": "n_outputs", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "multi_output_regressor_predict", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "MultiOutputRegressor" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "multi_output_regressor_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "MultiOutputRegressor" + } + ] + } + }, + "parameters": [ + "X", + "Y", + "n_outputs" + ], + "bucket": "runnable", + "sklearn_estimator": "MultiOutputRegressor" + }, + { + "flow_estimator": "multiclass_logistic", + "module": "linear.flow", + "fit": { + "name": "multiclass_logistic_fit", + "returns": "MultiClassLogisticRegression", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_classes", + "type": "i32" + }, + { + "name": "epochs", + "type": "i32" + }, + { + "name": "lr", + "type": "f32" + }, + { + "name": "penalty", + "type": "Penalty" + } + ] + }, + "companions": { + "predict": { + "name": "multiclass_logistic_predict", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "MultiClassLogisticRegression" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "multiclass_logistic_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "MultiClassLogisticRegression" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_classes", + "epochs", + "lr", + "penalty" + ], + "bucket": "runnable", + "sklearn_estimator": "LogisticRegression" + }, + { + "flow_estimator": "multilabel_binarizer", + "module": "preprocessing.flow", + "fit": { + "name": "multilabel_binarizer_fit", + "returns": "MultiLabelBinarizer", + "parameters": [ + { + "name": "y", + "type": "ptr >" + }, + { + "name": "n_samples", + "type": "i32" + }, + { + "name": "n_labels_per_sample", + "type": "ptr" + }, + { + "name": "n_classes", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "multilabel_binarizer_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "MultiLabelBinarizer" + }, + { + "name": "y", + "type": "ptr >" + }, + { + "name": "n_samples", + "type": "i32" + }, + { + "name": "n_labels_per_sample", + "type": "ptr" + } + ] + }, + "free": { + "name": "multilabel_binarizer_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "MultiLabelBinarizer" + } + ] + } + }, + "parameters": [ + "y", + "n_samples", + "n_labels_per_sample", + "n_classes" + ], + "bucket": "different_shape", + "reason": "takes ptr > first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "MultiLabelBinarizer" + }, + { + "flow_estimator": "multinomial_nb", + "module": "naive_bayes.flow", + "fit": { + "name": "multinomial_nb_fit", + "returns": "MultinomialNB", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_classes", + "type": "i32" + }, + { + "name": "alpha", + "type": "f32" + } + ] + }, + "companions": { + "predict": { + "name": "multinomial_nb_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "MultinomialNB" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "predict_proba": { + "name": "multinomial_nb_predict_proba", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "MultinomialNB" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "multinomial_nb_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "MultinomialNB" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_classes", + "alpha" + ], + "bucket": "runnable", + "sklearn_estimator": "MultinomialNB" + }, + { + "flow_estimator": "multitask_elastic_net_cv", + "module": "linear.flow", + "fit": { + "name": "multitask_elastic_net_cv_fit", + "returns": "MultiTaskElasticNetCV", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "Y", + "type": "Matrix" + }, + { + "name": "n_alphas", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "multitask_elastic_net_cv_predict", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "MultiTaskElasticNetCV" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "multitask_elastic_net_cv_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "MultiTaskElasticNetCV" + } + ] + } + }, + "parameters": [ + "X", + "Y", + "n_alphas" + ], + "bucket": "runnable", + "sklearn_estimator": "MultiTaskElasticNetCV" + }, + { + "flow_estimator": "multitask_elastic_net", + "module": "linear.flow", + "fit": { + "name": "multitask_elastic_net_fit", + "returns": "MultiTaskElasticNet", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "Y", + "type": "Matrix" + }, + { + "name": "alpha", + "type": "f32" + }, + { + "name": "l1_ratio", + "type": "f32" + }, + { + "name": "epochs", + "type": "i32" + }, + { + "name": "lr", + "type": "f32" + } + ] + }, + "companions": { + "predict": { + "name": "multitask_elastic_net_predict", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "MultiTaskElasticNet" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "multitask_elastic_net_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "MultiTaskElasticNet" + } + ] + } + }, + "parameters": [ + "X", + "Y", + "alpha", + "l1_ratio", + "epochs", + "lr" + ], + "bucket": "runnable", + "sklearn_estimator": "MultiTaskElasticNet" + }, + { + "flow_estimator": "multitask_lasso_cv", + "module": "linear.flow", + "fit": { + "name": "multitask_lasso_cv_fit", + "returns": "MultiTaskLassoCV", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "Y", + "type": "Matrix" + }, + { + "name": "n_alphas", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "multitask_lasso_cv_predict", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "MultiTaskLassoCV" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "multitask_lasso_cv_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "MultiTaskLassoCV" + } + ] + } + }, + "parameters": [ + "X", + "Y", + "n_alphas" + ], + "bucket": "runnable", + "sklearn_estimator": "MultiTaskLassoCV" + }, + { + "flow_estimator": "multitask_lasso", + "module": "linear.flow", + "fit": { + "name": "multitask_lasso_fit", + "returns": "MultiTaskLasso", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "Y", + "type": "Matrix" + }, + { + "name": "alpha", + "type": "f32" + }, + { + "name": "epochs", + "type": "i32" + }, + { + "name": "lr", + "type": "f32" + } + ] + }, + "companions": { + "predict": { + "name": "multitask_lasso_predict", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "MultiTaskLasso" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "multitask_lasso_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "MultiTaskLasso" + } + ] + } + }, + "parameters": [ + "X", + "Y", + "alpha", + "epochs", + "lr" + ], + "bucket": "runnable", + "sklearn_estimator": "MultiTaskLasso" + }, + { + "flow_estimator": "nca", + "module": "extra_estimators.flow", + "fit": { + "name": "nca_fit", + "returns": "NeighborhoodComponentsAnalysis", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_components", + "type": "i32" + }, + { + "name": "epochs", + "type": "i32" + }, + { + "name": "lr", + "type": "f32" + } + ] + }, + "companions": { + "transform": { + "name": "nca_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "NeighborhoodComponentsAnalysis" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "nca_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "NeighborhoodComponentsAnalysis" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_components", + "epochs", + "lr" + ], + "bucket": "simplified", + "reason": "the implementation's own comments call it simplified, so timing it against scikit-learn's algorithm compares two different things", + "sklearn_estimator": "NeighborhoodComponentsAnalysis" + }, + { + "flow_estimator": "nearest_centroid", + "module": "neighbors.flow", + "fit": { + "name": "nearest_centroid_fit", + "returns": "NearestCentroid", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_classes", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "nearest_centroid_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "NearestCentroid" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "nearest_centroid_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "NearestCentroid" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_classes" + ], + "bucket": "runnable", + "sklearn_estimator": "NearestCentroid" + }, + { + "flow_estimator": "nearest_neighbors", + "module": "neighbors.flow", + "fit": { + "name": "nearest_neighbors_fit", + "returns": "NearestNeighbors", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "n_neighbors", + "type": "i32" + } + ] + }, + "companions": { + "free": { + "name": "nearest_neighbors_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "NearestNeighbors" + } + ] + } + }, + "parameters": [ + "X", + "n_neighbors" + ], + "bucket": "runnable", + "sklearn_estimator": "NearestNeighbors" + }, + { + "flow_estimator": "nmf", + "module": "decomposition.flow", + "fit": { + "name": "nmf_fit", + "returns": "NMF", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "n_components", + "type": "i32" + }, + { + "name": "max_iter", + "type": "i32" + }, + { + "name": "tol", + "type": "f32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "nmf_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "NMF" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "nmf_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "NMF" + } + ] + } + }, + "parameters": [ + "X", + "n_components", + "max_iter", + "tol", + "seed" + ], + "bucket": "runnable", + "sklearn_estimator": "NMF" + }, + { + "flow_estimator": "nu_svc", + "module": "svm.flow", + "fit": { + "name": "nu_svc_fit", + "returns": "NuSVC", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "nu", + "type": "f32" + }, + { + "name": "gamma", + "type": "f32" + }, + { + "name": "max_iter", + "type": "i32" + }, + { + "name": "C", + "type": "f32" + } + ] + }, + "companions": { + "predict": { + "name": "nu_svc_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "NuSVC" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "nu_svc_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "NuSVC" + } + ] + } + }, + "parameters": [ + "X", + "y", + "nu", + "gamma", + "max_iter", + "C" + ], + "bucket": "runnable", + "sklearn_estimator": "NuSVC" + }, + { + "flow_estimator": "nu_svr", + "module": "svm.flow", + "fit": { + "name": "nu_svr_fit", + "returns": "NuSVR", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "nu", + "type": "f32" + }, + { + "name": "C", + "type": "f32" + }, + { + "name": "gamma", + "type": "f32" + }, + { + "name": "max_iter", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "nu_svr_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "NuSVR" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "nu_svr_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "NuSVR" + } + ] + } + }, + "parameters": [ + "X", + "y", + "nu", + "C", + "gamma", + "max_iter" + ], + "bucket": "runnable", + "sklearn_estimator": "NuSVR" + }, + { + "flow_estimator": "nystroem", + "module": "kernel_approximation.flow", + "fit": { + "name": "nystroem_fit", + "returns": "Nystroem", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "n_components", + "type": "i32" + }, + { + "name": "gamma", + "type": "f32" + }, + { + "name": "kernel", + "type": "i32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "nystroem_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "Nystroem" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "nystroem_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "Nystroem" + } + ] + } + }, + "parameters": [ + "X", + "n_components", + "gamma", + "kernel", + "seed" + ], + "bucket": "runnable", + "sklearn_estimator": "Nystroem" + }, + { + "flow_estimator": "oas_cv", + "module": "covariance.flow", + "fit": { + "name": "oas_cv_fit", + "returns": "OAS", + "parameters": [ + { + "name": "X", + "type": "Matrix" + } + ] + }, + "companions": { + "free": { + "name": "oas_cv_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "OAS" + } + ] + } + }, + "parameters": [ + "X" + ], + "bucket": "flow_only", + "reason": "cross-validated variant scikit-learn does not expose" + }, + { + "flow_estimator": "oas_estimator", + "module": "covariance.flow", + "fit": { + "name": "oas_estimator_fit", + "returns": "ShrunkCovariance", + "parameters": [ + { + "name": "X", + "type": "Matrix" + } + ] + }, + "companions": {}, + "parameters": [ + "X" + ], + "bucket": "runnable", + "sklearn_estimator": "OAS" + }, + { + "flow_estimator": "ols_inference", + "module": "linear.flow", + "fit": { + "name": "ols_inference_fit", + "returns": "OLSInference", + "parameters": [ + { + "name": "model", + "type": "LinearRegression" + }, + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + } + ] + }, + "companions": { + "free": { + "name": "ols_inference_free", + "returns": "void", + "parameters": [ + { + "name": "inf", + "type": "OLSInference" + } + ] + } + }, + "parameters": [ + "model", + "X", + "y" + ], + "bucket": "flow_only", + "reason": "inference summary for ordinary least squares; statsmodels territory" + }, + { + "flow_estimator": "omp_cv", + "module": "extra_estimators.flow", + "fit": { + "name": "omp_cv_fit", + "returns": "OrthogonalMatchingPursuitCV", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "max_coefs", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "omp_cv_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "OrthogonalMatchingPursuitCV" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "omp_cv_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "OrthogonalMatchingPursuitCV" + } + ] + } + }, + "parameters": [ + "X", + "y", + "max_coefs" + ], + "bucket": "runnable", + "sklearn_estimator": "OrthogonalMatchingPursuitCV" + }, + { + "flow_estimator": "one_class_svm", + "module": "svm.flow", + "fit": { + "name": "one_class_svm_fit", + "returns": "OneClassSVM", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "nu", + "type": "f32" + }, + { + "name": "gamma", + "type": "f32" + }, + { + "name": "max_iter", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "one_class_svm_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "OneClassSVM" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "one_class_svm_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "OneClassSVM" + } + ] + } + }, + "parameters": [ + "X", + "nu", + "gamma", + "max_iter" + ], + "bucket": "runnable", + "sklearn_estimator": "OneClassSVM" + }, + { + "flow_estimator": "one_vs_one", + "module": "meta_estimators.flow", + "fit": { + "name": "one_vs_one_fit", + "returns": "OneVsOneClassifier", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_classes", + "type": "i32" + }, + { + "name": "epochs", + "type": "i32" + }, + { + "name": "lr", + "type": "f32" + }, + { + "name": "penalty", + "type": "Penalty" + } + ] + }, + "companions": { + "predict": { + "name": "one_vs_one_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "OneVsOneClassifier" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "one_vs_one_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "OneVsOneClassifier" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_classes", + "epochs", + "lr", + "penalty" + ], + "bucket": "runnable", + "sklearn_estimator": "OneVsOneClassifier" + }, + { + "flow_estimator": "one_vs_rest", + "module": "meta_estimators.flow", + "fit": { + "name": "one_vs_rest_fit", + "returns": "OneVsRestClassifier", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_classes", + "type": "i32" + }, + { + "name": "epochs", + "type": "i32" + }, + { + "name": "lr", + "type": "f32" + }, + { + "name": "penalty", + "type": "Penalty" + } + ] + }, + "companions": { + "predict": { + "name": "one_vs_rest_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "OneVsRestClassifier" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "one_vs_rest_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "OneVsRestClassifier" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_classes", + "epochs", + "lr", + "penalty" + ], + "bucket": "runnable", + "sklearn_estimator": "OneVsRestClassifier" + }, + { + "flow_estimator": "onehot_encoder", + "module": "preprocessing.flow", + "fit": { + "name": "onehot_encoder_fit", + "returns": "OneHotEncoder", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "handle_unknown", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "onehot_encoder_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "encoder", + "type": "OneHotEncoder" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "onehot_encoder_free", + "returns": "void", + "parameters": [ + { + "name": "encoder", + "type": "OneHotEncoder" + } + ] + } + }, + "parameters": [ + "X", + "handle_unknown" + ], + "bucket": "runnable", + "sklearn_estimator": "OneHotEncoder" + }, + { + "flow_estimator": "optics", + "module": "cluster.flow", + "fit": { + "name": "optics_fit", + "returns": "OPTICS", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "eps", + "type": "f32" + }, + { + "name": "min_samples", + "type": "i32" + } + ] + }, + "companions": { + "free": { + "name": "optics_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "OPTICS" + } + ] + } + }, + "parameters": [ + "X", + "eps", + "min_samples" + ], + "bucket": "runnable", + "sklearn_estimator": "OPTICS" + }, + { + "flow_estimator": "ordinal_encoder", + "module": "preprocessing.flow", + "fit": { + "name": "ordinal_encoder_fit", + "returns": "OrdinalEncoder", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "handle_unknown", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "ordinal_encoder_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "encoder", + "type": "OrdinalEncoder" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "ordinal_encoder_free", + "returns": "void", + "parameters": [ + { + "name": "encoder", + "type": "OrdinalEncoder" + } + ] + } + }, + "parameters": [ + "X", + "handle_unknown" + ], + "bucket": "runnable", + "sklearn_estimator": "OrdinalEncoder" + }, + { + "flow_estimator": "orthogonal_matching_pursuit", + "module": "linear.flow", + "fit": { + "name": "orthogonal_matching_pursuit_fit", + "returns": "OrthogonalMatchingPursuit", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_nonzero_coefs", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "orthogonal_matching_pursuit_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "OrthogonalMatchingPursuit" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "orthogonal_matching_pursuit_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "OrthogonalMatchingPursuit" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_nonzero_coefs" + ], + "bucket": "runnable", + "sklearn_estimator": "OrthogonalMatchingPursuit" + }, + { + "flow_estimator": "output_code", + "module": "meta_estimators.flow", + "fit": { + "name": "output_code_fit", + "returns": "OutputCodeClassifier", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_classes", + "type": "i32" + }, + { + "name": "n_bits", + "type": "i32" + }, + { + "name": "epochs", + "type": "i32" + }, + { + "name": "lr", + "type": "f32" + }, + { + "name": "penalty", + "type": "Penalty" + } + ] + }, + "companions": { + "predict": { + "name": "output_code_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "OutputCodeClassifier" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "output_code_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "OutputCodeClassifier" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_classes", + "n_bits", + "epochs", + "lr", + "penalty" + ], + "bucket": "runnable", + "sklearn_estimator": "OutputCodeClassifier" + }, + { + "flow_estimator": "passive_aggressive_classifier", + "module": "linear.flow", + "fit": { + "name": "passive_aggressive_classifier_fit", + "returns": "PassiveAggressiveClassifier", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_iter", + "type": "i32" + }, + { + "name": "C", + "type": "f32" + } + ] + }, + "companions": { + "predict": { + "name": "passive_aggressive_classifier_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "PassiveAggressiveClassifier" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "passive_aggressive_classifier_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "PassiveAggressiveClassifier" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_iter", + "C" + ], + "bucket": "runnable", + "sklearn_estimator": "PassiveAggressiveClassifier" + }, + { + "flow_estimator": "passive_aggressive_regressor", + "module": "linear.flow", + "fit": { + "name": "passive_aggressive_regressor_fit", + "returns": "PassiveAggressiveRegressor", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "C", + "type": "f32" + }, + { + "name": "max_iter", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "passive_aggressive_regressor_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "PassiveAggressiveRegressor" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "passive_aggressive_regressor_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "PassiveAggressiveRegressor" + } + ] + } + }, + "parameters": [ + "X", + "y", + "C", + "max_iter" + ], + "bucket": "runnable", + "sklearn_estimator": "PassiveAggressiveRegressor" + }, + { + "flow_estimator": "pca", + "module": "decomposition.flow", + "fit": { + "name": "pca_fit", + "returns": "PCA", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "n_components", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "pca_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "PCA" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "pca_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "PCA" + } + ] + } + }, + "parameters": [ + "X", + "n_components" + ], + "bucket": "runnable", + "sklearn_estimator": "PCA" + }, + { + "flow_estimator": "perceptron", + "module": "linear.flow", + "fit": { + "name": "perceptron_fit", + "returns": "Perceptron", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_iter", + "type": "i32" + }, + { + "name": "lr", + "type": "f32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "perceptron_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "Perceptron" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "perceptron_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "Perceptron" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_iter", + "lr", + "seed" + ], + "bucket": "runnable", + "sklearn_estimator": "Perceptron" + }, + { + "flow_estimator": "pipeline", + "module": "pipeline.flow", + "fit": { + "name": "pipeline_fit", + "returns": "Pipeline", + "parameters": [ + { + "name": "pipeline", + "type": "Pipeline" + }, + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + } + ] + }, + "companions": { + "predict": { + "name": "pipeline_predict", + "returns": "ptr", + "parameters": [ + { + "name": "pipeline", + "type": "Pipeline" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "predict_proba": { + "name": "pipeline_predict_proba", + "returns": "ptr", + "parameters": [ + { + "name": "pipeline", + "type": "Pipeline" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "pipeline_free", + "returns": "void", + "parameters": [ + { + "name": "pipeline", + "type": "Pipeline" + } + ] + } + }, + "parameters": [ + "pipeline", + "X", + "y" + ], + "bucket": "different_shape", + "reason": "takes Pipeline first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "Pipeline" + }, + { + "flow_estimator": "pls_canonical", + "module": "cross_decomposition.flow", + "fit": { + "name": "pls_canonical_fit", + "returns": "PLSCanonical", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "Y", + "type": "Matrix" + }, + { + "name": "n_components", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "pls_canonical_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "PLSCanonical" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "pls_canonical_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "PLSCanonical" + } + ] + } + }, + "parameters": [ + "X", + "Y", + "n_components" + ], + "bucket": "runnable", + "sklearn_estimator": "PLSCanonical" + }, + { + "flow_estimator": "pls", + "module": "cross_decomposition.flow", + "fit": { + "name": "pls_fit", + "returns": "PLSRegression", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "Y", + "type": "Matrix" + }, + { + "name": "n_components", + "type": "i32" + }, + { + "name": "max_iter", + "type": "i32" + }, + { + "name": "tol", + "type": "f32" + } + ] + }, + "companions": { + "predict": { + "name": "pls_predict", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "PLSRegression" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "transform": { + "name": "pls_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "PLSRegression" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "pls_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "PLSRegression" + } + ] + } + }, + "parameters": [ + "X", + "Y", + "n_components", + "max_iter", + "tol" + ], + "bucket": "runnable", + "sklearn_estimator": "PLSRegression" + }, + { + "flow_estimator": "pls_svd", + "module": "cross_decomposition.flow", + "fit": { + "name": "pls_svd_fit", + "returns": "PLSSVD", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "Y", + "type": "Matrix" + }, + { + "name": "n_components", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "pls_svd_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "PLSSVD" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "pls_svd_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "PLSSVD" + } + ] + } + }, + "parameters": [ + "X", + "Y", + "n_components" + ], + "bucket": "runnable", + "sklearn_estimator": "PLSSVD" + }, + { + "flow_estimator": "poisson_regressor", + "module": "linear.flow", + "fit": { + "name": "poisson_regressor_fit", + "returns": "PoissonRegressor", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "alpha", + "type": "f32" + }, + { + "name": "max_iter", + "type": "i32" + }, + { + "name": "lr", + "type": "f32" + } + ] + }, + "companions": { + "predict": { + "name": "poisson_regressor_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "PoissonRegressor" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "poisson_regressor_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "PoissonRegressor" + } + ] + } + }, + "parameters": [ + "X", + "y", + "alpha", + "max_iter", + "lr" + ], + "bucket": "runnable", + "sklearn_estimator": "PoissonRegressor" + }, + { + "flow_estimator": "polynomial_count_sketch", + "module": "extra_estimators.flow", + "fit": { + "name": "polynomial_count_sketch_fit", + "returns": "PolynomialCountSketch", + "parameters": [ + { + "name": "n_features", + "type": "i32" + }, + { + "name": "n_components", + "type": "i32" + }, + { + "name": "degree", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "polynomial_count_sketch_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "PolynomialCountSketch" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "polynomial_count_sketch_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "PolynomialCountSketch" + } + ] + } + }, + "parameters": [ + "n_features", + "n_components", + "degree" + ], + "bucket": "different_shape", + "reason": "takes i32 first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "PolynomialCountSketch" + }, + { + "flow_estimator": "polynomial_features", + "module": "preprocessing.flow", + "fit": { + "name": "polynomial_features_fit", + "returns": "PolynomialFeatures", + "parameters": [ + { + "name": "n_features", + "type": "i32" + }, + { + "name": "degree", + "type": "i32" + }, + { + "name": "interaction_only", + "type": "bool" + }, + { + "name": "include_bias", + "type": "bool" + } + ] + }, + "companions": { + "transform": { + "name": "polynomial_features_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "pf", + "type": "PolynomialFeatures" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "polynomial_features_free", + "returns": "void", + "parameters": [ + { + "name": "pf", + "type": "PolynomialFeatures" + } + ] + } + }, + "parameters": [ + "n_features", + "degree", + "interaction_only", + "include_bias" + ], + "bucket": "different_shape", + "reason": "takes i32 first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "PolynomialFeatures" + }, + { + "flow_estimator": "power_transformer", + "module": "preprocessing.flow", + "fit": { + "name": "power_transformer_fit", + "returns": "PowerTransformer", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "method", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "power_transformer_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "PowerTransformer" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "power_transformer_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "PowerTransformer" + } + ] + } + }, + "parameters": [ + "X", + "method" + ], + "bucket": "runnable", + "sklearn_estimator": "PowerTransformer" + }, + { + "flow_estimator": "qda", + "module": "discriminant_analysis.flow", + "fit": { + "name": "qda_fit", + "returns": "QuadraticDiscriminantAnalysis", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_classes", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "qda_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "QuadraticDiscriminantAnalysis" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "qda_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "QuadraticDiscriminantAnalysis" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_classes" + ], + "bucket": "runnable", + "sklearn_estimator": "QuadraticDiscriminantAnalysis" + }, + { + "flow_estimator": "quantile_regressor", + "module": "linear.flow", + "fit": { + "name": "quantile_regressor_fit", + "returns": "QuantileRegressor", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "quantile", + "type": "f32" + }, + { + "name": "alpha", + "type": "f32" + }, + { + "name": "epochs", + "type": "i32" + }, + { + "name": "lr", + "type": "f32" + } + ] + }, + "companions": { + "predict": { + "name": "quantile_regressor_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "QuantileRegressor" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "quantile_regressor_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "QuantileRegressor" + } + ] + } + }, + "parameters": [ + "X", + "y", + "quantile", + "alpha", + "epochs", + "lr" + ], + "bucket": "runnable", + "sklearn_estimator": "QuantileRegressor" + }, + { + "flow_estimator": "quantile_transformer", + "module": "preprocessing.flow", + "fit": { + "name": "quantile_transformer_fit", + "returns": "QuantileTransformer", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "n_quantiles", + "type": "i32" + }, + { + "name": "output_distribution", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "quantile_transformer_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "QuantileTransformer" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "quantile_transformer_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "QuantileTransformer" + } + ] + } + }, + "parameters": [ + "X", + "n_quantiles", + "output_distribution" + ], + "bucket": "runnable", + "sklearn_estimator": "QuantileTransformer" + }, + { + "flow_estimator": "radius_neighbors_classifier", + "module": "neighbors.flow", + "fit": { + "name": "radius_neighbors_classifier_fit", + "returns": "RadiusNeighborsClassifier", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "radius", + "type": "f32" + }, + { + "name": "n_classes", + "type": "i32" + }, + { + "name": "outlier_label", + "type": "f32" + } + ] + }, + "companions": { + "predict": { + "name": "radius_neighbors_classifier_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "RadiusNeighborsClassifier" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "radius_neighbors_classifier_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "RadiusNeighborsClassifier" + } + ] + } + }, + "parameters": [ + "X", + "y", + "radius", + "n_classes", + "outlier_label" + ], + "bucket": "runnable", + "sklearn_estimator": "RadiusNeighborsClassifier" + }, + { + "flow_estimator": "radius_neighbors_regressor", + "module": "neighbors.flow", + "fit": { + "name": "radius_neighbors_regressor_fit", + "returns": "RadiusNeighborsRegressor", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "radius", + "type": "f32" + } + ] + }, + "companions": { + "predict": { + "name": "radius_neighbors_regressor_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "RadiusNeighborsRegressor" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "radius_neighbors_regressor_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "RadiusNeighborsRegressor" + } + ] + } + }, + "parameters": [ + "X", + "y", + "radius" + ], + "bucket": "runnable", + "sklearn_estimator": "RadiusNeighborsRegressor" + }, + { + "flow_estimator": "radius_neighbors_transformer", + "module": "neighbors.flow", + "fit": { + "name": "radius_neighbors_transformer_fit", + "returns": "RadiusNeighborsTransformer", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "radius", + "type": "f32" + }, + { + "name": "mode", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "radius_neighbors_transformer_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "RadiusNeighborsTransformer" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "radius_neighbors_transformer_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "RadiusNeighborsTransformer" + } + ] + } + }, + "parameters": [ + "X", + "radius", + "mode" + ], + "bucket": "runnable", + "sklearn_estimator": "RadiusNeighborsTransformer" + }, + { + "flow_estimator": "random_forest_classifier", + "module": "ensemble.flow", + "fit": { + "name": "random_forest_classifier_fit", + "returns": "RandomForestClassifier", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_classes", + "type": "i32" + }, + { + "name": "n_trees", + "type": "i32" + }, + { + "name": "max_depth", + "type": "i32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "random_forest_classifier_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "RandomForestClassifier" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "predict_proba": { + "name": "random_forest_classifier_predict_proba", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "RandomForestClassifier" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "random_forest_classifier_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "RandomForestClassifier" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_classes", + "n_trees", + "max_depth", + "seed" + ], + "bucket": "runnable", + "sklearn_estimator": "RandomForestClassifier" + }, + { + "flow_estimator": "random_forest_regressor", + "module": "ensemble.flow", + "fit": { + "name": "random_forest_regressor_fit", + "returns": "RandomForestRegressor", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_trees", + "type": "i32" + }, + { + "name": "max_depth", + "type": "i32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "random_forest_regressor_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "RandomForestRegressor" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "random_forest_regressor_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "RandomForestRegressor" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_trees", + "max_depth", + "seed" + ], + "bucket": "runnable", + "sklearn_estimator": "RandomForestRegressor" + }, + { + "flow_estimator": "random_trees_embedding", + "module": "ensemble.flow", + "fit": { + "name": "random_trees_embedding_fit", + "returns": "RandomTreesEmbedding", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "n_trees", + "type": "i32" + }, + { + "name": "max_depth", + "type": "i32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "random_trees_embedding_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "RandomTreesEmbedding" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "random_trees_embedding_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "RandomTreesEmbedding" + } + ] + } + }, + "parameters": [ + "X", + "n_trees", + "max_depth", + "seed" + ], + "bucket": "runnable", + "sklearn_estimator": "RandomTreesEmbedding" + }, + { + "flow_estimator": "ransac_regressor", + "module": "linear.flow", + "fit": { + "name": "ransac_regressor_fit", + "returns": "RANSACRegressor", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "min_samples", + "type": "i32" + }, + { + "name": "max_trials", + "type": "i32" + }, + { + "name": "residual_threshold", + "type": "f32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "ransac_regressor_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "RANSACRegressor" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "ransac_regressor_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "RANSACRegressor" + } + ] + } + }, + "parameters": [ + "X", + "y", + "min_samples", + "max_trials", + "residual_threshold", + "seed" + ], + "bucket": "runnable", + "sklearn_estimator": "RANSACRegressor" + }, + { + "flow_estimator": "rbf_sampler", + "module": "kernel_approximation.flow", + "fit": { + "name": "rbf_sampler_fit", + "returns": "RBFSampler", + "parameters": [ + { + "name": "n_features", + "type": "i32" + }, + { + "name": "gamma", + "type": "f32" + }, + { + "name": "n_components", + "type": "i32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "rbf_sampler_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "RBFSampler" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "rbf_sampler_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "RBFSampler" + } + ] + } + }, + "parameters": [ + "n_features", + "gamma", + "n_components", + "seed" + ], + "bucket": "different_shape", + "reason": "takes i32 first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "RBFSampler" + }, + { + "flow_estimator": "regressor_chain", + "module": "multioutput.flow", + "fit": { + "name": "regressor_chain_fit", + "returns": "RegressorChain", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "Y", + "type": "ptr >" + }, + { + "name": "n_samples", + "type": "i32" + }, + { + "name": "n_features", + "type": "i32" + }, + { + "name": "n_outputs", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "regressor_chain_predict", + "returns": "ptr >", + "parameters": [ + { + "name": "model", + "type": "RegressorChain" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "regressor_chain_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "RegressorChain" + } + ] + } + }, + "parameters": [ + "X", + "Y", + "n_samples", + "n_features", + "n_outputs" + ], + "bucket": "runnable", + "sklearn_estimator": "RegressorChain" + }, + { + "flow_estimator": "rfe", + "module": "feature_selection.flow", + "fit": { + "name": "rfe_fit", + "returns": "RFE", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_features_to_select", + "type": "i32" + }, + { + "name": "importance_fn", + "type": "ptr" + }, + { + "name": "n_samples", + "type": "i32" + }, + { + "name": "n_features", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "rfe_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "selector", + "type": "RFE" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "rfe_free", + "returns": "void", + "parameters": [ + { + "name": "selector", + "type": "RFE" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_features_to_select", + "importance_fn", + "n_samples", + "n_features" + ], + "bucket": "runnable", + "sklearn_estimator": "RFE" + }, + { + "flow_estimator": "rfecv", + "module": "feature_selection.flow", + "fit": { + "name": "rfecv_fit", + "returns": "RFECV", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_samples", + "type": "i32" + }, + { + "name": "n_features", + "type": "i32" + }, + { + "name": "cv_folds", + "type": "i32" + }, + { + "name": "score_fn", + "type": "ptr" + } + ] + }, + "companions": { + "free": { + "name": "rfecv_free", + "returns": "void", + "parameters": [ + { + "name": "selector", + "type": "RFECV" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_samples", + "n_features", + "cv_folds", + "score_fn" + ], + "bucket": "runnable", + "sklearn_estimator": "RFECV" + }, + { + "flow_estimator": "ridge_classifier_cv", + "module": "linear.flow", + "fit": { + "name": "ridge_classifier_cv_fit", + "returns": "RidgeClassifierCV", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_classes", + "type": "i32" + }, + { + "name": "n_alphas", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "ridge_classifier_cv_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "RidgeClassifierCV" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "ridge_classifier_cv_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "RidgeClassifierCV" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_classes", + "n_alphas" + ], + "bucket": "runnable", + "sklearn_estimator": "RidgeClassifierCV" + }, + { + "flow_estimator": "ridge_classifier", + "module": "linear.flow", + "fit": { + "name": "ridge_classifier_fit", + "returns": "RidgeClassifier", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "alpha", + "type": "f32" + }, + { + "name": "epochs", + "type": "i32" + }, + { + "name": "lr", + "type": "f32" + } + ] + }, + "companions": { + "predict": { + "name": "ridge_classifier_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "RidgeClassifier" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "decision_function": { + "name": "ridge_classifier_decision_function", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "RidgeClassifier" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "ridge_classifier_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "RidgeClassifier" + } + ] + } + }, + "parameters": [ + "X", + "y", + "alpha", + "epochs", + "lr" + ], + "bucket": "runnable", + "sklearn_estimator": "RidgeClassifier" + }, + { + "flow_estimator": "ridge_cv", + "module": "linear.flow", + "fit": { + "name": "ridge_cv_fit", + "returns": "RidgeCV", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_alphas", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "ridge_cv_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "RidgeCV" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "ridge_cv_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "RidgeCV" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_alphas" + ], + "bucket": "runnable", + "sklearn_estimator": "RidgeCV" + }, + { + "flow_estimator": "ridge", + "module": "linear.flow", + "fit": { + "name": "ridge_fit", + "returns": "Ridge", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "alpha", + "type": "f32" + }, + { + "name": "epochs", + "type": "i32" + }, + { + "name": "lr", + "type": "f32" + } + ] + }, + "companions": { + "predict": { + "name": "ridge_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "Ridge" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "ridge_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "Ridge" + } + ] + } + }, + "parameters": [ + "X", + "y", + "alpha", + "epochs", + "lr" + ], + "bucket": "runnable", + "sklearn_estimator": "Ridge" + }, + { + "flow_estimator": "robust_scaler", + "module": "preprocessing.flow", + "fit": { + "name": "robust_scaler_fit", + "returns": "RobustScaler", + "parameters": [ + { + "name": "X", + "type": "Matrix" + } + ] + }, + "companions": { + "transform": { + "name": "robust_scaler_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "scaler", + "type": "RobustScaler" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "robust_scaler_free", + "returns": "void", + "parameters": [ + { + "name": "scaler", + "type": "RobustScaler" + } + ] + } + }, + "parameters": [ + "X" + ], + "bucket": "runnable", + "sklearn_estimator": "RobustScaler" + }, + { + "flow_estimator": "select_fdr", + "module": "feature_selection.flow", + "fit": { + "name": "select_fdr_fit", + "returns": "SelectFdr", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "alpha", + "type": "f32" + } + ] + }, + "companions": { + "transform": { + "name": "select_fdr_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "SelectFdr" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "select_fdr_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "SelectFdr" + } + ] + } + }, + "parameters": [ + "X", + "y", + "alpha" + ], + "bucket": "runnable", + "sklearn_estimator": "SelectFdr" + }, + { + "flow_estimator": "select_fpr", + "module": "feature_selection.flow", + "fit": { + "name": "select_fpr_fit", + "returns": "SelectFpr", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "alpha", + "type": "f32" + } + ] + }, + "companions": { + "transform": { + "name": "select_fpr_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "SelectFpr" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "select_fpr_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "SelectFpr" + } + ] + } + }, + "parameters": [ + "X", + "y", + "alpha" + ], + "bucket": "runnable", + "sklearn_estimator": "SelectFpr" + }, + { + "flow_estimator": "select_from_model", + "module": "feature_selection.flow", + "fit": { + "name": "select_from_model_fit", + "returns": "SelectFromModel", + "parameters": [ + { + "name": "weights", + "type": "ptr" + }, + { + "name": "n_features", + "type": "i32" + }, + { + "name": "threshold", + "type": "f32" + } + ] + }, + "companions": { + "transform": { + "name": "select_from_model_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "selector", + "type": "SelectFromModel" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "select_from_model_free", + "returns": "void", + "parameters": [ + { + "name": "selector", + "type": "SelectFromModel" + } + ] + } + }, + "parameters": [ + "weights", + "n_features", + "threshold" + ], + "bucket": "different_shape", + "reason": "takes ptr first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "SelectFromModel" + }, + { + "flow_estimator": "select_fwe", + "module": "feature_selection.flow", + "fit": { + "name": "select_fwe_fit", + "returns": "SelectFwe", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "alpha", + "type": "f32" + } + ] + }, + "companions": { + "transform": { + "name": "select_fwe_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "SelectFwe" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "select_fwe_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "SelectFwe" + } + ] + } + }, + "parameters": [ + "X", + "y", + "alpha" + ], + "bucket": "runnable", + "sklearn_estimator": "SelectFwe" + }, + { + "flow_estimator": "select_k_best", + "module": "feature_selection.flow", + "fit": { + "name": "select_k_best_fit", + "returns": "SelectKBest", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "k", + "type": "i32" + }, + { + "name": "score_func", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "select_k_best_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "selector", + "type": "SelectKBest" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "select_k_best_free", + "returns": "void", + "parameters": [ + { + "name": "selector", + "type": "SelectKBest" + } + ] + } + }, + "parameters": [ + "X", + "y", + "k", + "score_func" + ], + "bucket": "runnable", + "sklearn_estimator": "SelectKBest" + }, + { + "flow_estimator": "select_percentile", + "module": "feature_selection.flow", + "fit": { + "name": "select_percentile_fit", + "returns": "SelectPercentile", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "percentile", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "select_percentile_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "SelectPercentile" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "select_percentile_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "SelectPercentile" + } + ] + } + }, + "parameters": [ + "X", + "y", + "percentile" + ], + "bucket": "runnable", + "sklearn_estimator": "SelectPercentile" + }, + { + "flow_estimator": "self_training_classifier", + "module": "semi_supervised.flow", + "fit": { + "name": "self_training_classifier_fit", + "returns": "SelfTrainingClassifier", + "parameters": [ + { + "name": "y_initial", + "type": "ptr" + }, + { + "name": "probabilities", + "type": "ptr >" + }, + { + "name": "n_samples", + "type": "i32" + }, + { + "name": "n_classes", + "type": "i32" + }, + { + "name": "threshold", + "type": "f32" + }, + { + "name": "max_iter", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "self_training_classifier_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "SelfTrainingClassifier" + } + ] + }, + "free": { + "name": "self_training_classifier_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "SelfTrainingClassifier" + } + ] + } + }, + "parameters": [ + "y_initial", + "probabilities", + "n_samples", + "n_classes", + "threshold", + "max_iter" + ], + "bucket": "different_shape", + "reason": "takes ptr first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "SelfTrainingClassifier" + }, + { + "flow_estimator": "sequential_feature_selector", + "module": "feature_selection.flow", + "fit": { + "name": "sequential_feature_selector_fit", + "returns": "SequentialFeatureSelector", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_features_to_select", + "type": "i32" + }, + { + "name": "direction", + "type": "i32" + }, + { + "name": "n_samples", + "type": "i32" + }, + { + "name": "n_features", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "sequential_feature_selector_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "selector", + "type": "SequentialFeatureSelector" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "sequential_feature_selector_free", + "returns": "void", + "parameters": [ + { + "name": "selector", + "type": "SequentialFeatureSelector" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_features_to_select", + "direction", + "n_samples", + "n_features" + ], + "bucket": "runnable", + "sklearn_estimator": "SequentialFeatureSelector" + }, + { + "flow_estimator": "sgd_classifier", + "module": "linear.flow", + "fit": { + "name": "sgd_classifier_fit", + "returns": "SGDClassifier", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_classes", + "type": "i32" + }, + { + "name": "loss", + "type": "i32" + }, + { + "name": "alpha", + "type": "f32" + }, + { + "name": "epochs", + "type": "i32" + }, + { + "name": "lr", + "type": "f32" + } + ] + }, + "companions": { + "predict": { + "name": "sgd_classifier_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "SGDClassifier" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "sgd_classifier_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "SGDClassifier" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_classes", + "loss", + "alpha", + "epochs", + "lr" + ], + "bucket": "runnable", + "sklearn_estimator": "SGDClassifier" + }, + { + "flow_estimator": "sgd_one_class_svm", + "module": "extra_estimators.flow", + "fit": { + "name": "sgd_one_class_svm_fit", + "returns": "SGDOneClassSVM", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "nu", + "type": "f32" + }, + { + "name": "epochs", + "type": "i32" + }, + { + "name": "lr", + "type": "f32" + } + ] + }, + "companions": { + "predict": { + "name": "sgd_one_class_svm_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "SGDOneClassSVM" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "decision_function": { + "name": "sgd_one_class_svm_decision_function", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "SGDOneClassSVM" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "sgd_one_class_svm_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "SGDOneClassSVM" + } + ] + } + }, + "parameters": [ + "X", + "nu", + "epochs", + "lr" + ], + "bucket": "runnable", + "sklearn_estimator": "SGDOneClassSVM" + }, + { + "flow_estimator": "sgd_regressor", + "module": "linear.flow", + "fit": { + "name": "sgd_regressor_fit", + "returns": "SGDRegressor", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "loss", + "type": "i32" + }, + { + "name": "alpha", + "type": "f32" + }, + { + "name": "epsilon", + "type": "f32" + }, + { + "name": "epochs", + "type": "i32" + }, + { + "name": "lr", + "type": "f32" + } + ] + }, + "companions": { + "predict": { + "name": "sgd_regressor_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "SGDRegressor" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "sgd_regressor_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "SGDRegressor" + } + ] + } + }, + "parameters": [ + "X", + "y", + "loss", + "alpha", + "epsilon", + "epochs", + "lr" + ], + "bucket": "runnable", + "sklearn_estimator": "SGDRegressor" + }, + { + "flow_estimator": "shrunk_covariance", + "module": "covariance.flow", + "fit": { + "name": "shrunk_covariance_fit", + "returns": "ShrunkCovariance", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "shrinkage", + "type": "f32" + } + ] + }, + "companions": { + "free": { + "name": "shrunk_covariance_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "ShrunkCovariance" + } + ] + } + }, + "parameters": [ + "X", + "shrinkage" + ], + "bucket": "runnable", + "sklearn_estimator": "ShrunkCovariance" + }, + { + "flow_estimator": "simple_imputer", + "module": "preprocessing.flow", + "fit": { + "name": "simple_imputer_fit", + "returns": "SimpleImputer", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "strategy", + "type": "i32" + }, + { + "name": "constant_value", + "type": "f32" + } + ] + }, + "companions": { + "transform": { + "name": "simple_imputer_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "imputer", + "type": "SimpleImputer" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "simple_imputer_free", + "returns": "void", + "parameters": [ + { + "name": "imputer", + "type": "SimpleImputer" + } + ] + } + }, + "parameters": [ + "X", + "strategy", + "constant_value" + ], + "bucket": "runnable", + "sklearn_estimator": "SimpleImputer" + }, + { + "flow_estimator": "skewed_chi2_sampler", + "module": "kernel_approximation.flow", + "fit": { + "name": "skewed_chi2_sampler_fit", + "returns": "SkewedChi2Sampler", + "parameters": [ + { + "name": "n_features", + "type": "i32" + }, + { + "name": "skewedness", + "type": "f32" + }, + { + "name": "n_components", + "type": "i32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "skewed_chi2_sampler_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "SkewedChi2Sampler" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "skewed_chi2_sampler_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "SkewedChi2Sampler" + } + ] + } + }, + "parameters": [ + "n_features", + "skewedness", + "n_components", + "seed" + ], + "bucket": "different_shape", + "reason": "takes i32 first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "SkewedChi2Sampler" + }, + { + "flow_estimator": "sparse_coder", + "module": "extra_estimators.flow", + "fit": { + "name": "sparse_coder_fit", + "returns": "SparseCoder", + "parameters": [ + { + "name": "dictionary", + "type": "Matrix" + }, + { + "name": "n_nonzero_coefs", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "sparse_coder_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "SparseCoder" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "sparse_coder_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "SparseCoder" + } + ] + } + }, + "parameters": [ + "dictionary", + "n_nonzero_coefs" + ], + "bucket": "runnable", + "sklearn_estimator": "SparseCoder" + }, + { + "flow_estimator": "sparse_pca", + "module": "decomposition.flow", + "fit": { + "name": "sparse_pca_fit", + "returns": "SparsePCA", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "n_components", + "type": "i32" + }, + { + "name": "alpha", + "type": "f32" + }, + { + "name": "max_iter", + "type": "i32" + }, + { + "name": "tol", + "type": "f32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "sparse_pca_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "SparsePCA" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "sparse_pca_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "SparsePCA" + } + ] + } + }, + "parameters": [ + "X", + "n_components", + "alpha", + "max_iter", + "tol", + "seed" + ], + "bucket": "runnable", + "sklearn_estimator": "SparsePCA" + }, + { + "flow_estimator": "sparse_random_projection", + "module": "random_projection.flow", + "fit": { + "name": "sparse_random_projection_fit", + "returns": "SparseRandomProjection", + "parameters": [ + { + "name": "n_features", + "type": "i32" + }, + { + "name": "n_components", + "type": "i32" + }, + { + "name": "density", + "type": "f32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "sparse_random_projection_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "SparseRandomProjection" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "sparse_random_projection_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "SparseRandomProjection" + } + ] + } + }, + "parameters": [ + "n_features", + "n_components", + "density", + "seed" + ], + "bucket": "different_shape", + "reason": "takes i32 first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "SparseRandomProjection" + }, + { + "flow_estimator": "spectral_biclustering", + "module": "extra_estimators.flow", + "fit": { + "name": "spectral_biclustering_fit", + "returns": "SpectralBiclustering", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "n_row_clusters", + "type": "i32" + }, + { + "name": "n_col_clusters", + "type": "i32" + } + ] + }, + "companions": { + "free": { + "name": "spectral_biclustering_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "SpectralBiclustering" + } + ] + } + }, + "parameters": [ + "X", + "n_row_clusters", + "n_col_clusters" + ], + "bucket": "simplified", + "reason": "the implementation's own comments call it simplified, so timing it against scikit-learn's algorithm compares two different things", + "sklearn_estimator": "SpectralBiclustering" + }, + { + "flow_estimator": "spectral_clustering", + "module": "cluster.flow", + "fit": { + "name": "spectral_clustering_fit", + "returns": "SpectralClustering", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "n_clusters", + "type": "i32" + }, + { + "name": "gamma", + "type": "f32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "free": { + "name": "spectral_clustering_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "SpectralClustering" + } + ] + } + }, + "parameters": [ + "X", + "n_clusters", + "gamma", + "seed" + ], + "bucket": "runnable", + "sklearn_estimator": "SpectralClustering" + }, + { + "flow_estimator": "spectral_coclustering", + "module": "extra_estimators.flow", + "fit": { + "name": "spectral_coclustering_fit", + "returns": "SpectralCoclustering", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "n_clusters", + "type": "i32" + } + ] + }, + "companions": { + "free": { + "name": "spectral_coclustering_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "SpectralCoclustering" + } + ] + } + }, + "parameters": [ + "X", + "n_clusters" + ], + "bucket": "simplified", + "reason": "the implementation's own comments call it simplified, so timing it against scikit-learn's algorithm compares two different things", + "sklearn_estimator": "SpectralCoclustering" + }, + { + "flow_estimator": "spectral_embedding", + "module": "manifold.flow", + "fit": { + "name": "spectral_embedding_fit", + "returns": "SpectralEmbedding", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "n_components", + "type": "i32" + }, + { + "name": "gamma", + "type": "f32" + } + ] + }, + "companions": { + "free": { + "name": "spectral_embedding_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "SpectralEmbedding" + } + ] + } + }, + "parameters": [ + "X", + "n_components", + "gamma" + ], + "bucket": "runnable", + "sklearn_estimator": "SpectralEmbedding" + }, + { + "flow_estimator": "spline_transformer", + "module": "preprocessing.flow", + "fit": { + "name": "spline_transformer_fit", + "returns": "SplineTransformer", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "n_knots", + "type": "i32" + }, + { + "name": "degree", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "spline_transformer_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "SplineTransformer" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "spline_transformer_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "SplineTransformer" + } + ] + } + }, + "parameters": [ + "X", + "n_knots", + "degree" + ], + "bucket": "runnable", + "sklearn_estimator": "SplineTransformer" + }, + { + "flow_estimator": "stacking_classifier", + "module": "ensemble.flow", + "fit": { + "name": "stacking_classifier_fit", + "returns": "StackingClassifier", + "parameters": [ + { + "name": "base_preds", + "type": "ptr >" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_samples", + "type": "i32" + }, + { + "name": "n_base", + "type": "i32" + }, + { + "name": "n_classes", + "type": "i32" + }, + { + "name": "lr", + "type": "f32" + }, + { + "name": "n_iter", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "stacking_classifier_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "StackingClassifier" + }, + { + "name": "base_preds", + "type": "ptr >" + }, + { + "name": "n_samples", + "type": "i32" + } + ] + }, + "free": { + "name": "stacking_classifier_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "StackingClassifier" + } + ] + } + }, + "parameters": [ + "base_preds", + "y", + "n_samples", + "n_base", + "n_classes", + "lr", + "n_iter" + ], + "bucket": "different_shape", + "reason": "takes ptr > first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "StackingClassifier" + }, + { + "flow_estimator": "stacking_regressor", + "module": "ensemble.flow", + "fit": { + "name": "stacking_regressor_fit", + "returns": "StackingRegressor", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_base", + "type": "i32" + }, + { + "name": "max_depth", + "type": "i32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "stacking_regressor_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "StackingRegressor" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "stacking_regressor_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "StackingRegressor" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_base", + "max_depth", + "seed" + ], + "bucket": "runnable", + "sklearn_estimator": "StackingRegressor" + }, + { + "flow_estimator": "standard_scaler", + "module": "preprocessing.flow", + "fit": { + "name": "standard_scaler_fit", + "returns": "StandardScaler", + "parameters": [ + { + "name": "X", + "type": "Matrix" + } + ] + }, + "companions": { + "transform": { + "name": "standard_scaler_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "scaler", + "type": "StandardScaler" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "standard_scaler_free", + "returns": "void", + "parameters": [ + { + "name": "scaler", + "type": "StandardScaler" + } + ] + } + }, + "parameters": [ + "X" + ], + "bucket": "runnable", + "sklearn_estimator": "StandardScaler" + }, + { + "flow_estimator": "svc", + "module": "svm.flow", + "fit": { + "name": "svc_fit", + "returns": "SVC", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_classes", + "type": "i32" + }, + { + "name": "C", + "type": "f32" + }, + { + "name": "kernel", + "type": "i32" + }, + { + "name": "gamma", + "type": "f32" + }, + { + "name": "degree", + "type": "i32" + }, + { + "name": "coef0", + "type": "f32" + } + ] + }, + "companions": { + "predict": { + "name": "svc_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "SVC" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "decision_function": { + "name": "svc_decision_function", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "SVC" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "svc_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "SVC" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_classes", + "C", + "kernel", + "gamma", + "degree", + "coef0" + ], + "bucket": "runnable", + "sklearn_estimator": "SVC" + }, + { + "flow_estimator": "svr", + "module": "svm.flow", + "fit": { + "name": "svr_fit", + "returns": "SVR", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "C", + "type": "f32" + }, + { + "name": "epsilon", + "type": "f32" + }, + { + "name": "kernel", + "type": "i32" + }, + { + "name": "gamma", + "type": "f32" + }, + { + "name": "degree", + "type": "i32" + }, + { + "name": "coef0", + "type": "f32" + } + ] + }, + "companions": { + "predict": { + "name": "svr_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "SVR" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "svr_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "SVR" + } + ] + } + }, + "parameters": [ + "X", + "y", + "C", + "epsilon", + "kernel", + "gamma", + "degree", + "coef0" + ], + "bucket": "runnable", + "sklearn_estimator": "SVR" + }, + { + "flow_estimator": "target_encoder", + "module": "preprocessing.flow", + "fit": { + "name": "target_encoder_fit", + "returns": "TargetEncoder", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_samples", + "type": "i32" + }, + { + "name": "smoothing", + "type": "f32" + } + ] + }, + "companions": { + "transform": { + "name": "target_encoder_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "TargetEncoder" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "target_encoder_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "TargetEncoder" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_samples", + "smoothing" + ], + "bucket": "runnable", + "sklearn_estimator": "TargetEncoder" + }, + { + "flow_estimator": "tfidf_vectorizer", + "module": "feature_extraction.flow", + "fit": { + "name": "tfidf_vectorizer_fit", + "returns": "TfidfVectorizer", + "parameters": [ + { + "name": "documents", + "type": "ptr" + }, + { + "name": "n_docs", + "type": "i32" + }, + { + "name": "max_features", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "tfidf_vectorizer_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "TfidfVectorizer" + }, + { + "name": "documents", + "type": "ptr" + }, + { + "name": "n_docs", + "type": "i32" + } + ] + }, + "free": { + "name": "tfidf_vectorizer_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "TfidfVectorizer" + } + ] + } + }, + "parameters": [ + "documents", + "n_docs", + "max_features" + ], + "bucket": "different_shape", + "reason": "takes ptr first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "TfidfVectorizer" + }, + { + "flow_estimator": "theil_sen_regressor", + "module": "linear.flow", + "fit": { + "name": "theil_sen_regressor_fit", + "returns": "TheilSenRegressor", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "n_subsamples", + "type": "i32" + }, + { + "name": "max_iter", + "type": "i32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "theil_sen_regressor_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "TheilSenRegressor" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "theil_sen_regressor_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "TheilSenRegressor" + } + ] + } + }, + "parameters": [ + "X", + "y", + "n_subsamples", + "max_iter", + "seed" + ], + "bucket": "runnable", + "sklearn_estimator": "TheilSenRegressor" + }, + { + "flow_estimator": "transformed_target_regressor", + "module": "compose.flow", + "fit": { + "name": "transformed_target_regressor_fit", + "returns": "TransformedTargetRegressor", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "transformer_type", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "transformed_target_regressor_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "TransformedTargetRegressor" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "transformed_target_regressor_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "TransformedTargetRegressor" + } + ] + } + }, + "parameters": [ + "X", + "y", + "transformer_type" + ], + "bucket": "runnable", + "sklearn_estimator": "TransformedTargetRegressor" + }, + { + "flow_estimator": "truncated_svd", + "module": "decomposition.flow", + "fit": { + "name": "truncated_svd_fit", + "returns": "TruncatedSVD", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "n_components", + "type": "i32" + } + ] + }, + "companions": { + "transform": { + "name": "truncated_svd_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "model", + "type": "TruncatedSVD" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "truncated_svd_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "TruncatedSVD" + } + ] + } + }, + "parameters": [ + "X", + "n_components" + ], + "bucket": "runnable", + "sklearn_estimator": "TruncatedSVD" + }, + { + "flow_estimator": "tsne", + "module": "manifold.flow", + "fit": { + "name": "tsne_fit", + "returns": "TSNE", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "n_components", + "type": "i32" + }, + { + "name": "perplexity", + "type": "f32" + }, + { + "name": "learning_rate", + "type": "f32" + }, + { + "name": "n_iter", + "type": "i32" + }, + { + "name": "seed", + "type": "i32" + } + ] + }, + "companions": { + "free": { + "name": "tsne_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "TSNE" + } + ] + } + }, + "parameters": [ + "X", + "n_components", + "perplexity", + "learning_rate", + "n_iter", + "seed" + ], + "bucket": "runnable", + "sklearn_estimator": "TSNE" + }, + { + "flow_estimator": "tweedie_regressor", + "module": "linear.flow", + "fit": { + "name": "tweedie_regressor_fit", + "returns": "TweedieRegressor", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "y", + "type": "ptr" + }, + { + "name": "alpha", + "type": "f32" + }, + { + "name": "power", + "type": "f32" + }, + { + "name": "max_iter", + "type": "i32" + }, + { + "name": "lr", + "type": "f32" + } + ] + }, + "companions": { + "predict": { + "name": "tweedie_regressor_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "TweedieRegressor" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "tweedie_regressor_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "TweedieRegressor" + } + ] + } + }, + "parameters": [ + "X", + "y", + "alpha", + "power", + "max_iter", + "lr" + ], + "bucket": "runnable", + "sklearn_estimator": "TweedieRegressor" + }, + { + "flow_estimator": "variance_threshold", + "module": "feature_selection.flow", + "fit": { + "name": "variance_threshold_fit", + "returns": "VarianceThreshold", + "parameters": [ + { + "name": "X", + "type": "Matrix" + }, + { + "name": "threshold", + "type": "f32" + } + ] + }, + "companions": { + "transform": { + "name": "variance_threshold_transform", + "returns": "Matrix", + "parameters": [ + { + "name": "selector", + "type": "VarianceThreshold" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "variance_threshold_free", + "returns": "void", + "parameters": [ + { + "name": "selector", + "type": "VarianceThreshold" + } + ] + } + }, + "parameters": [ + "X", + "threshold" + ], + "bucket": "runnable", + "sklearn_estimator": "VarianceThreshold" + }, + { + "flow_estimator": "voting_classifier", + "module": "ensemble.flow", + "fit": { + "name": "voting_classifier_fit", + "returns": "VotingClassifier", + "parameters": [ + { + "name": "estimators", + "type": "ptr" + }, + { + "name": "n_estimators", + "type": "i32" + }, + { + "name": "n_classes", + "type": "i32" + }, + { + "name": "classes", + "type": "ptr" + }, + { + "name": "voting", + "type": "i32" + } + ] + }, + "companions": { + "predict": { + "name": "voting_classifier_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "VotingClassifier" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "voting_classifier_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "VotingClassifier" + } + ] + } + }, + "parameters": [ + "estimators", + "n_estimators", + "n_classes", + "classes", + "voting" + ], + "bucket": "different_shape", + "reason": "takes ptr first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "VotingClassifier" + }, + { + "flow_estimator": "voting_regressor", + "module": "ensemble.flow", + "fit": { + "name": "voting_regressor_fit", + "returns": "VotingRegressor", + "parameters": [ + { + "name": "estimators", + "type": "ptr" + }, + { + "name": "n_estimators", + "type": "i32" + }, + { + "name": "weights", + "type": "ptr" + } + ] + }, + "companions": { + "predict": { + "name": "voting_regressor_predict", + "returns": "ptr", + "parameters": [ + { + "name": "model", + "type": "VotingRegressor" + }, + { + "name": "X", + "type": "Matrix" + } + ] + }, + "free": { + "name": "voting_regressor_free", + "returns": "void", + "parameters": [ + { + "name": "model", + "type": "VotingRegressor" + } + ] + } + }, + "parameters": [ + "estimators", + "n_estimators", + "weights" + ], + "bucket": "different_shape", + "reason": "takes ptr first, so it is not an estimator over a feature matrix and needs its own harness", + "sklearn_estimator": "VotingRegressor" + } + ] +} diff --git a/benchmarks/estimator_coverage.py b/benchmarks/estimator_coverage.py new file mode 100644 index 0000000..ef436c5 --- /dev/null +++ b/benchmarks/estimator_coverage.py @@ -0,0 +1,368 @@ +#!/usr/bin/env python3 +"""Inventory every exported Flow estimator and say whether it can be raced. + +The canonical benchmark covers 12 estimators. `lib/scikit` exports far more +than that, and a claim about Flow beating scikit-learn means very little while +the rest are unmeasured. This builds the registry the wider benchmark runs +from: every exported `*_fit`, its companion predict/transform, the arguments +to call it with, and the scikit-learn class to race it against. + +An estimator lands in one of three buckets, and every one carries a reason: + + runnable arguments resolved and a scikit-learn counterpart exists + flow_only Flow implements it and scikit-learn has no equivalent + blocked something about the signature is not resolved yet + +Nothing is silently dropped. `blocked` entries name what is missing, so the +coverage number is honest about its own gaps. +""" +from __future__ import annotations + +import argparse +import json +import re +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] +LIB = ROOT / "lib" / "scikit" +OUT = ROOT / "benchmarks" / "estimator_coverage.json" + +FIT_RE = re.compile(r"^export function ([a-z_0-9]+)\(([^)]*)\)\s*->\s*([^{]+)\{", re.M) + +# An implementation that says in its own comments that it is a simplified +# stand-in is not racing scikit-learn's algorithm, and timing it against one +# produces a number that means nothing. spectral_biclustering thresholds row +# and column means where scikit-learn does an SVD and k-means, and came out at +# 25106x. Detected from the source rather than listed here, so the registry +# stays true as the implementations are filled in. +SIMPLIFIED_RE = re.compile( + r"#[^\n]*\b(simplified|simple approximation|placeholder|stub|not a real)\b", re.I +) + +# Values for a hyperparameter, by parameter name. Chosen to be small and valid. +# This measures call cost on a fixed workload, so what matters is that every +# estimator gets arguments it will accept. Tuning them would not make the +# comparison more honest. +ARGS: dict[str, str] = { + "seed": "42", + "max_iter": "100", + "n_components": "2", + "alpha": "1.0", + "lr": "0.01", + "epochs": "50", + "tol": "0.0001", + "max_depth": "5", + "gamma": "0.1", + "n_trees": "10", + "C": "1.0", + "n_alphas": "10", + "n_iter": "50", + "n_clusters": "3", + "degree": "2", + "learning_rate": "0.1", + "n_neighbors": "5", + "batch_size": "32", + "kernel": "0", + "kernel_type": "0", + "min_samples": "5", + "threshold": "0.5", + "coef0": "0.0", + "strategy": "0", + "n_estimators": "10", + "nu": "0.5", + "epsilon": "0.1", + "n_outputs": "2", + "criterion": "0", + "max_features": "2", + "k": "3", + "n_features_to_select": "2", + "radius": "1.0", + "n_folds": "3", + "method": "0", + "eps": "0.5", + "bandwidth": "1.0", + "constant_value": "1.0", + "n_base": "3", + "n_nonzero_coefs": "3", + "n_docs": "10", + "loss": "0", + "l1_ratio": "0.5", + "mode": "0", + "n_hidden": "1", + "activation": "0", + "momentum": "0.9", + "handle_unknown": "0", + "branching_factor": "50", + "n_init": "1", + "quantile": "0.5", + "power": "1.5", + "var_smoothing": "0.000000001", + "shrinkage": "0.1", + "reg_param": "0.1", + "n_quantiles": "10", + "n_bins": "4", + "leaf_size": "30", + "p": "2", + "min_samples_split": "2", + "min_samples_leaf": "1", + "subsample": "1.0", + "verbose": "0", + "warm_start": "0", + "n_jobs": "1", + "cv": "3", + "perplexity": "5.0", + "early_exaggeration": "4.0", + "n_neighbors_out": "5", + "damping": "0.5", + "preference": "0.0", + "linkage": "0", + "affinity": "0", + "covariance_type": "0", + "reg_covar": "0.000001", + "weight_concentration": "1.0", + "sample_steps": "2", + "convergence_iter": "15", + "contamination": "0.1", + "n_trials": "10", + "length_scale": "1.0", + "min_cluster_size": "5", + "n_topics": "3", + "eta": "0.1", + "n_Cs": "5", + "h_fraction": "0.75", + "missing_values": "0.0", + "features": "0", + "max_coefs": "3", + "n_bits": "4", + "output_distribution": "0", + "outlier_label": "0.0", + "max_trials": "10", + "residual_threshold": "1.0", + "cv_folds": "3", + "score_func": "0", + "percentile": "50", + "direction": "0", + "skewedness": "1.0", + "density": "0.3", + "n_row_clusters": "2", + "n_col_clusters": "2", + "n_knots": "4", + "smoothing": "1.0", + "n_subsamples": "10", + "transformer_type": "0", + "interaction_only": "false", + "include_bias": "true", + "increasing": "true", + "n_labels": "3", +} + +# Where one parameter name carries different meanings at different types, +# the type decides. `weights` is a scheme selector on KNNImputer and a vector +# of per-model weights on VotingRegressor. +TYPED_ARGS: dict[tuple[str, str], str] = { + ("weights", "i32"): "0", + ("penalty", "Penalty"): "penalty_none()", + # Passed as a null callback, which is what lib/scikit's own RFECV does + # when it calls rfe_fit internally. + ("importance_fn", "ptr"): "null", + ("score_fn", "ptr"): "null", +} + +# Parameters that need a local built before the call. The generated harness +# emits the declaration, then passes the name. +PREAMBLE_ARGS: dict[str, tuple[str, str]] = { + "hidden_sizes": ("array", "[8]"), + "n_categories": ("array", "[4, 4, 4, 4]"), +} + +# Parameters whose value depends on the dataset rather than on a constant. +DATASET_ARGS = {"n_classes", "n_samples", "n_features"} + +# Flow names whose scikit-learn counterpart is not a case change away. +ALIASES: dict[str, str] = { + # decomposition.flow, and it takes n_topics and eta, so this is Latent + # Dirichlet Allocation. discriminant_analysis.flow holds the other one. + "lda": "LatentDirichletAllocation", + "discriminant_lda": "LinearDiscriminantAnalysis", + "qda": "QuadraticDiscriminantAnalysis", + "knn_classifier": "KNeighborsClassifier", + "knn_regressor": "KNeighborsRegressor", + "kernel_svc": "SVC", + "kernel_svc_multi": "SVC", + "linear_svc_multi": "LinearSVC", + "lle": "LocallyLinearEmbedding", + "nca": "NeighborhoodComponentsAnalysis", + "omp_cv": "OrthogonalMatchingPursuitCV", + "one_vs_one": "OneVsOneClassifier", + "one_vs_rest": "OneVsRestClassifier", + "output_code": "OutputCodeClassifier", + "pls": "PLSRegression", + "isotonic": "IsotonicRegression", + "iterative_imputer": "IterativeImputer", + "incremental_pca_partial": "IncrementalPCA", + "ledoit_wolf_estimator": "LedoitWolf", + "oas_estimator": "OAS", + "multiclass_logistic": "LogisticRegression", +} + +# Flow estimators with no scikit-learn equivalent to race against. +FLOW_ONLY: dict[str, str] = { + "logistic_inference": "inference summary (coefficient standard errors, Wald tests); statsmodels territory", + "ols_inference": "inference summary for ordinary least squares; statsmodels territory", + "lssvm_classifier": "least-squares SVM; not in the scikit-learn public surface", + "ledoit_wolf_cv": "cross-validated variant scikit-learn does not expose", + "oas_cv": "cross-validated variant scikit-learn does not expose", + "graphical_lasso_cv": "scikit-learn spells this GraphicalLassoCV; covered by that entry", +} + + +def fit_body(text: str, name: str) -> str: + m = re.search(rf"^export function {re.escape(name)}\(.*?\n\}}", text, re.S | re.M) + return m.group(0) if m else "" + + +def parse_exports() -> dict[str, dict]: + out: dict[str, dict] = {} + for path in sorted(LIB.glob("*.flow")): + text = path.read_text() + for m in FIT_RE.finditer(text): + name, raw, ret = m.group(1), m.group(2), m.group(3).strip() + params = [] + for p in raw.split(","): + p = p.strip() + if not p: + continue + pname, _, ptype = p.partition(":") + params.append({"name": pname.strip(), "type": ptype.strip()}) + entry = {"module": path.name, "params": params, "returns": ret} + if name.endswith("_fit"): + found = SIMPLIFIED_RE.search(fit_body(text, name)) + if found: + entry["simplified"] = found.group(1).lower() + out[name] = entry + return out + + +def companions(exports: dict[str, dict], base: str) -> dict[str, dict]: + found = {} + for suffix in ("predict", "transform", "decision_function", "predict_proba", "free"): + name = f"{base}_{suffix}" + if name in exports: + spec = exports[name] + found[suffix] = { + "name": name, + "returns": spec["returns"], + "parameters": [{"name": q["name"], "type": q["type"]} for q in spec["params"]], + } + return found + + +def sklearn_name(base: str, known: set[str]) -> str | None: + if base in ALIASES: + return ALIASES[base] + flat = base.replace("_", "") + for candidate in known: + if candidate.lower() == flat: + return candidate + return None + + +def classify(base: str, spec: dict, known: set[str], exports: dict[str, dict]) -> dict: + entry = { + "flow_estimator": base, + "module": spec["module"], + "fit": { + "name": f"{base}_fit", + "returns": spec["returns"], + "parameters": [{"name": p["name"], "type": p["type"]} for p in spec["params"]], + }, + "companions": companions(exports, base), + "parameters": [p["name"] for p in spec["params"]], + } + if base in FLOW_ONLY: + entry.update(bucket="flow_only", reason=FLOW_ONLY[base]) + return entry + if "simplified" in spec: + entry.update( + bucket="simplified", + reason=f"the implementation's own comments call it {spec['simplified']}, " + "so timing it against scikit-learn's algorithm compares two different things", + sklearn_estimator=sklearn_name(base, known), + ) + return entry + + unresolved = [] + for p in spec["params"][1:]: + if p["name"] in ("y", "Y", "X"): + continue + if p["name"] in DATASET_ARGS: + continue + if (p["name"], p["type"]) in TYPED_ARGS or p["name"] in ARGS or p["name"] in PREAMBLE_ARGS: + continue + unresolved.append(f"{p['name']}: {p['type']}") + head = spec["params"][0]["type"] if spec["params"] else "absent" + sk = sklearn_name(base, known) + if head != "Matrix": + entry.update( + bucket="different_shape", + reason=f"takes {head} first, so it is not an estimator over a feature matrix and needs its own harness", + sklearn_estimator=sk, + ) + return entry + if unresolved: + entry.update(bucket="blocked", reason="no value for " + "; ".join(unresolved), sklearn_estimator=sk) + return entry + if sk is None: + entry.update(bucket="blocked", reason="no scikit-learn counterpart found by name; add an alias or a flow_only reason") + return entry + entry.update(bucket="runnable", sklearn_estimator=sk) + return entry + + +def argument_tables() -> dict: + return {"ARGS": ARGS, "TYPED_ARGS": TYPED_ARGS, "PREAMBLE_ARGS": PREAMBLE_ARGS, "DATASET_ARGS": sorted(DATASET_ARGS)} + + +def main() -> int: + ap = argparse.ArgumentParser() + ap.add_argument("--out", type=Path, default=OUT) + ap.add_argument("--show", choices=["blocked", "flow_only", "runnable", "simplified", "different_shape"], help="list one bucket and exit") + args = ap.parse_args() + + exports = parse_exports() + inventory = json.loads((ROOT / "benchmarks" / "sklearn_execution_inventory.json").read_text()) + rows = inventory["rows"] if isinstance(inventory, dict) else inventory + known = {r["estimator"] for r in rows} + + entries = [] + for name, spec in sorted(exports.items()): + if not name.endswith("_fit"): + continue + entries.append(classify(name[:-4], spec, known, exports)) + + counts: dict[str, int] = {} + for e in entries: + counts[e["bucket"]] = counts.get(e["bucket"], 0) + 1 + + payload = { + "schema_version": 1, + "counts": {"estimators": len(entries), **counts}, + "sklearn_surface": len(known), + "entries": entries, + } + args.out.write_text(json.dumps(payload, indent=2) + "\n") + + if args.show: + for e in entries: + if e["bucket"] == args.show: + print(f"{e['flow_estimator']:34s} {e.get('reason', e.get('sklearn_estimator',''))}") + return 0 + + print(f"{len(entries)} exported Flow estimators against a {len(known)}-estimator scikit-learn surface") + for bucket in ("runnable", "simplified", "different_shape", "flow_only", "blocked"): + print(f" {bucket:10s} {counts.get(bucket, 0)}") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/benchmarks/estimators_sklearn.json b/benchmarks/estimators_sklearn.json new file mode 100644 index 0000000..b5e26d2 --- /dev/null +++ b/benchmarks/estimators_sklearn.json @@ -0,0 +1,1558 @@ +{ + "schema_version": 1, + "repeats": 3, + "counts": { + "rows": 172, + "ok": 172 + }, + "rows": [ + { + "flow_estimator": "adaboost_classifier", + "sklearn_estimator": "AdaBoostClassifier", + "dataset": "classification", + "fit_ms": 21.526500000618398, + "pred_ms": 1.768750007613562, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "adaboost_regressor", + "sklearn_estimator": "AdaBoostRegressor", + "dataset": "regression", + "fit_ms": 12.87933399726171, + "pred_ms": 2.329875002033077, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "affinity_propagation", + "sklearn_estimator": "AffinityPropagation", + "dataset": "unsupervised", + "fit_ms": 3.3742090017767623, + "pred_ms": 0.702042001648806, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "agglomerative_clustering", + "sklearn_estimator": "AgglomerativeClustering", + "dataset": "unsupervised", + "fit_ms": 0.3488329966785386, + "pred_ms": 0.0, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "ard_regression", + "sklearn_estimator": "ARDRegression", + "dataset": "regression", + "fit_ms": 3.2732499967096373, + "pred_ms": 0.033958989661186934, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "bagging_classifier", + "sklearn_estimator": "BaggingClassifier", + "dataset": "classification", + "fit_ms": 6.214499997440726, + "pred_ms": 0.4326670023147017, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "bagging_regressor", + "sklearn_estimator": "BaggingRegressor", + "dataset": "regression", + "fit_ms": 11.730874990462326, + "pred_ms": 0.7345839985646307, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "bayesian_gaussian_mixture", + "sklearn_estimator": "BayesianGaussianMixture", + "dataset": "unsupervised", + "fit_ms": 1.0715829994296655, + "pred_ms": 0.07137500506360084, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "bayesian_ridge", + "sklearn_estimator": "BayesianRidge", + "dataset": "regression", + "fit_ms": 0.4151250032009557, + "pred_ms": 0.03720899985637516, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "bernoulli_nb", + "sklearn_estimator": "BernoulliNB", + "dataset": "classification", + "fit_ms": 0.4657079989556223, + "pred_ms": 0.09958300506696105, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "bernoulli_rbm", + "sklearn_estimator": "BernoulliRBM", + "dataset": "unsupervised", + "fit_ms": 8.622250010375865, + "pred_ms": 0.15341700054705143, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "birch", + "sklearn_estimator": "Birch", + "dataset": "unsupervised", + "fit_ms": 0.840458000311628, + "pred_ms": 0.25662498956080526, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "bisecting_kmeans", + "sklearn_estimator": "BisectingKMeans", + "dataset": "unsupervised", + "fit_ms": 4.308542003855109, + "pred_ms": 0.15733300824649632, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "calibrated_classifier_cv", + "sklearn_estimator": "CalibratedClassifierCV", + "dataset": "classification", + "fit_ms": 9.41249998868443, + "pred_ms": 1.0962079977616668, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "categorical_nb", + "sklearn_estimator": "CategoricalNB", + "dataset": "classification", + "fit_ms": 0.6367500027408823, + "pred_ms": 0.04699999408330768, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "cca", + "sklearn_estimator": "CCA", + "dataset": "multioutput", + "fit_ms": 0.4142079997109249, + "pred_ms": 0.04770899249706417, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "classifier_chain", + "sklearn_estimator": "ClassifierChain", + "dataset": "multioutput_class", + "fit_ms": 0.5865419952897355, + "pred_ms": 0.3751250042114407, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "complement_nb", + "sklearn_estimator": "ComplementNB", + "dataset": "classification", + "fit_ms": 0.3906250058207661, + "pred_ms": 0.03195799945387989, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "dbscan", + "sklearn_estimator": "DBSCAN", + "dataset": "unsupervised", + "fit_ms": 0.4175000067334622, + "pred_ms": 0.0, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "decision_tree_classifier", + "sklearn_estimator": "DecisionTreeClassifier", + "dataset": "classification", + "fit_ms": 0.26191600773017853, + "pred_ms": 0.03999999898951501, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "decision_tree_regressor", + "sklearn_estimator": "DecisionTreeRegressor", + "dataset": "regression", + "fit_ms": 1.1715830041794106, + "pred_ms": 0.04270899808034301, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "dictionary_learning", + "sklearn_estimator": "DictionaryLearning", + "dataset": "unsupervised", + "fit_ms": 198.1439999944996, + "pred_ms": 1.038749993313104, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "discriminant_lda", + "sklearn_estimator": "LinearDiscriminantAnalysis", + "dataset": "classification", + "fit_ms": 0.2998340060003102, + "pred_ms": 0.047708002966828644, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "elastic_net_cv", + "sklearn_estimator": "ElasticNetCV", + "dataset": "regression", + "fit_ms": 13.445750009850599, + "pred_ms": 0.036375000490807, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "elastic_net", + "sklearn_estimator": "ElasticNet", + "dataset": "regression", + "fit_ms": 0.17662500613369048, + "pred_ms": 0.03762501000892371, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "elliptic_envelope", + "sklearn_estimator": "EllipticEnvelope", + "dataset": "unsupervised", + "fit_ms": 8.801250005490147, + "pred_ms": 0.14320800255518407, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "empirical_covariance", + "sklearn_estimator": "EmpiricalCovariance", + "dataset": "unsupervised", + "fit_ms": 0.12658300693146884, + "pred_ms": 0.0, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "extra_tree_classifier", + "sklearn_estimator": "ExtraTreeClassifier", + "dataset": "classification", + "fit_ms": 0.2129159984178841, + "pred_ms": 0.03758301318157464, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "extra_tree_regressor", + "sklearn_estimator": "ExtraTreeRegressor", + "dataset": "regression", + "fit_ms": 0.6256659980863333, + "pred_ms": 0.052041999879293144, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "extra_trees_classifier", + "sklearn_estimator": "ExtraTreesClassifier", + "dataset": "classification", + "fit_ms": 27.618958003586158, + "pred_ms": 2.439207994029857, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "extra_trees_regressor", + "sklearn_estimator": "ExtraTreesRegressor", + "dataset": "regression", + "fit_ms": 63.45508300000802, + "pred_ms": 5.237417004536837, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "factor_analysis", + "sklearn_estimator": "FactorAnalysis", + "dataset": "unsupervised", + "fit_ms": 0.43433399696368724, + "pred_ms": 0.0466250057797879, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "fast_ica", + "sklearn_estimator": "FastICA", + "dataset": "unsupervised", + "fit_ms": 0.6530410028062761, + "pred_ms": 0.032791998819448054, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "feature_agglomeration", + "sklearn_estimator": "FeatureAgglomeration", + "dataset": "unsupervised", + "fit_ms": 0.1066659897333011, + "pred_ms": 0.13712499639950693, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "gamma_regressor", + "sklearn_estimator": "GammaRegressor", + "dataset": "regression", + "fit_ms": 0.6624579982599244, + "pred_ms": 0.04954100586473942, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "gaussian_mixture", + "sklearn_estimator": "GaussianMixture", + "dataset": "unsupervised", + "fit_ms": 0.8224999910453334, + "pred_ms": 0.05604101170320064, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "gaussian_nb", + "sklearn_estimator": "GaussianNB", + "dataset": "classification", + "fit_ms": 0.2733330038608983, + "pred_ms": 0.06195799505803734, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "gaussian_process_classifier", + "sklearn_estimator": "GaussianProcessClassifier", + "dataset": "regression", + "fit_ms": 2415.014332989813, + "pred_ms": 390.27524999983143, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "gaussian_process_regressor", + "sklearn_estimator": "GaussianProcessRegressor", + "dataset": "regression", + "fit_ms": 71.80841700755991, + "pred_ms": 0.6932499964023009, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "gradient_boosting_classifier", + "sklearn_estimator": "GradientBoostingClassifier", + "dataset": "classification", + "fit_ms": 61.85595800343435, + "pred_ms": 0.3888749924954027, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "gradient_boosting_regressor", + "sklearn_estimator": "GradientBoostingRegressor", + "dataset": "regression", + "fit_ms": 46.95212499063928, + "pred_ms": 0.42537499393802136, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "graphical_lasso", + "sklearn_estimator": "GraphicalLasso", + "dataset": "unsupervised", + "fit_ms": 0.3913340042345226, + "pred_ms": 0.0, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "hdbscan", + "sklearn_estimator": "HDBSCAN", + "dataset": "unsupervised", + "fit_ms": 0.7184580026660115, + "pred_ms": 0.0, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "hist_gradient_boosting_classifier", + "sklearn_estimator": "HistGradientBoostingClassifier", + "dataset": "classification", + "fit_ms": 691.7245420045219, + "pred_ms": 31.16549999685958, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "hist_gradient_boosting_regressor", + "sklearn_estimator": "HistGradientBoostingRegressor", + "dataset": "regression", + "fit_ms": 830.1787499949569, + "pred_ms": 9.42004201351665, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "huber_regressor", + "sklearn_estimator": "HuberRegressor", + "dataset": "regression", + "fit_ms": 6.414291012333706, + "pred_ms": 0.03616600588429719, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "isolation_forest", + "sklearn_estimator": "IsolationForest", + "dataset": "unsupervised", + "fit_ms": 44.771750006475486, + "pred_ms": 3.546292005921714, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "isomap", + "sklearn_estimator": "Isomap", + "dataset": "unsupervised", + "fit_ms": 4.243792005581781, + "pred_ms": 1.1558330006664619, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "iterative_imputer", + "sklearn_estimator": "IterativeImputer", + "dataset": "unsupervised", + "fit_ms": 2.435165995848365, + "pred_ms": 0.33916700340341777, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "kbins_discretizer", + "sklearn_estimator": "KBinsDiscretizer", + "dataset": "unsupervised", + "fit_ms": 0.5584580067079514, + "pred_ms": 0.36358401121106, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "kernel_density", + "sklearn_estimator": "KernelDensity", + "dataset": "unsupervised", + "fit_ms": 0.10016700252890587, + "pred_ms": 0.0, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "kernel_pca", + "sklearn_estimator": "KernelPCA", + "dataset": "unsupervised", + "fit_ms": 1.56787499145139, + "pred_ms": 0.24125000345520675, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "kernel_ridge", + "sklearn_estimator": "KernelRidge", + "dataset": "regression", + "fit_ms": 1.1393750028219074, + "pred_ms": 0.2375839976593852, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "kernel_svc", + "sklearn_estimator": "SVC", + "dataset": "classification", + "fit_ms": 0.39362499956041574, + "pred_ms": 0.27049999334849417, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "kernel_svc_multi", + "sklearn_estimator": "SVC", + "dataset": "classification", + "fit_ms": 0.35620899870991707, + "pred_ms": 0.2690419933060184, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "kmeans", + "sklearn_estimator": "KMeans", + "dataset": "unsupervised", + "fit_ms": 0.6812089995946735, + "pred_ms": 0.04833299317397177, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "kneighbors_transformer", + "sklearn_estimator": "KNeighborsTransformer", + "dataset": "unsupervised", + "fit_ms": 0.10029198892880231, + "pred_ms": 0.260791988694109, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "knn_classifier", + "sklearn_estimator": "KNeighborsClassifier", + "dataset": "classification", + "fit_ms": 0.21258299238979816, + "pred_ms": 0.8801670046523213, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "knn_imputer", + "sklearn_estimator": "KNNImputer", + "dataset": "unsupervised", + "fit_ms": 0.11808300041593611, + "pred_ms": 0.07766700582578778, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "knn_regressor", + "sklearn_estimator": "KNeighborsRegressor", + "dataset": "regression", + "fit_ms": 0.23525000142399222, + "pred_ms": 1.054708001902327, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "label_propagation", + "sklearn_estimator": "LabelPropagation", + "dataset": "classification", + "fit_ms": 0.6667079869657755, + "pred_ms": 0.271833996521309, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "label_spreading", + "sklearn_estimator": "LabelSpreading", + "dataset": "classification", + "fit_ms": 0.5641659954562783, + "pred_ms": 0.3032079985132441, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "lars_cv", + "sklearn_estimator": "LarsCV", + "dataset": "regression", + "fit_ms": 2.6605409948388115, + "pred_ms": 0.034958997275680304, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "lars", + "sklearn_estimator": "Lars", + "dataset": "regression", + "fit_ms": 0.5107080069137737, + "pred_ms": 0.034249998861923814, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "lasso_cv", + "sklearn_estimator": "LassoCV", + "dataset": "regression", + "fit_ms": 19.690582994371653, + "pred_ms": 0.036708006518892944, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "lasso", + "sklearn_estimator": "Lasso", + "dataset": "regression", + "fit_ms": 0.21429199841804802, + "pred_ms": 0.03950000973418355, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "lasso_lars_cv", + "sklearn_estimator": "LassoLarsCV", + "dataset": "regression", + "fit_ms": 3.7610419967677444, + "pred_ms": 0.037541991332545877, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "lasso_lars", + "sklearn_estimator": "LassoLars", + "dataset": "regression", + "fit_ms": 0.37141700158827007, + "pred_ms": 0.03791700873989612, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "lasso_lars_ic", + "sklearn_estimator": "LassoLarsIC", + "dataset": "regression", + "fit_ms": 0.9982500050682575, + "pred_ms": 0.03679099609144032, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "lda", + "sklearn_estimator": "LatentDirichletAllocation", + "dataset": "unsupervised", + "fit_ms": 94.88433299702592, + "pred_ms": 8.573500002967194, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "ledoit_wolf_estimator", + "sklearn_estimator": "LedoitWolf", + "dataset": "unsupervised", + "fit_ms": 0.17004100664053112, + "pred_ms": 0.0, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "linear_regression", + "sklearn_estimator": "LinearRegression", + "dataset": "regression", + "fit_ms": 0.23187499027699232, + "pred_ms": 0.03329200262669474, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "linear_svc", + "sklearn_estimator": "LinearSVC", + "dataset": "classification", + "fit_ms": 0.34545800008345395, + "pred_ms": 0.04991699825040996, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "linear_svc_multi", + "sklearn_estimator": "LinearSVC", + "dataset": "classification", + "fit_ms": 0.32787499367259443, + "pred_ms": 0.04933300078846514, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "linear_svr", + "sklearn_estimator": "LinearSVR", + "dataset": "regression", + "fit_ms": 0.20270899403840303, + "pred_ms": 0.03354200453031808, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "lle", + "sklearn_estimator": "LocallyLinearEmbedding", + "dataset": "unsupervised", + "fit_ms": 4.141000012168661, + "pred_ms": 3.195874989614822, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "local_outlier_factor", + "sklearn_estimator": "LocalOutlierFactor", + "dataset": "unsupervised", + "fit_ms": 0.395416995161213, + "pred_ms": 0.0, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "logistic_regression_cv", + "sklearn_estimator": "LogisticRegressionCV", + "dataset": "classification", + "fit_ms": 92.88212499814108, + "pred_ms": 0.04820800677407533, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "logistic_regression", + "sklearn_estimator": "LogisticRegression", + "dataset": "classification", + "fit_ms": 4.227999990689568, + "pred_ms": 0.04837500455323607, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "maxabs_scaler", + "sklearn_estimator": "MaxAbsScaler", + "dataset": "unsupervised", + "fit_ms": 0.06274999759625643, + "pred_ms": 0.042625004425644875, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "mds", + "sklearn_estimator": "MDS", + "dataset": "unsupervised", + "fit_ms": 19.73887500935234, + "pred_ms": 0.0, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "mean_shift", + "sklearn_estimator": "MeanShift", + "dataset": "unsupervised", + "fit_ms": 124.32579199958127, + "pred_ms": 0.2981249999720603, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "min_cov_det", + "sklearn_estimator": "MinCovDet", + "dataset": "unsupervised", + "fit_ms": 8.830791994114406, + "pred_ms": 0.0, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "minibatch_dictionary_learning", + "sklearn_estimator": "MiniBatchDictionaryLearning", + "dataset": "unsupervised", + "fit_ms": 190.04408399632666, + "pred_ms": 1.0352090030210093, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "minibatch_kmeans", + "sklearn_estimator": "MiniBatchKMeans", + "dataset": "unsupervised", + "fit_ms": 6.938249993254431, + "pred_ms": 0.05083299765828997, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "minibatch_nmf", + "sklearn_estimator": "MiniBatchNMF", + "dataset": "unsupervised", + "fit_ms": 5.672750005032867, + "pred_ms": 1.905458004330285, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "minibatch_sparse_pca", + "sklearn_estimator": "MiniBatchSparsePCA", + "dataset": "unsupervised", + "fit_ms": 5.367624995415099, + "pred_ms": 0.24074999964796007, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "minmax_scaler", + "sklearn_estimator": "MinMaxScaler", + "dataset": "unsupervised", + "fit_ms": 0.05862499529030174, + "pred_ms": 0.03820900747086853, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "missing_indicator", + "sklearn_estimator": "MissingIndicator", + "dataset": "unsupervised", + "fit_ms": 0.15312500181607902, + "pred_ms": 0.14545800513587892, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "mlp_classifier", + "sklearn_estimator": "MLPClassifier", + "dataset": "classification", + "fit_ms": 31.965916001354344, + "pred_ms": 0.11833300231955945, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "mlp_regressor", + "sklearn_estimator": "MLPRegressor", + "dataset": "regression", + "fit_ms": 78.97175000107381, + "pred_ms": 0.08762499783188105, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "multi_output_classifier", + "sklearn_estimator": "MultiOutputClassifier", + "dataset": "multioutput_class", + "fit_ms": 0.6964170024730265, + "pred_ms": 0.14966700109653175, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "multi_output_regressor", + "sklearn_estimator": "MultiOutputRegressor", + "dataset": "multioutput", + "fit_ms": 1.2234579917276278, + "pred_ms": 0.1346669887425378, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "multiclass_logistic", + "sklearn_estimator": "LogisticRegression", + "dataset": "classification", + "fit_ms": 4.640500003006309, + "pred_ms": 0.05041599797550589, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "multinomial_nb", + "sklearn_estimator": "MultinomialNB", + "dataset": "classification", + "fit_ms": 0.4004160000476986, + "pred_ms": 0.03708299482241273, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "multitask_elastic_net_cv", + "sklearn_estimator": "MultiTaskElasticNetCV", + "dataset": "multioutput", + "fit_ms": 17.617125005926937, + "pred_ms": 0.04141598765272647, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "multitask_elastic_net", + "sklearn_estimator": "MultiTaskElasticNet", + "dataset": "multioutput", + "fit_ms": 0.1660410052863881, + "pred_ms": 0.04279200220480561, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "multitask_lasso_cv", + "sklearn_estimator": "MultiTaskLassoCV", + "dataset": "multioutput", + "fit_ms": 111.09995799779426, + "pred_ms": 0.04137499490752816, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "multitask_lasso", + "sklearn_estimator": "MultiTaskLasso", + "dataset": "multioutput", + "fit_ms": 0.17275000573135912, + "pred_ms": 0.04425000224728137, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "nca", + "sklearn_estimator": "NeighborhoodComponentsAnalysis", + "dataset": "regression", + "fit_ms": 36.13787500944454, + "pred_ms": 0.04454200097825378, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "nearest_centroid", + "sklearn_estimator": "NearestCentroid", + "dataset": "classification", + "fit_ms": 0.23249999503605068, + "pred_ms": 0.396334013203159, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "nearest_neighbors", + "sklearn_estimator": "NearestNeighbors", + "dataset": "unsupervised", + "fit_ms": 0.1144580019172281, + "pred_ms": 0.0, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "nmf", + "sklearn_estimator": "NMF", + "dataset": "unsupervised", + "fit_ms": 1.8606250087032095, + "pred_ms": 0.9039169963216409, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "nu_svc", + "sklearn_estimator": "NuSVC", + "dataset": "regression", + "fit_ms": 34.70416700292844, + "pred_ms": 28.50808299263008, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "nu_svr", + "sklearn_estimator": "NuSVR", + "dataset": "regression", + "fit_ms": 2.2352499945554882, + "pred_ms": 2.0772080024471506, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "nystroem", + "sklearn_estimator": "Nystroem", + "dataset": "unsupervised", + "fit_ms": 1.0292500082869083, + "pred_ms": 0.2373329916736111, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "oas_estimator", + "sklearn_estimator": "OAS", + "dataset": "unsupervised", + "fit_ms": 0.11737500608433038, + "pred_ms": 0.0, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "omp_cv", + "sklearn_estimator": "OrthogonalMatchingPursuitCV", + "dataset": "regression", + "fit_ms": 1.1912499903701246, + "pred_ms": 0.03545799700077623, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "one_class_svm", + "sklearn_estimator": "OneClassSVM", + "dataset": "unsupervised", + "fit_ms": 0.2614169934531674, + "pred_ms": 0.2536670072004199, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "one_vs_one", + "sklearn_estimator": "OneVsOneClassifier", + "dataset": "classification", + "fit_ms": 1.0546660050749779, + "pred_ms": 0.261875000433065, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "one_vs_rest", + "sklearn_estimator": "OneVsRestClassifier", + "dataset": "classification", + "fit_ms": 1.2373339995974675, + "pred_ms": 0.12208399130031466, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "onehot_encoder", + "sklearn_estimator": "OneHotEncoder", + "dataset": "unsupervised", + "fit_ms": 0.19645900465548038, + "pred_ms": 0.31279200629796833, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "optics", + "sklearn_estimator": "OPTICS", + "dataset": "unsupervised", + "fit_ms": 33.755458003724925, + "pred_ms": 0.0, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "ordinal_encoder", + "sklearn_estimator": "OrdinalEncoder", + "dataset": "unsupervised", + "fit_ms": 0.19041699124500155, + "pred_ms": 0.2847920113708824, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "orthogonal_matching_pursuit", + "sklearn_estimator": "OrthogonalMatchingPursuit", + "dataset": "regression", + "fit_ms": 0.163792006787844, + "pred_ms": 0.03566600207705051, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "output_code", + "sklearn_estimator": "OutputCodeClassifier", + "dataset": "classification", + "fit_ms": 1.1059580137953162, + "pred_ms": 0.6682499952148646, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "passive_aggressive_classifier", + "sklearn_estimator": "PassiveAggressiveClassifier", + "dataset": "regression", + "fit_ms": 29.842375006410293, + "pred_ms": 0.14250000822357833, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "passive_aggressive_regressor", + "sklearn_estimator": "PassiveAggressiveRegressor", + "dataset": "regression", + "fit_ms": 0.7141670066630468, + "pred_ms": 0.044666987378150225, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "pca", + "sklearn_estimator": "PCA", + "dataset": "unsupervised", + "fit_ms": 0.12437500117812306, + "pred_ms": 0.03454201214481145, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "perceptron", + "sklearn_estimator": "Perceptron", + "dataset": "regression", + "fit_ms": 26.729792007245123, + "pred_ms": 0.1406660012435168, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "pls_canonical", + "sklearn_estimator": "PLSCanonical", + "dataset": "multioutput", + "fit_ms": 0.27099999715574086, + "pred_ms": 0.04829200042877346, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "pls", + "sklearn_estimator": "PLSRegression", + "dataset": "multioutput", + "fit_ms": 0.30429101025220007, + "pred_ms": 0.04700000863522291, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "pls_svd", + "sklearn_estimator": "PLSSVD", + "dataset": "multioutput", + "fit_ms": 0.16183400293812156, + "pred_ms": 0.04591699689626694, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "poisson_regressor", + "sklearn_estimator": "PoissonRegressor", + "dataset": "regression", + "fit_ms": 3.4219579974887893, + "pred_ms": 0.046707995352335274, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "power_transformer", + "sklearn_estimator": "PowerTransformer", + "dataset": "unsupervised", + "fit_ms": 7.24483399244491, + "pred_ms": 0.14595799439121038, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "qda", + "sklearn_estimator": "QuadraticDiscriminantAnalysis", + "dataset": "classification", + "fit_ms": 0.24308300635311753, + "pred_ms": 0.05541701102629304, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "quantile_regressor", + "sklearn_estimator": "QuantileRegressor", + "dataset": "regression", + "fit_ms": 7.912083005066961, + "pred_ms": 0.036124998587183654, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "quantile_transformer", + "sklearn_estimator": "QuantileTransformer", + "dataset": "unsupervised", + "fit_ms": 0.23933300690259784, + "pred_ms": 0.0967499945545569, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "radius_neighbors_classifier", + "sklearn_estimator": "RadiusNeighborsClassifier", + "dataset": "classification", + "fit_ms": 0.21437500254251063, + "pred_ms": 0.4346670029917732, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "radius_neighbors_regressor", + "sklearn_estimator": "RadiusNeighborsRegressor", + "dataset": "regression", + "fit_ms": 0.13745900650974363, + "pred_ms": 2.245166993816383, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "radius_neighbors_transformer", + "sklearn_estimator": "RadiusNeighborsTransformer", + "dataset": "unsupervised", + "fit_ms": 0.09800000407267362, + "pred_ms": 0.3641660005087033, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "random_forest_classifier", + "sklearn_estimator": "RandomForestClassifier", + "dataset": "classification", + "fit_ms": 40.36312499374617, + "pred_ms": 2.061082996078767, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "random_forest_regressor", + "sklearn_estimator": "RandomForestRegressor", + "dataset": "regression", + "fit_ms": 101.38191700389143, + "pred_ms": 5.04854200698901, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "random_trees_embedding", + "sklearn_estimator": "RandomTreesEmbedding", + "dataset": "unsupervised", + "fit_ms": 28.07333400414791, + "pred_ms": 7.932915992569178, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "ransac_regressor", + "sklearn_estimator": "RANSACRegressor", + "dataset": "regression", + "fit_ms": 24.158707994502038, + "pred_ms": 0.052499992307275534, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "regressor_chain", + "sklearn_estimator": "RegressorChain", + "dataset": "multioutput_class", + "fit_ms": 0.44129199523013085, + "pred_ms": 0.18250000721309334, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "rfe", + "sklearn_estimator": "RFE", + "dataset": "regression", + "fit_ms": 8.731499998248182, + "pred_ms": 0.20066701108589768, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "rfecv", + "sklearn_estimator": "RFECV", + "dataset": "regression", + "fit_ms": 74.50441700348165, + "pred_ms": 0.1781250030035153, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "ridge_classifier_cv", + "sklearn_estimator": "RidgeClassifierCV", + "dataset": "classification", + "fit_ms": 0.9015840041683987, + "pred_ms": 0.04933300078846514, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "ridge_classifier", + "sklearn_estimator": "RidgeClassifier", + "dataset": "regression", + "fit_ms": 1.1716669978341088, + "pred_ms": 0.13649999164044857, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "ridge_cv", + "sklearn_estimator": "RidgeCV", + "dataset": "regression", + "fit_ms": 0.32245900365523994, + "pred_ms": 0.03095800639130175, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "ridge", + "sklearn_estimator": "Ridge", + "dataset": "regression", + "fit_ms": 0.2145000034943223, + "pred_ms": 0.031167000997811556, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "robust_scaler", + "sklearn_estimator": "RobustScaler", + "dataset": "unsupervised", + "fit_ms": 0.24712500453460962, + "pred_ms": 0.03325000579934567, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "select_fdr", + "sklearn_estimator": "SelectFdr", + "dataset": "regression", + "fit_ms": 4.67358400055673, + "pred_ms": 0.06299999949987978, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "select_fpr", + "sklearn_estimator": "SelectFpr", + "dataset": "regression", + "fit_ms": 4.496124995057471, + "pred_ms": 0.047124995035119355, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "select_fwe", + "sklearn_estimator": "SelectFwe", + "dataset": "regression", + "fit_ms": 4.529624988208525, + "pred_ms": 0.04804201307706535, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "select_k_best", + "sklearn_estimator": "SelectKBest", + "dataset": "regression", + "fit_ms": 4.775291992700659, + "pred_ms": 0.06295900675468147, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "select_percentile", + "sklearn_estimator": "SelectPercentile", + "dataset": "regression", + "fit_ms": 4.5787910057697445, + "pred_ms": 0.07845899381209165, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "sequential_feature_selector", + "sklearn_estimator": "SequentialFeatureSelector", + "dataset": "regression", + "fit_ms": 78.2538749917876, + "pred_ms": 0.044083993998356164, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "sgd_classifier", + "sklearn_estimator": "SGDClassifier", + "dataset": "classification", + "fit_ms": 0.7614170026499778, + "pred_ms": 0.04858399915974587, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "sgd_one_class_svm", + "sklearn_estimator": "SGDOneClassSVM", + "dataset": "unsupervised", + "fit_ms": 0.7592909969389439, + "pred_ms": 0.030416005756706, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "sgd_regressor", + "sklearn_estimator": "SGDRegressor", + "dataset": "regression", + "fit_ms": 11.165165997226723, + "pred_ms": 0.04470901330932975, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "shrunk_covariance", + "sklearn_estimator": "ShrunkCovariance", + "dataset": "unsupervised", + "fit_ms": 0.16125000547617674, + "pred_ms": 0.0, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "simple_imputer", + "sklearn_estimator": "SimpleImputer", + "dataset": "unsupervised", + "fit_ms": 0.2184999902965501, + "pred_ms": 0.1990000018849969, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "sparse_coder", + "sklearn_estimator": "SparseCoder", + "dataset": "unsupervised", + "fit_ms": 0.05112499638926238, + "pred_ms": 0.9910000080708414, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "sparse_pca", + "sklearn_estimator": "SparsePCA", + "dataset": "unsupervised", + "fit_ms": 7.765166999888606, + "pred_ms": 0.19429200619924814, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "spectral_biclustering", + "sklearn_estimator": "SpectralBiclustering", + "dataset": "unsupervised", + "fit_ms": 23.599958993145265, + "pred_ms": 0.0, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "spectral_clustering", + "sklearn_estimator": "SpectralClustering", + "dataset": "unsupervised", + "fit_ms": 5.601083001238294, + "pred_ms": 0.0, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "spectral_coclustering", + "sklearn_estimator": "SpectralCoclustering", + "dataset": "unsupervised", + "fit_ms": 3.2601659913780168, + "pred_ms": 0.0, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "spectral_embedding", + "sklearn_estimator": "SpectralEmbedding", + "dataset": "unsupervised", + "fit_ms": 1.5892500086920336, + "pred_ms": 0.0, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "spline_transformer", + "sklearn_estimator": "SplineTransformer", + "dataset": "unsupervised", + "fit_ms": 0.1152909972006455, + "pred_ms": 0.2897079975809902, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "stacking_regressor", + "sklearn_estimator": "StackingRegressor", + "dataset": "regression", + "fit_ms": 2.960166006232612, + "pred_ms": 0.08087500464171171, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "standard_scaler", + "sklearn_estimator": "StandardScaler", + "dataset": "unsupervised", + "fit_ms": 0.09716599015519023, + "pred_ms": 0.03808400651905686, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "svc", + "sklearn_estimator": "SVC", + "dataset": "classification", + "fit_ms": 0.3906659985659644, + "pred_ms": 0.24525000480934978, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "svr", + "sklearn_estimator": "SVR", + "dataset": "regression", + "fit_ms": 2.543499998864718, + "pred_ms": 4.009917000075802, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "target_encoder", + "sklearn_estimator": "TargetEncoder", + "dataset": "regression", + "fit_ms": 9.338125004433095, + "pred_ms": 4.422082987730391, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "theil_sen_regressor", + "sklearn_estimator": "TheilSenRegressor", + "dataset": "regression", + "fit_ms": 185.33920800837222, + "pred_ms": 0.039874998037703335, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "transformed_target_regressor", + "sklearn_estimator": "TransformedTargetRegressor", + "dataset": "regression", + "fit_ms": 0.4072919982718304, + "pred_ms": 0.05774998862762004, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "truncated_svd", + "sklearn_estimator": "TruncatedSVD", + "dataset": "unsupervised", + "fit_ms": 0.2280000044265762, + "pred_ms": 0.028750000637955964, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "tsne", + "sklearn_estimator": "TSNE", + "dataset": "unsupervised", + "fit_ms": 415.7947920029983, + "pred_ms": 0.0, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "tweedie_regressor", + "sklearn_estimator": "TweedieRegressor", + "dataset": "regression", + "fit_ms": 0.7009170076344162, + "pred_ms": 0.05162500019650906, + "timing_unit": "ms", + "status": "ok" + }, + { + "flow_estimator": "variance_threshold", + "sklearn_estimator": "VarianceThreshold", + "dataset": "unsupervised", + "fit_ms": 0.08800000068731606, + "pred_ms": 0.046250002924352884, + "timing_unit": "ms", + "status": "ok" + } + ] +} diff --git a/benchmarks/generate_estimator_bench.py b/benchmarks/generate_estimator_bench.py new file mode 100644 index 0000000..52b601d --- /dev/null +++ b/benchmarks/generate_estimator_bench.py @@ -0,0 +1,298 @@ +#!/usr/bin/env python3 +"""Emit the Flow side of the wide estimator benchmark from the registry. + +Writing 172 timing blocks by hand invites 172 small mistakes, so they are +generated from `estimator_coverage.json`. The output is split across several +files because Flow issue #469 miscompiles some functions once a program grows +past a certain size, and the benchmark is the wrong place to find that out. + +Each block times a fit, times the companion predict or transform when one +exists, and prints one parseable line: + + ESTIMATOR||||ok +""" +from __future__ import annotations + +import argparse +import json +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] +REGISTRY = ROOT / "benchmarks" / "estimator_coverage.json" +OUTDIR = ROOT / "benchmarks" / "generated" + +import importlib.util + +_spec = importlib.util.spec_from_file_location("estimator_coverage", ROOT / "benchmarks" / "estimator_coverage.py") +_cov = importlib.util.module_from_spec(_spec) +_spec.loader.exec_module(_cov) + +HEADER = '''# Generated by benchmarks/generate_estimator_bench.py. Do not edit. +# +# One timing block per Flow estimator that the registry marks runnable. +# Regenerate with: +# python benchmarks/estimator_coverage.py +# python benchmarks/generate_estimator_bench.py + +import "lib/scikit/scikit.flow" + +extern { + function printf(fmt: string, ...) -> i32 + function flow_now_ns() -> i64 + function malloc(size: i64) -> ptr + function free(p: ptr) -> void + function fflush(stream: ptr) -> i32 +} + +function ms_between(start: i64, finish: i64) -> f32 { + return ((((finish - start) as f64) / 1000000.0) as f32) +} +''' + + +SUFFIX = { + "classification": "c", + "regression": "r", + "unsupervised": "c", + "multioutput": "r", + "multioutput_class": "c", +} + + +def dataset_kind(entry: dict) -> str: + names = [p["name"] for p in entry["fit"]["parameters"]] + types = {p["name"]: p["type"] for p in entry["fit"]["parameters"]} + if "y" not in names and "Y" not in names: + return "unsupervised" + if "Y" in names: + # A multi-output classifier handed continuous targets is not doing the + # work its scikit-learn counterpart would refuse to do at all. + return "multioutput_class" if "classifier" in entry["flow_estimator"] or "chain" in entry["flow_estimator"] else "multioutput" + if "n_classes" in names: + return "classification" + if types.get("y") == "ptr": + return "classification" + return "regression" + + +def fit_arguments(entry: dict, kind: str) -> tuple[list[str], list[str]]: + """Returns (preamble lines, argument expressions).""" + pre: list[str] = [] + args: list[str] = [] + suffix = SUFFIX[kind] + for i, p in enumerate(entry["fit"]["parameters"]): + name, ptype = p["name"], p["type"] + if i == 0: + args.append(f"X_{suffix}") + continue + if name == "y": + args.append(f"y_{suffix}" if ptype == "ptr" else f"yi_{suffix}") + continue + if name == "Y": + multi = "Y_labels" if kind == "multioutput_class" else "Y_multi" + if ptype == "Matrix": + args.append(multi) + continue + if ptype.replace(" ", "") == "ptr>": + args.append("Y_label_rows" if kind == "multioutput_class" else "Y_rows") + continue + if name == "n_classes": + args.append("3" if kind == "classification" else "1") + continue + if name == "n_samples": + args.append(f"n_{suffix}") + continue + if name == "n_features": + args.append(f"f_{suffix}") + continue + if name in _cov.PREAMBLE_ARGS: + decl_type, literal = _cov.PREAMBLE_ARGS[name] + local = f"{name}_{entry['flow_estimator']}" + pre.append(f" let {local}: {decl_type} = {literal}") + args.append(local) + continue + typed = _cov.TYPED_ARGS.get((name, ptype)) + if typed is not None: + args.append(typed) + continue + args.append(_cov.ARGS[name]) + return pre, args + + +def block(entry: dict) -> str: + """One estimator: probe, choose a repeat count, then time fit and predict. + + A single shot does not measure an estimator that finishes in microseconds. + StandardScaler on 150 rows by 4 came back as 0.001 ms, which is the clock's + resolution rather than its cost, and spectral_biclustering produced a + 23600x ratio out of the same rounding. The probe below picks a repeat count + so every estimator is timed over something the clock can see. + """ + name = entry["flow_estimator"] + kind = dataset_kind(entry) + suffix = SUFFIX[kind] + pre, args = fit_arguments(entry, kind) + ret = entry["fit"]["returns"] + call = f"{entry['fit']['name']}({', '.join(args)})" + free = entry["companions"].get("free") + free_ok = free is not None and len(free["parameters"]) == 1 + + comp = entry["companions"].get("predict") or entry["companions"].get("transform") + comp_ok = ( + comp is not None + and len(comp["parameters"]) == 2 + and comp["parameters"][1]["type"] == "Matrix" + and comp["returns"] in ("Matrix", "ptr") + ) + + def release(var: str, kind_: str) -> str: + return "matrix_free(%s)" % var if kind_ == "Matrix" else "array_free_f32(%s)" % var + + lines = [f" # ---- {name} ({kind}) ----"] + lines += pre + # Probe once to size the repeat count, then release it. + lines.append(" t0 = flow_now_ns()") + lines.append(f" let probe_{name}: {ret} = {call}") + lines.append(" t1 = flow_now_ns()") + if free_ok: + lines.append(f" {free['name']}(probe_{name})") + lines.append(" reps = 1") + lines.append(" if (t1 - t0) < 200000 { reps = 200 }") + lines.append(" elif (t1 - t0) < 2000000 { reps = 20 }") + + # Timed fit loop. + lines.append(" t0 = flow_now_ns()") + lines.append(" for rep in 0 to reps {") + lines.append(f" let m_{name}: {ret} = {call}") + if free_ok: + lines.append(f" {free['name']}(m_{name})") + lines.append(" }") + lines.append(" t1 = flow_now_ns()") + + # Timed predict loop against one fitted model. + if comp_ok: + lines.append(f" let fitted_{name}: {ret} = {call}") + lines.append(" t2 = flow_now_ns()") + lines.append(" for rep2 in 0 to reps {") + lines.append(f" let o_{name}: {comp['returns']} = {comp['name']}(fitted_{name}, X_{suffix})") + lines.append(" " + release(f"o_{name}", comp["returns"])) + lines.append(" }") + lines.append(" t3 = flow_now_ns()") + if free_ok: + lines.append(f" {free['name']}(fitted_{name})") + pred_expr = "ms_between(t2, t3) / (reps as f32)" + else: + pred_expr = "0.0" + + lines.append( + f' printf("ESTIMATOR|{name}|%.9f|%.9f|%d|ok\\n", ' + f"ms_between(t0, t1) / (reps as f32), {pred_expr}, reps)" + ) + # Without this, one estimator trapping takes the whole file's buffered + # output with it and the run looks empty rather than partial. + lines.append(" fflush(null)") + return "\n".join(lines) + "\n" + + +def chunk_file(index: int, entries: list[dict]) -> str: + body = "\n".join(block(e) for e in entries) + return f'''{HEADER} +function main() -> i32 {{ + let iris: Dataset = load_iris() + let diabetes: Dataset = load_diabetes() + let X_c: Matrix = iris.X + let y_c: ptr = iris.y + let n_c: i32 = X_c.rows + let f_c: i32 = X_c.cols + let X_r: Matrix = diabetes.X + let y_r: ptr = diabetes.y + let n_r: i32 = X_r.rows + let f_r: i32 = X_r.cols + + let yi_c: ptr = malloc((n_c as i64) * 4) as ptr + for i in 0 to n_c {{ yi_c[i] = y_c[i] as i32 }} + let yi_r: ptr = malloc((n_r as i64) * 4) as ptr + for i in 0 to n_r {{ yi_r[i] = y_r[i] as i32 }} + + # A two-column target for the cross-decomposition estimators. + let Y_multi: Matrix = matrix_new(n_r, 2) + for i in 0 to n_r {{ + matrix_set(Y_multi, i, 0, y_r[i]) + matrix_set(Y_multi, i, 1, y_r[i] * 0.5) + }} + + # Label-valued targets for the multi-output classifiers. + let Y_labels: Matrix = matrix_new(n_c, 2) + let Y_label_rows: ptr > = malloc((n_c as i64) * 8) as ptr > + for i in 0 to n_c {{ + let a: f32 = y_c[i] + let b: f32 = ((((y_c[i] as i32) + 1) % 3) as f32) + matrix_set(Y_labels, i, 0, a) + matrix_set(Y_labels, i, 1, b) + let lrow: ptr = array_new_f32(2) + lrow[0] = a + lrow[1] = b + Y_label_rows[i] = lrow + }} + + let Y_rows: ptr > = malloc((n_r as i64) * 8) as ptr > + for i in 0 to n_r {{ + let row: ptr = array_new_f32(2) + row[0] = y_r[i] + row[1] = y_r[i] * 0.5 + Y_rows[i] = row + }} + + let mut t0: i64 = 0 + let mut t1: i64 = 0 + let mut t2: i64 = 0 + let mut t3: i64 = 0 + let mut reps: i32 = 1 + +{body} + for i in 0 to n_c {{ array_free_f32(Y_label_rows[i]) }} + free(Y_label_rows as ptr) + matrix_free(Y_labels) + for i in 0 to n_r {{ array_free_f32(Y_rows[i]) }} + free(Y_rows as ptr) + matrix_free(Y_multi) + free(yi_c as ptr) + free(yi_r as ptr) + return 0 +}} +''' + + +def main() -> int: + ap = argparse.ArgumentParser() + ap.add_argument("--per-file", type=int, default=20) + ap.add_argument("--outdir", type=Path, default=OUTDIR) + ap.add_argument("--only", help="generate a single estimator, for bisecting a compile failure") + args = ap.parse_args() + + registry = json.loads(REGISTRY.read_text()) + runnable = [e for e in registry["entries"] if e["bucket"] == "runnable"] + if args.only: + runnable = [e for e in runnable if e["flow_estimator"] == args.only] + if not runnable: + raise SystemExit(f"{args.only} is not a runnable registry entry") + + args.outdir.mkdir(parents=True, exist_ok=True) + for stale in args.outdir.glob("bench_estimators_*.flow"): + stale.unlink() + + written = [] + for i in range(0, len(runnable), args.per_file): + part = runnable[i : i + args.per_file] + path = args.outdir / f"bench_estimators_{i // args.per_file:02d}.flow" + path.write_text(chunk_file(i // args.per_file, part)) + written.append((path.name, len(part))) + + for name, n in written: + print(f"{name}: {n} estimators") + print(f"total {len(runnable)} estimators across {len(written)} files") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/benchmarks/generated/bench_estimators_00.flow b/benchmarks/generated/bench_estimators_00.flow new file mode 100644 index 0000000..f67f874 --- /dev/null +++ b/benchmarks/generated/bench_estimators_00.flow @@ -0,0 +1,536 @@ +# Generated by benchmarks/generate_estimator_bench.py. Do not edit. +# +# One timing block per Flow estimator that the registry marks runnable. +# Regenerate with: +# python benchmarks/estimator_coverage.py +# python benchmarks/generate_estimator_bench.py + +import "lib/scikit/scikit.flow" + +extern { + function printf(fmt: string, ...) -> i32 + function flow_now_ns() -> i64 + function malloc(size: i64) -> ptr + function free(p: ptr) -> void + function fflush(stream: ptr) -> i32 +} + +function ms_between(start: i64, finish: i64) -> f32 { + return ((((finish - start) as f64) / 1000000.0) as f32) +} + +function main() -> i32 { + let iris: Dataset = load_iris() + let diabetes: Dataset = load_diabetes() + let X_c: Matrix = iris.X + let y_c: ptr = iris.y + let n_c: i32 = X_c.rows + let f_c: i32 = X_c.cols + let X_r: Matrix = diabetes.X + let y_r: ptr = diabetes.y + let n_r: i32 = X_r.rows + let f_r: i32 = X_r.cols + + let yi_c: ptr = malloc((n_c as i64) * 4) as ptr + for i in 0 to n_c { yi_c[i] = y_c[i] as i32 } + let yi_r: ptr = malloc((n_r as i64) * 4) as ptr + for i in 0 to n_r { yi_r[i] = y_r[i] as i32 } + + # A two-column target for the cross-decomposition estimators. + let Y_multi: Matrix = matrix_new(n_r, 2) + for i in 0 to n_r { + matrix_set(Y_multi, i, 0, y_r[i]) + matrix_set(Y_multi, i, 1, y_r[i] * 0.5) + } + + # Label-valued targets for the multi-output classifiers. + let Y_labels: Matrix = matrix_new(n_c, 2) + let Y_label_rows: ptr > = malloc((n_c as i64) * 8) as ptr > + for i in 0 to n_c { + let a: f32 = y_c[i] + let b: f32 = ((((y_c[i] as i32) + 1) % 3) as f32) + matrix_set(Y_labels, i, 0, a) + matrix_set(Y_labels, i, 1, b) + let lrow: ptr = array_new_f32(2) + lrow[0] = a + lrow[1] = b + Y_label_rows[i] = lrow + } + + let Y_rows: ptr > = malloc((n_r as i64) * 8) as ptr > + for i in 0 to n_r { + let row: ptr = array_new_f32(2) + row[0] = y_r[i] + row[1] = y_r[i] * 0.5 + Y_rows[i] = row + } + + let mut t0: i64 = 0 + let mut t1: i64 = 0 + let mut t2: i64 = 0 + let mut t3: i64 = 0 + let mut reps: i32 = 1 + + # ---- adaboost_classifier (classification) ---- + t0 = flow_now_ns() + let probe_adaboost_classifier: AdaBoostClassifier = adaboost_classifier_fit(X_c, y_c, 3, 10, 5, 42) + t1 = flow_now_ns() + adaboost_classifier_free(probe_adaboost_classifier) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_adaboost_classifier: AdaBoostClassifier = adaboost_classifier_fit(X_c, y_c, 3, 10, 5, 42) + adaboost_classifier_free(m_adaboost_classifier) + } + t1 = flow_now_ns() + let fitted_adaboost_classifier: AdaBoostClassifier = adaboost_classifier_fit(X_c, y_c, 3, 10, 5, 42) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_adaboost_classifier: ptr = adaboost_classifier_predict(fitted_adaboost_classifier, X_c) + array_free_f32(o_adaboost_classifier) + } + t3 = flow_now_ns() + adaboost_classifier_free(fitted_adaboost_classifier) + printf("ESTIMATOR|adaboost_classifier|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- adaboost_regressor (regression) ---- + t0 = flow_now_ns() + let probe_adaboost_regressor: AdaBoostRegressor = adaboost_regressor_fit(X_r, y_r, 10, 5, 42) + t1 = flow_now_ns() + adaboost_regressor_free(probe_adaboost_regressor) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_adaboost_regressor: AdaBoostRegressor = adaboost_regressor_fit(X_r, y_r, 10, 5, 42) + adaboost_regressor_free(m_adaboost_regressor) + } + t1 = flow_now_ns() + let fitted_adaboost_regressor: AdaBoostRegressor = adaboost_regressor_fit(X_r, y_r, 10, 5, 42) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_adaboost_regressor: ptr = adaboost_regressor_predict(fitted_adaboost_regressor, X_r) + array_free_f32(o_adaboost_regressor) + } + t3 = flow_now_ns() + adaboost_regressor_free(fitted_adaboost_regressor) + printf("ESTIMATOR|adaboost_regressor|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- affinity_propagation (unsupervised) ---- + t0 = flow_now_ns() + let probe_affinity_propagation: AffinityPropagation = affinity_propagation_fit(X_c, 0.5, 100, 15) + t1 = flow_now_ns() + affinity_propagation_free(probe_affinity_propagation) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_affinity_propagation: AffinityPropagation = affinity_propagation_fit(X_c, 0.5, 100, 15) + affinity_propagation_free(m_affinity_propagation) + } + t1 = flow_now_ns() + printf("ESTIMATOR|affinity_propagation|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), 0.0, reps) + fflush(null) + + # ---- agglomerative_clustering (unsupervised) ---- + t0 = flow_now_ns() + let probe_agglomerative_clustering: AgglomerativeClustering = agglomerative_clustering_fit(X_c, 3) + t1 = flow_now_ns() + agglomerative_clustering_free(probe_agglomerative_clustering) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_agglomerative_clustering: AgglomerativeClustering = agglomerative_clustering_fit(X_c, 3) + agglomerative_clustering_free(m_agglomerative_clustering) + } + t1 = flow_now_ns() + printf("ESTIMATOR|agglomerative_clustering|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), 0.0, reps) + fflush(null) + + # ---- ard_regression (regression) ---- + t0 = flow_now_ns() + let probe_ard_regression: ARDRegression = ard_regression_fit(X_r, y_r, 100, 0.0001) + t1 = flow_now_ns() + ard_regression_free(probe_ard_regression) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_ard_regression: ARDRegression = ard_regression_fit(X_r, y_r, 100, 0.0001) + ard_regression_free(m_ard_regression) + } + t1 = flow_now_ns() + let fitted_ard_regression: ARDRegression = ard_regression_fit(X_r, y_r, 100, 0.0001) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_ard_regression: ptr = ard_regression_predict(fitted_ard_regression, X_r) + array_free_f32(o_ard_regression) + } + t3 = flow_now_ns() + ard_regression_free(fitted_ard_regression) + printf("ESTIMATOR|ard_regression|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- bagging_classifier (classification) ---- + t0 = flow_now_ns() + let probe_bagging_classifier: BaggingClassifier = bagging_classifier_fit(X_c, y_c, 3, 10, 5, 42) + t1 = flow_now_ns() + bagging_classifier_free(probe_bagging_classifier) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_bagging_classifier: BaggingClassifier = bagging_classifier_fit(X_c, y_c, 3, 10, 5, 42) + bagging_classifier_free(m_bagging_classifier) + } + t1 = flow_now_ns() + let fitted_bagging_classifier: BaggingClassifier = bagging_classifier_fit(X_c, y_c, 3, 10, 5, 42) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_bagging_classifier: ptr = bagging_classifier_predict(fitted_bagging_classifier, X_c) + array_free_f32(o_bagging_classifier) + } + t3 = flow_now_ns() + bagging_classifier_free(fitted_bagging_classifier) + printf("ESTIMATOR|bagging_classifier|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- bagging_regressor (regression) ---- + t0 = flow_now_ns() + let probe_bagging_regressor: BaggingRegressor = bagging_regressor_fit(X_r, y_r, 10, 5, 42) + t1 = flow_now_ns() + bagging_regressor_free(probe_bagging_regressor) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_bagging_regressor: BaggingRegressor = bagging_regressor_fit(X_r, y_r, 10, 5, 42) + bagging_regressor_free(m_bagging_regressor) + } + t1 = flow_now_ns() + let fitted_bagging_regressor: BaggingRegressor = bagging_regressor_fit(X_r, y_r, 10, 5, 42) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_bagging_regressor: ptr = bagging_regressor_predict(fitted_bagging_regressor, X_r) + array_free_f32(o_bagging_regressor) + } + t3 = flow_now_ns() + bagging_regressor_free(fitted_bagging_regressor) + printf("ESTIMATOR|bagging_regressor|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- bayesian_gaussian_mixture (unsupervised) ---- + t0 = flow_now_ns() + let probe_bayesian_gaussian_mixture: BayesianGaussianMixture = bayesian_gaussian_mixture_fit(X_c, 2, 100, 0.0001, 42) + t1 = flow_now_ns() + bayesian_gaussian_mixture_free(probe_bayesian_gaussian_mixture) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_bayesian_gaussian_mixture: BayesianGaussianMixture = bayesian_gaussian_mixture_fit(X_c, 2, 100, 0.0001, 42) + bayesian_gaussian_mixture_free(m_bayesian_gaussian_mixture) + } + t1 = flow_now_ns() + let fitted_bayesian_gaussian_mixture: BayesianGaussianMixture = bayesian_gaussian_mixture_fit(X_c, 2, 100, 0.0001, 42) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_bayesian_gaussian_mixture: ptr = bayesian_gaussian_mixture_predict(fitted_bayesian_gaussian_mixture, X_c) + array_free_f32(o_bayesian_gaussian_mixture) + } + t3 = flow_now_ns() + bayesian_gaussian_mixture_free(fitted_bayesian_gaussian_mixture) + printf("ESTIMATOR|bayesian_gaussian_mixture|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- bayesian_ridge (regression) ---- + t0 = flow_now_ns() + let probe_bayesian_ridge: BayesianRidge = bayesian_ridge_fit(X_r, y_r, 100, 0.0001) + t1 = flow_now_ns() + bayesian_ridge_free(probe_bayesian_ridge) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_bayesian_ridge: BayesianRidge = bayesian_ridge_fit(X_r, y_r, 100, 0.0001) + bayesian_ridge_free(m_bayesian_ridge) + } + t1 = flow_now_ns() + let fitted_bayesian_ridge: BayesianRidge = bayesian_ridge_fit(X_r, y_r, 100, 0.0001) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_bayesian_ridge: ptr = bayesian_ridge_predict(fitted_bayesian_ridge, X_r) + array_free_f32(o_bayesian_ridge) + } + t3 = flow_now_ns() + bayesian_ridge_free(fitted_bayesian_ridge) + printf("ESTIMATOR|bayesian_ridge|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- bernoulli_nb (classification) ---- + t0 = flow_now_ns() + let probe_bernoulli_nb: BernoulliNB = bernoulli_nb_fit(X_c, y_c, 3, 1.0) + t1 = flow_now_ns() + bernoulli_nb_free(probe_bernoulli_nb) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_bernoulli_nb: BernoulliNB = bernoulli_nb_fit(X_c, y_c, 3, 1.0) + bernoulli_nb_free(m_bernoulli_nb) + } + t1 = flow_now_ns() + let fitted_bernoulli_nb: BernoulliNB = bernoulli_nb_fit(X_c, y_c, 3, 1.0) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_bernoulli_nb: ptr = bernoulli_nb_predict(fitted_bernoulli_nb, X_c) + array_free_f32(o_bernoulli_nb) + } + t3 = flow_now_ns() + bernoulli_nb_free(fitted_bernoulli_nb) + printf("ESTIMATOR|bernoulli_nb|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- bernoulli_rbm (unsupervised) ---- + t0 = flow_now_ns() + let probe_bernoulli_rbm: BernoulliRBM = bernoulli_rbm_fit(X_c, 2, 0.1, 50, 32, 42) + t1 = flow_now_ns() + bernoulli_rbm_free(probe_bernoulli_rbm) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_bernoulli_rbm: BernoulliRBM = bernoulli_rbm_fit(X_c, 2, 0.1, 50, 32, 42) + bernoulli_rbm_free(m_bernoulli_rbm) + } + t1 = flow_now_ns() + let fitted_bernoulli_rbm: BernoulliRBM = bernoulli_rbm_fit(X_c, 2, 0.1, 50, 32, 42) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_bernoulli_rbm: Matrix = bernoulli_rbm_transform(fitted_bernoulli_rbm, X_c) + matrix_free(o_bernoulli_rbm) + } + t3 = flow_now_ns() + bernoulli_rbm_free(fitted_bernoulli_rbm) + printf("ESTIMATOR|bernoulli_rbm|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- birch (unsupervised) ---- + t0 = flow_now_ns() + let probe_birch: Birch = birch_fit(X_c, 0.5, 50) + t1 = flow_now_ns() + birch_free(probe_birch) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_birch: Birch = birch_fit(X_c, 0.5, 50) + birch_free(m_birch) + } + t1 = flow_now_ns() + printf("ESTIMATOR|birch|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), 0.0, reps) + fflush(null) + + # ---- bisecting_kmeans (unsupervised) ---- + t0 = flow_now_ns() + let probe_bisecting_kmeans: BisectingKMeans = bisecting_kmeans_fit(X_c, 3, 100, 42) + t1 = flow_now_ns() + bisecting_kmeans_free(probe_bisecting_kmeans) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_bisecting_kmeans: BisectingKMeans = bisecting_kmeans_fit(X_c, 3, 100, 42) + bisecting_kmeans_free(m_bisecting_kmeans) + } + t1 = flow_now_ns() + printf("ESTIMATOR|bisecting_kmeans|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), 0.0, reps) + fflush(null) + + # ---- calibrated_classifier_cv (classification) ---- + t0 = flow_now_ns() + let probe_calibrated_classifier_cv: CalibratedClassifierCV = calibrated_classifier_cv_fit(X_c, y_c, 3, 3, 0) + t1 = flow_now_ns() + calibrated_classifier_cv_free(probe_calibrated_classifier_cv) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_calibrated_classifier_cv: CalibratedClassifierCV = calibrated_classifier_cv_fit(X_c, y_c, 3, 3, 0) + calibrated_classifier_cv_free(m_calibrated_classifier_cv) + } + t1 = flow_now_ns() + let fitted_calibrated_classifier_cv: CalibratedClassifierCV = calibrated_classifier_cv_fit(X_c, y_c, 3, 3, 0) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_calibrated_classifier_cv: ptr = calibrated_classifier_cv_predict(fitted_calibrated_classifier_cv, X_c) + array_free_f32(o_calibrated_classifier_cv) + } + t3 = flow_now_ns() + calibrated_classifier_cv_free(fitted_calibrated_classifier_cv) + printf("ESTIMATOR|calibrated_classifier_cv|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- categorical_nb (classification) ---- + let n_categories_categorical_nb: array = [4, 4, 4, 4] + t0 = flow_now_ns() + let probe_categorical_nb: CategoricalNB = categorical_nb_fit(X_c, y_c, 3, n_categories_categorical_nb, 1.0) + t1 = flow_now_ns() + categorical_nb_free(probe_categorical_nb) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_categorical_nb: CategoricalNB = categorical_nb_fit(X_c, y_c, 3, n_categories_categorical_nb, 1.0) + categorical_nb_free(m_categorical_nb) + } + t1 = flow_now_ns() + let fitted_categorical_nb: CategoricalNB = categorical_nb_fit(X_c, y_c, 3, n_categories_categorical_nb, 1.0) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_categorical_nb: ptr = categorical_nb_predict(fitted_categorical_nb, X_c) + array_free_f32(o_categorical_nb) + } + t3 = flow_now_ns() + categorical_nb_free(fitted_categorical_nb) + printf("ESTIMATOR|categorical_nb|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- cca (multioutput) ---- + t0 = flow_now_ns() + let probe_cca: CCA = cca_fit(X_r, Y_multi, 2, 100, 0.0001) + t1 = flow_now_ns() + cca_free(probe_cca) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_cca: CCA = cca_fit(X_r, Y_multi, 2, 100, 0.0001) + cca_free(m_cca) + } + t1 = flow_now_ns() + let fitted_cca: CCA = cca_fit(X_r, Y_multi, 2, 100, 0.0001) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_cca: Matrix = cca_transform(fitted_cca, X_r) + matrix_free(o_cca) + } + t3 = flow_now_ns() + cca_free(fitted_cca) + printf("ESTIMATOR|cca|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- classifier_chain (multioutput_class) ---- + t0 = flow_now_ns() + let probe_classifier_chain: ClassifierChain = classifier_chain_fit(X_c, Y_label_rows, n_c, f_c, 2, 0.01, 50) + t1 = flow_now_ns() + classifier_chain_free(probe_classifier_chain) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_classifier_chain: ClassifierChain = classifier_chain_fit(X_c, Y_label_rows, n_c, f_c, 2, 0.01, 50) + classifier_chain_free(m_classifier_chain) + } + t1 = flow_now_ns() + printf("ESTIMATOR|classifier_chain|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), 0.0, reps) + fflush(null) + + # ---- complement_nb (classification) ---- + t0 = flow_now_ns() + let probe_complement_nb: ComplementNB = complement_nb_fit(X_c, y_c, 3, 1.0) + t1 = flow_now_ns() + complement_nb_free(probe_complement_nb) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_complement_nb: ComplementNB = complement_nb_fit(X_c, y_c, 3, 1.0) + complement_nb_free(m_complement_nb) + } + t1 = flow_now_ns() + let fitted_complement_nb: ComplementNB = complement_nb_fit(X_c, y_c, 3, 1.0) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_complement_nb: ptr = complement_nb_predict(fitted_complement_nb, X_c) + array_free_f32(o_complement_nb) + } + t3 = flow_now_ns() + complement_nb_free(fitted_complement_nb) + printf("ESTIMATOR|complement_nb|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- dbscan (unsupervised) ---- + t0 = flow_now_ns() + let probe_dbscan: DBSCAN = dbscan_fit(X_c, 0.5, 5) + t1 = flow_now_ns() + dbscan_free(probe_dbscan) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_dbscan: DBSCAN = dbscan_fit(X_c, 0.5, 5) + dbscan_free(m_dbscan) + } + t1 = flow_now_ns() + printf("ESTIMATOR|dbscan|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), 0.0, reps) + fflush(null) + + # ---- decision_tree_classifier (classification) ---- + t0 = flow_now_ns() + let probe_decision_tree_classifier: DecisionTreeClassifier = decision_tree_classifier_fit(X_c, y_c, 3, 5, 0) + t1 = flow_now_ns() + decision_tree_classifier_free(probe_decision_tree_classifier) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_decision_tree_classifier: DecisionTreeClassifier = decision_tree_classifier_fit(X_c, y_c, 3, 5, 0) + decision_tree_classifier_free(m_decision_tree_classifier) + } + t1 = flow_now_ns() + let fitted_decision_tree_classifier: DecisionTreeClassifier = decision_tree_classifier_fit(X_c, y_c, 3, 5, 0) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_decision_tree_classifier: ptr = decision_tree_classifier_predict(fitted_decision_tree_classifier, X_c) + array_free_f32(o_decision_tree_classifier) + } + t3 = flow_now_ns() + decision_tree_classifier_free(fitted_decision_tree_classifier) + printf("ESTIMATOR|decision_tree_classifier|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + for i in 0 to n_c { array_free_f32(Y_label_rows[i]) } + free(Y_label_rows as ptr) + matrix_free(Y_labels) + for i in 0 to n_r { array_free_f32(Y_rows[i]) } + free(Y_rows as ptr) + matrix_free(Y_multi) + free(yi_c as ptr) + free(yi_r as ptr) + return 0 +} diff --git a/benchmarks/generated/bench_estimators_01.flow b/benchmarks/generated/bench_estimators_01.flow new file mode 100644 index 0000000..4101be8 --- /dev/null +++ b/benchmarks/generated/bench_estimators_01.flow @@ -0,0 +1,567 @@ +# Generated by benchmarks/generate_estimator_bench.py. Do not edit. +# +# One timing block per Flow estimator that the registry marks runnable. +# Regenerate with: +# python benchmarks/estimator_coverage.py +# python benchmarks/generate_estimator_bench.py + +import "lib/scikit/scikit.flow" + +extern { + function printf(fmt: string, ...) -> i32 + function flow_now_ns() -> i64 + function malloc(size: i64) -> ptr + function free(p: ptr) -> void + function fflush(stream: ptr) -> i32 +} + +function ms_between(start: i64, finish: i64) -> f32 { + return ((((finish - start) as f64) / 1000000.0) as f32) +} + +function main() -> i32 { + let iris: Dataset = load_iris() + let diabetes: Dataset = load_diabetes() + let X_c: Matrix = iris.X + let y_c: ptr = iris.y + let n_c: i32 = X_c.rows + let f_c: i32 = X_c.cols + let X_r: Matrix = diabetes.X + let y_r: ptr = diabetes.y + let n_r: i32 = X_r.rows + let f_r: i32 = X_r.cols + + let yi_c: ptr = malloc((n_c as i64) * 4) as ptr + for i in 0 to n_c { yi_c[i] = y_c[i] as i32 } + let yi_r: ptr = malloc((n_r as i64) * 4) as ptr + for i in 0 to n_r { yi_r[i] = y_r[i] as i32 } + + # A two-column target for the cross-decomposition estimators. + let Y_multi: Matrix = matrix_new(n_r, 2) + for i in 0 to n_r { + matrix_set(Y_multi, i, 0, y_r[i]) + matrix_set(Y_multi, i, 1, y_r[i] * 0.5) + } + + # Label-valued targets for the multi-output classifiers. + let Y_labels: Matrix = matrix_new(n_c, 2) + let Y_label_rows: ptr > = malloc((n_c as i64) * 8) as ptr > + for i in 0 to n_c { + let a: f32 = y_c[i] + let b: f32 = ((((y_c[i] as i32) + 1) % 3) as f32) + matrix_set(Y_labels, i, 0, a) + matrix_set(Y_labels, i, 1, b) + let lrow: ptr = array_new_f32(2) + lrow[0] = a + lrow[1] = b + Y_label_rows[i] = lrow + } + + let Y_rows: ptr > = malloc((n_r as i64) * 8) as ptr > + for i in 0 to n_r { + let row: ptr = array_new_f32(2) + row[0] = y_r[i] + row[1] = y_r[i] * 0.5 + Y_rows[i] = row + } + + let mut t0: i64 = 0 + let mut t1: i64 = 0 + let mut t2: i64 = 0 + let mut t3: i64 = 0 + let mut reps: i32 = 1 + + # ---- decision_tree_regressor (regression) ---- + t0 = flow_now_ns() + let probe_decision_tree_regressor: DecisionTreeRegressor = decision_tree_regressor_fit(X_r, y_r, 5, 0) + t1 = flow_now_ns() + decision_tree_regressor_free(probe_decision_tree_regressor) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_decision_tree_regressor: DecisionTreeRegressor = decision_tree_regressor_fit(X_r, y_r, 5, 0) + decision_tree_regressor_free(m_decision_tree_regressor) + } + t1 = flow_now_ns() + let fitted_decision_tree_regressor: DecisionTreeRegressor = decision_tree_regressor_fit(X_r, y_r, 5, 0) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_decision_tree_regressor: ptr = decision_tree_regressor_predict(fitted_decision_tree_regressor, X_r) + array_free_f32(o_decision_tree_regressor) + } + t3 = flow_now_ns() + decision_tree_regressor_free(fitted_decision_tree_regressor) + printf("ESTIMATOR|decision_tree_regressor|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- dictionary_learning (unsupervised) ---- + t0 = flow_now_ns() + let probe_dictionary_learning: DictionaryLearning = dictionary_learning_fit(X_c, 2, 1.0, 100, 0.0001, 42) + t1 = flow_now_ns() + dictionary_learning_free(probe_dictionary_learning) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_dictionary_learning: DictionaryLearning = dictionary_learning_fit(X_c, 2, 1.0, 100, 0.0001, 42) + dictionary_learning_free(m_dictionary_learning) + } + t1 = flow_now_ns() + let fitted_dictionary_learning: DictionaryLearning = dictionary_learning_fit(X_c, 2, 1.0, 100, 0.0001, 42) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_dictionary_learning: Matrix = dictionary_learning_transform(fitted_dictionary_learning, X_c) + matrix_free(o_dictionary_learning) + } + t3 = flow_now_ns() + dictionary_learning_free(fitted_dictionary_learning) + printf("ESTIMATOR|dictionary_learning|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- discriminant_lda (classification) ---- + t0 = flow_now_ns() + let probe_discriminant_lda: LinearDiscriminantAnalysis = discriminant_lda_fit(X_c, y_c, 3) + t1 = flow_now_ns() + discriminant_lda_free(probe_discriminant_lda) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_discriminant_lda: LinearDiscriminantAnalysis = discriminant_lda_fit(X_c, y_c, 3) + discriminant_lda_free(m_discriminant_lda) + } + t1 = flow_now_ns() + let fitted_discriminant_lda: LinearDiscriminantAnalysis = discriminant_lda_fit(X_c, y_c, 3) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_discriminant_lda: ptr = discriminant_lda_predict(fitted_discriminant_lda, X_c) + array_free_f32(o_discriminant_lda) + } + t3 = flow_now_ns() + discriminant_lda_free(fitted_discriminant_lda) + printf("ESTIMATOR|discriminant_lda|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- elastic_net_cv (regression) ---- + t0 = flow_now_ns() + let probe_elastic_net_cv: ElasticNetCV = elastic_net_cv_fit(X_r, y_r, 10) + t1 = flow_now_ns() + elastic_net_cv_free(probe_elastic_net_cv) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_elastic_net_cv: ElasticNetCV = elastic_net_cv_fit(X_r, y_r, 10) + elastic_net_cv_free(m_elastic_net_cv) + } + t1 = flow_now_ns() + let fitted_elastic_net_cv: ElasticNetCV = elastic_net_cv_fit(X_r, y_r, 10) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_elastic_net_cv: ptr = elastic_net_cv_predict(fitted_elastic_net_cv, X_r) + array_free_f32(o_elastic_net_cv) + } + t3 = flow_now_ns() + elastic_net_cv_free(fitted_elastic_net_cv) + printf("ESTIMATOR|elastic_net_cv|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- elastic_net (regression) ---- + t0 = flow_now_ns() + let probe_elastic_net: ElasticNet = elastic_net_fit(X_r, y_r, 1.0, 0.5, 50, 0.01) + t1 = flow_now_ns() + elastic_net_free(probe_elastic_net) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_elastic_net: ElasticNet = elastic_net_fit(X_r, y_r, 1.0, 0.5, 50, 0.01) + elastic_net_free(m_elastic_net) + } + t1 = flow_now_ns() + let fitted_elastic_net: ElasticNet = elastic_net_fit(X_r, y_r, 1.0, 0.5, 50, 0.01) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_elastic_net: ptr = elastic_net_predict(fitted_elastic_net, X_r) + array_free_f32(o_elastic_net) + } + t3 = flow_now_ns() + elastic_net_free(fitted_elastic_net) + printf("ESTIMATOR|elastic_net|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- elliptic_envelope (unsupervised) ---- + t0 = flow_now_ns() + let probe_elliptic_envelope: EllipticEnvelope = elliptic_envelope_fit(X_c, 0.1, 10) + t1 = flow_now_ns() + elliptic_envelope_free(probe_elliptic_envelope) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_elliptic_envelope: EllipticEnvelope = elliptic_envelope_fit(X_c, 0.1, 10) + elliptic_envelope_free(m_elliptic_envelope) + } + t1 = flow_now_ns() + let fitted_elliptic_envelope: EllipticEnvelope = elliptic_envelope_fit(X_c, 0.1, 10) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_elliptic_envelope: ptr = elliptic_envelope_predict(fitted_elliptic_envelope, X_c) + array_free_f32(o_elliptic_envelope) + } + t3 = flow_now_ns() + elliptic_envelope_free(fitted_elliptic_envelope) + printf("ESTIMATOR|elliptic_envelope|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- empirical_covariance (unsupervised) ---- + t0 = flow_now_ns() + let probe_empirical_covariance: EmpiricalCovariance = empirical_covariance_fit(X_c) + t1 = flow_now_ns() + empirical_covariance_free(probe_empirical_covariance) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_empirical_covariance: EmpiricalCovariance = empirical_covariance_fit(X_c) + empirical_covariance_free(m_empirical_covariance) + } + t1 = flow_now_ns() + printf("ESTIMATOR|empirical_covariance|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), 0.0, reps) + fflush(null) + + # ---- extra_tree_classifier (classification) ---- + t0 = flow_now_ns() + let probe_extra_tree_classifier: ExtraTreeClassifier = extra_tree_classifier_fit(X_c, y_c, n_c, f_c, 3, 5, 42) + t1 = flow_now_ns() + extra_tree_classifier_free(probe_extra_tree_classifier) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_extra_tree_classifier: ExtraTreeClassifier = extra_tree_classifier_fit(X_c, y_c, n_c, f_c, 3, 5, 42) + extra_tree_classifier_free(m_extra_tree_classifier) + } + t1 = flow_now_ns() + let fitted_extra_tree_classifier: ExtraTreeClassifier = extra_tree_classifier_fit(X_c, y_c, n_c, f_c, 3, 5, 42) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_extra_tree_classifier: ptr = extra_tree_classifier_predict(fitted_extra_tree_classifier, X_c) + array_free_f32(o_extra_tree_classifier) + } + t3 = flow_now_ns() + extra_tree_classifier_free(fitted_extra_tree_classifier) + printf("ESTIMATOR|extra_tree_classifier|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- extra_tree_regressor (regression) ---- + t0 = flow_now_ns() + let probe_extra_tree_regressor: ExtraTreeRegressor = extra_tree_regressor_fit(X_r, y_r, n_r, f_r, 5, 42) + t1 = flow_now_ns() + extra_tree_regressor_free(probe_extra_tree_regressor) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_extra_tree_regressor: ExtraTreeRegressor = extra_tree_regressor_fit(X_r, y_r, n_r, f_r, 5, 42) + extra_tree_regressor_free(m_extra_tree_regressor) + } + t1 = flow_now_ns() + let fitted_extra_tree_regressor: ExtraTreeRegressor = extra_tree_regressor_fit(X_r, y_r, n_r, f_r, 5, 42) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_extra_tree_regressor: ptr = extra_tree_regressor_predict(fitted_extra_tree_regressor, X_r) + array_free_f32(o_extra_tree_regressor) + } + t3 = flow_now_ns() + extra_tree_regressor_free(fitted_extra_tree_regressor) + printf("ESTIMATOR|extra_tree_regressor|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- extra_trees_classifier (classification) ---- + t0 = flow_now_ns() + let probe_extra_trees_classifier: ExtraTreesClassifier = extra_trees_classifier_fit(X_c, y_c, 3, 10, 5, 42) + t1 = flow_now_ns() + extra_trees_classifier_free(probe_extra_trees_classifier) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_extra_trees_classifier: ExtraTreesClassifier = extra_trees_classifier_fit(X_c, y_c, 3, 10, 5, 42) + extra_trees_classifier_free(m_extra_trees_classifier) + } + t1 = flow_now_ns() + let fitted_extra_trees_classifier: ExtraTreesClassifier = extra_trees_classifier_fit(X_c, y_c, 3, 10, 5, 42) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_extra_trees_classifier: ptr = extra_trees_classifier_predict(fitted_extra_trees_classifier, X_c) + array_free_f32(o_extra_trees_classifier) + } + t3 = flow_now_ns() + extra_trees_classifier_free(fitted_extra_trees_classifier) + printf("ESTIMATOR|extra_trees_classifier|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- extra_trees_regressor (regression) ---- + t0 = flow_now_ns() + let probe_extra_trees_regressor: ExtraTreesRegressor = extra_trees_regressor_fit(X_r, y_r, 10, 5, 42) + t1 = flow_now_ns() + extra_trees_regressor_free(probe_extra_trees_regressor) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_extra_trees_regressor: ExtraTreesRegressor = extra_trees_regressor_fit(X_r, y_r, 10, 5, 42) + extra_trees_regressor_free(m_extra_trees_regressor) + } + t1 = flow_now_ns() + let fitted_extra_trees_regressor: ExtraTreesRegressor = extra_trees_regressor_fit(X_r, y_r, 10, 5, 42) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_extra_trees_regressor: ptr = extra_trees_regressor_predict(fitted_extra_trees_regressor, X_r) + array_free_f32(o_extra_trees_regressor) + } + t3 = flow_now_ns() + extra_trees_regressor_free(fitted_extra_trees_regressor) + printf("ESTIMATOR|extra_trees_regressor|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- factor_analysis (unsupervised) ---- + t0 = flow_now_ns() + let probe_factor_analysis: FactorAnalysis = factor_analysis_fit(X_c, 2, 100, 0.0001) + t1 = flow_now_ns() + factor_analysis_free(probe_factor_analysis) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_factor_analysis: FactorAnalysis = factor_analysis_fit(X_c, 2, 100, 0.0001) + factor_analysis_free(m_factor_analysis) + } + t1 = flow_now_ns() + let fitted_factor_analysis: FactorAnalysis = factor_analysis_fit(X_c, 2, 100, 0.0001) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_factor_analysis: Matrix = factor_analysis_transform(fitted_factor_analysis, X_c) + matrix_free(o_factor_analysis) + } + t3 = flow_now_ns() + factor_analysis_free(fitted_factor_analysis) + printf("ESTIMATOR|factor_analysis|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- fast_ica (unsupervised) ---- + t0 = flow_now_ns() + let probe_fast_ica: FastICA = fast_ica_fit(X_c, 2, 100, 0.0001, 42) + t1 = flow_now_ns() + fast_ica_free(probe_fast_ica) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_fast_ica: FastICA = fast_ica_fit(X_c, 2, 100, 0.0001, 42) + fast_ica_free(m_fast_ica) + } + t1 = flow_now_ns() + let fitted_fast_ica: FastICA = fast_ica_fit(X_c, 2, 100, 0.0001, 42) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_fast_ica: Matrix = fast_ica_transform(fitted_fast_ica, X_c) + matrix_free(o_fast_ica) + } + t3 = flow_now_ns() + fast_ica_free(fitted_fast_ica) + printf("ESTIMATOR|fast_ica|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- feature_agglomeration (unsupervised) ---- + t0 = flow_now_ns() + let probe_feature_agglomeration: FeatureAgglomeration = feature_agglomeration_fit(X_c, 3) + t1 = flow_now_ns() + feature_agglomeration_free(probe_feature_agglomeration) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_feature_agglomeration: FeatureAgglomeration = feature_agglomeration_fit(X_c, 3) + feature_agglomeration_free(m_feature_agglomeration) + } + t1 = flow_now_ns() + let fitted_feature_agglomeration: FeatureAgglomeration = feature_agglomeration_fit(X_c, 3) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_feature_agglomeration: Matrix = feature_agglomeration_transform(fitted_feature_agglomeration, X_c) + matrix_free(o_feature_agglomeration) + } + t3 = flow_now_ns() + feature_agglomeration_free(fitted_feature_agglomeration) + printf("ESTIMATOR|feature_agglomeration|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- gamma_regressor (regression) ---- + t0 = flow_now_ns() + let probe_gamma_regressor: GammaRegressor = gamma_regressor_fit(X_r, y_r, 1.0, 100, 0.01) + t1 = flow_now_ns() + gamma_regressor_free(probe_gamma_regressor) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_gamma_regressor: GammaRegressor = gamma_regressor_fit(X_r, y_r, 1.0, 100, 0.01) + gamma_regressor_free(m_gamma_regressor) + } + t1 = flow_now_ns() + let fitted_gamma_regressor: GammaRegressor = gamma_regressor_fit(X_r, y_r, 1.0, 100, 0.01) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_gamma_regressor: ptr = gamma_regressor_predict(fitted_gamma_regressor, X_r) + array_free_f32(o_gamma_regressor) + } + t3 = flow_now_ns() + gamma_regressor_free(fitted_gamma_regressor) + printf("ESTIMATOR|gamma_regressor|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- gaussian_mixture (unsupervised) ---- + t0 = flow_now_ns() + let probe_gaussian_mixture: GaussianMixture = gaussian_mixture_fit(X_c, 2, 100, 0.0001, 42) + t1 = flow_now_ns() + gaussian_mixture_free(probe_gaussian_mixture) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_gaussian_mixture: GaussianMixture = gaussian_mixture_fit(X_c, 2, 100, 0.0001, 42) + gaussian_mixture_free(m_gaussian_mixture) + } + t1 = flow_now_ns() + printf("ESTIMATOR|gaussian_mixture|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), 0.0, reps) + fflush(null) + + # ---- gaussian_nb (classification) ---- + t0 = flow_now_ns() + let probe_gaussian_nb: GaussianNB = gaussian_nb_fit(X_c, y_c, 3, 0.000000001) + t1 = flow_now_ns() + gaussian_nb_free(probe_gaussian_nb) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_gaussian_nb: GaussianNB = gaussian_nb_fit(X_c, y_c, 3, 0.000000001) + gaussian_nb_free(m_gaussian_nb) + } + t1 = flow_now_ns() + let fitted_gaussian_nb: GaussianNB = gaussian_nb_fit(X_c, y_c, 3, 0.000000001) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_gaussian_nb: ptr = gaussian_nb_predict(fitted_gaussian_nb, X_c) + array_free_f32(o_gaussian_nb) + } + t3 = flow_now_ns() + gaussian_nb_free(fitted_gaussian_nb) + printf("ESTIMATOR|gaussian_nb|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- gaussian_process_classifier (regression) ---- + t0 = flow_now_ns() + let probe_gaussian_process_classifier: GaussianProcessClassifier = gaussian_process_classifier_fit(X_r, y_r, 0.1, 100) + t1 = flow_now_ns() + gaussian_process_classifier_free(probe_gaussian_process_classifier) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_gaussian_process_classifier: GaussianProcessClassifier = gaussian_process_classifier_fit(X_r, y_r, 0.1, 100) + gaussian_process_classifier_free(m_gaussian_process_classifier) + } + t1 = flow_now_ns() + let fitted_gaussian_process_classifier: GaussianProcessClassifier = gaussian_process_classifier_fit(X_r, y_r, 0.1, 100) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_gaussian_process_classifier: ptr = gaussian_process_classifier_predict(fitted_gaussian_process_classifier, X_r) + array_free_f32(o_gaussian_process_classifier) + } + t3 = flow_now_ns() + gaussian_process_classifier_free(fitted_gaussian_process_classifier) + printf("ESTIMATOR|gaussian_process_classifier|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- gaussian_process_regressor (regression) ---- + t0 = flow_now_ns() + let probe_gaussian_process_regressor: GaussianProcessRegressor = gaussian_process_regressor_fit(X_r, y_r, 1.0, 1.0, 0) + t1 = flow_now_ns() + gaussian_process_regressor_free(probe_gaussian_process_regressor) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_gaussian_process_regressor: GaussianProcessRegressor = gaussian_process_regressor_fit(X_r, y_r, 1.0, 1.0, 0) + gaussian_process_regressor_free(m_gaussian_process_regressor) + } + t1 = flow_now_ns() + let fitted_gaussian_process_regressor: GaussianProcessRegressor = gaussian_process_regressor_fit(X_r, y_r, 1.0, 1.0, 0) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_gaussian_process_regressor: ptr = gaussian_process_regressor_predict(fitted_gaussian_process_regressor, X_r) + array_free_f32(o_gaussian_process_regressor) + } + t3 = flow_now_ns() + gaussian_process_regressor_free(fitted_gaussian_process_regressor) + printf("ESTIMATOR|gaussian_process_regressor|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- gradient_boosting_classifier (classification) ---- + t0 = flow_now_ns() + let probe_gradient_boosting_classifier: GradientBoostingClassifier = gradient_boosting_classifier_fit(X_c, y_c, 3, 10, 0.1, 5, 42) + t1 = flow_now_ns() + gradient_boosting_classifier_free(probe_gradient_boosting_classifier) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_gradient_boosting_classifier: GradientBoostingClassifier = gradient_boosting_classifier_fit(X_c, y_c, 3, 10, 0.1, 5, 42) + gradient_boosting_classifier_free(m_gradient_boosting_classifier) + } + t1 = flow_now_ns() + let fitted_gradient_boosting_classifier: GradientBoostingClassifier = gradient_boosting_classifier_fit(X_c, y_c, 3, 10, 0.1, 5, 42) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_gradient_boosting_classifier: ptr = gradient_boosting_classifier_predict(fitted_gradient_boosting_classifier, X_c) + array_free_f32(o_gradient_boosting_classifier) + } + t3 = flow_now_ns() + gradient_boosting_classifier_free(fitted_gradient_boosting_classifier) + printf("ESTIMATOR|gradient_boosting_classifier|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + for i in 0 to n_c { array_free_f32(Y_label_rows[i]) } + free(Y_label_rows as ptr) + matrix_free(Y_labels) + for i in 0 to n_r { array_free_f32(Y_rows[i]) } + free(Y_rows as ptr) + matrix_free(Y_multi) + free(yi_c as ptr) + free(yi_r as ptr) + return 0 +} diff --git a/benchmarks/generated/bench_estimators_02.flow b/benchmarks/generated/bench_estimators_02.flow new file mode 100644 index 0000000..79de920 --- /dev/null +++ b/benchmarks/generated/bench_estimators_02.flow @@ -0,0 +1,535 @@ +# Generated by benchmarks/generate_estimator_bench.py. Do not edit. +# +# One timing block per Flow estimator that the registry marks runnable. +# Regenerate with: +# python benchmarks/estimator_coverage.py +# python benchmarks/generate_estimator_bench.py + +import "lib/scikit/scikit.flow" + +extern { + function printf(fmt: string, ...) -> i32 + function flow_now_ns() -> i64 + function malloc(size: i64) -> ptr + function free(p: ptr) -> void + function fflush(stream: ptr) -> i32 +} + +function ms_between(start: i64, finish: i64) -> f32 { + return ((((finish - start) as f64) / 1000000.0) as f32) +} + +function main() -> i32 { + let iris: Dataset = load_iris() + let diabetes: Dataset = load_diabetes() + let X_c: Matrix = iris.X + let y_c: ptr = iris.y + let n_c: i32 = X_c.rows + let f_c: i32 = X_c.cols + let X_r: Matrix = diabetes.X + let y_r: ptr = diabetes.y + let n_r: i32 = X_r.rows + let f_r: i32 = X_r.cols + + let yi_c: ptr = malloc((n_c as i64) * 4) as ptr + for i in 0 to n_c { yi_c[i] = y_c[i] as i32 } + let yi_r: ptr = malloc((n_r as i64) * 4) as ptr + for i in 0 to n_r { yi_r[i] = y_r[i] as i32 } + + # A two-column target for the cross-decomposition estimators. + let Y_multi: Matrix = matrix_new(n_r, 2) + for i in 0 to n_r { + matrix_set(Y_multi, i, 0, y_r[i]) + matrix_set(Y_multi, i, 1, y_r[i] * 0.5) + } + + # Label-valued targets for the multi-output classifiers. + let Y_labels: Matrix = matrix_new(n_c, 2) + let Y_label_rows: ptr > = malloc((n_c as i64) * 8) as ptr > + for i in 0 to n_c { + let a: f32 = y_c[i] + let b: f32 = ((((y_c[i] as i32) + 1) % 3) as f32) + matrix_set(Y_labels, i, 0, a) + matrix_set(Y_labels, i, 1, b) + let lrow: ptr = array_new_f32(2) + lrow[0] = a + lrow[1] = b + Y_label_rows[i] = lrow + } + + let Y_rows: ptr > = malloc((n_r as i64) * 8) as ptr > + for i in 0 to n_r { + let row: ptr = array_new_f32(2) + row[0] = y_r[i] + row[1] = y_r[i] * 0.5 + Y_rows[i] = row + } + + let mut t0: i64 = 0 + let mut t1: i64 = 0 + let mut t2: i64 = 0 + let mut t3: i64 = 0 + let mut reps: i32 = 1 + + # ---- gradient_boosting_regressor (regression) ---- + t0 = flow_now_ns() + let probe_gradient_boosting_regressor: GradientBoostingRegressor = gradient_boosting_regressor_fit(X_r, y_r, 10, 0.1, 5) + t1 = flow_now_ns() + gradient_boosting_regressor_free(probe_gradient_boosting_regressor) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_gradient_boosting_regressor: GradientBoostingRegressor = gradient_boosting_regressor_fit(X_r, y_r, 10, 0.1, 5) + gradient_boosting_regressor_free(m_gradient_boosting_regressor) + } + t1 = flow_now_ns() + let fitted_gradient_boosting_regressor: GradientBoostingRegressor = gradient_boosting_regressor_fit(X_r, y_r, 10, 0.1, 5) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_gradient_boosting_regressor: ptr = gradient_boosting_regressor_predict(fitted_gradient_boosting_regressor, X_r) + array_free_f32(o_gradient_boosting_regressor) + } + t3 = flow_now_ns() + gradient_boosting_regressor_free(fitted_gradient_boosting_regressor) + printf("ESTIMATOR|gradient_boosting_regressor|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- graphical_lasso (unsupervised) ---- + t0 = flow_now_ns() + let probe_graphical_lasso: GraphicalLasso = graphical_lasso_fit(X_c, 1.0, 100, 0.0001) + t1 = flow_now_ns() + graphical_lasso_free(probe_graphical_lasso) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_graphical_lasso: GraphicalLasso = graphical_lasso_fit(X_c, 1.0, 100, 0.0001) + graphical_lasso_free(m_graphical_lasso) + } + t1 = flow_now_ns() + printf("ESTIMATOR|graphical_lasso|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), 0.0, reps) + fflush(null) + + # ---- hdbscan (unsupervised) ---- + t0 = flow_now_ns() + let probe_hdbscan: HDBSCAN = hdbscan_fit(X_c, 5, 5) + t1 = flow_now_ns() + hdbscan_free(probe_hdbscan) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_hdbscan: HDBSCAN = hdbscan_fit(X_c, 5, 5) + hdbscan_free(m_hdbscan) + } + t1 = flow_now_ns() + printf("ESTIMATOR|hdbscan|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), 0.0, reps) + fflush(null) + + # ---- hist_gradient_boosting_classifier (classification) ---- + t0 = flow_now_ns() + let probe_hist_gradient_boosting_classifier: HistGradientBoostingClassifier = hist_gradient_boosting_classifier_fit(X_c, y_c, n_c, 3, 10, 0.1, 5) + t1 = flow_now_ns() + hist_gradient_boosting_classifier_free(probe_hist_gradient_boosting_classifier) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_hist_gradient_boosting_classifier: HistGradientBoostingClassifier = hist_gradient_boosting_classifier_fit(X_c, y_c, n_c, 3, 10, 0.1, 5) + hist_gradient_boosting_classifier_free(m_hist_gradient_boosting_classifier) + } + t1 = flow_now_ns() + let fitted_hist_gradient_boosting_classifier: HistGradientBoostingClassifier = hist_gradient_boosting_classifier_fit(X_c, y_c, n_c, 3, 10, 0.1, 5) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_hist_gradient_boosting_classifier: ptr = hist_gradient_boosting_classifier_predict(fitted_hist_gradient_boosting_classifier, X_c) + array_free_f32(o_hist_gradient_boosting_classifier) + } + t3 = flow_now_ns() + hist_gradient_boosting_classifier_free(fitted_hist_gradient_boosting_classifier) + printf("ESTIMATOR|hist_gradient_boosting_classifier|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- hist_gradient_boosting_regressor (regression) ---- + t0 = flow_now_ns() + let probe_hist_gradient_boosting_regressor: HistGradientBoostingRegressor = hist_gradient_boosting_regressor_fit(X_r, y_r, n_r, 10, 0.1, 5) + t1 = flow_now_ns() + hist_gradient_boosting_regressor_free(probe_hist_gradient_boosting_regressor) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_hist_gradient_boosting_regressor: HistGradientBoostingRegressor = hist_gradient_boosting_regressor_fit(X_r, y_r, n_r, 10, 0.1, 5) + hist_gradient_boosting_regressor_free(m_hist_gradient_boosting_regressor) + } + t1 = flow_now_ns() + let fitted_hist_gradient_boosting_regressor: HistGradientBoostingRegressor = hist_gradient_boosting_regressor_fit(X_r, y_r, n_r, 10, 0.1, 5) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_hist_gradient_boosting_regressor: ptr = hist_gradient_boosting_regressor_predict(fitted_hist_gradient_boosting_regressor, X_r) + array_free_f32(o_hist_gradient_boosting_regressor) + } + t3 = flow_now_ns() + hist_gradient_boosting_regressor_free(fitted_hist_gradient_boosting_regressor) + printf("ESTIMATOR|hist_gradient_boosting_regressor|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- huber_regressor (regression) ---- + t0 = flow_now_ns() + let probe_huber_regressor: HuberRegressor = huber_regressor_fit(X_r, y_r, 0.1, 100, 0.01) + t1 = flow_now_ns() + huber_regressor_free(probe_huber_regressor) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_huber_regressor: HuberRegressor = huber_regressor_fit(X_r, y_r, 0.1, 100, 0.01) + huber_regressor_free(m_huber_regressor) + } + t1 = flow_now_ns() + let fitted_huber_regressor: HuberRegressor = huber_regressor_fit(X_r, y_r, 0.1, 100, 0.01) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_huber_regressor: ptr = huber_regressor_predict(fitted_huber_regressor, X_r) + array_free_f32(o_huber_regressor) + } + t3 = flow_now_ns() + huber_regressor_free(fitted_huber_regressor) + printf("ESTIMATOR|huber_regressor|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- isolation_forest (unsupervised) ---- + t0 = flow_now_ns() + let probe_isolation_forest: IsolationForest = isolation_forest_fit(X_c, 10, 5, 42) + t1 = flow_now_ns() + isolation_forest_free(probe_isolation_forest) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_isolation_forest: IsolationForest = isolation_forest_fit(X_c, 10, 5, 42) + isolation_forest_free(m_isolation_forest) + } + t1 = flow_now_ns() + printf("ESTIMATOR|isolation_forest|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), 0.0, reps) + fflush(null) + + # ---- isomap (unsupervised) ---- + t0 = flow_now_ns() + let probe_isomap: Isomap = isomap_fit(X_c, 2, 5) + t1 = flow_now_ns() + isomap_free(probe_isomap) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_isomap: Isomap = isomap_fit(X_c, 2, 5) + isomap_free(m_isomap) + } + t1 = flow_now_ns() + printf("ESTIMATOR|isomap|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), 0.0, reps) + fflush(null) + + # ---- iterative_imputer (unsupervised) ---- + t0 = flow_now_ns() + let probe_iterative_imputer: IterativeImputer = iterative_imputer_fit(X_c, 100, 0.0001, 42) + t1 = flow_now_ns() + iterative_imputer_free(probe_iterative_imputer) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_iterative_imputer: IterativeImputer = iterative_imputer_fit(X_c, 100, 0.0001, 42) + iterative_imputer_free(m_iterative_imputer) + } + t1 = flow_now_ns() + let fitted_iterative_imputer: IterativeImputer = iterative_imputer_fit(X_c, 100, 0.0001, 42) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_iterative_imputer: Matrix = iterative_imputer_transform(fitted_iterative_imputer, X_c) + matrix_free(o_iterative_imputer) + } + t3 = flow_now_ns() + iterative_imputer_free(fitted_iterative_imputer) + printf("ESTIMATOR|iterative_imputer|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- kbins_discretizer (unsupervised) ---- + t0 = flow_now_ns() + let probe_kbins_discretizer: KBinsDiscretizer = kbins_discretizer_fit(X_c, 4, 0) + t1 = flow_now_ns() + kbins_discretizer_free(probe_kbins_discretizer) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_kbins_discretizer: KBinsDiscretizer = kbins_discretizer_fit(X_c, 4, 0) + kbins_discretizer_free(m_kbins_discretizer) + } + t1 = flow_now_ns() + let fitted_kbins_discretizer: KBinsDiscretizer = kbins_discretizer_fit(X_c, 4, 0) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_kbins_discretizer: Matrix = kbins_discretizer_transform(fitted_kbins_discretizer, X_c) + matrix_free(o_kbins_discretizer) + } + t3 = flow_now_ns() + kbins_discretizer_free(fitted_kbins_discretizer) + printf("ESTIMATOR|kbins_discretizer|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- kernel_density (unsupervised) ---- + t0 = flow_now_ns() + let probe_kernel_density: KernelDensity = kernel_density_fit(X_c, 1.0, 0) + t1 = flow_now_ns() + kernel_density_free(probe_kernel_density) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_kernel_density: KernelDensity = kernel_density_fit(X_c, 1.0, 0) + kernel_density_free(m_kernel_density) + } + t1 = flow_now_ns() + printf("ESTIMATOR|kernel_density|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), 0.0, reps) + fflush(null) + + # ---- kernel_pca (unsupervised) ---- + t0 = flow_now_ns() + let probe_kernel_pca: KernelPCA = kernel_pca_fit(X_c, 2, 0, 0.1, 2, 0.0) + t1 = flow_now_ns() + kernel_pca_free(probe_kernel_pca) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_kernel_pca: KernelPCA = kernel_pca_fit(X_c, 2, 0, 0.1, 2, 0.0) + kernel_pca_free(m_kernel_pca) + } + t1 = flow_now_ns() + let fitted_kernel_pca: KernelPCA = kernel_pca_fit(X_c, 2, 0, 0.1, 2, 0.0) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_kernel_pca: Matrix = kernel_pca_transform(fitted_kernel_pca, X_c) + matrix_free(o_kernel_pca) + } + t3 = flow_now_ns() + kernel_pca_free(fitted_kernel_pca) + printf("ESTIMATOR|kernel_pca|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- kernel_ridge (regression) ---- + t0 = flow_now_ns() + let probe_kernel_ridge: KernelRidge = kernel_ridge_fit(X_r, y_r, 1.0, 0, 0.1, 2, 0.0) + t1 = flow_now_ns() + kernel_ridge_free(probe_kernel_ridge) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_kernel_ridge: KernelRidge = kernel_ridge_fit(X_r, y_r, 1.0, 0, 0.1, 2, 0.0) + kernel_ridge_free(m_kernel_ridge) + } + t1 = flow_now_ns() + let fitted_kernel_ridge: KernelRidge = kernel_ridge_fit(X_r, y_r, 1.0, 0, 0.1, 2, 0.0) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_kernel_ridge: ptr = kernel_ridge_predict(fitted_kernel_ridge, X_r) + array_free_f32(o_kernel_ridge) + } + t3 = flow_now_ns() + kernel_ridge_free(fitted_kernel_ridge) + printf("ESTIMATOR|kernel_ridge|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- kernel_svc (classification) ---- + t0 = flow_now_ns() + let probe_kernel_svc: KernelSVC = kernel_svc_fit(X_c, y_c, 3, 1.0, 0.1, 100) + t1 = flow_now_ns() + kernel_svc_free(probe_kernel_svc) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_kernel_svc: KernelSVC = kernel_svc_fit(X_c, y_c, 3, 1.0, 0.1, 100) + kernel_svc_free(m_kernel_svc) + } + t1 = flow_now_ns() + let fitted_kernel_svc: KernelSVC = kernel_svc_fit(X_c, y_c, 3, 1.0, 0.1, 100) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_kernel_svc: ptr = kernel_svc_predict(fitted_kernel_svc, X_c) + array_free_f32(o_kernel_svc) + } + t3 = flow_now_ns() + kernel_svc_free(fitted_kernel_svc) + printf("ESTIMATOR|kernel_svc|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- kernel_svc_multi (classification) ---- + t0 = flow_now_ns() + let probe_kernel_svc_multi: KernelSVCMulti = kernel_svc_multi_fit(X_c, y_c, 3, 0.1, 1.0, 100) + t1 = flow_now_ns() + kernel_svc_multi_free(probe_kernel_svc_multi) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_kernel_svc_multi: KernelSVCMulti = kernel_svc_multi_fit(X_c, y_c, 3, 0.1, 1.0, 100) + kernel_svc_multi_free(m_kernel_svc_multi) + } + t1 = flow_now_ns() + let fitted_kernel_svc_multi: KernelSVCMulti = kernel_svc_multi_fit(X_c, y_c, 3, 0.1, 1.0, 100) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_kernel_svc_multi: ptr = kernel_svc_multi_predict(fitted_kernel_svc_multi, X_c) + array_free_f32(o_kernel_svc_multi) + } + t3 = flow_now_ns() + kernel_svc_multi_free(fitted_kernel_svc_multi) + printf("ESTIMATOR|kernel_svc_multi|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- kmeans (unsupervised) ---- + t0 = flow_now_ns() + let probe_kmeans: KMeans = kmeans_fit(X_c, 3, 100, 0.0001, 42) + t1 = flow_now_ns() + kmeans_free(probe_kmeans) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_kmeans: KMeans = kmeans_fit(X_c, 3, 100, 0.0001, 42) + kmeans_free(m_kmeans) + } + t1 = flow_now_ns() + printf("ESTIMATOR|kmeans|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), 0.0, reps) + fflush(null) + + # ---- kneighbors_transformer (unsupervised) ---- + t0 = flow_now_ns() + let probe_kneighbors_transformer: KNeighborsTransformer = kneighbors_transformer_fit(X_c, 5, 0) + t1 = flow_now_ns() + kneighbors_transformer_free(probe_kneighbors_transformer) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_kneighbors_transformer: KNeighborsTransformer = kneighbors_transformer_fit(X_c, 5, 0) + kneighbors_transformer_free(m_kneighbors_transformer) + } + t1 = flow_now_ns() + let fitted_kneighbors_transformer: KNeighborsTransformer = kneighbors_transformer_fit(X_c, 5, 0) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_kneighbors_transformer: Matrix = kneighbors_transformer_transform(fitted_kneighbors_transformer, X_c) + matrix_free(o_kneighbors_transformer) + } + t3 = flow_now_ns() + kneighbors_transformer_free(fitted_kneighbors_transformer) + printf("ESTIMATOR|kneighbors_transformer|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- knn_classifier (classification) ---- + t0 = flow_now_ns() + let probe_knn_classifier: KNNClassifier = knn_classifier_fit(X_c, y_c, 3, 3) + t1 = flow_now_ns() + knn_classifier_free(probe_knn_classifier) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_knn_classifier: KNNClassifier = knn_classifier_fit(X_c, y_c, 3, 3) + knn_classifier_free(m_knn_classifier) + } + t1 = flow_now_ns() + let fitted_knn_classifier: KNNClassifier = knn_classifier_fit(X_c, y_c, 3, 3) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_knn_classifier: Matrix = knn_classifier_predict(fitted_knn_classifier, X_c) + matrix_free(o_knn_classifier) + } + t3 = flow_now_ns() + knn_classifier_free(fitted_knn_classifier) + printf("ESTIMATOR|knn_classifier|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- knn_imputer (unsupervised) ---- + t0 = flow_now_ns() + let probe_knn_imputer: KNNImputer = knn_imputer_fit(X_c, 5, 0) + t1 = flow_now_ns() + knn_imputer_free(probe_knn_imputer) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_knn_imputer: KNNImputer = knn_imputer_fit(X_c, 5, 0) + knn_imputer_free(m_knn_imputer) + } + t1 = flow_now_ns() + let fitted_knn_imputer: KNNImputer = knn_imputer_fit(X_c, 5, 0) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_knn_imputer: Matrix = knn_imputer_transform(fitted_knn_imputer, X_c) + matrix_free(o_knn_imputer) + } + t3 = flow_now_ns() + knn_imputer_free(fitted_knn_imputer) + printf("ESTIMATOR|knn_imputer|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- knn_regressor (regression) ---- + t0 = flow_now_ns() + let probe_knn_regressor: KNNRegressor = knn_regressor_fit(X_r, y_r, 3) + t1 = flow_now_ns() + knn_regressor_free(probe_knn_regressor) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_knn_regressor: KNNRegressor = knn_regressor_fit(X_r, y_r, 3) + knn_regressor_free(m_knn_regressor) + } + t1 = flow_now_ns() + let fitted_knn_regressor: KNNRegressor = knn_regressor_fit(X_r, y_r, 3) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_knn_regressor: ptr = knn_regressor_predict(fitted_knn_regressor, X_r) + array_free_f32(o_knn_regressor) + } + t3 = flow_now_ns() + knn_regressor_free(fitted_knn_regressor) + printf("ESTIMATOR|knn_regressor|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + for i in 0 to n_c { array_free_f32(Y_label_rows[i]) } + free(Y_label_rows as ptr) + matrix_free(Y_labels) + for i in 0 to n_r { array_free_f32(Y_rows[i]) } + free(Y_rows as ptr) + matrix_free(Y_multi) + free(yi_c as ptr) + free(yi_r as ptr) + return 0 +} diff --git a/benchmarks/generated/bench_estimators_03.flow b/benchmarks/generated/bench_estimators_03.flow new file mode 100644 index 0000000..542372f --- /dev/null +++ b/benchmarks/generated/bench_estimators_03.flow @@ -0,0 +1,533 @@ +# Generated by benchmarks/generate_estimator_bench.py. Do not edit. +# +# One timing block per Flow estimator that the registry marks runnable. +# Regenerate with: +# python benchmarks/estimator_coverage.py +# python benchmarks/generate_estimator_bench.py + +import "lib/scikit/scikit.flow" + +extern { + function printf(fmt: string, ...) -> i32 + function flow_now_ns() -> i64 + function malloc(size: i64) -> ptr + function free(p: ptr) -> void + function fflush(stream: ptr) -> i32 +} + +function ms_between(start: i64, finish: i64) -> f32 { + return ((((finish - start) as f64) / 1000000.0) as f32) +} + +function main() -> i32 { + let iris: Dataset = load_iris() + let diabetes: Dataset = load_diabetes() + let X_c: Matrix = iris.X + let y_c: ptr = iris.y + let n_c: i32 = X_c.rows + let f_c: i32 = X_c.cols + let X_r: Matrix = diabetes.X + let y_r: ptr = diabetes.y + let n_r: i32 = X_r.rows + let f_r: i32 = X_r.cols + + let yi_c: ptr = malloc((n_c as i64) * 4) as ptr + for i in 0 to n_c { yi_c[i] = y_c[i] as i32 } + let yi_r: ptr = malloc((n_r as i64) * 4) as ptr + for i in 0 to n_r { yi_r[i] = y_r[i] as i32 } + + # A two-column target for the cross-decomposition estimators. + let Y_multi: Matrix = matrix_new(n_r, 2) + for i in 0 to n_r { + matrix_set(Y_multi, i, 0, y_r[i]) + matrix_set(Y_multi, i, 1, y_r[i] * 0.5) + } + + # Label-valued targets for the multi-output classifiers. + let Y_labels: Matrix = matrix_new(n_c, 2) + let Y_label_rows: ptr > = malloc((n_c as i64) * 8) as ptr > + for i in 0 to n_c { + let a: f32 = y_c[i] + let b: f32 = ((((y_c[i] as i32) + 1) % 3) as f32) + matrix_set(Y_labels, i, 0, a) + matrix_set(Y_labels, i, 1, b) + let lrow: ptr = array_new_f32(2) + lrow[0] = a + lrow[1] = b + Y_label_rows[i] = lrow + } + + let Y_rows: ptr > = malloc((n_r as i64) * 8) as ptr > + for i in 0 to n_r { + let row: ptr = array_new_f32(2) + row[0] = y_r[i] + row[1] = y_r[i] * 0.5 + Y_rows[i] = row + } + + let mut t0: i64 = 0 + let mut t1: i64 = 0 + let mut t2: i64 = 0 + let mut t3: i64 = 0 + let mut reps: i32 = 1 + + # ---- label_propagation (classification) ---- + t0 = flow_now_ns() + let probe_label_propagation: LabelPropagation = label_propagation_fit(X_c, yi_c, 3, 0.1, 100, 0.0001) + t1 = flow_now_ns() + label_propagation_free(probe_label_propagation) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_label_propagation: LabelPropagation = label_propagation_fit(X_c, yi_c, 3, 0.1, 100, 0.0001) + label_propagation_free(m_label_propagation) + } + t1 = flow_now_ns() + printf("ESTIMATOR|label_propagation|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), 0.0, reps) + fflush(null) + + # ---- label_spreading (classification) ---- + t0 = flow_now_ns() + let probe_label_spreading: LabelSpreading = label_spreading_fit(X_c, yi_c, 3, 0.1, 1.0, 100, 0.0001) + t1 = flow_now_ns() + label_spreading_free(probe_label_spreading) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_label_spreading: LabelSpreading = label_spreading_fit(X_c, yi_c, 3, 0.1, 1.0, 100, 0.0001) + label_spreading_free(m_label_spreading) + } + t1 = flow_now_ns() + printf("ESTIMATOR|label_spreading|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), 0.0, reps) + fflush(null) + + # ---- lars_cv (regression) ---- + t0 = flow_now_ns() + let probe_lars_cv: LarsCV = lars_cv_fit(X_r, y_r, 2) + t1 = flow_now_ns() + lars_cv_free(probe_lars_cv) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_lars_cv: LarsCV = lars_cv_fit(X_r, y_r, 2) + lars_cv_free(m_lars_cv) + } + t1 = flow_now_ns() + let fitted_lars_cv: LarsCV = lars_cv_fit(X_r, y_r, 2) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_lars_cv: ptr = lars_cv_predict(fitted_lars_cv, X_r) + array_free_f32(o_lars_cv) + } + t3 = flow_now_ns() + lars_cv_free(fitted_lars_cv) + printf("ESTIMATOR|lars_cv|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- lars (regression) ---- + t0 = flow_now_ns() + let probe_lars: Lars = lars_fit(X_r, y_r, 2) + t1 = flow_now_ns() + lars_free(probe_lars) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_lars: Lars = lars_fit(X_r, y_r, 2) + lars_free(m_lars) + } + t1 = flow_now_ns() + let fitted_lars: Lars = lars_fit(X_r, y_r, 2) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_lars: ptr = lars_predict(fitted_lars, X_r) + array_free_f32(o_lars) + } + t3 = flow_now_ns() + lars_free(fitted_lars) + printf("ESTIMATOR|lars|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- lasso_cv (regression) ---- + t0 = flow_now_ns() + let probe_lasso_cv: LassoCV = lasso_cv_fit(X_r, y_r, 10) + t1 = flow_now_ns() + lasso_cv_free(probe_lasso_cv) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_lasso_cv: LassoCV = lasso_cv_fit(X_r, y_r, 10) + lasso_cv_free(m_lasso_cv) + } + t1 = flow_now_ns() + let fitted_lasso_cv: LassoCV = lasso_cv_fit(X_r, y_r, 10) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_lasso_cv: ptr = lasso_cv_predict(fitted_lasso_cv, X_r) + array_free_f32(o_lasso_cv) + } + t3 = flow_now_ns() + lasso_cv_free(fitted_lasso_cv) + printf("ESTIMATOR|lasso_cv|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- lasso (regression) ---- + t0 = flow_now_ns() + let probe_lasso: Lasso = lasso_fit(X_r, y_r, 1.0, 50, 0.01) + t1 = flow_now_ns() + lasso_free(probe_lasso) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_lasso: Lasso = lasso_fit(X_r, y_r, 1.0, 50, 0.01) + lasso_free(m_lasso) + } + t1 = flow_now_ns() + let fitted_lasso: Lasso = lasso_fit(X_r, y_r, 1.0, 50, 0.01) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_lasso: ptr = lasso_predict(fitted_lasso, X_r) + array_free_f32(o_lasso) + } + t3 = flow_now_ns() + lasso_free(fitted_lasso) + printf("ESTIMATOR|lasso|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- lasso_lars_cv (regression) ---- + t0 = flow_now_ns() + let probe_lasso_lars_cv: LassoLarsCV = lasso_lars_cv_fit(X_r, y_r, 10) + t1 = flow_now_ns() + lasso_lars_cv_free(probe_lasso_lars_cv) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_lasso_lars_cv: LassoLarsCV = lasso_lars_cv_fit(X_r, y_r, 10) + lasso_lars_cv_free(m_lasso_lars_cv) + } + t1 = flow_now_ns() + let fitted_lasso_lars_cv: LassoLarsCV = lasso_lars_cv_fit(X_r, y_r, 10) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_lasso_lars_cv: ptr = lasso_lars_cv_predict(fitted_lasso_lars_cv, X_r) + array_free_f32(o_lasso_lars_cv) + } + t3 = flow_now_ns() + lasso_lars_cv_free(fitted_lasso_lars_cv) + printf("ESTIMATOR|lasso_lars_cv|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- lasso_lars (regression) ---- + t0 = flow_now_ns() + let probe_lasso_lars: LassoLars = lasso_lars_fit(X_r, y_r, 1.0, 100) + t1 = flow_now_ns() + lasso_lars_free(probe_lasso_lars) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_lasso_lars: LassoLars = lasso_lars_fit(X_r, y_r, 1.0, 100) + lasso_lars_free(m_lasso_lars) + } + t1 = flow_now_ns() + let fitted_lasso_lars: LassoLars = lasso_lars_fit(X_r, y_r, 1.0, 100) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_lasso_lars: ptr = lasso_lars_predict(fitted_lasso_lars, X_r) + array_free_f32(o_lasso_lars) + } + t3 = flow_now_ns() + lasso_lars_free(fitted_lasso_lars) + printf("ESTIMATOR|lasso_lars|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- lasso_lars_ic (regression) ---- + t0 = flow_now_ns() + let probe_lasso_lars_ic: LassoLarsIC = lasso_lars_ic_fit(X_r, y_r, 0) + t1 = flow_now_ns() + lasso_lars_ic_free(probe_lasso_lars_ic) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_lasso_lars_ic: LassoLarsIC = lasso_lars_ic_fit(X_r, y_r, 0) + lasso_lars_ic_free(m_lasso_lars_ic) + } + t1 = flow_now_ns() + let fitted_lasso_lars_ic: LassoLarsIC = lasso_lars_ic_fit(X_r, y_r, 0) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_lasso_lars_ic: ptr = lasso_lars_ic_predict(fitted_lasso_lars_ic, X_r) + array_free_f32(o_lasso_lars_ic) + } + t3 = flow_now_ns() + lasso_lars_ic_free(fitted_lasso_lars_ic) + printf("ESTIMATOR|lasso_lars_ic|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- lda (unsupervised) ---- + t0 = flow_now_ns() + let probe_lda: LatentDirichletAllocation = lda_fit(X_c, 3, 50, 1.0, 0.1, 42) + t1 = flow_now_ns() + lda_free(probe_lda) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_lda: LatentDirichletAllocation = lda_fit(X_c, 3, 50, 1.0, 0.1, 42) + lda_free(m_lda) + } + t1 = flow_now_ns() + let fitted_lda: LatentDirichletAllocation = lda_fit(X_c, 3, 50, 1.0, 0.1, 42) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_lda: Matrix = lda_transform(fitted_lda, X_c) + matrix_free(o_lda) + } + t3 = flow_now_ns() + lda_free(fitted_lda) + printf("ESTIMATOR|lda|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- ledoit_wolf_estimator (unsupervised) ---- + t0 = flow_now_ns() + let probe_ledoit_wolf_estimator: ShrunkCovariance = ledoit_wolf_estimator_fit(X_c) + t1 = flow_now_ns() + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_ledoit_wolf_estimator: ShrunkCovariance = ledoit_wolf_estimator_fit(X_c) + } + t1 = flow_now_ns() + printf("ESTIMATOR|ledoit_wolf_estimator|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), 0.0, reps) + fflush(null) + + # ---- linear_regression (regression) ---- + t0 = flow_now_ns() + let probe_linear_regression: LinearRegression = linear_regression_fit(X_r, y_r, penalty_none()) + t1 = flow_now_ns() + linear_regression_free(probe_linear_regression) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_linear_regression: LinearRegression = linear_regression_fit(X_r, y_r, penalty_none()) + linear_regression_free(m_linear_regression) + } + t1 = flow_now_ns() + let fitted_linear_regression: LinearRegression = linear_regression_fit(X_r, y_r, penalty_none()) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_linear_regression: ptr = linear_regression_predict(fitted_linear_regression, X_r) + array_free_f32(o_linear_regression) + } + t3 = flow_now_ns() + linear_regression_free(fitted_linear_regression) + printf("ESTIMATOR|linear_regression|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- linear_svc (classification) ---- + t0 = flow_now_ns() + let probe_linear_svc: LinearSVC = linear_svc_fit(X_c, y_c, 3, 1.0, 50, 0.01) + t1 = flow_now_ns() + linear_svc_free(probe_linear_svc) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_linear_svc: LinearSVC = linear_svc_fit(X_c, y_c, 3, 1.0, 50, 0.01) + linear_svc_free(m_linear_svc) + } + t1 = flow_now_ns() + let fitted_linear_svc: LinearSVC = linear_svc_fit(X_c, y_c, 3, 1.0, 50, 0.01) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_linear_svc: ptr = linear_svc_predict(fitted_linear_svc, X_c) + array_free_f32(o_linear_svc) + } + t3 = flow_now_ns() + linear_svc_free(fitted_linear_svc) + printf("ESTIMATOR|linear_svc|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- linear_svc_multi (classification) ---- + t0 = flow_now_ns() + let probe_linear_svc_multi: LinearSVCMulti = linear_svc_multi_fit(X_c, y_c, 3, 1.0, 50) + t1 = flow_now_ns() + linear_svc_multi_free(probe_linear_svc_multi) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_linear_svc_multi: LinearSVCMulti = linear_svc_multi_fit(X_c, y_c, 3, 1.0, 50) + linear_svc_multi_free(m_linear_svc_multi) + } + t1 = flow_now_ns() + let fitted_linear_svc_multi: LinearSVCMulti = linear_svc_multi_fit(X_c, y_c, 3, 1.0, 50) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_linear_svc_multi: ptr = linear_svc_multi_predict(fitted_linear_svc_multi, X_c) + array_free_f32(o_linear_svc_multi) + } + t3 = flow_now_ns() + linear_svc_multi_free(fitted_linear_svc_multi) + printf("ESTIMATOR|linear_svc_multi|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- linear_svr (regression) ---- + t0 = flow_now_ns() + let probe_linear_svr: LinearSVR = linear_svr_fit(X_r, y_r, 1.0, 0.1, 50, 0.01) + t1 = flow_now_ns() + linear_svr_free(probe_linear_svr) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_linear_svr: LinearSVR = linear_svr_fit(X_r, y_r, 1.0, 0.1, 50, 0.01) + linear_svr_free(m_linear_svr) + } + t1 = flow_now_ns() + let fitted_linear_svr: LinearSVR = linear_svr_fit(X_r, y_r, 1.0, 0.1, 50, 0.01) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_linear_svr: ptr = linear_svr_predict(fitted_linear_svr, X_r) + array_free_f32(o_linear_svr) + } + t3 = flow_now_ns() + linear_svr_free(fitted_linear_svr) + printf("ESTIMATOR|linear_svr|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- lle (unsupervised) ---- + t0 = flow_now_ns() + let probe_lle: LLE = lle_fit(X_c, 2, 5) + t1 = flow_now_ns() + lle_free(probe_lle) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_lle: LLE = lle_fit(X_c, 2, 5) + lle_free(m_lle) + } + t1 = flow_now_ns() + printf("ESTIMATOR|lle|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), 0.0, reps) + fflush(null) + + # ---- local_outlier_factor (unsupervised) ---- + t0 = flow_now_ns() + let probe_local_outlier_factor: LocalOutlierFactor = local_outlier_factor_fit(X_c, 5) + t1 = flow_now_ns() + local_outlier_factor_free(probe_local_outlier_factor) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_local_outlier_factor: LocalOutlierFactor = local_outlier_factor_fit(X_c, 5) + local_outlier_factor_free(m_local_outlier_factor) + } + t1 = flow_now_ns() + printf("ESTIMATOR|local_outlier_factor|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), 0.0, reps) + fflush(null) + + # ---- logistic_regression_cv (classification) ---- + t0 = flow_now_ns() + let probe_logistic_regression_cv: LogisticRegressionCV = logistic_regression_cv_fit(X_c, y_c, 3, 5) + t1 = flow_now_ns() + logistic_regression_cv_free(probe_logistic_regression_cv) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_logistic_regression_cv: LogisticRegressionCV = logistic_regression_cv_fit(X_c, y_c, 3, 5) + logistic_regression_cv_free(m_logistic_regression_cv) + } + t1 = flow_now_ns() + let fitted_logistic_regression_cv: LogisticRegressionCV = logistic_regression_cv_fit(X_c, y_c, 3, 5) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_logistic_regression_cv: ptr = logistic_regression_cv_predict(fitted_logistic_regression_cv, X_c) + array_free_f32(o_logistic_regression_cv) + } + t3 = flow_now_ns() + logistic_regression_cv_free(fitted_logistic_regression_cv) + printf("ESTIMATOR|logistic_regression_cv|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- logistic_regression (classification) ---- + t0 = flow_now_ns() + let probe_logistic_regression: LogisticRegression = logistic_regression_fit(X_c, y_c, 3, 50, 0.01, penalty_none()) + t1 = flow_now_ns() + logistic_regression_free(probe_logistic_regression) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_logistic_regression: LogisticRegression = logistic_regression_fit(X_c, y_c, 3, 50, 0.01, penalty_none()) + logistic_regression_free(m_logistic_regression) + } + t1 = flow_now_ns() + printf("ESTIMATOR|logistic_regression|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), 0.0, reps) + fflush(null) + + # ---- maxabs_scaler (unsupervised) ---- + t0 = flow_now_ns() + let probe_maxabs_scaler: MaxAbsScaler = maxabs_scaler_fit(X_c) + t1 = flow_now_ns() + maxabs_scaler_free(probe_maxabs_scaler) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_maxabs_scaler: MaxAbsScaler = maxabs_scaler_fit(X_c) + maxabs_scaler_free(m_maxabs_scaler) + } + t1 = flow_now_ns() + let fitted_maxabs_scaler: MaxAbsScaler = maxabs_scaler_fit(X_c) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_maxabs_scaler: Matrix = maxabs_scaler_transform(fitted_maxabs_scaler, X_c) + matrix_free(o_maxabs_scaler) + } + t3 = flow_now_ns() + maxabs_scaler_free(fitted_maxabs_scaler) + printf("ESTIMATOR|maxabs_scaler|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + for i in 0 to n_c { array_free_f32(Y_label_rows[i]) } + free(Y_label_rows as ptr) + matrix_free(Y_labels) + for i in 0 to n_r { array_free_f32(Y_rows[i]) } + free(Y_rows as ptr) + matrix_free(Y_multi) + free(yi_c as ptr) + free(yi_r as ptr) + return 0 +} diff --git a/benchmarks/generated/bench_estimators_04.flow b/benchmarks/generated/bench_estimators_04.flow new file mode 100644 index 0000000..132eb6e --- /dev/null +++ b/benchmarks/generated/bench_estimators_04.flow @@ -0,0 +1,553 @@ +# Generated by benchmarks/generate_estimator_bench.py. Do not edit. +# +# One timing block per Flow estimator that the registry marks runnable. +# Regenerate with: +# python benchmarks/estimator_coverage.py +# python benchmarks/generate_estimator_bench.py + +import "lib/scikit/scikit.flow" + +extern { + function printf(fmt: string, ...) -> i32 + function flow_now_ns() -> i64 + function malloc(size: i64) -> ptr + function free(p: ptr) -> void + function fflush(stream: ptr) -> i32 +} + +function ms_between(start: i64, finish: i64) -> f32 { + return ((((finish - start) as f64) / 1000000.0) as f32) +} + +function main() -> i32 { + let iris: Dataset = load_iris() + let diabetes: Dataset = load_diabetes() + let X_c: Matrix = iris.X + let y_c: ptr = iris.y + let n_c: i32 = X_c.rows + let f_c: i32 = X_c.cols + let X_r: Matrix = diabetes.X + let y_r: ptr = diabetes.y + let n_r: i32 = X_r.rows + let f_r: i32 = X_r.cols + + let yi_c: ptr = malloc((n_c as i64) * 4) as ptr + for i in 0 to n_c { yi_c[i] = y_c[i] as i32 } + let yi_r: ptr = malloc((n_r as i64) * 4) as ptr + for i in 0 to n_r { yi_r[i] = y_r[i] as i32 } + + # A two-column target for the cross-decomposition estimators. + let Y_multi: Matrix = matrix_new(n_r, 2) + for i in 0 to n_r { + matrix_set(Y_multi, i, 0, y_r[i]) + matrix_set(Y_multi, i, 1, y_r[i] * 0.5) + } + + # Label-valued targets for the multi-output classifiers. + let Y_labels: Matrix = matrix_new(n_c, 2) + let Y_label_rows: ptr > = malloc((n_c as i64) * 8) as ptr > + for i in 0 to n_c { + let a: f32 = y_c[i] + let b: f32 = ((((y_c[i] as i32) + 1) % 3) as f32) + matrix_set(Y_labels, i, 0, a) + matrix_set(Y_labels, i, 1, b) + let lrow: ptr = array_new_f32(2) + lrow[0] = a + lrow[1] = b + Y_label_rows[i] = lrow + } + + let Y_rows: ptr > = malloc((n_r as i64) * 8) as ptr > + for i in 0 to n_r { + let row: ptr = array_new_f32(2) + row[0] = y_r[i] + row[1] = y_r[i] * 0.5 + Y_rows[i] = row + } + + let mut t0: i64 = 0 + let mut t1: i64 = 0 + let mut t2: i64 = 0 + let mut t3: i64 = 0 + let mut reps: i32 = 1 + + # ---- mds (unsupervised) ---- + t0 = flow_now_ns() + let probe_mds: MDS = mds_fit(X_c, 2, 100, 0.0001, 42) + t1 = flow_now_ns() + mds_free(probe_mds) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_mds: MDS = mds_fit(X_c, 2, 100, 0.0001, 42) + mds_free(m_mds) + } + t1 = flow_now_ns() + printf("ESTIMATOR|mds|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), 0.0, reps) + fflush(null) + + # ---- mean_shift (unsupervised) ---- + t0 = flow_now_ns() + let probe_mean_shift: MeanShift = mean_shift_fit(X_c, 1.0, 100) + t1 = flow_now_ns() + mean_shift_free(probe_mean_shift) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_mean_shift: MeanShift = mean_shift_fit(X_c, 1.0, 100) + mean_shift_free(m_mean_shift) + } + t1 = flow_now_ns() + printf("ESTIMATOR|mean_shift|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), 0.0, reps) + fflush(null) + + # ---- min_cov_det (unsupervised) ---- + t0 = flow_now_ns() + let probe_min_cov_det: MinCovDet = min_cov_det_fit(X_c, 0.75, 100, 42) + t1 = flow_now_ns() + min_cov_det_free(probe_min_cov_det) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_min_cov_det: MinCovDet = min_cov_det_fit(X_c, 0.75, 100, 42) + min_cov_det_free(m_min_cov_det) + } + t1 = flow_now_ns() + printf("ESTIMATOR|min_cov_det|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), 0.0, reps) + fflush(null) + + # ---- minibatch_dictionary_learning (unsupervised) ---- + t0 = flow_now_ns() + let probe_minibatch_dictionary_learning: MiniBatchDictionaryLearning = minibatch_dictionary_learning_fit(X_c, 2, 50, 32, 42) + t1 = flow_now_ns() + minibatch_dictionary_learning_free(probe_minibatch_dictionary_learning) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_minibatch_dictionary_learning: MiniBatchDictionaryLearning = minibatch_dictionary_learning_fit(X_c, 2, 50, 32, 42) + minibatch_dictionary_learning_free(m_minibatch_dictionary_learning) + } + t1 = flow_now_ns() + let fitted_minibatch_dictionary_learning: MiniBatchDictionaryLearning = minibatch_dictionary_learning_fit(X_c, 2, 50, 32, 42) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_minibatch_dictionary_learning: Matrix = minibatch_dictionary_learning_transform(fitted_minibatch_dictionary_learning, X_c) + matrix_free(o_minibatch_dictionary_learning) + } + t3 = flow_now_ns() + minibatch_dictionary_learning_free(fitted_minibatch_dictionary_learning) + printf("ESTIMATOR|minibatch_dictionary_learning|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- minibatch_kmeans (unsupervised) ---- + t0 = flow_now_ns() + let probe_minibatch_kmeans: MiniBatchKMeans = minibatch_kmeans_fit(X_c, 3, 32, 100, 0.0001, 42) + t1 = flow_now_ns() + minibatch_kmeans_free(probe_minibatch_kmeans) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_minibatch_kmeans: MiniBatchKMeans = minibatch_kmeans_fit(X_c, 3, 32, 100, 0.0001, 42) + minibatch_kmeans_free(m_minibatch_kmeans) + } + t1 = flow_now_ns() + printf("ESTIMATOR|minibatch_kmeans|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), 0.0, reps) + fflush(null) + + # ---- minibatch_nmf (unsupervised) ---- + t0 = flow_now_ns() + let probe_minibatch_nmf: MiniBatchNMF = minibatch_nmf_fit(X_c, 2, 100, 32, 0.0001, 42) + t1 = flow_now_ns() + minibatch_nmf_free(probe_minibatch_nmf) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_minibatch_nmf: MiniBatchNMF = minibatch_nmf_fit(X_c, 2, 100, 32, 0.0001, 42) + minibatch_nmf_free(m_minibatch_nmf) + } + t1 = flow_now_ns() + let fitted_minibatch_nmf: MiniBatchNMF = minibatch_nmf_fit(X_c, 2, 100, 32, 0.0001, 42) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_minibatch_nmf: Matrix = minibatch_nmf_transform(fitted_minibatch_nmf, X_c) + matrix_free(o_minibatch_nmf) + } + t3 = flow_now_ns() + minibatch_nmf_free(fitted_minibatch_nmf) + printf("ESTIMATOR|minibatch_nmf|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- minibatch_sparse_pca (unsupervised) ---- + t0 = flow_now_ns() + let probe_minibatch_sparse_pca: MiniBatchSparsePCA = minibatch_sparse_pca_fit(X_c, 2, 1.0, 100, 32, 0.0001, 42) + t1 = flow_now_ns() + minibatch_sparse_pca_free(probe_minibatch_sparse_pca) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_minibatch_sparse_pca: MiniBatchSparsePCA = minibatch_sparse_pca_fit(X_c, 2, 1.0, 100, 32, 0.0001, 42) + minibatch_sparse_pca_free(m_minibatch_sparse_pca) + } + t1 = flow_now_ns() + let fitted_minibatch_sparse_pca: MiniBatchSparsePCA = minibatch_sparse_pca_fit(X_c, 2, 1.0, 100, 32, 0.0001, 42) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_minibatch_sparse_pca: Matrix = minibatch_sparse_pca_transform(fitted_minibatch_sparse_pca, X_c) + matrix_free(o_minibatch_sparse_pca) + } + t3 = flow_now_ns() + minibatch_sparse_pca_free(fitted_minibatch_sparse_pca) + printf("ESTIMATOR|minibatch_sparse_pca|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- minmax_scaler (unsupervised) ---- + t0 = flow_now_ns() + let probe_minmax_scaler: MinMaxScaler = minmax_scaler_fit(X_c) + t1 = flow_now_ns() + minmax_scaler_free(probe_minmax_scaler) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_minmax_scaler: MinMaxScaler = minmax_scaler_fit(X_c) + minmax_scaler_free(m_minmax_scaler) + } + t1 = flow_now_ns() + let fitted_minmax_scaler: MinMaxScaler = minmax_scaler_fit(X_c) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_minmax_scaler: Matrix = minmax_scaler_transform(fitted_minmax_scaler, X_c) + matrix_free(o_minmax_scaler) + } + t3 = flow_now_ns() + minmax_scaler_free(fitted_minmax_scaler) + printf("ESTIMATOR|minmax_scaler|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- missing_indicator (unsupervised) ---- + t0 = flow_now_ns() + let probe_missing_indicator: MissingIndicator = missing_indicator_fit(X_c, 0.0, 0) + t1 = flow_now_ns() + missing_indicator_free(probe_missing_indicator) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_missing_indicator: MissingIndicator = missing_indicator_fit(X_c, 0.0, 0) + missing_indicator_free(m_missing_indicator) + } + t1 = flow_now_ns() + let fitted_missing_indicator: MissingIndicator = missing_indicator_fit(X_c, 0.0, 0) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_missing_indicator: Matrix = missing_indicator_transform(fitted_missing_indicator, X_c) + matrix_free(o_missing_indicator) + } + t3 = flow_now_ns() + missing_indicator_free(fitted_missing_indicator) + printf("ESTIMATOR|missing_indicator|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- mlp_classifier (classification) ---- + let hidden_sizes_mlp_classifier: array = [8] + t0 = flow_now_ns() + let probe_mlp_classifier: MLPClassifier = mlp_classifier_fit(X_c, y_c, 3, hidden_sizes_mlp_classifier, 1, 0, 50, 0.01, 0.9, 42) + t1 = flow_now_ns() + mlp_classifier_free(probe_mlp_classifier) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_mlp_classifier: MLPClassifier = mlp_classifier_fit(X_c, y_c, 3, hidden_sizes_mlp_classifier, 1, 0, 50, 0.01, 0.9, 42) + mlp_classifier_free(m_mlp_classifier) + } + t1 = flow_now_ns() + let fitted_mlp_classifier: MLPClassifier = mlp_classifier_fit(X_c, y_c, 3, hidden_sizes_mlp_classifier, 1, 0, 50, 0.01, 0.9, 42) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_mlp_classifier: Matrix = mlp_classifier_predict(fitted_mlp_classifier, X_c) + matrix_free(o_mlp_classifier) + } + t3 = flow_now_ns() + mlp_classifier_free(fitted_mlp_classifier) + printf("ESTIMATOR|mlp_classifier|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- mlp_regressor (regression) ---- + let hidden_sizes_mlp_regressor: array = [8] + t0 = flow_now_ns() + let probe_mlp_regressor: MLPRegressor = mlp_regressor_fit(X_r, y_r, hidden_sizes_mlp_regressor, 1, 0, 50, 0.01, 0.9, 42) + t1 = flow_now_ns() + mlp_regressor_free(probe_mlp_regressor) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_mlp_regressor: MLPRegressor = mlp_regressor_fit(X_r, y_r, hidden_sizes_mlp_regressor, 1, 0, 50, 0.01, 0.9, 42) + mlp_regressor_free(m_mlp_regressor) + } + t1 = flow_now_ns() + let fitted_mlp_regressor: MLPRegressor = mlp_regressor_fit(X_r, y_r, hidden_sizes_mlp_regressor, 1, 0, 50, 0.01, 0.9, 42) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_mlp_regressor: ptr = mlp_regressor_predict(fitted_mlp_regressor, X_r) + array_free_f32(o_mlp_regressor) + } + t3 = flow_now_ns() + mlp_regressor_free(fitted_mlp_regressor) + printf("ESTIMATOR|mlp_regressor|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- multi_output_classifier (multioutput_class) ---- + t0 = flow_now_ns() + let probe_multi_output_classifier: MultiOutputClassifier = multi_output_classifier_fit(X_c, Y_labels, 2, 1, 50, 0.01) + t1 = flow_now_ns() + multi_output_classifier_free(probe_multi_output_classifier) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_multi_output_classifier: MultiOutputClassifier = multi_output_classifier_fit(X_c, Y_labels, 2, 1, 50, 0.01) + multi_output_classifier_free(m_multi_output_classifier) + } + t1 = flow_now_ns() + let fitted_multi_output_classifier: MultiOutputClassifier = multi_output_classifier_fit(X_c, Y_labels, 2, 1, 50, 0.01) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_multi_output_classifier: Matrix = multi_output_classifier_predict(fitted_multi_output_classifier, X_c) + matrix_free(o_multi_output_classifier) + } + t3 = flow_now_ns() + multi_output_classifier_free(fitted_multi_output_classifier) + printf("ESTIMATOR|multi_output_classifier|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- multi_output_regressor (multioutput) ---- + t0 = flow_now_ns() + let probe_multi_output_regressor: MultiOutputRegressor = multi_output_regressor_fit(X_r, Y_multi, 2) + t1 = flow_now_ns() + multi_output_regressor_free(probe_multi_output_regressor) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_multi_output_regressor: MultiOutputRegressor = multi_output_regressor_fit(X_r, Y_multi, 2) + multi_output_regressor_free(m_multi_output_regressor) + } + t1 = flow_now_ns() + let fitted_multi_output_regressor: MultiOutputRegressor = multi_output_regressor_fit(X_r, Y_multi, 2) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_multi_output_regressor: Matrix = multi_output_regressor_predict(fitted_multi_output_regressor, X_r) + matrix_free(o_multi_output_regressor) + } + t3 = flow_now_ns() + multi_output_regressor_free(fitted_multi_output_regressor) + printf("ESTIMATOR|multi_output_regressor|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- multiclass_logistic (classification) ---- + t0 = flow_now_ns() + let probe_multiclass_logistic: MultiClassLogisticRegression = multiclass_logistic_fit(X_c, y_c, 3, 50, 0.01, penalty_none()) + t1 = flow_now_ns() + multiclass_logistic_free(probe_multiclass_logistic) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_multiclass_logistic: MultiClassLogisticRegression = multiclass_logistic_fit(X_c, y_c, 3, 50, 0.01, penalty_none()) + multiclass_logistic_free(m_multiclass_logistic) + } + t1 = flow_now_ns() + let fitted_multiclass_logistic: MultiClassLogisticRegression = multiclass_logistic_fit(X_c, y_c, 3, 50, 0.01, penalty_none()) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_multiclass_logistic: Matrix = multiclass_logistic_predict(fitted_multiclass_logistic, X_c) + matrix_free(o_multiclass_logistic) + } + t3 = flow_now_ns() + multiclass_logistic_free(fitted_multiclass_logistic) + printf("ESTIMATOR|multiclass_logistic|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- multinomial_nb (classification) ---- + t0 = flow_now_ns() + let probe_multinomial_nb: MultinomialNB = multinomial_nb_fit(X_c, y_c, 3, 1.0) + t1 = flow_now_ns() + multinomial_nb_free(probe_multinomial_nb) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_multinomial_nb: MultinomialNB = multinomial_nb_fit(X_c, y_c, 3, 1.0) + multinomial_nb_free(m_multinomial_nb) + } + t1 = flow_now_ns() + let fitted_multinomial_nb: MultinomialNB = multinomial_nb_fit(X_c, y_c, 3, 1.0) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_multinomial_nb: ptr = multinomial_nb_predict(fitted_multinomial_nb, X_c) + array_free_f32(o_multinomial_nb) + } + t3 = flow_now_ns() + multinomial_nb_free(fitted_multinomial_nb) + printf("ESTIMATOR|multinomial_nb|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- multitask_elastic_net_cv (multioutput) ---- + t0 = flow_now_ns() + let probe_multitask_elastic_net_cv: MultiTaskElasticNetCV = multitask_elastic_net_cv_fit(X_r, Y_multi, 10) + t1 = flow_now_ns() + multitask_elastic_net_cv_free(probe_multitask_elastic_net_cv) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_multitask_elastic_net_cv: MultiTaskElasticNetCV = multitask_elastic_net_cv_fit(X_r, Y_multi, 10) + multitask_elastic_net_cv_free(m_multitask_elastic_net_cv) + } + t1 = flow_now_ns() + let fitted_multitask_elastic_net_cv: MultiTaskElasticNetCV = multitask_elastic_net_cv_fit(X_r, Y_multi, 10) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_multitask_elastic_net_cv: Matrix = multitask_elastic_net_cv_predict(fitted_multitask_elastic_net_cv, X_r) + matrix_free(o_multitask_elastic_net_cv) + } + t3 = flow_now_ns() + multitask_elastic_net_cv_free(fitted_multitask_elastic_net_cv) + printf("ESTIMATOR|multitask_elastic_net_cv|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- multitask_elastic_net (multioutput) ---- + t0 = flow_now_ns() + let probe_multitask_elastic_net: MultiTaskElasticNet = multitask_elastic_net_fit(X_r, Y_multi, 1.0, 0.5, 50, 0.01) + t1 = flow_now_ns() + multitask_elastic_net_free(probe_multitask_elastic_net) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_multitask_elastic_net: MultiTaskElasticNet = multitask_elastic_net_fit(X_r, Y_multi, 1.0, 0.5, 50, 0.01) + multitask_elastic_net_free(m_multitask_elastic_net) + } + t1 = flow_now_ns() + let fitted_multitask_elastic_net: MultiTaskElasticNet = multitask_elastic_net_fit(X_r, Y_multi, 1.0, 0.5, 50, 0.01) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_multitask_elastic_net: Matrix = multitask_elastic_net_predict(fitted_multitask_elastic_net, X_r) + matrix_free(o_multitask_elastic_net) + } + t3 = flow_now_ns() + multitask_elastic_net_free(fitted_multitask_elastic_net) + printf("ESTIMATOR|multitask_elastic_net|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- multitask_lasso_cv (multioutput) ---- + t0 = flow_now_ns() + let probe_multitask_lasso_cv: MultiTaskLassoCV = multitask_lasso_cv_fit(X_r, Y_multi, 10) + t1 = flow_now_ns() + multitask_lasso_cv_free(probe_multitask_lasso_cv) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_multitask_lasso_cv: MultiTaskLassoCV = multitask_lasso_cv_fit(X_r, Y_multi, 10) + multitask_lasso_cv_free(m_multitask_lasso_cv) + } + t1 = flow_now_ns() + let fitted_multitask_lasso_cv: MultiTaskLassoCV = multitask_lasso_cv_fit(X_r, Y_multi, 10) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_multitask_lasso_cv: Matrix = multitask_lasso_cv_predict(fitted_multitask_lasso_cv, X_r) + matrix_free(o_multitask_lasso_cv) + } + t3 = flow_now_ns() + multitask_lasso_cv_free(fitted_multitask_lasso_cv) + printf("ESTIMATOR|multitask_lasso_cv|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- multitask_lasso (multioutput) ---- + t0 = flow_now_ns() + let probe_multitask_lasso: MultiTaskLasso = multitask_lasso_fit(X_r, Y_multi, 1.0, 50, 0.01) + t1 = flow_now_ns() + multitask_lasso_free(probe_multitask_lasso) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_multitask_lasso: MultiTaskLasso = multitask_lasso_fit(X_r, Y_multi, 1.0, 50, 0.01) + multitask_lasso_free(m_multitask_lasso) + } + t1 = flow_now_ns() + let fitted_multitask_lasso: MultiTaskLasso = multitask_lasso_fit(X_r, Y_multi, 1.0, 50, 0.01) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_multitask_lasso: Matrix = multitask_lasso_predict(fitted_multitask_lasso, X_r) + matrix_free(o_multitask_lasso) + } + t3 = flow_now_ns() + multitask_lasso_free(fitted_multitask_lasso) + printf("ESTIMATOR|multitask_lasso|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- nca (regression) ---- + t0 = flow_now_ns() + let probe_nca: NeighborhoodComponentsAnalysis = nca_fit(X_r, y_r, 2, 50, 0.01) + t1 = flow_now_ns() + nca_free(probe_nca) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_nca: NeighborhoodComponentsAnalysis = nca_fit(X_r, y_r, 2, 50, 0.01) + nca_free(m_nca) + } + t1 = flow_now_ns() + let fitted_nca: NeighborhoodComponentsAnalysis = nca_fit(X_r, y_r, 2, 50, 0.01) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_nca: Matrix = nca_transform(fitted_nca, X_r) + matrix_free(o_nca) + } + t3 = flow_now_ns() + nca_free(fitted_nca) + printf("ESTIMATOR|nca|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + for i in 0 to n_c { array_free_f32(Y_label_rows[i]) } + free(Y_label_rows as ptr) + matrix_free(Y_labels) + for i in 0 to n_r { array_free_f32(Y_rows[i]) } + free(Y_rows as ptr) + matrix_free(Y_multi) + free(yi_c as ptr) + free(yi_r as ptr) + return 0 +} diff --git a/benchmarks/generated/bench_estimators_05.flow b/benchmarks/generated/bench_estimators_05.flow new file mode 100644 index 0000000..37b3de4 --- /dev/null +++ b/benchmarks/generated/bench_estimators_05.flow @@ -0,0 +1,549 @@ +# Generated by benchmarks/generate_estimator_bench.py. Do not edit. +# +# One timing block per Flow estimator that the registry marks runnable. +# Regenerate with: +# python benchmarks/estimator_coverage.py +# python benchmarks/generate_estimator_bench.py + +import "lib/scikit/scikit.flow" + +extern { + function printf(fmt: string, ...) -> i32 + function flow_now_ns() -> i64 + function malloc(size: i64) -> ptr + function free(p: ptr) -> void + function fflush(stream: ptr) -> i32 +} + +function ms_between(start: i64, finish: i64) -> f32 { + return ((((finish - start) as f64) / 1000000.0) as f32) +} + +function main() -> i32 { + let iris: Dataset = load_iris() + let diabetes: Dataset = load_diabetes() + let X_c: Matrix = iris.X + let y_c: ptr = iris.y + let n_c: i32 = X_c.rows + let f_c: i32 = X_c.cols + let X_r: Matrix = diabetes.X + let y_r: ptr = diabetes.y + let n_r: i32 = X_r.rows + let f_r: i32 = X_r.cols + + let yi_c: ptr = malloc((n_c as i64) * 4) as ptr + for i in 0 to n_c { yi_c[i] = y_c[i] as i32 } + let yi_r: ptr = malloc((n_r as i64) * 4) as ptr + for i in 0 to n_r { yi_r[i] = y_r[i] as i32 } + + # A two-column target for the cross-decomposition estimators. + let Y_multi: Matrix = matrix_new(n_r, 2) + for i in 0 to n_r { + matrix_set(Y_multi, i, 0, y_r[i]) + matrix_set(Y_multi, i, 1, y_r[i] * 0.5) + } + + # Label-valued targets for the multi-output classifiers. + let Y_labels: Matrix = matrix_new(n_c, 2) + let Y_label_rows: ptr > = malloc((n_c as i64) * 8) as ptr > + for i in 0 to n_c { + let a: f32 = y_c[i] + let b: f32 = ((((y_c[i] as i32) + 1) % 3) as f32) + matrix_set(Y_labels, i, 0, a) + matrix_set(Y_labels, i, 1, b) + let lrow: ptr = array_new_f32(2) + lrow[0] = a + lrow[1] = b + Y_label_rows[i] = lrow + } + + let Y_rows: ptr > = malloc((n_r as i64) * 8) as ptr > + for i in 0 to n_r { + let row: ptr = array_new_f32(2) + row[0] = y_r[i] + row[1] = y_r[i] * 0.5 + Y_rows[i] = row + } + + let mut t0: i64 = 0 + let mut t1: i64 = 0 + let mut t2: i64 = 0 + let mut t3: i64 = 0 + let mut reps: i32 = 1 + + # ---- nearest_centroid (classification) ---- + t0 = flow_now_ns() + let probe_nearest_centroid: NearestCentroid = nearest_centroid_fit(X_c, y_c, 3) + t1 = flow_now_ns() + nearest_centroid_free(probe_nearest_centroid) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_nearest_centroid: NearestCentroid = nearest_centroid_fit(X_c, y_c, 3) + nearest_centroid_free(m_nearest_centroid) + } + t1 = flow_now_ns() + let fitted_nearest_centroid: NearestCentroid = nearest_centroid_fit(X_c, y_c, 3) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_nearest_centroid: ptr = nearest_centroid_predict(fitted_nearest_centroid, X_c) + array_free_f32(o_nearest_centroid) + } + t3 = flow_now_ns() + nearest_centroid_free(fitted_nearest_centroid) + printf("ESTIMATOR|nearest_centroid|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- nearest_neighbors (unsupervised) ---- + t0 = flow_now_ns() + let probe_nearest_neighbors: NearestNeighbors = nearest_neighbors_fit(X_c, 5) + t1 = flow_now_ns() + nearest_neighbors_free(probe_nearest_neighbors) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_nearest_neighbors: NearestNeighbors = nearest_neighbors_fit(X_c, 5) + nearest_neighbors_free(m_nearest_neighbors) + } + t1 = flow_now_ns() + printf("ESTIMATOR|nearest_neighbors|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), 0.0, reps) + fflush(null) + + # ---- nmf (unsupervised) ---- + t0 = flow_now_ns() + let probe_nmf: NMF = nmf_fit(X_c, 2, 100, 0.0001, 42) + t1 = flow_now_ns() + nmf_free(probe_nmf) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_nmf: NMF = nmf_fit(X_c, 2, 100, 0.0001, 42) + nmf_free(m_nmf) + } + t1 = flow_now_ns() + let fitted_nmf: NMF = nmf_fit(X_c, 2, 100, 0.0001, 42) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_nmf: Matrix = nmf_transform(fitted_nmf, X_c) + matrix_free(o_nmf) + } + t3 = flow_now_ns() + nmf_free(fitted_nmf) + printf("ESTIMATOR|nmf|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- nu_svc (regression) ---- + t0 = flow_now_ns() + let probe_nu_svc: NuSVC = nu_svc_fit(X_r, y_r, 0.5, 0.1, 100, 1.0) + t1 = flow_now_ns() + nu_svc_free(probe_nu_svc) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_nu_svc: NuSVC = nu_svc_fit(X_r, y_r, 0.5, 0.1, 100, 1.0) + nu_svc_free(m_nu_svc) + } + t1 = flow_now_ns() + let fitted_nu_svc: NuSVC = nu_svc_fit(X_r, y_r, 0.5, 0.1, 100, 1.0) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_nu_svc: ptr = nu_svc_predict(fitted_nu_svc, X_r) + array_free_f32(o_nu_svc) + } + t3 = flow_now_ns() + nu_svc_free(fitted_nu_svc) + printf("ESTIMATOR|nu_svc|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- nu_svr (regression) ---- + t0 = flow_now_ns() + let probe_nu_svr: NuSVR = nu_svr_fit(X_r, y_r, 0.5, 1.0, 0.1, 100) + t1 = flow_now_ns() + nu_svr_free(probe_nu_svr) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_nu_svr: NuSVR = nu_svr_fit(X_r, y_r, 0.5, 1.0, 0.1, 100) + nu_svr_free(m_nu_svr) + } + t1 = flow_now_ns() + let fitted_nu_svr: NuSVR = nu_svr_fit(X_r, y_r, 0.5, 1.0, 0.1, 100) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_nu_svr: ptr = nu_svr_predict(fitted_nu_svr, X_r) + array_free_f32(o_nu_svr) + } + t3 = flow_now_ns() + nu_svr_free(fitted_nu_svr) + printf("ESTIMATOR|nu_svr|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- nystroem (unsupervised) ---- + t0 = flow_now_ns() + let probe_nystroem: Nystroem = nystroem_fit(X_c, 2, 0.1, 0, 42) + t1 = flow_now_ns() + nystroem_free(probe_nystroem) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_nystroem: Nystroem = nystroem_fit(X_c, 2, 0.1, 0, 42) + nystroem_free(m_nystroem) + } + t1 = flow_now_ns() + let fitted_nystroem: Nystroem = nystroem_fit(X_c, 2, 0.1, 0, 42) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_nystroem: Matrix = nystroem_transform(fitted_nystroem, X_c) + matrix_free(o_nystroem) + } + t3 = flow_now_ns() + nystroem_free(fitted_nystroem) + printf("ESTIMATOR|nystroem|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- oas_estimator (unsupervised) ---- + t0 = flow_now_ns() + let probe_oas_estimator: ShrunkCovariance = oas_estimator_fit(X_c) + t1 = flow_now_ns() + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_oas_estimator: ShrunkCovariance = oas_estimator_fit(X_c) + } + t1 = flow_now_ns() + printf("ESTIMATOR|oas_estimator|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), 0.0, reps) + fflush(null) + + # ---- omp_cv (regression) ---- + t0 = flow_now_ns() + let probe_omp_cv: OrthogonalMatchingPursuitCV = omp_cv_fit(X_r, y_r, 3) + t1 = flow_now_ns() + omp_cv_free(probe_omp_cv) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_omp_cv: OrthogonalMatchingPursuitCV = omp_cv_fit(X_r, y_r, 3) + omp_cv_free(m_omp_cv) + } + t1 = flow_now_ns() + let fitted_omp_cv: OrthogonalMatchingPursuitCV = omp_cv_fit(X_r, y_r, 3) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_omp_cv: ptr = omp_cv_predict(fitted_omp_cv, X_r) + array_free_f32(o_omp_cv) + } + t3 = flow_now_ns() + omp_cv_free(fitted_omp_cv) + printf("ESTIMATOR|omp_cv|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- one_class_svm (unsupervised) ---- + t0 = flow_now_ns() + let probe_one_class_svm: OneClassSVM = one_class_svm_fit(X_c, 0.5, 0.1, 100) + t1 = flow_now_ns() + one_class_svm_free(probe_one_class_svm) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_one_class_svm: OneClassSVM = one_class_svm_fit(X_c, 0.5, 0.1, 100) + one_class_svm_free(m_one_class_svm) + } + t1 = flow_now_ns() + printf("ESTIMATOR|one_class_svm|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), 0.0, reps) + fflush(null) + + # ---- one_vs_one (classification) ---- + t0 = flow_now_ns() + let probe_one_vs_one: OneVsOneClassifier = one_vs_one_fit(X_c, y_c, 3, 50, 0.01, penalty_none()) + t1 = flow_now_ns() + one_vs_one_free(probe_one_vs_one) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_one_vs_one: OneVsOneClassifier = one_vs_one_fit(X_c, y_c, 3, 50, 0.01, penalty_none()) + one_vs_one_free(m_one_vs_one) + } + t1 = flow_now_ns() + let fitted_one_vs_one: OneVsOneClassifier = one_vs_one_fit(X_c, y_c, 3, 50, 0.01, penalty_none()) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_one_vs_one: ptr = one_vs_one_predict(fitted_one_vs_one, X_c) + array_free_f32(o_one_vs_one) + } + t3 = flow_now_ns() + one_vs_one_free(fitted_one_vs_one) + printf("ESTIMATOR|one_vs_one|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- one_vs_rest (classification) ---- + t0 = flow_now_ns() + let probe_one_vs_rest: OneVsRestClassifier = one_vs_rest_fit(X_c, y_c, 3, 50, 0.01, penalty_none()) + t1 = flow_now_ns() + one_vs_rest_free(probe_one_vs_rest) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_one_vs_rest: OneVsRestClassifier = one_vs_rest_fit(X_c, y_c, 3, 50, 0.01, penalty_none()) + one_vs_rest_free(m_one_vs_rest) + } + t1 = flow_now_ns() + let fitted_one_vs_rest: OneVsRestClassifier = one_vs_rest_fit(X_c, y_c, 3, 50, 0.01, penalty_none()) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_one_vs_rest: ptr = one_vs_rest_predict(fitted_one_vs_rest, X_c) + array_free_f32(o_one_vs_rest) + } + t3 = flow_now_ns() + one_vs_rest_free(fitted_one_vs_rest) + printf("ESTIMATOR|one_vs_rest|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- onehot_encoder (unsupervised) ---- + t0 = flow_now_ns() + let probe_onehot_encoder: OneHotEncoder = onehot_encoder_fit(X_c, 0) + t1 = flow_now_ns() + onehot_encoder_free(probe_onehot_encoder) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_onehot_encoder: OneHotEncoder = onehot_encoder_fit(X_c, 0) + onehot_encoder_free(m_onehot_encoder) + } + t1 = flow_now_ns() + let fitted_onehot_encoder: OneHotEncoder = onehot_encoder_fit(X_c, 0) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_onehot_encoder: Matrix = onehot_encoder_transform(fitted_onehot_encoder, X_c) + matrix_free(o_onehot_encoder) + } + t3 = flow_now_ns() + onehot_encoder_free(fitted_onehot_encoder) + printf("ESTIMATOR|onehot_encoder|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- optics (unsupervised) ---- + t0 = flow_now_ns() + let probe_optics: OPTICS = optics_fit(X_c, 0.5, 5) + t1 = flow_now_ns() + optics_free(probe_optics) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_optics: OPTICS = optics_fit(X_c, 0.5, 5) + optics_free(m_optics) + } + t1 = flow_now_ns() + printf("ESTIMATOR|optics|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), 0.0, reps) + fflush(null) + + # ---- ordinal_encoder (unsupervised) ---- + t0 = flow_now_ns() + let probe_ordinal_encoder: OrdinalEncoder = ordinal_encoder_fit(X_c, 0) + t1 = flow_now_ns() + ordinal_encoder_free(probe_ordinal_encoder) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_ordinal_encoder: OrdinalEncoder = ordinal_encoder_fit(X_c, 0) + ordinal_encoder_free(m_ordinal_encoder) + } + t1 = flow_now_ns() + let fitted_ordinal_encoder: OrdinalEncoder = ordinal_encoder_fit(X_c, 0) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_ordinal_encoder: Matrix = ordinal_encoder_transform(fitted_ordinal_encoder, X_c) + matrix_free(o_ordinal_encoder) + } + t3 = flow_now_ns() + ordinal_encoder_free(fitted_ordinal_encoder) + printf("ESTIMATOR|ordinal_encoder|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- orthogonal_matching_pursuit (regression) ---- + t0 = flow_now_ns() + let probe_orthogonal_matching_pursuit: OrthogonalMatchingPursuit = orthogonal_matching_pursuit_fit(X_r, y_r, 3) + t1 = flow_now_ns() + orthogonal_matching_pursuit_free(probe_orthogonal_matching_pursuit) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_orthogonal_matching_pursuit: OrthogonalMatchingPursuit = orthogonal_matching_pursuit_fit(X_r, y_r, 3) + orthogonal_matching_pursuit_free(m_orthogonal_matching_pursuit) + } + t1 = flow_now_ns() + let fitted_orthogonal_matching_pursuit: OrthogonalMatchingPursuit = orthogonal_matching_pursuit_fit(X_r, y_r, 3) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_orthogonal_matching_pursuit: ptr = orthogonal_matching_pursuit_predict(fitted_orthogonal_matching_pursuit, X_r) + array_free_f32(o_orthogonal_matching_pursuit) + } + t3 = flow_now_ns() + orthogonal_matching_pursuit_free(fitted_orthogonal_matching_pursuit) + printf("ESTIMATOR|orthogonal_matching_pursuit|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- output_code (classification) ---- + t0 = flow_now_ns() + let probe_output_code: OutputCodeClassifier = output_code_fit(X_c, y_c, 3, 4, 50, 0.01, penalty_none()) + t1 = flow_now_ns() + output_code_free(probe_output_code) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_output_code: OutputCodeClassifier = output_code_fit(X_c, y_c, 3, 4, 50, 0.01, penalty_none()) + output_code_free(m_output_code) + } + t1 = flow_now_ns() + let fitted_output_code: OutputCodeClassifier = output_code_fit(X_c, y_c, 3, 4, 50, 0.01, penalty_none()) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_output_code: ptr = output_code_predict(fitted_output_code, X_c) + array_free_f32(o_output_code) + } + t3 = flow_now_ns() + output_code_free(fitted_output_code) + printf("ESTIMATOR|output_code|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- passive_aggressive_classifier (regression) ---- + t0 = flow_now_ns() + let probe_passive_aggressive_classifier: PassiveAggressiveClassifier = passive_aggressive_classifier_fit(X_r, y_r, 50, 1.0) + t1 = flow_now_ns() + passive_aggressive_classifier_free(probe_passive_aggressive_classifier) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_passive_aggressive_classifier: PassiveAggressiveClassifier = passive_aggressive_classifier_fit(X_r, y_r, 50, 1.0) + passive_aggressive_classifier_free(m_passive_aggressive_classifier) + } + t1 = flow_now_ns() + let fitted_passive_aggressive_classifier: PassiveAggressiveClassifier = passive_aggressive_classifier_fit(X_r, y_r, 50, 1.0) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_passive_aggressive_classifier: ptr = passive_aggressive_classifier_predict(fitted_passive_aggressive_classifier, X_r) + array_free_f32(o_passive_aggressive_classifier) + } + t3 = flow_now_ns() + passive_aggressive_classifier_free(fitted_passive_aggressive_classifier) + printf("ESTIMATOR|passive_aggressive_classifier|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- passive_aggressive_regressor (regression) ---- + t0 = flow_now_ns() + let probe_passive_aggressive_regressor: PassiveAggressiveRegressor = passive_aggressive_regressor_fit(X_r, y_r, 1.0, 100) + t1 = flow_now_ns() + passive_aggressive_regressor_free(probe_passive_aggressive_regressor) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_passive_aggressive_regressor: PassiveAggressiveRegressor = passive_aggressive_regressor_fit(X_r, y_r, 1.0, 100) + passive_aggressive_regressor_free(m_passive_aggressive_regressor) + } + t1 = flow_now_ns() + let fitted_passive_aggressive_regressor: PassiveAggressiveRegressor = passive_aggressive_regressor_fit(X_r, y_r, 1.0, 100) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_passive_aggressive_regressor: ptr = passive_aggressive_regressor_predict(fitted_passive_aggressive_regressor, X_r) + array_free_f32(o_passive_aggressive_regressor) + } + t3 = flow_now_ns() + passive_aggressive_regressor_free(fitted_passive_aggressive_regressor) + printf("ESTIMATOR|passive_aggressive_regressor|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- pca (unsupervised) ---- + t0 = flow_now_ns() + let probe_pca: PCA = pca_fit(X_c, 2) + t1 = flow_now_ns() + pca_free(probe_pca) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_pca: PCA = pca_fit(X_c, 2) + pca_free(m_pca) + } + t1 = flow_now_ns() + let fitted_pca: PCA = pca_fit(X_c, 2) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_pca: Matrix = pca_transform(fitted_pca, X_c) + matrix_free(o_pca) + } + t3 = flow_now_ns() + pca_free(fitted_pca) + printf("ESTIMATOR|pca|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- perceptron (regression) ---- + t0 = flow_now_ns() + let probe_perceptron: Perceptron = perceptron_fit(X_r, y_r, 50, 0.01, 42) + t1 = flow_now_ns() + perceptron_free(probe_perceptron) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_perceptron: Perceptron = perceptron_fit(X_r, y_r, 50, 0.01, 42) + perceptron_free(m_perceptron) + } + t1 = flow_now_ns() + let fitted_perceptron: Perceptron = perceptron_fit(X_r, y_r, 50, 0.01, 42) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_perceptron: ptr = perceptron_predict(fitted_perceptron, X_r) + array_free_f32(o_perceptron) + } + t3 = flow_now_ns() + perceptron_free(fitted_perceptron) + printf("ESTIMATOR|perceptron|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + for i in 0 to n_c { array_free_f32(Y_label_rows[i]) } + free(Y_label_rows as ptr) + matrix_free(Y_labels) + for i in 0 to n_r { array_free_f32(Y_rows[i]) } + free(Y_rows as ptr) + matrix_free(Y_multi) + free(yi_c as ptr) + free(yi_r as ptr) + return 0 +} diff --git a/benchmarks/generated/bench_estimators_06.flow b/benchmarks/generated/bench_estimators_06.flow new file mode 100644 index 0000000..043cadb --- /dev/null +++ b/benchmarks/generated/bench_estimators_06.flow @@ -0,0 +1,567 @@ +# Generated by benchmarks/generate_estimator_bench.py. Do not edit. +# +# One timing block per Flow estimator that the registry marks runnable. +# Regenerate with: +# python benchmarks/estimator_coverage.py +# python benchmarks/generate_estimator_bench.py + +import "lib/scikit/scikit.flow" + +extern { + function printf(fmt: string, ...) -> i32 + function flow_now_ns() -> i64 + function malloc(size: i64) -> ptr + function free(p: ptr) -> void + function fflush(stream: ptr) -> i32 +} + +function ms_between(start: i64, finish: i64) -> f32 { + return ((((finish - start) as f64) / 1000000.0) as f32) +} + +function main() -> i32 { + let iris: Dataset = load_iris() + let diabetes: Dataset = load_diabetes() + let X_c: Matrix = iris.X + let y_c: ptr = iris.y + let n_c: i32 = X_c.rows + let f_c: i32 = X_c.cols + let X_r: Matrix = diabetes.X + let y_r: ptr = diabetes.y + let n_r: i32 = X_r.rows + let f_r: i32 = X_r.cols + + let yi_c: ptr = malloc((n_c as i64) * 4) as ptr + for i in 0 to n_c { yi_c[i] = y_c[i] as i32 } + let yi_r: ptr = malloc((n_r as i64) * 4) as ptr + for i in 0 to n_r { yi_r[i] = y_r[i] as i32 } + + # A two-column target for the cross-decomposition estimators. + let Y_multi: Matrix = matrix_new(n_r, 2) + for i in 0 to n_r { + matrix_set(Y_multi, i, 0, y_r[i]) + matrix_set(Y_multi, i, 1, y_r[i] * 0.5) + } + + # Label-valued targets for the multi-output classifiers. + let Y_labels: Matrix = matrix_new(n_c, 2) + let Y_label_rows: ptr > = malloc((n_c as i64) * 8) as ptr > + for i in 0 to n_c { + let a: f32 = y_c[i] + let b: f32 = ((((y_c[i] as i32) + 1) % 3) as f32) + matrix_set(Y_labels, i, 0, a) + matrix_set(Y_labels, i, 1, b) + let lrow: ptr = array_new_f32(2) + lrow[0] = a + lrow[1] = b + Y_label_rows[i] = lrow + } + + let Y_rows: ptr > = malloc((n_r as i64) * 8) as ptr > + for i in 0 to n_r { + let row: ptr = array_new_f32(2) + row[0] = y_r[i] + row[1] = y_r[i] * 0.5 + Y_rows[i] = row + } + + let mut t0: i64 = 0 + let mut t1: i64 = 0 + let mut t2: i64 = 0 + let mut t3: i64 = 0 + let mut reps: i32 = 1 + + # ---- pls_canonical (multioutput) ---- + t0 = flow_now_ns() + let probe_pls_canonical: PLSCanonical = pls_canonical_fit(X_r, Y_multi, 2) + t1 = flow_now_ns() + pls_canonical_free(probe_pls_canonical) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_pls_canonical: PLSCanonical = pls_canonical_fit(X_r, Y_multi, 2) + pls_canonical_free(m_pls_canonical) + } + t1 = flow_now_ns() + let fitted_pls_canonical: PLSCanonical = pls_canonical_fit(X_r, Y_multi, 2) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_pls_canonical: Matrix = pls_canonical_transform(fitted_pls_canonical, X_r) + matrix_free(o_pls_canonical) + } + t3 = flow_now_ns() + pls_canonical_free(fitted_pls_canonical) + printf("ESTIMATOR|pls_canonical|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- pls (multioutput) ---- + t0 = flow_now_ns() + let probe_pls: PLSRegression = pls_fit(X_r, Y_multi, 2, 100, 0.0001) + t1 = flow_now_ns() + pls_free(probe_pls) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_pls: PLSRegression = pls_fit(X_r, Y_multi, 2, 100, 0.0001) + pls_free(m_pls) + } + t1 = flow_now_ns() + let fitted_pls: PLSRegression = pls_fit(X_r, Y_multi, 2, 100, 0.0001) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_pls: Matrix = pls_predict(fitted_pls, X_r) + matrix_free(o_pls) + } + t3 = flow_now_ns() + pls_free(fitted_pls) + printf("ESTIMATOR|pls|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- pls_svd (multioutput) ---- + t0 = flow_now_ns() + let probe_pls_svd: PLSSVD = pls_svd_fit(X_r, Y_multi, 2) + t1 = flow_now_ns() + pls_svd_free(probe_pls_svd) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_pls_svd: PLSSVD = pls_svd_fit(X_r, Y_multi, 2) + pls_svd_free(m_pls_svd) + } + t1 = flow_now_ns() + let fitted_pls_svd: PLSSVD = pls_svd_fit(X_r, Y_multi, 2) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_pls_svd: Matrix = pls_svd_transform(fitted_pls_svd, X_r) + matrix_free(o_pls_svd) + } + t3 = flow_now_ns() + pls_svd_free(fitted_pls_svd) + printf("ESTIMATOR|pls_svd|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- poisson_regressor (regression) ---- + t0 = flow_now_ns() + let probe_poisson_regressor: PoissonRegressor = poisson_regressor_fit(X_r, y_r, 1.0, 100, 0.01) + t1 = flow_now_ns() + poisson_regressor_free(probe_poisson_regressor) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_poisson_regressor: PoissonRegressor = poisson_regressor_fit(X_r, y_r, 1.0, 100, 0.01) + poisson_regressor_free(m_poisson_regressor) + } + t1 = flow_now_ns() + let fitted_poisson_regressor: PoissonRegressor = poisson_regressor_fit(X_r, y_r, 1.0, 100, 0.01) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_poisson_regressor: ptr = poisson_regressor_predict(fitted_poisson_regressor, X_r) + array_free_f32(o_poisson_regressor) + } + t3 = flow_now_ns() + poisson_regressor_free(fitted_poisson_regressor) + printf("ESTIMATOR|poisson_regressor|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- power_transformer (unsupervised) ---- + t0 = flow_now_ns() + let probe_power_transformer: PowerTransformer = power_transformer_fit(X_c, 0) + t1 = flow_now_ns() + power_transformer_free(probe_power_transformer) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_power_transformer: PowerTransformer = power_transformer_fit(X_c, 0) + power_transformer_free(m_power_transformer) + } + t1 = flow_now_ns() + let fitted_power_transformer: PowerTransformer = power_transformer_fit(X_c, 0) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_power_transformer: Matrix = power_transformer_transform(fitted_power_transformer, X_c) + matrix_free(o_power_transformer) + } + t3 = flow_now_ns() + power_transformer_free(fitted_power_transformer) + printf("ESTIMATOR|power_transformer|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- qda (classification) ---- + t0 = flow_now_ns() + let probe_qda: QuadraticDiscriminantAnalysis = qda_fit(X_c, y_c, 3) + t1 = flow_now_ns() + qda_free(probe_qda) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_qda: QuadraticDiscriminantAnalysis = qda_fit(X_c, y_c, 3) + qda_free(m_qda) + } + t1 = flow_now_ns() + let fitted_qda: QuadraticDiscriminantAnalysis = qda_fit(X_c, y_c, 3) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_qda: ptr = qda_predict(fitted_qda, X_c) + array_free_f32(o_qda) + } + t3 = flow_now_ns() + qda_free(fitted_qda) + printf("ESTIMATOR|qda|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- quantile_regressor (regression) ---- + t0 = flow_now_ns() + let probe_quantile_regressor: QuantileRegressor = quantile_regressor_fit(X_r, y_r, 0.5, 1.0, 50, 0.01) + t1 = flow_now_ns() + quantile_regressor_free(probe_quantile_regressor) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_quantile_regressor: QuantileRegressor = quantile_regressor_fit(X_r, y_r, 0.5, 1.0, 50, 0.01) + quantile_regressor_free(m_quantile_regressor) + } + t1 = flow_now_ns() + let fitted_quantile_regressor: QuantileRegressor = quantile_regressor_fit(X_r, y_r, 0.5, 1.0, 50, 0.01) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_quantile_regressor: ptr = quantile_regressor_predict(fitted_quantile_regressor, X_r) + array_free_f32(o_quantile_regressor) + } + t3 = flow_now_ns() + quantile_regressor_free(fitted_quantile_regressor) + printf("ESTIMATOR|quantile_regressor|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- quantile_transformer (unsupervised) ---- + t0 = flow_now_ns() + let probe_quantile_transformer: QuantileTransformer = quantile_transformer_fit(X_c, 10, 0) + t1 = flow_now_ns() + quantile_transformer_free(probe_quantile_transformer) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_quantile_transformer: QuantileTransformer = quantile_transformer_fit(X_c, 10, 0) + quantile_transformer_free(m_quantile_transformer) + } + t1 = flow_now_ns() + let fitted_quantile_transformer: QuantileTransformer = quantile_transformer_fit(X_c, 10, 0) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_quantile_transformer: Matrix = quantile_transformer_transform(fitted_quantile_transformer, X_c) + matrix_free(o_quantile_transformer) + } + t3 = flow_now_ns() + quantile_transformer_free(fitted_quantile_transformer) + printf("ESTIMATOR|quantile_transformer|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- radius_neighbors_classifier (classification) ---- + t0 = flow_now_ns() + let probe_radius_neighbors_classifier: RadiusNeighborsClassifier = radius_neighbors_classifier_fit(X_c, y_c, 1.0, 3, 0.0) + t1 = flow_now_ns() + radius_neighbors_classifier_free(probe_radius_neighbors_classifier) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_radius_neighbors_classifier: RadiusNeighborsClassifier = radius_neighbors_classifier_fit(X_c, y_c, 1.0, 3, 0.0) + radius_neighbors_classifier_free(m_radius_neighbors_classifier) + } + t1 = flow_now_ns() + let fitted_radius_neighbors_classifier: RadiusNeighborsClassifier = radius_neighbors_classifier_fit(X_c, y_c, 1.0, 3, 0.0) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_radius_neighbors_classifier: ptr = radius_neighbors_classifier_predict(fitted_radius_neighbors_classifier, X_c) + array_free_f32(o_radius_neighbors_classifier) + } + t3 = flow_now_ns() + radius_neighbors_classifier_free(fitted_radius_neighbors_classifier) + printf("ESTIMATOR|radius_neighbors_classifier|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- radius_neighbors_regressor (regression) ---- + t0 = flow_now_ns() + let probe_radius_neighbors_regressor: RadiusNeighborsRegressor = radius_neighbors_regressor_fit(X_r, y_r, 1.0) + t1 = flow_now_ns() + radius_neighbors_regressor_free(probe_radius_neighbors_regressor) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_radius_neighbors_regressor: RadiusNeighborsRegressor = radius_neighbors_regressor_fit(X_r, y_r, 1.0) + radius_neighbors_regressor_free(m_radius_neighbors_regressor) + } + t1 = flow_now_ns() + let fitted_radius_neighbors_regressor: RadiusNeighborsRegressor = radius_neighbors_regressor_fit(X_r, y_r, 1.0) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_radius_neighbors_regressor: ptr = radius_neighbors_regressor_predict(fitted_radius_neighbors_regressor, X_r) + array_free_f32(o_radius_neighbors_regressor) + } + t3 = flow_now_ns() + radius_neighbors_regressor_free(fitted_radius_neighbors_regressor) + printf("ESTIMATOR|radius_neighbors_regressor|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- radius_neighbors_transformer (unsupervised) ---- + t0 = flow_now_ns() + let probe_radius_neighbors_transformer: RadiusNeighborsTransformer = radius_neighbors_transformer_fit(X_c, 1.0, 0) + t1 = flow_now_ns() + radius_neighbors_transformer_free(probe_radius_neighbors_transformer) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_radius_neighbors_transformer: RadiusNeighborsTransformer = radius_neighbors_transformer_fit(X_c, 1.0, 0) + radius_neighbors_transformer_free(m_radius_neighbors_transformer) + } + t1 = flow_now_ns() + let fitted_radius_neighbors_transformer: RadiusNeighborsTransformer = radius_neighbors_transformer_fit(X_c, 1.0, 0) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_radius_neighbors_transformer: Matrix = radius_neighbors_transformer_transform(fitted_radius_neighbors_transformer, X_c) + matrix_free(o_radius_neighbors_transformer) + } + t3 = flow_now_ns() + radius_neighbors_transformer_free(fitted_radius_neighbors_transformer) + printf("ESTIMATOR|radius_neighbors_transformer|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- random_forest_classifier (classification) ---- + t0 = flow_now_ns() + let probe_random_forest_classifier: RandomForestClassifier = random_forest_classifier_fit(X_c, y_c, 3, 10, 5, 42) + t1 = flow_now_ns() + random_forest_classifier_free(probe_random_forest_classifier) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_random_forest_classifier: RandomForestClassifier = random_forest_classifier_fit(X_c, y_c, 3, 10, 5, 42) + random_forest_classifier_free(m_random_forest_classifier) + } + t1 = flow_now_ns() + let fitted_random_forest_classifier: RandomForestClassifier = random_forest_classifier_fit(X_c, y_c, 3, 10, 5, 42) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_random_forest_classifier: ptr = random_forest_classifier_predict(fitted_random_forest_classifier, X_c) + array_free_f32(o_random_forest_classifier) + } + t3 = flow_now_ns() + random_forest_classifier_free(fitted_random_forest_classifier) + printf("ESTIMATOR|random_forest_classifier|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- random_forest_regressor (regression) ---- + t0 = flow_now_ns() + let probe_random_forest_regressor: RandomForestRegressor = random_forest_regressor_fit(X_r, y_r, 10, 5, 42) + t1 = flow_now_ns() + random_forest_regressor_free(probe_random_forest_regressor) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_random_forest_regressor: RandomForestRegressor = random_forest_regressor_fit(X_r, y_r, 10, 5, 42) + random_forest_regressor_free(m_random_forest_regressor) + } + t1 = flow_now_ns() + let fitted_random_forest_regressor: RandomForestRegressor = random_forest_regressor_fit(X_r, y_r, 10, 5, 42) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_random_forest_regressor: ptr = random_forest_regressor_predict(fitted_random_forest_regressor, X_r) + array_free_f32(o_random_forest_regressor) + } + t3 = flow_now_ns() + random_forest_regressor_free(fitted_random_forest_regressor) + printf("ESTIMATOR|random_forest_regressor|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- random_trees_embedding (unsupervised) ---- + t0 = flow_now_ns() + let probe_random_trees_embedding: RandomTreesEmbedding = random_trees_embedding_fit(X_c, 10, 5, 42) + t1 = flow_now_ns() + random_trees_embedding_free(probe_random_trees_embedding) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_random_trees_embedding: RandomTreesEmbedding = random_trees_embedding_fit(X_c, 10, 5, 42) + random_trees_embedding_free(m_random_trees_embedding) + } + t1 = flow_now_ns() + let fitted_random_trees_embedding: RandomTreesEmbedding = random_trees_embedding_fit(X_c, 10, 5, 42) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_random_trees_embedding: Matrix = random_trees_embedding_transform(fitted_random_trees_embedding, X_c) + matrix_free(o_random_trees_embedding) + } + t3 = flow_now_ns() + random_trees_embedding_free(fitted_random_trees_embedding) + printf("ESTIMATOR|random_trees_embedding|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- ransac_regressor (regression) ---- + t0 = flow_now_ns() + let probe_ransac_regressor: RANSACRegressor = ransac_regressor_fit(X_r, y_r, 5, 10, 1.0, 42) + t1 = flow_now_ns() + ransac_regressor_free(probe_ransac_regressor) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_ransac_regressor: RANSACRegressor = ransac_regressor_fit(X_r, y_r, 5, 10, 1.0, 42) + ransac_regressor_free(m_ransac_regressor) + } + t1 = flow_now_ns() + let fitted_ransac_regressor: RANSACRegressor = ransac_regressor_fit(X_r, y_r, 5, 10, 1.0, 42) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_ransac_regressor: ptr = ransac_regressor_predict(fitted_ransac_regressor, X_r) + array_free_f32(o_ransac_regressor) + } + t3 = flow_now_ns() + ransac_regressor_free(fitted_ransac_regressor) + printf("ESTIMATOR|ransac_regressor|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- regressor_chain (multioutput_class) ---- + t0 = flow_now_ns() + let probe_regressor_chain: RegressorChain = regressor_chain_fit(X_c, Y_label_rows, n_c, f_c, 2) + t1 = flow_now_ns() + regressor_chain_free(probe_regressor_chain) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_regressor_chain: RegressorChain = regressor_chain_fit(X_c, Y_label_rows, n_c, f_c, 2) + regressor_chain_free(m_regressor_chain) + } + t1 = flow_now_ns() + printf("ESTIMATOR|regressor_chain|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), 0.0, reps) + fflush(null) + + # ---- rfe (regression) ---- + t0 = flow_now_ns() + let probe_rfe: RFE = rfe_fit(X_r, y_r, 2, null, n_r, f_r) + t1 = flow_now_ns() + rfe_free(probe_rfe) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_rfe: RFE = rfe_fit(X_r, y_r, 2, null, n_r, f_r) + rfe_free(m_rfe) + } + t1 = flow_now_ns() + let fitted_rfe: RFE = rfe_fit(X_r, y_r, 2, null, n_r, f_r) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_rfe: Matrix = rfe_transform(fitted_rfe, X_r) + matrix_free(o_rfe) + } + t3 = flow_now_ns() + rfe_free(fitted_rfe) + printf("ESTIMATOR|rfe|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- rfecv (regression) ---- + t0 = flow_now_ns() + let probe_rfecv: RFECV = rfecv_fit(X_r, y_r, n_r, f_r, 3, null) + t1 = flow_now_ns() + rfecv_free(probe_rfecv) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_rfecv: RFECV = rfecv_fit(X_r, y_r, n_r, f_r, 3, null) + rfecv_free(m_rfecv) + } + t1 = flow_now_ns() + printf("ESTIMATOR|rfecv|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), 0.0, reps) + fflush(null) + + # ---- ridge_classifier_cv (classification) ---- + t0 = flow_now_ns() + let probe_ridge_classifier_cv: RidgeClassifierCV = ridge_classifier_cv_fit(X_c, y_c, 3, 10) + t1 = flow_now_ns() + ridge_classifier_cv_free(probe_ridge_classifier_cv) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_ridge_classifier_cv: RidgeClassifierCV = ridge_classifier_cv_fit(X_c, y_c, 3, 10) + ridge_classifier_cv_free(m_ridge_classifier_cv) + } + t1 = flow_now_ns() + let fitted_ridge_classifier_cv: RidgeClassifierCV = ridge_classifier_cv_fit(X_c, y_c, 3, 10) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_ridge_classifier_cv: ptr = ridge_classifier_cv_predict(fitted_ridge_classifier_cv, X_c) + array_free_f32(o_ridge_classifier_cv) + } + t3 = flow_now_ns() + ridge_classifier_cv_free(fitted_ridge_classifier_cv) + printf("ESTIMATOR|ridge_classifier_cv|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- ridge_classifier (regression) ---- + t0 = flow_now_ns() + let probe_ridge_classifier: RidgeClassifier = ridge_classifier_fit(X_r, y_r, 1.0, 50, 0.01) + t1 = flow_now_ns() + ridge_classifier_free(probe_ridge_classifier) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_ridge_classifier: RidgeClassifier = ridge_classifier_fit(X_r, y_r, 1.0, 50, 0.01) + ridge_classifier_free(m_ridge_classifier) + } + t1 = flow_now_ns() + let fitted_ridge_classifier: RidgeClassifier = ridge_classifier_fit(X_r, y_r, 1.0, 50, 0.01) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_ridge_classifier: ptr = ridge_classifier_predict(fitted_ridge_classifier, X_r) + array_free_f32(o_ridge_classifier) + } + t3 = flow_now_ns() + ridge_classifier_free(fitted_ridge_classifier) + printf("ESTIMATOR|ridge_classifier|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + for i in 0 to n_c { array_free_f32(Y_label_rows[i]) } + free(Y_label_rows as ptr) + matrix_free(Y_labels) + for i in 0 to n_r { array_free_f32(Y_rows[i]) } + free(Y_rows as ptr) + matrix_free(Y_multi) + free(yi_c as ptr) + free(yi_r as ptr) + return 0 +} diff --git a/benchmarks/generated/bench_estimators_07.flow b/benchmarks/generated/bench_estimators_07.flow new file mode 100644 index 0000000..17e83a2 --- /dev/null +++ b/benchmarks/generated/bench_estimators_07.flow @@ -0,0 +1,543 @@ +# Generated by benchmarks/generate_estimator_bench.py. Do not edit. +# +# One timing block per Flow estimator that the registry marks runnable. +# Regenerate with: +# python benchmarks/estimator_coverage.py +# python benchmarks/generate_estimator_bench.py + +import "lib/scikit/scikit.flow" + +extern { + function printf(fmt: string, ...) -> i32 + function flow_now_ns() -> i64 + function malloc(size: i64) -> ptr + function free(p: ptr) -> void + function fflush(stream: ptr) -> i32 +} + +function ms_between(start: i64, finish: i64) -> f32 { + return ((((finish - start) as f64) / 1000000.0) as f32) +} + +function main() -> i32 { + let iris: Dataset = load_iris() + let diabetes: Dataset = load_diabetes() + let X_c: Matrix = iris.X + let y_c: ptr = iris.y + let n_c: i32 = X_c.rows + let f_c: i32 = X_c.cols + let X_r: Matrix = diabetes.X + let y_r: ptr = diabetes.y + let n_r: i32 = X_r.rows + let f_r: i32 = X_r.cols + + let yi_c: ptr = malloc((n_c as i64) * 4) as ptr + for i in 0 to n_c { yi_c[i] = y_c[i] as i32 } + let yi_r: ptr = malloc((n_r as i64) * 4) as ptr + for i in 0 to n_r { yi_r[i] = y_r[i] as i32 } + + # A two-column target for the cross-decomposition estimators. + let Y_multi: Matrix = matrix_new(n_r, 2) + for i in 0 to n_r { + matrix_set(Y_multi, i, 0, y_r[i]) + matrix_set(Y_multi, i, 1, y_r[i] * 0.5) + } + + # Label-valued targets for the multi-output classifiers. + let Y_labels: Matrix = matrix_new(n_c, 2) + let Y_label_rows: ptr > = malloc((n_c as i64) * 8) as ptr > + for i in 0 to n_c { + let a: f32 = y_c[i] + let b: f32 = ((((y_c[i] as i32) + 1) % 3) as f32) + matrix_set(Y_labels, i, 0, a) + matrix_set(Y_labels, i, 1, b) + let lrow: ptr = array_new_f32(2) + lrow[0] = a + lrow[1] = b + Y_label_rows[i] = lrow + } + + let Y_rows: ptr > = malloc((n_r as i64) * 8) as ptr > + for i in 0 to n_r { + let row: ptr = array_new_f32(2) + row[0] = y_r[i] + row[1] = y_r[i] * 0.5 + Y_rows[i] = row + } + + let mut t0: i64 = 0 + let mut t1: i64 = 0 + let mut t2: i64 = 0 + let mut t3: i64 = 0 + let mut reps: i32 = 1 + + # ---- ridge_cv (regression) ---- + t0 = flow_now_ns() + let probe_ridge_cv: RidgeCV = ridge_cv_fit(X_r, y_r, 10) + t1 = flow_now_ns() + ridge_cv_free(probe_ridge_cv) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_ridge_cv: RidgeCV = ridge_cv_fit(X_r, y_r, 10) + ridge_cv_free(m_ridge_cv) + } + t1 = flow_now_ns() + let fitted_ridge_cv: RidgeCV = ridge_cv_fit(X_r, y_r, 10) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_ridge_cv: ptr = ridge_cv_predict(fitted_ridge_cv, X_r) + array_free_f32(o_ridge_cv) + } + t3 = flow_now_ns() + ridge_cv_free(fitted_ridge_cv) + printf("ESTIMATOR|ridge_cv|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- ridge (regression) ---- + t0 = flow_now_ns() + let probe_ridge: Ridge = ridge_fit(X_r, y_r, 1.0, 50, 0.01) + t1 = flow_now_ns() + ridge_free(probe_ridge) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_ridge: Ridge = ridge_fit(X_r, y_r, 1.0, 50, 0.01) + ridge_free(m_ridge) + } + t1 = flow_now_ns() + let fitted_ridge: Ridge = ridge_fit(X_r, y_r, 1.0, 50, 0.01) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_ridge: ptr = ridge_predict(fitted_ridge, X_r) + array_free_f32(o_ridge) + } + t3 = flow_now_ns() + ridge_free(fitted_ridge) + printf("ESTIMATOR|ridge|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- robust_scaler (unsupervised) ---- + t0 = flow_now_ns() + let probe_robust_scaler: RobustScaler = robust_scaler_fit(X_c) + t1 = flow_now_ns() + robust_scaler_free(probe_robust_scaler) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_robust_scaler: RobustScaler = robust_scaler_fit(X_c) + robust_scaler_free(m_robust_scaler) + } + t1 = flow_now_ns() + let fitted_robust_scaler: RobustScaler = robust_scaler_fit(X_c) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_robust_scaler: Matrix = robust_scaler_transform(fitted_robust_scaler, X_c) + matrix_free(o_robust_scaler) + } + t3 = flow_now_ns() + robust_scaler_free(fitted_robust_scaler) + printf("ESTIMATOR|robust_scaler|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- select_fdr (regression) ---- + t0 = flow_now_ns() + let probe_select_fdr: SelectFdr = select_fdr_fit(X_r, y_r, 1.0) + t1 = flow_now_ns() + select_fdr_free(probe_select_fdr) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_select_fdr: SelectFdr = select_fdr_fit(X_r, y_r, 1.0) + select_fdr_free(m_select_fdr) + } + t1 = flow_now_ns() + let fitted_select_fdr: SelectFdr = select_fdr_fit(X_r, y_r, 1.0) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_select_fdr: Matrix = select_fdr_transform(fitted_select_fdr, X_r) + matrix_free(o_select_fdr) + } + t3 = flow_now_ns() + select_fdr_free(fitted_select_fdr) + printf("ESTIMATOR|select_fdr|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- select_fpr (regression) ---- + t0 = flow_now_ns() + let probe_select_fpr: SelectFpr = select_fpr_fit(X_r, y_r, 1.0) + t1 = flow_now_ns() + select_fpr_free(probe_select_fpr) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_select_fpr: SelectFpr = select_fpr_fit(X_r, y_r, 1.0) + select_fpr_free(m_select_fpr) + } + t1 = flow_now_ns() + let fitted_select_fpr: SelectFpr = select_fpr_fit(X_r, y_r, 1.0) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_select_fpr: Matrix = select_fpr_transform(fitted_select_fpr, X_r) + matrix_free(o_select_fpr) + } + t3 = flow_now_ns() + select_fpr_free(fitted_select_fpr) + printf("ESTIMATOR|select_fpr|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- select_fwe (regression) ---- + t0 = flow_now_ns() + let probe_select_fwe: SelectFwe = select_fwe_fit(X_r, y_r, 1.0) + t1 = flow_now_ns() + select_fwe_free(probe_select_fwe) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_select_fwe: SelectFwe = select_fwe_fit(X_r, y_r, 1.0) + select_fwe_free(m_select_fwe) + } + t1 = flow_now_ns() + let fitted_select_fwe: SelectFwe = select_fwe_fit(X_r, y_r, 1.0) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_select_fwe: Matrix = select_fwe_transform(fitted_select_fwe, X_r) + matrix_free(o_select_fwe) + } + t3 = flow_now_ns() + select_fwe_free(fitted_select_fwe) + printf("ESTIMATOR|select_fwe|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- select_k_best (regression) ---- + t0 = flow_now_ns() + let probe_select_k_best: SelectKBest = select_k_best_fit(X_r, y_r, 3, 0) + t1 = flow_now_ns() + select_k_best_free(probe_select_k_best) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_select_k_best: SelectKBest = select_k_best_fit(X_r, y_r, 3, 0) + select_k_best_free(m_select_k_best) + } + t1 = flow_now_ns() + let fitted_select_k_best: SelectKBest = select_k_best_fit(X_r, y_r, 3, 0) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_select_k_best: Matrix = select_k_best_transform(fitted_select_k_best, X_r) + matrix_free(o_select_k_best) + } + t3 = flow_now_ns() + select_k_best_free(fitted_select_k_best) + printf("ESTIMATOR|select_k_best|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- select_percentile (regression) ---- + t0 = flow_now_ns() + let probe_select_percentile: SelectPercentile = select_percentile_fit(X_r, y_r, 50) + t1 = flow_now_ns() + select_percentile_free(probe_select_percentile) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_select_percentile: SelectPercentile = select_percentile_fit(X_r, y_r, 50) + select_percentile_free(m_select_percentile) + } + t1 = flow_now_ns() + let fitted_select_percentile: SelectPercentile = select_percentile_fit(X_r, y_r, 50) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_select_percentile: Matrix = select_percentile_transform(fitted_select_percentile, X_r) + matrix_free(o_select_percentile) + } + t3 = flow_now_ns() + select_percentile_free(fitted_select_percentile) + printf("ESTIMATOR|select_percentile|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- sequential_feature_selector (regression) ---- + t0 = flow_now_ns() + let probe_sequential_feature_selector: SequentialFeatureSelector = sequential_feature_selector_fit(X_r, y_r, 2, 0, n_r, f_r) + t1 = flow_now_ns() + sequential_feature_selector_free(probe_sequential_feature_selector) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_sequential_feature_selector: SequentialFeatureSelector = sequential_feature_selector_fit(X_r, y_r, 2, 0, n_r, f_r) + sequential_feature_selector_free(m_sequential_feature_selector) + } + t1 = flow_now_ns() + let fitted_sequential_feature_selector: SequentialFeatureSelector = sequential_feature_selector_fit(X_r, y_r, 2, 0, n_r, f_r) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_sequential_feature_selector: Matrix = sequential_feature_selector_transform(fitted_sequential_feature_selector, X_r) + matrix_free(o_sequential_feature_selector) + } + t3 = flow_now_ns() + sequential_feature_selector_free(fitted_sequential_feature_selector) + printf("ESTIMATOR|sequential_feature_selector|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- sgd_classifier (classification) ---- + t0 = flow_now_ns() + let probe_sgd_classifier: SGDClassifier = sgd_classifier_fit(X_c, y_c, 3, 0, 1.0, 50, 0.01) + t1 = flow_now_ns() + sgd_classifier_free(probe_sgd_classifier) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_sgd_classifier: SGDClassifier = sgd_classifier_fit(X_c, y_c, 3, 0, 1.0, 50, 0.01) + sgd_classifier_free(m_sgd_classifier) + } + t1 = flow_now_ns() + let fitted_sgd_classifier: SGDClassifier = sgd_classifier_fit(X_c, y_c, 3, 0, 1.0, 50, 0.01) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_sgd_classifier: ptr = sgd_classifier_predict(fitted_sgd_classifier, X_c) + array_free_f32(o_sgd_classifier) + } + t3 = flow_now_ns() + sgd_classifier_free(fitted_sgd_classifier) + printf("ESTIMATOR|sgd_classifier|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- sgd_one_class_svm (unsupervised) ---- + t0 = flow_now_ns() + let probe_sgd_one_class_svm: SGDOneClassSVM = sgd_one_class_svm_fit(X_c, 0.5, 50, 0.01) + t1 = flow_now_ns() + sgd_one_class_svm_free(probe_sgd_one_class_svm) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_sgd_one_class_svm: SGDOneClassSVM = sgd_one_class_svm_fit(X_c, 0.5, 50, 0.01) + sgd_one_class_svm_free(m_sgd_one_class_svm) + } + t1 = flow_now_ns() + let fitted_sgd_one_class_svm: SGDOneClassSVM = sgd_one_class_svm_fit(X_c, 0.5, 50, 0.01) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_sgd_one_class_svm: ptr = sgd_one_class_svm_predict(fitted_sgd_one_class_svm, X_c) + array_free_f32(o_sgd_one_class_svm) + } + t3 = flow_now_ns() + sgd_one_class_svm_free(fitted_sgd_one_class_svm) + printf("ESTIMATOR|sgd_one_class_svm|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- sgd_regressor (regression) ---- + t0 = flow_now_ns() + let probe_sgd_regressor: SGDRegressor = sgd_regressor_fit(X_r, y_r, 0, 1.0, 0.1, 50, 0.01) + t1 = flow_now_ns() + sgd_regressor_free(probe_sgd_regressor) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_sgd_regressor: SGDRegressor = sgd_regressor_fit(X_r, y_r, 0, 1.0, 0.1, 50, 0.01) + sgd_regressor_free(m_sgd_regressor) + } + t1 = flow_now_ns() + let fitted_sgd_regressor: SGDRegressor = sgd_regressor_fit(X_r, y_r, 0, 1.0, 0.1, 50, 0.01) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_sgd_regressor: ptr = sgd_regressor_predict(fitted_sgd_regressor, X_r) + array_free_f32(o_sgd_regressor) + } + t3 = flow_now_ns() + sgd_regressor_free(fitted_sgd_regressor) + printf("ESTIMATOR|sgd_regressor|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- shrunk_covariance (unsupervised) ---- + t0 = flow_now_ns() + let probe_shrunk_covariance: ShrunkCovariance = shrunk_covariance_fit(X_c, 0.1) + t1 = flow_now_ns() + shrunk_covariance_free(probe_shrunk_covariance) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_shrunk_covariance: ShrunkCovariance = shrunk_covariance_fit(X_c, 0.1) + shrunk_covariance_free(m_shrunk_covariance) + } + t1 = flow_now_ns() + printf("ESTIMATOR|shrunk_covariance|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), 0.0, reps) + fflush(null) + + # ---- simple_imputer (unsupervised) ---- + t0 = flow_now_ns() + let probe_simple_imputer: SimpleImputer = simple_imputer_fit(X_c, 0, 1.0) + t1 = flow_now_ns() + simple_imputer_free(probe_simple_imputer) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_simple_imputer: SimpleImputer = simple_imputer_fit(X_c, 0, 1.0) + simple_imputer_free(m_simple_imputer) + } + t1 = flow_now_ns() + let fitted_simple_imputer: SimpleImputer = simple_imputer_fit(X_c, 0, 1.0) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_simple_imputer: Matrix = simple_imputer_transform(fitted_simple_imputer, X_c) + matrix_free(o_simple_imputer) + } + t3 = flow_now_ns() + simple_imputer_free(fitted_simple_imputer) + printf("ESTIMATOR|simple_imputer|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- sparse_coder (unsupervised) ---- + t0 = flow_now_ns() + let probe_sparse_coder: SparseCoder = sparse_coder_fit(X_c, 3) + t1 = flow_now_ns() + sparse_coder_free(probe_sparse_coder) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_sparse_coder: SparseCoder = sparse_coder_fit(X_c, 3) + sparse_coder_free(m_sparse_coder) + } + t1 = flow_now_ns() + let fitted_sparse_coder: SparseCoder = sparse_coder_fit(X_c, 3) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_sparse_coder: Matrix = sparse_coder_transform(fitted_sparse_coder, X_c) + matrix_free(o_sparse_coder) + } + t3 = flow_now_ns() + sparse_coder_free(fitted_sparse_coder) + printf("ESTIMATOR|sparse_coder|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- sparse_pca (unsupervised) ---- + t0 = flow_now_ns() + let probe_sparse_pca: SparsePCA = sparse_pca_fit(X_c, 2, 1.0, 100, 0.0001, 42) + t1 = flow_now_ns() + sparse_pca_free(probe_sparse_pca) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_sparse_pca: SparsePCA = sparse_pca_fit(X_c, 2, 1.0, 100, 0.0001, 42) + sparse_pca_free(m_sparse_pca) + } + t1 = flow_now_ns() + let fitted_sparse_pca: SparsePCA = sparse_pca_fit(X_c, 2, 1.0, 100, 0.0001, 42) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_sparse_pca: Matrix = sparse_pca_transform(fitted_sparse_pca, X_c) + matrix_free(o_sparse_pca) + } + t3 = flow_now_ns() + sparse_pca_free(fitted_sparse_pca) + printf("ESTIMATOR|sparse_pca|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- spectral_biclustering (unsupervised) ---- + t0 = flow_now_ns() + let probe_spectral_biclustering: SpectralBiclustering = spectral_biclustering_fit(X_c, 2, 2) + t1 = flow_now_ns() + spectral_biclustering_free(probe_spectral_biclustering) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_spectral_biclustering: SpectralBiclustering = spectral_biclustering_fit(X_c, 2, 2) + spectral_biclustering_free(m_spectral_biclustering) + } + t1 = flow_now_ns() + printf("ESTIMATOR|spectral_biclustering|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), 0.0, reps) + fflush(null) + + # ---- spectral_clustering (unsupervised) ---- + t0 = flow_now_ns() + let probe_spectral_clustering: SpectralClustering = spectral_clustering_fit(X_c, 3, 0.1, 42) + t1 = flow_now_ns() + spectral_clustering_free(probe_spectral_clustering) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_spectral_clustering: SpectralClustering = spectral_clustering_fit(X_c, 3, 0.1, 42) + spectral_clustering_free(m_spectral_clustering) + } + t1 = flow_now_ns() + printf("ESTIMATOR|spectral_clustering|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), 0.0, reps) + fflush(null) + + # ---- spectral_coclustering (unsupervised) ---- + t0 = flow_now_ns() + let probe_spectral_coclustering: SpectralCoclustering = spectral_coclustering_fit(X_c, 3) + t1 = flow_now_ns() + spectral_coclustering_free(probe_spectral_coclustering) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_spectral_coclustering: SpectralCoclustering = spectral_coclustering_fit(X_c, 3) + spectral_coclustering_free(m_spectral_coclustering) + } + t1 = flow_now_ns() + printf("ESTIMATOR|spectral_coclustering|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), 0.0, reps) + fflush(null) + + # ---- spectral_embedding (unsupervised) ---- + t0 = flow_now_ns() + let probe_spectral_embedding: SpectralEmbedding = spectral_embedding_fit(X_c, 2, 0.1) + t1 = flow_now_ns() + spectral_embedding_free(probe_spectral_embedding) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_spectral_embedding: SpectralEmbedding = spectral_embedding_fit(X_c, 2, 0.1) + spectral_embedding_free(m_spectral_embedding) + } + t1 = flow_now_ns() + printf("ESTIMATOR|spectral_embedding|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), 0.0, reps) + fflush(null) + + for i in 0 to n_c { array_free_f32(Y_label_rows[i]) } + free(Y_label_rows as ptr) + matrix_free(Y_labels) + for i in 0 to n_r { array_free_f32(Y_rows[i]) } + free(Y_rows as ptr) + matrix_free(Y_multi) + free(yi_c as ptr) + free(yi_r as ptr) + return 0 +} diff --git a/benchmarks/generated/bench_estimators_08.flow b/benchmarks/generated/bench_estimators_08.flow new file mode 100644 index 0000000..a654f22 --- /dev/null +++ b/benchmarks/generated/bench_estimators_08.flow @@ -0,0 +1,375 @@ +# Generated by benchmarks/generate_estimator_bench.py. Do not edit. +# +# One timing block per Flow estimator that the registry marks runnable. +# Regenerate with: +# python benchmarks/estimator_coverage.py +# python benchmarks/generate_estimator_bench.py + +import "lib/scikit/scikit.flow" + +extern { + function printf(fmt: string, ...) -> i32 + function flow_now_ns() -> i64 + function malloc(size: i64) -> ptr + function free(p: ptr) -> void + function fflush(stream: ptr) -> i32 +} + +function ms_between(start: i64, finish: i64) -> f32 { + return ((((finish - start) as f64) / 1000000.0) as f32) +} + +function main() -> i32 { + let iris: Dataset = load_iris() + let diabetes: Dataset = load_diabetes() + let X_c: Matrix = iris.X + let y_c: ptr = iris.y + let n_c: i32 = X_c.rows + let f_c: i32 = X_c.cols + let X_r: Matrix = diabetes.X + let y_r: ptr = diabetes.y + let n_r: i32 = X_r.rows + let f_r: i32 = X_r.cols + + let yi_c: ptr = malloc((n_c as i64) * 4) as ptr + for i in 0 to n_c { yi_c[i] = y_c[i] as i32 } + let yi_r: ptr = malloc((n_r as i64) * 4) as ptr + for i in 0 to n_r { yi_r[i] = y_r[i] as i32 } + + # A two-column target for the cross-decomposition estimators. + let Y_multi: Matrix = matrix_new(n_r, 2) + for i in 0 to n_r { + matrix_set(Y_multi, i, 0, y_r[i]) + matrix_set(Y_multi, i, 1, y_r[i] * 0.5) + } + + # Label-valued targets for the multi-output classifiers. + let Y_labels: Matrix = matrix_new(n_c, 2) + let Y_label_rows: ptr > = malloc((n_c as i64) * 8) as ptr > + for i in 0 to n_c { + let a: f32 = y_c[i] + let b: f32 = ((((y_c[i] as i32) + 1) % 3) as f32) + matrix_set(Y_labels, i, 0, a) + matrix_set(Y_labels, i, 1, b) + let lrow: ptr = array_new_f32(2) + lrow[0] = a + lrow[1] = b + Y_label_rows[i] = lrow + } + + let Y_rows: ptr > = malloc((n_r as i64) * 8) as ptr > + for i in 0 to n_r { + let row: ptr = array_new_f32(2) + row[0] = y_r[i] + row[1] = y_r[i] * 0.5 + Y_rows[i] = row + } + + let mut t0: i64 = 0 + let mut t1: i64 = 0 + let mut t2: i64 = 0 + let mut t3: i64 = 0 + let mut reps: i32 = 1 + + # ---- spline_transformer (unsupervised) ---- + t0 = flow_now_ns() + let probe_spline_transformer: SplineTransformer = spline_transformer_fit(X_c, 4, 2) + t1 = flow_now_ns() + spline_transformer_free(probe_spline_transformer) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_spline_transformer: SplineTransformer = spline_transformer_fit(X_c, 4, 2) + spline_transformer_free(m_spline_transformer) + } + t1 = flow_now_ns() + let fitted_spline_transformer: SplineTransformer = spline_transformer_fit(X_c, 4, 2) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_spline_transformer: Matrix = spline_transformer_transform(fitted_spline_transformer, X_c) + matrix_free(o_spline_transformer) + } + t3 = flow_now_ns() + spline_transformer_free(fitted_spline_transformer) + printf("ESTIMATOR|spline_transformer|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- stacking_regressor (regression) ---- + t0 = flow_now_ns() + let probe_stacking_regressor: StackingRegressor = stacking_regressor_fit(X_r, y_r, 3, 5, 42) + t1 = flow_now_ns() + stacking_regressor_free(probe_stacking_regressor) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_stacking_regressor: StackingRegressor = stacking_regressor_fit(X_r, y_r, 3, 5, 42) + stacking_regressor_free(m_stacking_regressor) + } + t1 = flow_now_ns() + let fitted_stacking_regressor: StackingRegressor = stacking_regressor_fit(X_r, y_r, 3, 5, 42) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_stacking_regressor: ptr = stacking_regressor_predict(fitted_stacking_regressor, X_r) + array_free_f32(o_stacking_regressor) + } + t3 = flow_now_ns() + stacking_regressor_free(fitted_stacking_regressor) + printf("ESTIMATOR|stacking_regressor|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- standard_scaler (unsupervised) ---- + t0 = flow_now_ns() + let probe_standard_scaler: StandardScaler = standard_scaler_fit(X_c) + t1 = flow_now_ns() + standard_scaler_free(probe_standard_scaler) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_standard_scaler: StandardScaler = standard_scaler_fit(X_c) + standard_scaler_free(m_standard_scaler) + } + t1 = flow_now_ns() + let fitted_standard_scaler: StandardScaler = standard_scaler_fit(X_c) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_standard_scaler: Matrix = standard_scaler_transform(fitted_standard_scaler, X_c) + matrix_free(o_standard_scaler) + } + t3 = flow_now_ns() + standard_scaler_free(fitted_standard_scaler) + printf("ESTIMATOR|standard_scaler|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- svc (classification) ---- + t0 = flow_now_ns() + let probe_svc: SVC = svc_fit(X_c, y_c, 3, 1.0, 0, 0.1, 2, 0.0) + t1 = flow_now_ns() + svc_free(probe_svc) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_svc: SVC = svc_fit(X_c, y_c, 3, 1.0, 0, 0.1, 2, 0.0) + svc_free(m_svc) + } + t1 = flow_now_ns() + let fitted_svc: SVC = svc_fit(X_c, y_c, 3, 1.0, 0, 0.1, 2, 0.0) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_svc: ptr = svc_predict(fitted_svc, X_c) + array_free_f32(o_svc) + } + t3 = flow_now_ns() + svc_free(fitted_svc) + printf("ESTIMATOR|svc|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- svr (regression) ---- + t0 = flow_now_ns() + let probe_svr: SVR = svr_fit(X_r, y_r, 1.0, 0.1, 0, 0.1, 2, 0.0) + t1 = flow_now_ns() + svr_free(probe_svr) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_svr: SVR = svr_fit(X_r, y_r, 1.0, 0.1, 0, 0.1, 2, 0.0) + svr_free(m_svr) + } + t1 = flow_now_ns() + let fitted_svr: SVR = svr_fit(X_r, y_r, 1.0, 0.1, 0, 0.1, 2, 0.0) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_svr: ptr = svr_predict(fitted_svr, X_r) + array_free_f32(o_svr) + } + t3 = flow_now_ns() + svr_free(fitted_svr) + printf("ESTIMATOR|svr|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- target_encoder (regression) ---- + t0 = flow_now_ns() + let probe_target_encoder: TargetEncoder = target_encoder_fit(X_r, y_r, n_r, 1.0) + t1 = flow_now_ns() + target_encoder_free(probe_target_encoder) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_target_encoder: TargetEncoder = target_encoder_fit(X_r, y_r, n_r, 1.0) + target_encoder_free(m_target_encoder) + } + t1 = flow_now_ns() + let fitted_target_encoder: TargetEncoder = target_encoder_fit(X_r, y_r, n_r, 1.0) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_target_encoder: Matrix = target_encoder_transform(fitted_target_encoder, X_r) + matrix_free(o_target_encoder) + } + t3 = flow_now_ns() + target_encoder_free(fitted_target_encoder) + printf("ESTIMATOR|target_encoder|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- theil_sen_regressor (regression) ---- + t0 = flow_now_ns() + let probe_theil_sen_regressor: TheilSenRegressor = theil_sen_regressor_fit(X_r, y_r, 10, 100, 42) + t1 = flow_now_ns() + theil_sen_regressor_free(probe_theil_sen_regressor) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_theil_sen_regressor: TheilSenRegressor = theil_sen_regressor_fit(X_r, y_r, 10, 100, 42) + theil_sen_regressor_free(m_theil_sen_regressor) + } + t1 = flow_now_ns() + let fitted_theil_sen_regressor: TheilSenRegressor = theil_sen_regressor_fit(X_r, y_r, 10, 100, 42) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_theil_sen_regressor: ptr = theil_sen_regressor_predict(fitted_theil_sen_regressor, X_r) + array_free_f32(o_theil_sen_regressor) + } + t3 = flow_now_ns() + theil_sen_regressor_free(fitted_theil_sen_regressor) + printf("ESTIMATOR|theil_sen_regressor|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- transformed_target_regressor (regression) ---- + t0 = flow_now_ns() + let probe_transformed_target_regressor: TransformedTargetRegressor = transformed_target_regressor_fit(X_r, y_r, 0) + t1 = flow_now_ns() + transformed_target_regressor_free(probe_transformed_target_regressor) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_transformed_target_regressor: TransformedTargetRegressor = transformed_target_regressor_fit(X_r, y_r, 0) + transformed_target_regressor_free(m_transformed_target_regressor) + } + t1 = flow_now_ns() + let fitted_transformed_target_regressor: TransformedTargetRegressor = transformed_target_regressor_fit(X_r, y_r, 0) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_transformed_target_regressor: ptr = transformed_target_regressor_predict(fitted_transformed_target_regressor, X_r) + array_free_f32(o_transformed_target_regressor) + } + t3 = flow_now_ns() + transformed_target_regressor_free(fitted_transformed_target_regressor) + printf("ESTIMATOR|transformed_target_regressor|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- truncated_svd (unsupervised) ---- + t0 = flow_now_ns() + let probe_truncated_svd: TruncatedSVD = truncated_svd_fit(X_c, 2) + t1 = flow_now_ns() + truncated_svd_free(probe_truncated_svd) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_truncated_svd: TruncatedSVD = truncated_svd_fit(X_c, 2) + truncated_svd_free(m_truncated_svd) + } + t1 = flow_now_ns() + let fitted_truncated_svd: TruncatedSVD = truncated_svd_fit(X_c, 2) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_truncated_svd: Matrix = truncated_svd_transform(fitted_truncated_svd, X_c) + matrix_free(o_truncated_svd) + } + t3 = flow_now_ns() + truncated_svd_free(fitted_truncated_svd) + printf("ESTIMATOR|truncated_svd|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- tsne (unsupervised) ---- + t0 = flow_now_ns() + let probe_tsne: TSNE = tsne_fit(X_c, 2, 5.0, 0.1, 50, 42) + t1 = flow_now_ns() + tsne_free(probe_tsne) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_tsne: TSNE = tsne_fit(X_c, 2, 5.0, 0.1, 50, 42) + tsne_free(m_tsne) + } + t1 = flow_now_ns() + printf("ESTIMATOR|tsne|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), 0.0, reps) + fflush(null) + + # ---- tweedie_regressor (regression) ---- + t0 = flow_now_ns() + let probe_tweedie_regressor: TweedieRegressor = tweedie_regressor_fit(X_r, y_r, 1.0, 1.5, 100, 0.01) + t1 = flow_now_ns() + tweedie_regressor_free(probe_tweedie_regressor) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_tweedie_regressor: TweedieRegressor = tweedie_regressor_fit(X_r, y_r, 1.0, 1.5, 100, 0.01) + tweedie_regressor_free(m_tweedie_regressor) + } + t1 = flow_now_ns() + let fitted_tweedie_regressor: TweedieRegressor = tweedie_regressor_fit(X_r, y_r, 1.0, 1.5, 100, 0.01) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_tweedie_regressor: ptr = tweedie_regressor_predict(fitted_tweedie_regressor, X_r) + array_free_f32(o_tweedie_regressor) + } + t3 = flow_now_ns() + tweedie_regressor_free(fitted_tweedie_regressor) + printf("ESTIMATOR|tweedie_regressor|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + # ---- variance_threshold (unsupervised) ---- + t0 = flow_now_ns() + let probe_variance_threshold: VarianceThreshold = variance_threshold_fit(X_c, 0.5) + t1 = flow_now_ns() + variance_threshold_free(probe_variance_threshold) + reps = 1 + if (t1 - t0) < 200000 { reps = 200 } + elif (t1 - t0) < 2000000 { reps = 20 } + t0 = flow_now_ns() + for rep in 0 to reps { + let m_variance_threshold: VarianceThreshold = variance_threshold_fit(X_c, 0.5) + variance_threshold_free(m_variance_threshold) + } + t1 = flow_now_ns() + let fitted_variance_threshold: VarianceThreshold = variance_threshold_fit(X_c, 0.5) + t2 = flow_now_ns() + for rep2 in 0 to reps { + let o_variance_threshold: Matrix = variance_threshold_transform(fitted_variance_threshold, X_c) + matrix_free(o_variance_threshold) + } + t3 = flow_now_ns() + variance_threshold_free(fitted_variance_threshold) + printf("ESTIMATOR|variance_threshold|%.9f|%.9f|%d|ok\n", ms_between(t0, t1) / (reps as f32), ms_between(t2, t3) / (reps as f32), reps) + fflush(null) + + for i in 0 to n_c { array_free_f32(Y_label_rows[i]) } + free(Y_label_rows as ptr) + matrix_free(Y_labels) + for i in 0 to n_r { array_free_f32(Y_rows[i]) } + free(Y_rows as ptr) + matrix_free(Y_multi) + free(yi_c as ptr) + free(yi_r as ptr) + return 0 +} diff --git a/benchmarks/publish_headline_v2.py b/benchmarks/publish_headline_v2.py index e201e05..28e73d3 100644 --- a/benchmarks/publish_headline_v2.py +++ b/benchmarks/publish_headline_v2.py @@ -15,11 +15,13 @@ DISPARITY = BENCH / "disparity_report.json" HISTORY = BENCH / "disparity_history.json" SCALED = BENCH / "scaled_ci_history.json" +ESTIMATORS = BENCH / "estimator_comparison.json" DOC_RESULT = DOCS / "headline-result-v2.json" DOC_ARCH = DOCS / "architecture-performance-map.json" DOC_DISPARITY = DOCS / "disparity-report.json" DOC_HISTORY = DOCS / "disparity-history.json" DOC_SCALED = DOCS / "scaled-ci-history.json" +DOC_ESTIMATORS = DOCS / "estimator-comparison.json" def validate_headline(result: dict) -> None: @@ -76,6 +78,21 @@ def validate_scaled(scaled: dict) -> None: raise SystemExit(f"scaled row {row['algorithm']} has a stale observation list") +def validate_estimators(payload: dict) -> None: + counts = payload["counts"] + rows = payload["rows"] + if counts["registry_estimators"] != len(rows): + raise SystemExit("estimator comparison does not cover every registry entry") + ranked = [r for r in rows if r["status"] == "ok" and r.get("speedup") is not None] + if counts["compared"] != len(ranked): + raise SystemExit("estimator comparison compared count disagrees with its own rows") + wins = sum(1 for r in ranked if r["speedup"] >= 1.0) + if counts["flow_wins"] != wins: + raise SystemExit("estimator comparison win count disagrees with its own rows") + if not payload.get("contract"): + raise SystemExit("the estimator comparison must carry the contract it was measured under") + + def main() -> int: parser = argparse.ArgumentParser() parser.add_argument("--check", action="store_true") @@ -92,6 +109,11 @@ def main() -> int: validate_disparity(disparity, result["counts"]["total_rows"]) elif not args.check: raise SystemExit("disparity_report.json must be generated before Pages publication") + estimators = json.loads(ESTIMATORS.read_text()) if ESTIMATORS.exists() else None + if estimators is not None: + validate_estimators(estimators) + elif not args.check: + raise SystemExit("estimator_comparison.json must be present before Pages publication") scaled = json.loads(SCALED.read_text()) if SCALED.exists() else None if scaled is not None: validate_scaled(scaled) @@ -108,8 +130,9 @@ def main() -> int: shutil.copyfile(DISPARITY, DOC_DISPARITY) shutil.copyfile(HISTORY, DOC_HISTORY) shutil.copyfile(SCALED, DOC_SCALED) + shutil.copyfile(ESTIMATORS, DOC_ESTIMATORS) counts = result["counts"] - print(f"published evidence: {counts['flow_wins']}/{counts['eligible_comparisons']} Flow wins; {disparity['counts']['rows_with_tracked_disparity']} rows with tracked disparities; {len(history['snapshots'])} history snapshots; scaled matrix over {scaled['counts']['runs']} runs") + print(f"published evidence: {counts['flow_wins']}/{counts['eligible_comparisons']} Flow wins; {disparity['counts']['rows_with_tracked_disparity']} rows with tracked disparities; {len(history['snapshots'])} history snapshots; scaled matrix over {scaled['counts']['runs']} runs; {estimators['counts']['compared']} estimators ranked") else: print("canonical benchmark, architecture, disparity, and available history evidence are internally consistent") return 0 diff --git a/docs/benchmarks.html b/docs/benchmarks.html index f2ed530..f7754de 100644 --- a/docs/benchmarks.html +++ b/docs/benchmarks.html @@ -1,4 +1,4 @@ -Benchmarks, flow-scikit

canonical v2 / parity + disparity benchmark

Eligibility never means identity.

All 19 canonical rows are measured and currently eligible for comparison, but numerical, semantic and runtime disparities remain first-class evidence. This page renders the committed benchmark and disparity artifacts directly so differences cannot disappear merely because a row passes its contract.

Flow wins...

End-to-end fit + predict comparisons won by Flow.

sklearn wins...

End-to-end comparisons won by scikit-learn.

parity eligible...

Rows admitted to the competitive denominator.

substantive disparities...

Rows whose fitted state, score, configuration or semantics genuinely diverge, above float-noise floors. Runtime differences are tracked per row but not counted here.

TIMING_UNIT|msend-to-endseed=4280/20 persisted split2% practical tie thresholddisparity retained after eligibility
KMeans note: Digits KMeans is eligible under the same declared contract as every other clustering row. Its seeded k-means++ initialization now matches scikit-learn's, so the strict diagnostic and the final eligibility decision agree. The convergence statistic, the point at which inertia is reported, empty-cluster relocation and the n_init selection rule still differ and stay visible in the disparity artifact.

runtime overview

The plots are generated from the canonical JSON.

Each runtime plot shows end-to-end fit + predict time on a log scale. The plots use the same rows as the table below and therefore update whenever the frozen canonical result changes.

All 19 speed ratios

scikit-learn total time divided by Flow total time. The vertical 1× line separates Flow wins from scikit-learn wins.

Iris total runtime

scikit-learnFlow

Digits total runtime

scikit-learnFlow

Diabetes total runtime

scikit-learnFlow

persistent disparity

Passing parity does not erase the gap.

The disparity plot normalizes each row's principal numerical difference against its effective tolerance where a tolerance is available. A value near 1 means the row is close to the acceptance boundary. Semantic/configuration differences are tracked in the same artifact and remain visible in the table.

Numerical disparity relative to tolerance

The dashed line is the acceptance boundary. Values can remain non-zero even for eligible rows.

all canonical rows

No selected-win table.

Every row is shown below. Speedup is sklearn_ms / flow_ms; values above 1× favor Flow. Strict diagnostic status is kept separate from final eligibility.

AlgorithmDatasetFinal parityStrict diagnosticWinnerscore |Δ|sklearn msFlow msspeedup

larger data

The canonical rows all fit in 1797 samples.

A separate matrix runs five estimators at 100, 1000 and 10000 rows against 8 and 32 features. One run of it does not settle a row: at 100 and 1000 samples a fit finishes in well under a millisecond, and the CI runner moves that by more than the difference being measured. Lasso at 1000 rows and 32 features was recorded at 3.83x and at 0.92x on code that differs in nothing touching Lasso. The table is therefore the spread across consecutive runs rather than one run's number, sorted with the narrowest margins first.

Algorithmsamplesfeaturesruns wonmedianrange

This matrix is reported without gating the build. What does gate is benchmarks/scaled_flow_baseline.json, a Flow-against-itself comparison refreshed from a CI artifact.

methodology

Correctness, disparity and timing are separate dimensions.

The benchmark consumes the same persisted train/test indices in Python and Flow. Python uses high-resolution adaptive timing and the canonical runner aggregates repeated process measurements with medians and IQR. Flow timings are emitted in milliseconds and aggregated by the same runner.

Supervised rows compare predictive metrics under declared tolerances. PCA additionally checks explained variance, singular values, reconstruction error and sign-aligned components. KMeans uses permutation-invariant clustering quality and inertia. The persistent disparity artifact preserves raw numerical gaps and known semantic/configuration differences even after the estimator-specific eligibility contract succeeds.

historical deployment evidence

Footprint and startup remain separate experiments.

The repository also contains a historical deployment comparison recording a roughly 1.4 MB Flow native executable and a roughly 65× cold-start advantage (33 ms versus 2160 ms). Those figures come from a different deployment experiment and are intentionally not mixed into the canonical estimator timing denominator.

trajectory

Flow versus Python, across freezes.

Each row's speedup at the previous freeze and at the latest one. A speedup can move because Flow changed or because scikit-learn's side changed on that runner. When a row moves by more than 10%, the last column names which side's own time moved more, from the committed absolute timings.

reproduce

Read the source artifacts.

Canonical result ↗ Disparity report ↗

\ No newline at end of file +fetch('scaled-ci-history.json').then(r=>r.json()).then(h=>{const c=h.counts,tb=document.getElementById('scaled-rows');document.getElementById('scaled-summary').textContent=`Across ${c.runs} consecutive CI runs, Flow wins ${c.rows_won_in_every_run} of the ${c.rows} rows in every one of them. The remaining ${c.rows_lost_in_at_least_one_run} dipped below 1x in at least one run, and every one of those is at 1000 samples.`;h.rows.slice().sort((a,b)=>a.median_speedup-b.median_speedup).forEach(r=>{const tr=document.createElement('tr');const cells=[[r.algorithm.replace(/_/g,' '),''],[r.rows,'num'],[r.features,'num'],[`${r.runs_won} / ${r.runs_observed}`,r.runs_won===r.runs_observed?'num win':'num loss'],[`${r.median_speedup.toFixed(2)}x`,'num'],[`${r.min_speedup.toFixed(2)} to ${r.max_speedup.toFixed(2)}`,'num']];cells.forEach(([t,cl])=>{const td=document.createElement('td');if(cl)td.className=cl;td.textContent=t;tr.appendChild(td)});tb.appendChild(tr)})});fetch('estimator-comparison.json').then(r=>r.json()).then(e=>{const c=e.counts;const byStatus=t=>e.rows.filter(r=>r.status===t).length;document.getElementById('est-ranked').textContent=c.compared;document.getElementById('est-wins').textContent=`${c.flow_wins} / ${c.compared}`;document.getElementById('est-shape').textContent=byStatus('different_shape');document.getElementById('est-only').textContent=byStatus('flow_only');document.getElementById('est-note').textContent=e.note||'';const stub=e.rows.filter(r=>r.status==='simplified');if(stub.length){document.getElementById('est-excluded').textContent=`${stub.length} more are excluded from the count because the implementation's own comments call it simplified: ${stub.map(r=>r.flow_estimator.replace(/_/g,' ')).join(', ')}. `+`Timing a stand-in against scikit-learn's algorithm compares two different things, and three of those four would otherwise have been the widest wins on the page.`}const tb=document.getElementById('est-rows');e.rows.filter(r=>r.status==='ok'&&r.speedup!=null).sort((a,b)=>a.speedup-b.speedup).forEach(r=>{const tr=document.createElement('tr');const cells=[[r.flow_estimator.replace(/_/g,' '),''],[r.sklearn_estimator,''],[r.flow_ms.toFixed(3),'num'],[r.sklearn_ms.toFixed(3),'num'],[`${r.speedup.toFixed(2)}x`,r.speedup>=1?'num win':'num loss']];cells.forEach(([t,cl])=>{const td=document.createElement('td');if(cl)td.className=cl;td.textContent=t;tr.appendChild(td)});tb.appendChild(tr)})}); \ No newline at end of file diff --git a/docs/index.html b/docs/index.html index be7dd01..18b416d 100644 --- a/docs/index.html +++ b/docs/index.html @@ -1,4 +1,4 @@ -flow-scikit, compiled classical ML in Flow

compiled classical machine learning / Flow

Rebuild the estimator stack. Measure what actually changes.

flow-scikit reimplements classical ML in Flow and tests it against scikit-learn with parity-gated timings, explicit execution-substrate analysis, native deployment and reproducible benchmark artifacts.

  • 19/19 parity eligible
  • native binary
  • substrate-aware appraisal
canonical parity19 / 19

Every canonical row has resolved parity and measurement status.

Flow timing wins19 / 19

Current canonical end-to-end result; all rows remain visible.

estimator operations mapped491

Generated sklearn execution-substrate inventory.

runtime profiles32

Mixed-stack attribution rows feeding the optimization roadmap.

the important distinction

"Python versus compiled" is too crude a model for scikit-learn.

Its public interface is Python, but hot paths may execute in NumPy/SciPy, BLAS/LAPACK, sklearn-owned compiled code, or external native libraries such as liblinear and libsvm. The meaningful question is therefore not simply whether Flow beats Python, but which execution layer Flow is replacing, retaining, or compiling around.

Inspect the execution map →
Python-boundInterpreter and orchestration work can be direct compilation targets.
Mixed / boundary-heavyWhole-estimator compilation can remove crossings and temporary allocations.
BLAS / LAPACK-boundRetain mature kernels unless evidence supports replacement.
sklearn-owned nativeBenchmark Flow against the compiled implementation directly.
External nativeliblinear and libsvm are compiled competitors. Beating them is a different claim from beating an interpreter.

current appraisal

The architecture result is more useful than a blanket speed claim.

The committed map groups every canonical row by sklearn fit substrate. Flow wins all of them, and the margin follows the class. That is evidence that execution substrate is predictive enough to guide optimization work, while still being far from a causal proof by itself.

What the evidence supports today: Flow wins every canonical row, including the ones where scikit-learn calls liblinear or libsvm, and ships a much smaller native deployment. Those native solvers remain close baselines rather than beaten ones: the narrowest canonical margins sit near 1.5x, and a row that close can land either way on a different BLAS and core count. The roadmap ranks work from measured substrate, runtime attribution, benchmark results and implementation readiness rather than from source-language folklore.

evidence chain

Correctness → timing → substrate → attribution → roadmap.

The repository now keeps each stage machine-readable and reproducible.

01 / parity

Resolve all 19 canonical rows.

Competitive timing only follows estimator-specific numerical checks.

Parity evidence →
02 / architecture

Map what sklearn actually executes.

491 estimator-operation rows are classified with evidence and drift detection.

Architecture map →
03 / priorities

Turn evidence into an optimization queue.

Runtime profiles, native-hotspot dispositions and whole-estimator experiments feed a generated roadmap.

Roadmap ↗

minimal example

Fit, predict, inspect.

The library remains a native Flow implementation rather than a Python compatibility layer.

import "lib/scikit/scikit.flow"
+flow-scikit, compiled classical ML in Flow

compiled classical machine learning / Flow

Rebuild the estimator stack. Measure what actually changes.

flow-scikit reimplements classical ML in Flow and tests it against scikit-learn with parity-gated timings, explicit execution-substrate analysis, native deployment and reproducible benchmark artifacts.

  • 19/19 parity eligible
  • native binary
  • substrate-aware appraisal
canonical parity19 / 19

Every canonical row has resolved parity and measurement status.

Flow timing wins19 / 19

Current canonical end-to-end result; all rows remain visible.

estimator operations mapped491

Generated sklearn execution-substrate inventory.

runtime profiles32

Mixed-stack attribution rows feeding the optimization roadmap.

the important distinction

"Python versus compiled" is too crude a model for scikit-learn.

Its public interface is Python, but hot paths may execute in NumPy/SciPy, BLAS/LAPACK, sklearn-owned compiled code, or external native libraries such as liblinear and libsvm. The meaningful question is therefore not simply whether Flow beats Python, but which execution layer Flow is replacing, retaining, or compiling around.

Inspect the execution map →
Python-boundInterpreter and orchestration work can be direct compilation targets.
Mixed / boundary-heavyWhole-estimator compilation can remove crossings and temporary allocations.
BLAS / LAPACK-boundRetain mature kernels unless evidence supports replacement.
sklearn-owned nativeBenchmark Flow against the compiled implementation directly.
External nativeliblinear and libsvm are compiled competitors. Beating them is a different claim from beating an interpreter.

current appraisal

The architecture result is more useful than a blanket speed claim.

The committed map groups every canonical row by sklearn fit substrate. Flow wins all of them, and the margin follows the class. That is evidence that execution substrate is predictive enough to guide optimization work, while still being far from a causal proof by itself.

What the evidence supports today: Flow wins every canonical row, including the ones where scikit-learn calls liblinear or libsvm, and ships a much smaller native deployment. Those native solvers remain close baselines rather than beaten ones: the narrowest canonical margins sit near 1.5x, and a row that close can land either way on a different BLAS and core count. The roadmap ranks work from measured substrate, runtime attribution, benchmark results and implementation readiness rather than from source-language folklore.

evidence chain

Correctness → timing → substrate → attribution → roadmap.

The repository now keeps each stage machine-readable and reproducible.

01 / parity

Resolve all 19 canonical rows.

Competitive timing only follows estimator-specific numerical checks.

Parity evidence →
02 / coverage

Measure the whole library.

The canonical rows race twelve estimators. 203 are exported, and every one of them is now either raced, or carries a written reason why not.

Estimator coverage →
03 / architecture

Map what sklearn actually executes.

491 estimator-operation rows are classified with evidence and drift detection.

Architecture map →
04 / priorities

Turn evidence into an optimization queue.

Runtime profiles, native-hotspot dispositions and whole-estimator experiments feed a generated roadmap.

Roadmap ↗

minimal example

Fit, predict, inspect.

The library remains a native Flow implementation rather than a Python compatibility layer.

import "lib/scikit/scikit.flow"
 
 let model = knn_classifier_fit(X_train, y_train, 5)
 let predictions = knn_classifier_predict(model, X_test)
diff --git a/lib/scikit/ensemble.flow b/lib/scikit/ensemble.flow
index a6c3259..fb6aa6e 100644
--- a/lib/scikit/ensemble.flow
+++ b/lib/scikit/ensemble.flow
@@ -699,10 +699,16 @@ export struct VotingClassifier {
 
 export function voting_classifier_fit(estimators: ptr, n_estimators: i32, n_classes: i32, classes: ptr, voting: i32) -> VotingClassifier {
     return VotingClassifier {
+        # The model takes over the fitted sub-estimators and its free
+        # function releases them, so the caller must not free this array
+        # or the trees in it.
         estimators: estimators,
         n_estimators: n_estimators,
         n_classes: n_classes,
-        classes: classes,
+        # Freed by this model's free function, so it must be this model's
+        # own memory. Aliasing the caller's made fit and free destroy the
+        # input.
+        classes: array_copy_f32(classes, n_classes),
         voting: voting,
         fitted: true
     }
@@ -2696,10 +2702,16 @@ export struct VotingRegressor {
 
 export function voting_regressor_fit(estimators: ptr, n_estimators: i32, weights: ptr) -> VotingRegressor {
     let result: VotingRegressor = VotingRegressor {
+        # The model takes over the fitted sub-estimators and its free
+        # function releases them, so the caller must not free this array
+        # or the trees in it.
         estimators: estimators,
         n_estimators: n_estimators,
         n_features: 0,
-        weights: weights,
+        # Freed by this model's free function, so it must be this model's
+        # own memory. Aliasing the caller's made fit and free destroy the
+        # input.
+        weights: array_copy_f32(weights, n_estimators),
         fitted: true
     }
     return result
diff --git a/lib/scikit/extra_estimators.flow b/lib/scikit/extra_estimators.flow
index f28e2c2..4c13a99 100644
--- a/lib/scikit/extra_estimators.flow
+++ b/lib/scikit/extra_estimators.flow
@@ -35,7 +35,11 @@ export struct SparseCoder {
 
 export function sparse_coder_fit(dictionary: Matrix, n_nonzero_coefs: i32) -> SparseCoder {
     return SparseCoder {
-        dictionary: dictionary,
+        # The free function releases this, so it has to be this model's
+        # copy. Aliasing the caller's matrix made a fit and free pair
+        # destroy the input, which AddressSanitizer caught as a
+        # use-after-free on the second fit of the same data.
+        dictionary: matrix_copy(dictionary),
         n_components: dictionary.rows,
         n_features: dictionary.cols,
         transform_n_nonzero_coefs: n_nonzero_coefs,
diff --git a/lib/scikit/gaussian_process.flow b/lib/scikit/gaussian_process.flow
index 69f1458..361787b 100644
--- a/lib/scikit/gaussian_process.flow
+++ b/lib/scikit/gaussian_process.flow
@@ -329,7 +329,11 @@ export function gaussian_process_classifier_fit(X: Matrix, y: ptr, gamma: f
     free(K as ptr)
 
     return GaussianProcessClassifier {
-        X_train: X,
+        # The free function releases this, so it has to be this model's
+        # copy. Aliasing the caller's matrix made a fit and free pair
+        # destroy the input, which AddressSanitizer caught as a
+        # use-after-free on the second fit of the same data.
+        X_train: matrix_copy(X),
         y_train: y_dual,
         f: f,
         gamma: gamma,
diff --git a/lib/scikit/linear.flow b/lib/scikit/linear.flow
index 0c8d130..3795fad 100644
--- a/lib/scikit/linear.flow
+++ b/lib/scikit/linear.flow
@@ -187,7 +187,17 @@ export function _solve_f64(A: ptr, b: ptr, n: i32) -> void {
 #
 # On a full-rank design the relative cutoff is never reached, so the arithmetic
 # is untouched: same reflectors, same order, same divisions.
+# X is m rows by n columns, row-major, and y must have room for max(m, n)
+# entries because the solution is read back from y[0..n-1].
+#
+# A design with fewer rows than columns has at most m pivots, and the
+# coordinates past that are not determined by the data. Every sweep below runs
+# to `rank` rather than to n for that reason: reading to n walked off the end
+# of both X and y, which RANSAC reached by fitting 5 sampled rows against the
+# 10 features of diabetes. AddressSanitizer caught it there.
 export function _solve_lstsq_qr(X: ptr, y: ptr, m: i32, n: i32) -> void {
+    let mut rank: i32 = n
+    if m < n { rank = m }
     let dots: ptr = array_new_f64(n)
 
     # Apply n Householder transformations to reduce X to upper triangular
@@ -234,7 +244,7 @@ export function _solve_lstsq_qr(X: ptr, y: ptr, m: i32, n: i32) -> voi
     # Rank cutoff, relative to the largest pivot and floored at the old
     # absolute value so it can never be looser than the code it replaces.
     let mut maxdiag: f64 = 0.0
-    for d in 0 to n {
+    for d in 0 to rank {
         let dd: f64 = fabs(X[d * n + d])
         if dd > maxdiag { maxdiag = dd }
     }
@@ -244,7 +254,14 @@ export function _solve_lstsq_qr(X: ptr, y: ptr, m: i32, n: i32) -> voi
     # Back-substitution: R * w = y(0..n-1). A coordinate whose pivot is below
     # tol is set to zero, which drops its column out of every sum above it and
     # leaves the reduced problem to the surviving pivots.
-    let mut i: i32 = n - 1
+    # Coordinates with no pivot are undetermined, so they are zero here and
+    # drop out of the sums for the pivots above them.
+    let mut z: i32 = rank
+    while z < n {
+        y[z] = 0.0
+        z = z + 1
+    }
+    let mut i: i32 = rank - 1
     while i >= 0 {
         let mut s: f64 = y[i]
         let mut j: i32 = i + 1
@@ -412,7 +429,11 @@ export function linear_regression_fit(X: Matrix, y: ptr, penalty: Penalty)
     let qr_work: i64 = (m as i64) * (n as i64) * (n as i64)
 
     let Xc: ptr = malloc((m as i64) * (n as i64) * 8) as ptr
-    let yc: ptr = array_new_f64(m)
+    # The solution comes back in yc[0..n-1], so it needs room for n even when
+    # the design has fewer rows than that.
+    let mut yc_len: i32 = m
+    if n > yc_len { yc_len = n }
+    let yc: ptr = array_new_f64(yc_len)
     let w_out: ptr = array_new_f64(n)
     for i in 0 to m {
         yc[i] = (y[i] as f64) - y_mean
@@ -753,7 +774,10 @@ export function ols_inference_fit(model: LinearRegression, X: Matrix, y: ptr = malloc((m as i64) * (n as i64) * 8 + 128) as ptr
-    let yc: ptr = array_new_f64(m + 8)
+    # Same reason as the fit: the solution is read back from yc[0..n-1].
+    let mut yc_len2: i32 = m + 8
+    if n > yc_len2 { yc_len2 = n }
+    let yc: ptr = array_new_f64(yc_len2)
     for i in 0 to m {
         yc[i] = (y[i] as f64) - y_mean
         let base: i32 = i * n
@@ -767,8 +791,10 @@ export function ols_inference_fit(model: LinearRegression, X: Matrix, y: ptr maxdiag { maxdiag = dj }
     }
diff --git a/lib/scikit/neighbors.flow b/lib/scikit/neighbors.flow
index cfacd84..2743acb 100644
--- a/lib/scikit/neighbors.flow
+++ b/lib/scikit/neighbors.flow
@@ -366,8 +366,15 @@ export struct RadiusNeighborsClassifier {
 
 export function radius_neighbors_classifier_fit(X: Matrix, y: ptr, radius: f32, n_classes: i32, outlier_label: f32) -> RadiusNeighborsClassifier {
     return RadiusNeighborsClassifier {
-        X_train: X,
-        y_train: y,
+        # The free function releases this, so it has to be this model's
+        # copy. Aliasing the caller's matrix made a fit and free pair
+        # destroy the input, which AddressSanitizer caught as a
+        # use-after-free on the second fit of the same data.
+        X_train: matrix_copy(X),
+        # Freed by this model's free function, so it must be this model's
+        # own memory. Aliasing the caller's made fit and free destroy the
+        # input.
+        y_train: array_copy_f32(y, X.rows),
         radius: radius,
         n_features: X.cols,
         n_classes: n_classes,
@@ -658,7 +665,11 @@ export function local_outlier_factor_fit(X: Matrix, n_neighbors: i32) -> LocalOu
     array_free_f32(k_dist)
 
     return LocalOutlierFactor {
-        X_train: X,
+        # The free function releases this, so it has to be this model's
+        # copy. Aliasing the caller's matrix made a fit and free pair
+        # destroy the input, which AddressSanitizer caught as a
+        # use-after-free on the second fit of the same data.
+        X_train: matrix_copy(X),
         lrd: lrd,
         lof: lof,
         n_neighbors: k,
diff --git a/lib/scikit/svm.flow b/lib/scikit/svm.flow
index 3c555f1..e1d1e00 100644
--- a/lib/scikit/svm.flow
+++ b/lib/scikit/svm.flow
@@ -1903,8 +1903,15 @@ export function nu_svr_fit(X: Matrix, y: ptr, nu: f32, C: f32, gamma: f32,
     }
 
     return NuSVR {
-        X_train: X,
-        y_train: y,
+        # The free function releases this, so it has to be this model's
+        # copy. Aliasing the caller's matrix made a fit and free pair
+        # destroy the input, which AddressSanitizer caught as a
+        # use-after-free on the second fit of the same data.
+        X_train: matrix_copy(X),
+        # Freed by this model's free function, so it must be this model's
+        # own memory. Aliasing the caller's made fit and free destroy the
+        # input.
+        y_train: array_copy_f32(y, n),
         alphas: alphas,
         b: b,
         gamma: gamma,
@@ -2011,7 +2018,10 @@ export function one_class_svm_fit(X: Matrix, nu: f32, gamma: f32, max_iter: i32)
     }
 
     return OneClassSVM {
-        X_train: X,
+        # Freed by this model's free function, so it must be this model's
+        # own memory. Aliasing the caller's made fit and free destroy the
+        # input.
+        X_train: matrix_copy(X),
         alphas: alphas,
         b: b,
         gamma: gamma,
@@ -2166,124 +2176,225 @@ function _svm_kernel_matrix(X: Matrix, kernel: i32, gamma: f32, degree: i32, coe
     return K
 }
 
+# Second order working set selection over a materialized kernel, as in
+# libsvm's Solver (WSS3 of Fan, Chen and Lin 2005).
+#
+# The previous implementation recomputed every error term from scratch: an
+# O(n) sum per candidate, inside a scan over candidates to pick the partner,
+# inside a pass over all n, for up to max_iter passes. On iris that is billions
+# of operations and SVC took 4.8 seconds against scikit-learn's 0.6 ms. The
+# gradient is maintained incrementally here instead, which is what
+# _kernel_svc_smo_precomputed already does for KernelSVC.
+#
+# Q[i,j] is y_i * y_j * K[i,j] and G[i] is sum_j Q[i,j] * alpha_j - 1. Since
+# y squared is one, the curvature terms reduce to plain kernel entries:
+# QD_i + QD_j - 2 * y_i * y_j * Q_ij is K_ii + K_jj - 2 * K_ij.
 function _smo_train(K: Matrix, y_dual: ptr, n: i32, C: f32, max_iter: i32) -> SMOResult {
-    let alphas: ptr = array_new_f32(n)
-    let mut b: f32 = 0.0
-    let tol: f32 = 0.001
+    let kdata: ptr = K.data
+    let C_d: f64 = C as f64
 
-    let mut iter: i32 = 0
-    let mut passes_without_change: i32 = 0
-    while iter < max_iter && passes_without_change < 5 {
-        let mut num_changed: i32 = 0
+    let alphas_d: ptr = array_new_f64(n)
+    let y_d: ptr = array_new_f64(n)
+    let G: ptr = array_new_f64(n)
+    let QD: ptr = array_new_f64(n)
+    for i in 0 to n {
+        y_d[i] = y_dual[i] as f64
+        G[i] = -1.0
+        QD[i] = kdata[i * n + i] as f64
+    }
 
-        for i in 0 to n {
-            let mut ei: f32 = 0.0
-            for j in 0 to n {
-                ei = ei + alphas[j] * y_dual[j] * matrix_at(K, i, j)
-            }
-            ei = ei + b - y_dual[i]
-
-            let yi_ei: f32 = y_dual[i] * ei
-            if (yi_ei < -tol && alphas[i] < C) || (yi_ei > tol && alphas[i] > 0.0) {
-                # Platt-style second-choice heuristic: choose the partner
-                # with the largest |E_i - E_j| so each step has useful gain.
-                let mut j: i32 = -1
-                let mut best_gap: f32 = -1.0
-                for candidate in 0 to n {
-                    if candidate != i {
-                        let mut candidate_e: f32 = 0.0
-                        for k in 0 to n {
-                            candidate_e = candidate_e + alphas[k] * y_dual[k] * matrix_at(K, candidate, k)
-                        }
-                        candidate_e = candidate_e + b - y_dual[candidate]
-                        let mut gap: f32 = ei - candidate_e
-                        if gap < 0.0 { gap = -gap }
-                        if gap > best_gap {
-                            best_gap = gap
-                            j = candidate
-                        }
-                    }
-                }
-                if j < 0 { continue }
+    let tol: f64 = 0.001
+    let tau: f64 = 0.000000000001
+    let big: f64 = 1000000000000.0
+    # The old rule counted outer passes over n examples; keep the same total
+    # step budget so a caller passing a small cap behaves the same way.
+    let budget: i32 = max_iter * n
+    let mut steps: i32 = 0
 
-                let mut ej: f32 = 0.0
-                for k in 0 to n {
-                    ej = ej + alphas[k] * y_dual[k] * matrix_at(K, j, k)
-                }
-                ej = ej + b - y_dual[j]
-
-                let alpha_i_old: f32 = alphas[i]
-                let alpha_j_old: f32 = alphas[j]
-
-                let mut L: f32 = 0.0
-                let mut H: f32 = C
-                if y_dual[i] == y_dual[j] {
-                    L = alpha_i_old + alpha_j_old - C
-                    H = alpha_i_old + alpha_j_old
-                    if L < 0.0 {
-                        L = 0.0
-                    }
-                    if H > C {
-                        H = C
-                    }
-                } else {
-                    # Standard SMO bounds when y_i != y_j:
-                    # L=max(0, a_j-a_i), H=min(C, C+a_j-a_i).
-                    L = alpha_j_old - alpha_i_old
-                    H = alpha_j_old - alpha_i_old + C
-                    if L < 0.0 {
-                        L = 0.0
+    while steps < budget {
+        let mut gmax: f64 = 0.0 - big
+        let mut i_sel: i32 = -1
+        for t in 0 to n {
+            let at: f64 = alphas_d[t]
+            if y_d[t] > 0.0 {
+                if at < C_d {
+                    let v: f64 = 0.0 - G[t]
+                    if v >= gmax {
+                        gmax = v
+                        i_sel = t
                     }
-                    if H > C {
-                        H = C
+                }
+            } else {
+                if at > 0.0 {
+                    let v2: f64 = G[t]
+                    if v2 >= gmax {
+                        gmax = v2
+                        i_sel = t
                     }
                 }
+            }
+        }
+        if i_sel < 0 { break }
 
-                if L == H {
-                    continue
-                }
+        let y_i: f64 = y_d[i_sel]
+        let row_i: ptr = kdata + i_sel * n
+        let qd_i: f64 = QD[i_sel]
 
-                let eta: f32 = 2.0 * matrix_at(K, i, j) - matrix_at(K, i, i) - matrix_at(K, j, j)
-                if eta >= 0.0 {
-                    continue
+        let mut gmax2: f64 = 0.0 - big
+        let mut j_sel: i32 = -1
+        let mut obj_min: f64 = big
+        for t2 in 0 to n {
+            let a2: f64 = alphas_d[t2]
+            let y_t: f64 = y_d[t2]
+            let mut usable: bool = false
+            if y_t > 0.0 {
+                if a2 > 0.0 { usable = true }
+            } else {
+                if a2 < C_d { usable = true }
+            }
+            if usable {
+                let ygt: f64 = y_t * G[t2]
+                if ygt >= gmax2 { gmax2 = ygt }
+                let gdiff: f64 = gmax + ygt
+                if gdiff > 0.0 {
+                    let mut quad: f64 = qd_i + QD[t2] - 2.0 * (row_i[t2] as f64)
+                    if quad <= 0.0 { quad = tau }
+                    let odiff: f64 = 0.0 - (gdiff * gdiff) / quad
+                    if odiff <= obj_min {
+                        obj_min = odiff
+                        j_sel = t2
+                    }
                 }
+            }
+        }
+
+        # Maximal KKT violation, which is libsvm's stopping rule.
+        if gmax + gmax2 < tol { break }
+        if j_sel < 0 { break }
+
+        let y_j: f64 = y_d[j_sel]
+        let ai_old: f64 = alphas_d[i_sel]
+        let aj_old: f64 = alphas_d[j_sel]
+        let mut quad_c: f64 = qd_i + QD[j_sel] - 2.0 * (row_i[j_sel] as f64)
+        if quad_c <= 0.0 { quad_c = tau }
 
-                let mut alpha_j_new: f32 = alpha_j_old - y_dual[j] * (ei - ej) / eta
-                if alpha_j_new < L {
-                    alpha_j_new = L
+        let mut ai_new: f64 = 0.0
+        let mut aj_new: f64 = 0.0
+        if y_i != y_j {
+            let delta: f64 = (0.0 - G[i_sel] - G[j_sel]) / quad_c
+            let diff: f64 = ai_old - aj_old
+            ai_new = ai_old + delta
+            aj_new = aj_old + delta
+            if diff > 0.0 {
+                if aj_new < 0.0 {
+                    aj_new = 0.0
+                    ai_new = diff
                 }
-                if alpha_j_new > H {
-                    alpha_j_new = H
+            } else {
+                if ai_new < 0.0 {
+                    ai_new = 0.0
+                    aj_new = 0.0 - diff
                 }
+            }
+            if diff > 0.0 {
+                if ai_new > C_d {
+                    ai_new = C_d
+                    aj_new = C_d - diff
+                }
+            } else {
+                if aj_new > C_d {
+                    aj_new = C_d
+                    ai_new = C_d + diff
+                }
+            }
+        } else {
+            let delta2: f64 = (G[i_sel] - G[j_sel]) / quad_c
+            let sum_a: f64 = ai_old + aj_old
+            ai_new = ai_old - delta2
+            aj_new = aj_old + delta2
+            if sum_a > C_d {
+                if ai_new > C_d {
+                    ai_new = C_d
+                    aj_new = sum_a - C_d
+                }
+            } else {
+                if aj_new < 0.0 {
+                    aj_new = 0.0
+                    ai_new = sum_a
+                }
+            }
+            if sum_a > C_d {
+                if aj_new > C_d {
+                    aj_new = C_d
+                    ai_new = sum_a - C_d
+                }
+            } else {
+                if ai_new < 0.0 {
+                    ai_new = 0.0
+                    aj_new = sum_a
+                }
+            }
+        }
 
-                let alpha_i_new: f32 = alpha_i_old + y_dual[i] * y_dual[j] * (alpha_j_old - alpha_j_new)
-
-                alphas[i] = alpha_i_new
-                alphas[j] = alpha_j_new
+        alphas_d[i_sel] = ai_new
+        alphas_d[j_sel] = aj_new
+        let d_i: f64 = ai_new - ai_old
+        let d_j: f64 = aj_new - aj_old
 
-                let b1: f32 = b - ei - y_dual[i] * (alpha_i_new - alpha_i_old) * matrix_at(K, i, i) - y_dual[j] * (alpha_j_new - alpha_j_old) * matrix_at(K, i, j)
-                let b2: f32 = b - ej - y_dual[i] * (alpha_i_new - alpha_i_old) * matrix_at(K, i, j) - y_dual[j] * (alpha_j_new - alpha_j_old) * matrix_at(K, j, j)
+        # G[k] += Q[i,k] * d_i + Q[j,k] * d_j, and the kernel is symmetric so
+        # both come off a row rather than striding down a column.
+        let row_j: ptr = kdata + j_sel * n
+        for k in 0 to n {
+            let contrib: f64 = y_i * (row_i[k] as f64) * d_i + y_j * (row_j[k] as f64) * d_j
+            G[k] = G[k] + y_d[k] * contrib
+        }
+        steps = steps + 1
+    }
 
-                if alpha_i_new > 0.0 && alpha_i_new < C {
-                    b = b1
-                } elif alpha_j_new > 0.0 && alpha_j_new < C {
-                    b = b2
-                } else {
-                    b = (b1 + b2) / 2.0
+    # Bias from the free support vectors, which is where the decision function
+    # is exactly y_i. With none of them, every support vector is at a bound and
+    # their average is the best available estimate.
+    let mut b_sum: f64 = 0.0
+    let mut n_free: i32 = 0
+    let mut b_any: f64 = 0.0
+    let mut n_any: i32 = 0
+    for i in 0 to n {
+        let ai: f64 = alphas_d[i]
+        if ai > 0.0 {
+            let row: ptr = kdata + i * n
+            let mut fi: f64 = 0.0
+            for j in 0 to n {
+                let aj: f64 = alphas_d[j]
+                if aj > 0.0 {
+                    fi = fi + aj * y_d[j] * (row[j] as f64)
                 }
-
-                num_changed = num_changed + 1
+            }
+            let resid: f64 = y_d[i] - fi
+            b_any = b_any + resid
+            n_any = n_any + 1
+            if ai < C_d {
+                b_sum = b_sum + resid
+                n_free = n_free + 1
             }
         }
+    }
+    let mut b: f32 = 0.0
+    if n_free > 0 {
+        b = (b_sum / (n_free as f64)) as f32
+    } elif n_any > 0 {
+        b = (b_any / (n_any as f64)) as f32
+    }
 
-        iter = iter + 1
-        if num_changed == 0 {
-            passes_without_change = passes_without_change + 1
-        } else {
-            passes_without_change = 0
-        }
+    let alphas: ptr = array_new_f32(n)
+    for i in 0 to n {
+        alphas[i] = alphas_d[i] as f32
     }
 
+    array_free_f64(alphas_d)
+    array_free_f64(y_d)
+    array_free_f64(G)
+    array_free_f64(QD)
+
     let result: SMOResult = SMOResult { alphas: alphas, b: b }
     return result
 }