{"adoption": {"forks": 327, "observed_at": "2026-08-28T04:05:50.954150+00:00", "stars": 1895}, "canonical_url": "https://ross.abutalabs.com/products/benchm-ml", "card": {"archived": false, "artifact_type": "dataset", "description": "A minimal benchmark for scalability, speed and accuracy of commonly used open source implementations (R packages, Python scikit-learn, H2O, xgboost, Spark MLlib etc.) of the top machine learning algorithms for binary classification (random forests, gradient boosted trees, deep neural networks etc.).", "domain": ["machine-learning", "data-science", "performance"], "enriched": true, "function": ["benchmarking", "machine-learning", "data-science"], "health_score": 20, "homepage": null, "language": "R", "license": "MIT", "license_family": "permissive", "maturity": "maintenance", "member_repos": ["szilard/benchm-ml"], "name": "szilard/benchm-ml", "platform": ["python", "jvm", "cross-platform"], "pushed_at": "2022-09-16T14:01:14+00:00", "repo": "szilard/benchm-ml", "stars": 1895, "tags": ["binary-classification", "scikit-learn", "xgboost", "h2o", "spark-mllib", "r-packages", "gradient-boosting", "random-forest", "deep-neural-networks", "scalability-benchmark"], "topics": ["machine-learning", "data-science", "r", "python", "gradient-boosting-machine", "random-forest", "deep-learning", "xgboost", "h2o", "spark"], "urls": [], "use_cases": ["compare speed of xgboost vs lightgbm vs h2o on classification", "find out which ML library scales to 10M rows without running out of memory", "benchmark random forest implementations in R and Python", "evaluate accuracy of gradient boosting vs deep neural networks on tabular data", "choose an ML tool for credit scoring or fraud detection workloads", "reproduce a minimal ML benchmark with my own tool"], "what_it_is": "A minimal benchmark comparing scalability, speed, and accuracy of open-source machine learning implementations (R packages, scikit-learn, H2O, xgboost, lightgbm, Spark MLlib, Vowpal Wabbit) for binary classification. It varies dataset sizes from 10K to 10M rows across linear models, random forests, gradient boosting, and deep neural networks.", "when_to_avoid": ["you need benchmarks on sparse, high-cardinality, or missing-data scenarios", "you need distributed multi-node scaling results", "you need up-to-date results - much of the benchmark dates to 2015 and the author moved to a newer project", "you need benchmarks for tasks other than binary classification"], "when_to_choose": ["you need evidence-based comparisons of classification libraries on tabular data", "you are evaluating ML tools for medium-to-large business datasets (10K-10M rows)", "you want a simple template to benchmark your own ML implementation"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/benchm-ml", "repo": "szilard/benchm-ml", "role": "main", "score": 32}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:05:50.954150+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-30T03:12:52.185922+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "89e284c3a20e13f741283b341f61df1f78a30cd9188be4c707d13774aa876bd1", "fetched_at": "2026-08-28T04:05:50.954150+00:00", "kind": "readme", "missing": false, "url": "https://github.com/szilard/benchm-ml"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:05:50.954150+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-30T03:12:52.185922+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "89e284c3a20e13f741283b341f61df1f78a30cd9188be4c707d13774aa876bd1", "fetched_at": "2026-08-28T04:05:50.954150+00:00", "kind": "readme", "missing": false, "url": "https://github.com/szilard/benchm-ml"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-30T03:12:52.185922+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "89e284c3a20e13f741283b341f61df1f78a30cd9188be4c707d13774aa876bd1", "fetched_at": "2026-08-28T04:05:50.954150+00:00", "kind": "readme", "missing": false, "url": "https://github.com/szilard/benchm-ml"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:05:50.954150+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:05:50.954150+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:05:50.954150+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-30T03:12:52.185922+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "89e284c3a20e13f741283b341f61df1f78a30cd9188be4c707d13774aa876bd1", "fetched_at": "2026-08-28T04:05:50.954150+00:00", "kind": "readme", "missing": false, "url": "https://github.com/szilard/benchm-ml"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:05:50.954150+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:05:50.954150+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-30T03:12:52.185922+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "89e284c3a20e13f741283b341f61df1f78a30cd9188be4c707d13774aa876bd1", "fetched_at": "2026-08-28T04:05:50.954150+00:00", "kind": "readme", "missing": false, "url": "https://github.com/szilard/benchm-ml"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:05:50.954150+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:05:50.954150+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:05:50.954150+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-30T03:12:52.185922+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "89e284c3a20e13f741283b341f61df1f78a30cd9188be4c707d13774aa876bd1", "fetched_at": "2026-08-28T04:05:50.954150+00:00", "kind": "readme", "missing": false, "url": "https://github.com/szilard/benchm-ml"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:05:50.954150+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:05:50.954150+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-30T03:12:52.185922+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "89e284c3a20e13f741283b341f61df1f78a30cd9188be4c707d13774aa876bd1", "fetched_at": "2026-08-28T04:05:50.954150+00:00", "kind": "readme", "missing": false, "url": "https://github.com/szilard/benchm-ml"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-30T03:12:52.185922+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "89e284c3a20e13f741283b341f61df1f78a30cd9188be4c707d13774aa876bd1", "fetched_at": "2026-08-28T04:05:50.954150+00:00", "kind": "readme", "missing": false, "url": "https://github.com/szilard/benchm-ml"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-30T03:12:52.185922+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "89e284c3a20e13f741283b341f61df1f78a30cd9188be4c707d13774aa876bd1", "fetched_at": "2026-08-28T04:05:50.954150+00:00", "kind": "readme", "missing": false, "url": "https://github.com/szilard/benchm-ml"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-30T03:12:52.185922+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "89e284c3a20e13f741283b341f61df1f78a30cd9188be4c707d13774aa876bd1", "fetched_at": "2026-08-28T04:05:50.954150+00:00", "kind": "readme", "missing": false, "url": "https://github.com/szilard/benchm-ml"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 0, "longevity": 100, "rhythm": 35}, "computed_at": "2026-09-03T02:20:16.233290+00:00", "flags": ["no_releases"], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 4177, "days_push": 1447, "days_rel": null, "gap_med": null, "n_releases_24m": 0}, "score": 32, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}