{"adoption": {"forks": 506, "observed_at": "2026-08-28T04:08:55.054851+00:00", "stars": 4612}, "canonical_url": "https://ross.abutalabs.com/products/simple-evals", "card": {"archived": false, "artifact_type": "library", "description": null, "domain": ["large-language-models", "machine-learning", "developer-tools"], "enriched": true, "function": ["machine-learning", "benchmarking", "llm-inference"], "health_score": 67, "homepage": null, "language": "Python", "license": "MIT", "license_family": "permissive", "maturity": "maintenance", "member_repos": ["openai/simple-evals"], "name": "openai/simple-evals", "platform": ["python"], "pushed_at": "2026-04-22T22:16:18+00:00", "repo": "openai/simple-evals", "stars": 4612, "tags": ["llm-evaluation", "benchmarks", "openai", "simpleqa", "healthbench", "browsecomp"], "topics": [], "urls": [], "use_cases": ["evaluate an LLM on MMLU or GPQA benchmarks", "measure model factual accuracy with SimpleQA", "run HealthBench evaluations on a language model", "reproduce OpenAI's published benchmark numbers", "compare model performance on math and coding tasks", "benchmark a new model against o3 or o4-mini results"], "what_it_is": "A lightweight Python library from OpenAI for evaluating language models against benchmarks like MMLU, GPQA, MATH, HumanEval, SimpleQA, HealthBench, and BrowseComp. It hosts reference implementations used to transparently publish model accuracy numbers.", "when_to_avoid": ["you need a full-featured, actively maintained eval framework with new benchmarks", "you need evaluations updated for the latest models, since the repo is deprecated for new results", "you need a production evaluation pipeline with extensive integrations"], "when_to_choose": ["you want lightweight, reference-quality LLM evaluation code", "you need to reproduce OpenAI's published benchmark scores", "you want to evaluate models on SimpleQA, HealthBench, or BrowseComp"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/simple-evals", "repo": "openai/simple-evals", "role": "main", "score": 60}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:08:55.054851+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-29T18:19:42.744962+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "e97a732b51b2aba8f9fd2341844085c66b3c598864705ebc19d008442281e5a2", "fetched_at": "2026-08-28T04:08:55.054851+00:00", "kind": "readme", "missing": false, "url": "https://github.com/openai/simple-evals"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:08:55.054851+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-29T18:19:42.744962+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "e97a732b51b2aba8f9fd2341844085c66b3c598864705ebc19d008442281e5a2", "fetched_at": "2026-08-28T04:08:55.054851+00:00", "kind": "readme", "missing": false, "url": "https://github.com/openai/simple-evals"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-29T18:19:42.744962+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "e97a732b51b2aba8f9fd2341844085c66b3c598864705ebc19d008442281e5a2", "fetched_at": "2026-08-28T04:08:55.054851+00:00", "kind": "readme", "missing": false, "url": "https://github.com/openai/simple-evals"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:08:55.054851+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:08:55.054851+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:08:55.054851+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-29T18:19:42.744962+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "e97a732b51b2aba8f9fd2341844085c66b3c598864705ebc19d008442281e5a2", "fetched_at": "2026-08-28T04:08:55.054851+00:00", "kind": "readme", "missing": false, "url": "https://github.com/openai/simple-evals"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:08:55.054851+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:08:55.054851+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-29T18:19:42.744962+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "e97a732b51b2aba8f9fd2341844085c66b3c598864705ebc19d008442281e5a2", "fetched_at": "2026-08-28T04:08:55.054851+00:00", "kind": "readme", "missing": false, "url": "https://github.com/openai/simple-evals"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:08:55.054851+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:08:55.054851+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:08:55.054851+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-29T18:19:42.744962+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "e97a732b51b2aba8f9fd2341844085c66b3c598864705ebc19d008442281e5a2", "fetched_at": "2026-08-28T04:08:55.054851+00:00", "kind": "readme", "missing": false, "url": "https://github.com/openai/simple-evals"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:08:55.054851+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:08:55.054851+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-29T18:19:42.744962+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "e97a732b51b2aba8f9fd2341844085c66b3c598864705ebc19d008442281e5a2", "fetched_at": "2026-08-28T04:08:55.054851+00:00", "kind": "readme", "missing": false, "url": "https://github.com/openai/simple-evals"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-29T18:19:42.744962+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "e97a732b51b2aba8f9fd2341844085c66b3c598864705ebc19d008442281e5a2", "fetched_at": "2026-08-28T04:08:55.054851+00:00", "kind": "readme", "missing": false, "url": "https://github.com/openai/simple-evals"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-29T18:19:42.744962+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "e97a732b51b2aba8f9fd2341844085c66b3c598864705ebc19d008442281e5a2", "fetched_at": "2026-08-28T04:08:55.054851+00:00", "kind": "readme", "missing": false, "url": "https://github.com/openai/simple-evals"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-29T18:19:42.744962+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "e97a732b51b2aba8f9fd2341844085c66b3c598864705ebc19d008442281e5a2", "fetched_at": "2026-08-28T04:08:55.054851+00:00", "kind": "readme", "missing": false, "url": "https://github.com/openai/simple-evals"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 78, "longevity": 62, "rhythm": 35}, "computed_at": "2026-09-03T02:20:16.233290+00:00", "flags": ["no_releases"], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 874, "days_push": 133, "days_rel": null, "gap_med": null, "n_releases_24m": 0}, "score": 60, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}