{"adoption": {"forks": 408, "observed_at": "2026-08-28T04:07:28.251446+00:00", "stars": 2887}, "canonical_url": "https://ross.abutalabs.com/products/stanford-crfm-helm", "card": {"archived": false, "artifact_type": "framework", "description": "Holistic Evaluation of Language Models (HELM) is an open source Python framework created by the Center for Research on Foundation Models (CRFM) at Stanford for holistic, reproducible and transparent evaluation of foundation models, including large language models (LLMs) and multimodal models.", "domain": ["large-language-models", "machine-learning", "artificial-intelligence", "developer-tools", "analytics"], "enriched": true, "function": ["benchmarking", "machine-learning", "llm-inference", "data-science", "testing"], "health_score": 95, "homepage": "https://crfm.stanford.edu/helm", "language": "Python", "license": "Apache-2.0", "license_family": "permissive", "maturity": "maintenance", "member_repos": ["stanford-crfm/helm"], "name": "stanford-crfm/helm", "platform": ["python", "cli", "cross-platform"], "pushed_at": "2026-08-01T01:23:17+00:00", "repo": "stanford-crfm/helm", "stars": 2887, "tags": ["llm-evaluation", "foundation-models", "leaderboard", "benchmarks", "multimodal-evaluation", "reproducibility", "web-server"], "topics": [], "urls": [], "use_cases": ["evaluate LLMs on benchmarks like MMLU-Pro and GPQA", "compare models across providers on a leaderboard", "measure model bias, toxicity, and efficiency beyond accuracy", "inspect individual prompts and model responses in a web UI", "run reproducible model evaluation suites", "benchmark multimodal foundation models"], "what_it_is": "HELM is a Python framework from Stanford's CRFM for holistic, reproducible evaluation of foundation models including LLMs and multimodal models. It provides standardized benchmarks, a unified model interface across providers, metrics beyond accuracy, and a web UI/leaderboard for inspecting and comparing results.", "when_to_avoid": ["you need actively developed new features (HELM entered maintenance mode in June 2026)", "you only need lightweight eval harnesses for a single model", "you need non-Python tooling or real-time production monitoring"], "when_to_choose": ["you need rigorous, reproducible benchmarking of LLMs or multimodal models", "you want standardized datasets and metrics maintained by a research lab", "you need a unified interface to evaluate models from OpenAI, Anthropic, Google, and others", "you want a web leaderboard and UI for exploring evaluation results"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/stanford-crfm-helm", "repo": "stanford-crfm/helm", "role": "main", "score": 87}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:07:28.251446+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-30T07:35:22.532943+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "1740901d80cf4a7b1dd24eeced70c57af27f06eb7e52abe3b662d5ae35b45d72", "fetched_at": "2026-08-28T04:07:28.251446+00:00", "kind": "readme", "missing": false, "url": "https://github.com/stanford-crfm/helm"}, {"content_hash": "8bab84344c1601251d683c994d92d7222883726bb0e9935e86cd62a0fab72d17", "fetched_at": "2026-08-29T09:50:36.719675+00:00", "kind": "homepage", "missing": false, "url": "https://crfm.stanford.edu/helm"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:07:28.251446+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-30T07:35:22.532943+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "1740901d80cf4a7b1dd24eeced70c57af27f06eb7e52abe3b662d5ae35b45d72", "fetched_at": "2026-08-28T04:07:28.251446+00:00", "kind": "readme", "missing": false, "url": "https://github.com/stanford-crfm/helm"}, {"content_hash": "8bab84344c1601251d683c994d92d7222883726bb0e9935e86cd62a0fab72d17", "fetched_at": "2026-08-29T09:50:36.719675+00:00", "kind": "homepage", "missing": false, "url": "https://crfm.stanford.edu/helm"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-30T07:35:22.532943+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "1740901d80cf4a7b1dd24eeced70c57af27f06eb7e52abe3b662d5ae35b45d72", "fetched_at": "2026-08-28T04:07:28.251446+00:00", "kind": "readme", "missing": false, "url": "https://github.com/stanford-crfm/helm"}, {"content_hash": "8bab84344c1601251d683c994d92d7222883726bb0e9935e86cd62a0fab72d17", "fetched_at": "2026-08-29T09:50:36.719675+00:00", "kind": "homepage", "missing": false, "url": "https://crfm.stanford.edu/helm"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:07:28.251446+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:07:28.251446+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:07:28.251446+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-30T07:35:22.532943+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "1740901d80cf4a7b1dd24eeced70c57af27f06eb7e52abe3b662d5ae35b45d72", "fetched_at": "2026-08-28T04:07:28.251446+00:00", "kind": "readme", "missing": false, "url": "https://github.com/stanford-crfm/helm"}, {"content_hash": "8bab84344c1601251d683c994d92d7222883726bb0e9935e86cd62a0fab72d17", "fetched_at": "2026-08-29T09:50:36.719675+00:00", "kind": "homepage", "missing": false, "url": "https://crfm.stanford.edu/helm"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:07:28.251446+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:07:28.251446+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-30T07:35:22.532943+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "1740901d80cf4a7b1dd24eeced70c57af27f06eb7e52abe3b662d5ae35b45d72", "fetched_at": "2026-08-28T04:07:28.251446+00:00", "kind": "readme", "missing": false, "url": "https://github.com/stanford-crfm/helm"}, {"content_hash": "8bab84344c1601251d683c994d92d7222883726bb0e9935e86cd62a0fab72d17", "fetched_at": "2026-08-29T09:50:36.719675+00:00", "kind": "homepage", "missing": false, "url": "https://crfm.stanford.edu/helm"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:07:28.251446+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:07:28.251446+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:07:28.251446+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-30T07:35:22.532943+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "1740901d80cf4a7b1dd24eeced70c57af27f06eb7e52abe3b662d5ae35b45d72", "fetched_at": "2026-08-28T04:07:28.251446+00:00", "kind": "readme", "missing": false, "url": "https://github.com/stanford-crfm/helm"}, {"content_hash": "8bab84344c1601251d683c994d92d7222883726bb0e9935e86cd62a0fab72d17", "fetched_at": "2026-08-29T09:50:36.719675+00:00", "kind": "homepage", "missing": false, "url": "https://crfm.stanford.edu/helm"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:07:28.251446+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:07:28.251446+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-30T07:35:22.532943+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "1740901d80cf4a7b1dd24eeced70c57af27f06eb7e52abe3b662d5ae35b45d72", "fetched_at": "2026-08-28T04:07:28.251446+00:00", "kind": "readme", "missing": false, "url": "https://github.com/stanford-crfm/helm"}, {"content_hash": "8bab84344c1601251d683c994d92d7222883726bb0e9935e86cd62a0fab72d17", "fetched_at": "2026-08-29T09:50:36.719675+00:00", "kind": "homepage", "missing": false, "url": "https://crfm.stanford.edu/helm"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-30T07:35:22.532943+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "1740901d80cf4a7b1dd24eeced70c57af27f06eb7e52abe3b662d5ae35b45d72", "fetched_at": "2026-08-28T04:07:28.251446+00:00", "kind": "readme", "missing": false, "url": "https://github.com/stanford-crfm/helm"}, {"content_hash": "8bab84344c1601251d683c994d92d7222883726bb0e9935e86cd62a0fab72d17", "fetched_at": "2026-08-29T09:50:36.719675+00:00", "kind": "homepage", "missing": false, "url": "https://crfm.stanford.edu/helm"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-30T07:35:22.532943+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "1740901d80cf4a7b1dd24eeced70c57af27f06eb7e52abe3b662d5ae35b45d72", "fetched_at": "2026-08-28T04:07:28.251446+00:00", "kind": "readme", "missing": false, "url": "https://github.com/stanford-crfm/helm"}, {"content_hash": "8bab84344c1601251d683c994d92d7222883726bb0e9935e86cd62a0fab72d17", "fetched_at": "2026-08-29T09:50:36.719675+00:00", "kind": "homepage", "missing": false, "url": "https://crfm.stanford.edu/helm"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-30T07:35:22.532943+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "1740901d80cf4a7b1dd24eeced70c57af27f06eb7e52abe3b662d5ae35b45d72", "fetched_at": "2026-08-28T04:07:28.251446+00:00", "kind": "readme", "missing": false, "url": "https://github.com/stanford-crfm/helm"}, {"content_hash": "8bab84344c1601251d683c994d92d7222883726bb0e9935e86cd62a0fab72d17", "fetched_at": "2026-08-29T09:50:36.719675+00:00", "kind": "homepage", "missing": false, "url": "https://crfm.stanford.edu/helm"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 95, "longevity": 100, "rhythm": 69}, "computed_at": "2026-09-03T02:20:16.233290+00:00", "flags": [], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 1738, "days_push": 33, "days_rel": 125, "gap_med": 32, "n_releases_24m": 14}, "score": 87, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}