{"adoption": {"forks": 109, "observed_at": "2026-08-28T04:05:18.956263+00:00", "stars": 1665}, "canonical_url": "https://ross.abutalabs.com/products/hle", "card": {"archived": false, "artifact_type": "dataset", "description": "Humanity's Last Exam", "domain": ["artificial-intelligence", "large-language-models", "education", "mathematics"], "enriched": true, "function": ["benchmarking", "llm-inference", "machine-learning", "data-science"], "health_score": 77, "homepage": "https://lastexam.ai", "language": "Python", "license": "MIT", "license_family": "permissive", "maturity": "active", "member_repos": ["centerforaisafety/hle"], "name": "centerforaisafety/hle", "platform": ["python", "cross-platform"], "pushed_at": "2026-08-01T08:53:41+00:00", "repo": "centerforaisafety/hle", "stars": 1665, "tags": ["llm-benchmark", "evaluation", "multimodal", "academic-benchmark", "huggingface-dataset", "model-evaluation", "natural-language-processing"], "topics": [], "urls": [], "use_cases": ["evaluate frontier LLMs on expert-level academic questions", "benchmark reasoning models across math, humanities, and sciences", "compare model accuracy and calibration on a hard closed-ended benchmark", "filter benchmark data out of training corpora using the canary string", "run automated grading of model answers with an LLM judge", "contribute new questions to the rolling HLE-Rolling version"], "what_it_is": "Humanity's Last Exam (HLE) is a multi-modal benchmark of 2,500 expert-written questions across dozens of academic subjects, designed to test frontier AI models at the edge of human knowledge. The repository provides the dataset (hosted on Hugging Face) plus simple evaluation scripts for running models and grading their answers with an LLM judge.", "when_to_avoid": ["you need a benchmark for everyday or beginner-level tasks", "you want a training dataset - the benchmark data must not appear in training corpora", "you need cheap, fast evaluation - frontier questions require large token budgets and LLM-judge calls"], "when_to_choose": ["you need a challenging, broad-coverage benchmark for evaluating large language models", "you want automated, reproducible evaluation with multiple-choice and short-answer grading", "you need a citable, peer-reviewed benchmark (published in Nature) for research"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/hle", "repo": "centerforaisafety/hle", "role": "main", "score": 63}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:05:18.956263+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-30T03:43:01.483824+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "fd3bd0049e1a2b39244d8240a63d0ccfe037b0b6e5979d5826e1d1cf4203ee88", "fetched_at": "2026-08-28T04:05:18.956263+00:00", "kind": "readme", "missing": false, "url": "https://github.com/centerforaisafety/hle"}, {"content_hash": "8492effff88c024e61e16f5d0bfa5c9f522525ae0341a3de8666d9478b0d334b", "fetched_at": "2026-08-29T11:16:29.427275+00:00", "kind": "homepage", "missing": false, "url": "https://lastexam.ai"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:05:18.956263+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-30T03:43:01.483824+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "fd3bd0049e1a2b39244d8240a63d0ccfe037b0b6e5979d5826e1d1cf4203ee88", "fetched_at": "2026-08-28T04:05:18.956263+00:00", "kind": "readme", "missing": false, "url": "https://github.com/centerforaisafety/hle"}, {"content_hash": "8492effff88c024e61e16f5d0bfa5c9f522525ae0341a3de8666d9478b0d334b", "fetched_at": "2026-08-29T11:16:29.427275+00:00", "kind": "homepage", "missing": false, "url": "https://lastexam.ai"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-30T03:43:01.483824+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "fd3bd0049e1a2b39244d8240a63d0ccfe037b0b6e5979d5826e1d1cf4203ee88", "fetched_at": "2026-08-28T04:05:18.956263+00:00", "kind": "readme", "missing": false, "url": "https://github.com/centerforaisafety/hle"}, {"content_hash": "8492effff88c024e61e16f5d0bfa5c9f522525ae0341a3de8666d9478b0d334b", "fetched_at": "2026-08-29T11:16:29.427275+00:00", "kind": "homepage", "missing": false, "url": "https://lastexam.ai"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:05:18.956263+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:05:18.956263+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:05:18.956263+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-30T03:43:01.483824+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "fd3bd0049e1a2b39244d8240a63d0ccfe037b0b6e5979d5826e1d1cf4203ee88", "fetched_at": "2026-08-28T04:05:18.956263+00:00", "kind": "readme", "missing": false, "url": "https://github.com/centerforaisafety/hle"}, {"content_hash": "8492effff88c024e61e16f5d0bfa5c9f522525ae0341a3de8666d9478b0d334b", "fetched_at": "2026-08-29T11:16:29.427275+00:00", "kind": "homepage", "missing": false, "url": "https://lastexam.ai"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:05:18.956263+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:05:18.956263+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-30T03:43:01.483824+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "fd3bd0049e1a2b39244d8240a63d0ccfe037b0b6e5979d5826e1d1cf4203ee88", "fetched_at": "2026-08-28T04:05:18.956263+00:00", "kind": "readme", "missing": false, "url": "https://github.com/centerforaisafety/hle"}, {"content_hash": "8492effff88c024e61e16f5d0bfa5c9f522525ae0341a3de8666d9478b0d334b", "fetched_at": "2026-08-29T11:16:29.427275+00:00", "kind": "homepage", "missing": false, "url": "https://lastexam.ai"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:05:18.956263+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:05:18.956263+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:05:18.956263+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-30T03:43:01.483824+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "fd3bd0049e1a2b39244d8240a63d0ccfe037b0b6e5979d5826e1d1cf4203ee88", "fetched_at": "2026-08-28T04:05:18.956263+00:00", "kind": "readme", "missing": false, "url": "https://github.com/centerforaisafety/hle"}, {"content_hash": "8492effff88c024e61e16f5d0bfa5c9f522525ae0341a3de8666d9478b0d334b", "fetched_at": "2026-08-29T11:16:29.427275+00:00", "kind": "homepage", "missing": false, "url": "https://lastexam.ai"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:05:18.956263+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:05:18.956263+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-30T03:43:01.483824+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "fd3bd0049e1a2b39244d8240a63d0ccfe037b0b6e5979d5826e1d1cf4203ee88", "fetched_at": "2026-08-28T04:05:18.956263+00:00", "kind": "readme", "missing": false, "url": "https://github.com/centerforaisafety/hle"}, {"content_hash": "8492effff88c024e61e16f5d0bfa5c9f522525ae0341a3de8666d9478b0d334b", "fetched_at": "2026-08-29T11:16:29.427275+00:00", "kind": "homepage", "missing": false, "url": "https://lastexam.ai"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-30T03:43:01.483824+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "fd3bd0049e1a2b39244d8240a63d0ccfe037b0b6e5979d5826e1d1cf4203ee88", "fetched_at": "2026-08-28T04:05:18.956263+00:00", "kind": "readme", "missing": false, "url": "https://github.com/centerforaisafety/hle"}, {"content_hash": "8492effff88c024e61e16f5d0bfa5c9f522525ae0341a3de8666d9478b0d334b", "fetched_at": "2026-08-29T11:16:29.427275+00:00", "kind": "homepage", "missing": false, "url": "https://lastexam.ai"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-30T03:43:01.483824+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "fd3bd0049e1a2b39244d8240a63d0ccfe037b0b6e5979d5826e1d1cf4203ee88", "fetched_at": "2026-08-28T04:05:18.956263+00:00", "kind": "readme", "missing": false, "url": "https://github.com/centerforaisafety/hle"}, {"content_hash": "8492effff88c024e61e16f5d0bfa5c9f522525ae0341a3de8666d9478b0d334b", "fetched_at": "2026-08-29T11:16:29.427275+00:00", "kind": "homepage", "missing": false, "url": "https://lastexam.ai"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-30T03:43:01.483824+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "fd3bd0049e1a2b39244d8240a63d0ccfe037b0b6e5979d5826e1d1cf4203ee88", "fetched_at": "2026-08-28T04:05:18.956263+00:00", "kind": "readme", "missing": false, "url": "https://github.com/centerforaisafety/hle"}, {"content_hash": "8492effff88c024e61e16f5d0bfa5c9f522525ae0341a3de8666d9478b0d334b", "fetched_at": "2026-08-29T11:16:29.427275+00:00", "kind": "homepage", "missing": false, "url": "https://lastexam.ai"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 95, "longevity": 42, "rhythm": 35}, "computed_at": "2026-09-03T02:20:16.233290+00:00", "flags": ["no_releases"], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 588, "days_push": 32, "days_rel": null, "gap_med": null, "n_releases_24m": 0}, "score": 63, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}