{"adoption": {"forks": 22, "observed_at": "2026-08-28T04:05:05.020097+00:00", "stars": 1567}, "canonical_url": "https://ross.abutalabs.com/products/llm_benchmark", "card": {"archived": false, "artifact_type": "dataset", "description": null, "domain": ["large-language-models", "artificial-intelligence", "machine-learning"], "enriched": true, "function": ["benchmarking", "llm-inference"], "health_score": 80, "homepage": null, "language": null, "license": null, "license_family": "other", "maturity": "active", "member_repos": ["llm2014/llm_benchmark"], "name": "llm2014/llm_benchmark", "platform": ["python"], "pushed_at": "2026-08-26T16:28:45+00:00", "repo": "llm2014/llm_benchmark", "stars": 1567, "tags": ["llm-evaluation", "leaderboard", "private-question-bank", "reasoning-benchmark", "monthly-updates", "chinese-language", "evaluation", "web-server"], "topics": [], "urls": [], "use_cases": ["compare reasoning ability of different LLMs", "track how LLMs improve over time", "find which model is best at math and logic puzzles", "view a leaderboard of large language model scores", "evaluate models on programming and deduction tasks"], "what_it_is": "A personal, long-running benchmark that tracks large language model performance on logic, math, programming, and intuition tasks using a private, rolling question bank of ~28 questions. Results are published as a monthly leaderboard with scoring based on multi-point rubrics.", "when_to_avoid": ["you need a comprehensive or authoritative academic benchmark", "you require publicly available test questions to run yourself", "you need domain-specific evaluation outside reasoning tasks"], "when_to_choose": ["you want an independent, long-term view of LLM reasoning trends", "you need a leaderboard covering logic, math, and coding ability", "you want to see how specific models evolve month over month"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/llm_benchmark", "repo": "llm2014/llm_benchmark", "role": "main", "score": 65}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:05:05.020097+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-30T03:59:01.011433+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "b4bd052f13c45775f8cbaff398fbcf31f1bc9a56eb75701eb8e1aba34f1cdb53", "fetched_at": "2026-08-28T04:05:05.020097+00:00", "kind": "readme", "missing": false, "url": "https://github.com/llm2014/llm_benchmark"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:05:05.020097+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-30T03:59:01.011433+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "b4bd052f13c45775f8cbaff398fbcf31f1bc9a56eb75701eb8e1aba34f1cdb53", "fetched_at": "2026-08-28T04:05:05.020097+00:00", "kind": "readme", "missing": false, "url": "https://github.com/llm2014/llm_benchmark"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-30T03:59:01.011433+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "b4bd052f13c45775f8cbaff398fbcf31f1bc9a56eb75701eb8e1aba34f1cdb53", "fetched_at": "2026-08-28T04:05:05.020097+00:00", "kind": "readme", "missing": false, "url": "https://github.com/llm2014/llm_benchmark"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:05:05.020097+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:05:05.020097+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:05:05.020097+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-30T03:59:01.011433+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "b4bd052f13c45775f8cbaff398fbcf31f1bc9a56eb75701eb8e1aba34f1cdb53", "fetched_at": "2026-08-28T04:05:05.020097+00:00", "kind": "readme", "missing": false, "url": "https://github.com/llm2014/llm_benchmark"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:05:05.020097+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:05:05.020097+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-30T03:59:01.011433+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "b4bd052f13c45775f8cbaff398fbcf31f1bc9a56eb75701eb8e1aba34f1cdb53", "fetched_at": "2026-08-28T04:05:05.020097+00:00", "kind": "readme", "missing": false, "url": "https://github.com/llm2014/llm_benchmark"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:05:05.020097+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:05:05.020097+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:05:05.020097+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-30T03:59:01.011433+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "b4bd052f13c45775f8cbaff398fbcf31f1bc9a56eb75701eb8e1aba34f1cdb53", "fetched_at": "2026-08-28T04:05:05.020097+00:00", "kind": "readme", "missing": false, "url": "https://github.com/llm2014/llm_benchmark"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:05:05.020097+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:05:05.020097+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-30T03:59:01.011433+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "b4bd052f13c45775f8cbaff398fbcf31f1bc9a56eb75701eb8e1aba34f1cdb53", "fetched_at": "2026-08-28T04:05:05.020097+00:00", "kind": "readme", "missing": false, "url": "https://github.com/llm2014/llm_benchmark"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-30T03:59:01.011433+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "b4bd052f13c45775f8cbaff398fbcf31f1bc9a56eb75701eb8e1aba34f1cdb53", "fetched_at": "2026-08-28T04:05:05.020097+00:00", "kind": "readme", "missing": false, "url": "https://github.com/llm2014/llm_benchmark"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-30T03:59:01.011433+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "b4bd052f13c45775f8cbaff398fbcf31f1bc9a56eb75701eb8e1aba34f1cdb53", "fetched_at": "2026-08-28T04:05:05.020097+00:00", "kind": "readme", "missing": false, "url": "https://github.com/llm2014/llm_benchmark"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-30T03:59:01.011433+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "b4bd052f13c45775f8cbaff398fbcf31f1bc9a56eb75701eb8e1aba34f1cdb53", "fetched_at": "2026-08-28T04:05:05.020097+00:00", "kind": "readme", "missing": false, "url": "https://github.com/llm2014/llm_benchmark"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 99, "longevity": 40, "rhythm": 35}, "computed_at": "2026-09-02T17:46:02.011165+00:00", "flags": ["no_releases", "no_license"], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 572, "days_push": 7, "days_rel": null, "gap_med": null, "n_releases_24m": 0}, "score": 65, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}