{"adoption": {"forks": 73, "observed_at": "2026-08-28T04:05:42.656397+00:00", "stars": 1836}, "canonical_url": "https://ross.abutalabs.com/products/bullshit-benchmark", "card": {"archived": false, "artifact_type": "dataset", "description": "BullshitBench measures whether AI models challenge nonsensical prompts instead of confidently answering them, created by Peter Gostev.", "domain": ["large-language-models", "artificial-intelligence", "data-visualization", "analytics"], "enriched": true, "function": ["benchmarking", "llm-inference", "data-visualization", "analytics"], "health_score": 80, "homepage": "https://x.com/petergostev", "language": "Python", "license": "MIT", "license_family": "permissive", "maturity": "active", "member_repos": ["petergpt/bullshit-benchmark"], "name": "petergpt/bullshit-benchmark", "platform": ["python", "cross-platform"], "pushed_at": "2026-08-26T01:58:05+00:00", "repo": "petergpt/bullshit-benchmark", "stars": 1836, "tags": ["llm-evaluation", "nonsense-detection", "sycophancy", "leaderboard", "model-evaluation", "hallucination-testing", "web-server"], "topics": [], "urls": [], "use_cases": ["evaluate whether an LLM challenges nonsensical prompts", "compare AI models on sycophancy and pushback behavior", "check if reasoning mode helps or hurts nonsense detection", "find benchmark results for new LLM releases", "visualize model detection rates across domains like finance and medical", "build an LLM evaluation leaderboard"], "what_it_is": "BullshitBench is a benchmark dataset and evaluation harness that tests whether AI models detect and push back on nonsensical prompts rather than confidently answering them. It includes question sets across multiple domains, judge-based scoring, and a public leaderboard viewer with charts comparing model performance.", "when_to_avoid": ["you need general-purpose capability benchmarks like MMLU or coding evals", "you require formal statistical significance testing of model differences", "you need a benchmark for tasks other than nonsense detection"], "when_to_choose": ["you want to measure how honestly a model responds to invalid premises", "you need domain-specific nonsense detection scores across software, finance, legal, medical, and physics", "you want an existing leaderboard instead of building an eval from scratch"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/bullshit-benchmark", "repo": "petergpt/bullshit-benchmark", "role": "main", "score": 59}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:05:42.656397+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-30T03:18:29.021903+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "cd45e5fb8a8d1539a83624f5bb96f579f51f6ca5abb8828f33bffc701435747b", "fetched_at": "2026-08-28T04:05:42.656397+00:00", "kind": "readme", "missing": false, "url": "https://github.com/petergpt/bullshit-benchmark"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:05:42.656397+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-30T03:18:29.021903+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "cd45e5fb8a8d1539a83624f5bb96f579f51f6ca5abb8828f33bffc701435747b", "fetched_at": "2026-08-28T04:05:42.656397+00:00", "kind": "readme", "missing": false, "url": "https://github.com/petergpt/bullshit-benchmark"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-30T03:18:29.021903+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "cd45e5fb8a8d1539a83624f5bb96f579f51f6ca5abb8828f33bffc701435747b", "fetched_at": "2026-08-28T04:05:42.656397+00:00", "kind": "readme", "missing": false, "url": "https://github.com/petergpt/bullshit-benchmark"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:05:42.656397+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:05:42.656397+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:05:42.656397+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-30T03:18:29.021903+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "cd45e5fb8a8d1539a83624f5bb96f579f51f6ca5abb8828f33bffc701435747b", "fetched_at": "2026-08-28T04:05:42.656397+00:00", "kind": "readme", "missing": false, "url": "https://github.com/petergpt/bullshit-benchmark"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:05:42.656397+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:05:42.656397+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-30T03:18:29.021903+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "cd45e5fb8a8d1539a83624f5bb96f579f51f6ca5abb8828f33bffc701435747b", "fetched_at": "2026-08-28T04:05:42.656397+00:00", "kind": "readme", "missing": false, "url": "https://github.com/petergpt/bullshit-benchmark"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:05:42.656397+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:05:42.656397+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:05:42.656397+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-30T03:18:29.021903+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "cd45e5fb8a8d1539a83624f5bb96f579f51f6ca5abb8828f33bffc701435747b", "fetched_at": "2026-08-28T04:05:42.656397+00:00", "kind": "readme", "missing": false, "url": "https://github.com/petergpt/bullshit-benchmark"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:05:42.656397+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:05:42.656397+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-30T03:18:29.021903+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "cd45e5fb8a8d1539a83624f5bb96f579f51f6ca5abb8828f33bffc701435747b", "fetched_at": "2026-08-28T04:05:42.656397+00:00", "kind": "readme", "missing": false, "url": "https://github.com/petergpt/bullshit-benchmark"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-30T03:18:29.021903+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "cd45e5fb8a8d1539a83624f5bb96f579f51f6ca5abb8828f33bffc701435747b", "fetched_at": "2026-08-28T04:05:42.656397+00:00", "kind": "readme", "missing": false, "url": "https://github.com/petergpt/bullshit-benchmark"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-30T03:18:29.021903+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "cd45e5fb8a8d1539a83624f5bb96f579f51f6ca5abb8828f33bffc701435747b", "fetched_at": "2026-08-28T04:05:42.656397+00:00", "kind": "readme", "missing": false, "url": "https://github.com/petergpt/bullshit-benchmark"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-30T03:18:29.021903+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "cd45e5fb8a8d1539a83624f5bb96f579f51f6ca5abb8828f33bffc701435747b", "fetched_at": "2026-08-28T04:05:42.656397+00:00", "kind": "readme", "missing": false, "url": "https://github.com/petergpt/bullshit-benchmark"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 99, "longevity": 13, "rhythm": 35}, "computed_at": "2026-09-03T02:20:16.233290+00:00", "flags": ["no_releases"], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 190, "days_push": 8, "days_rel": null, "gap_med": null, "n_releases_24m": 0}, "score": 59, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}