{"adoption": {"forks": 221, "observed_at": "2026-08-28T04:07:23.935679+00:00", "stars": 2819}, "canonical_url": "https://ross.abutalabs.com/products/promptbench", "card": {"archived": true, "artifact_type": "framework", "description": "A unified evaluation framework for large language models", "domain": ["large-language-models", "machine-learning", "artificial-intelligence"], "enriched": true, "function": ["benchmarking", "llm-inference", "prompt-engineering", "testing"], "health_score": 10, "homepage": "http://aka.ms/promptbench", "language": "Python", "license": "MIT", "license_family": "permissive", "maturity": "maintenance", "member_repos": ["microsoftarchive/promptbench"], "name": "microsoftarchive/promptbench", "platform": ["python", "cross-platform"], "pushed_at": "2026-02-20T18:55:32+00:00", "repo": "microsoftarchive/promptbench", "stars": 2819, "tags": ["llm-evaluation", "adversarial-robustness", "prompt-robustness", "benchmarking-suite", "microsoft"], "topics": ["adversarial-attacks", "chatgpt", "evaluation", "large-language-models", "robustness", "prompt", "prompt-engineering", "benchmark"], "urls": [], "use_cases": ["benchmark llm performance across multiple datasets", "evaluate prompt robustness against adversarial attacks", "compare gpt-4o, gemini, mistral and open-source models", "measure how prompt wording affects model accuracy", "run multi-prompt evaluation efficiently", "evaluate multimodal llm capabilities"], "what_it_is": "PromptBench is a unified Python library from Microsoft for evaluating and understanding large language models across many datasets, models, and prompt techniques. It includes adversarial prompt attack modules and robustness analysis to measure how LLM performance degrades under perturbed prompts.", "when_to_avoid": ["you need a production serving or inference framework rather than evaluation", "you want lightweight ad-hoc testing of a single model without benchmark overhead", "you need a hosted leaderboard service rather than a local library"], "when_to_choose": ["you need a standardized harness to compare many LLMs on academic benchmarks", "you want to study prompt sensitivity and adversarial robustness of LLMs", "you need reproducible evaluation with support for major commercial and open models"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/promptbench", "repo": "microsoftarchive/promptbench", "role": "main", "score": 10}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:07:23.935679+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-30T07:38:33.588509+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "bda340c071b21de836ec75ce6f3d136316596132514ccdf5b00ea2349280b2bd", "fetched_at": "2026-08-28T04:07:23.935679+00:00", "kind": "readme", "missing": false, "url": "https://github.com/microsoftarchive/promptbench"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:07:23.935679+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-30T07:38:33.588509+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "bda340c071b21de836ec75ce6f3d136316596132514ccdf5b00ea2349280b2bd", "fetched_at": "2026-08-28T04:07:23.935679+00:00", "kind": "readme", "missing": false, "url": "https://github.com/microsoftarchive/promptbench"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-30T07:38:33.588509+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "bda340c071b21de836ec75ce6f3d136316596132514ccdf5b00ea2349280b2bd", "fetched_at": "2026-08-28T04:07:23.935679+00:00", "kind": "readme", "missing": false, "url": "https://github.com/microsoftarchive/promptbench"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:07:23.935679+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:07:23.935679+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:07:23.935679+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-30T07:38:33.588509+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "bda340c071b21de836ec75ce6f3d136316596132514ccdf5b00ea2349280b2bd", "fetched_at": "2026-08-28T04:07:23.935679+00:00", "kind": "readme", "missing": false, "url": "https://github.com/microsoftarchive/promptbench"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:07:23.935679+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:07:23.935679+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-30T07:38:33.588509+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "bda340c071b21de836ec75ce6f3d136316596132514ccdf5b00ea2349280b2bd", "fetched_at": "2026-08-28T04:07:23.935679+00:00", "kind": "readme", "missing": false, "url": "https://github.com/microsoftarchive/promptbench"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:07:23.935679+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:07:23.935679+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:07:23.935679+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-30T07:38:33.588509+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "bda340c071b21de836ec75ce6f3d136316596132514ccdf5b00ea2349280b2bd", "fetched_at": "2026-08-28T04:07:23.935679+00:00", "kind": "readme", "missing": false, "url": "https://github.com/microsoftarchive/promptbench"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:07:23.935679+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:07:23.935679+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-30T07:38:33.588509+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "bda340c071b21de836ec75ce6f3d136316596132514ccdf5b00ea2349280b2bd", "fetched_at": "2026-08-28T04:07:23.935679+00:00", "kind": "readme", "missing": false, "url": "https://github.com/microsoftarchive/promptbench"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-30T07:38:33.588509+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "bda340c071b21de836ec75ce6f3d136316596132514ccdf5b00ea2349280b2bd", "fetched_at": "2026-08-28T04:07:23.935679+00:00", "kind": "readme", "missing": false, "url": "https://github.com/microsoftarchive/promptbench"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-30T07:38:33.588509+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "bda340c071b21de836ec75ce6f3d136316596132514ccdf5b00ea2349280b2bd", "fetched_at": "2026-08-28T04:07:23.935679+00:00", "kind": "readme", "missing": false, "url": "https://github.com/microsoftarchive/promptbench"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-30T07:38:33.588509+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "bda340c071b21de836ec75ce6f3d136316596132514ccdf5b00ea2349280b2bd", "fetched_at": "2026-08-28T04:07:23.935679+00:00", "kind": "readme", "missing": false, "url": "https://github.com/microsoftarchive/promptbench"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 68, "longevity": 84, "rhythm": 35}, "computed_at": "2026-09-03T02:20:16.233290+00:00", "flags": ["no_releases", "archived"], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 1177, "days_push": 194, "days_rel": null, "gap_med": null, "n_releases_24m": 0}, "score": 10, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}