{"adoption": {"forks": 472, "observed_at": "2026-08-28T04:05:48.658890+00:00", "stars": 1882}, "canonical_url": "https://ross.abutalabs.com/products/tau-bench", "card": {"archived": false, "artifact_type": "dataset", "description": "τ-Bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains", "domain": ["artificial-intelligence", "large-language-models", "chatbots", "developer-tools", "tutorials"], "enriched": true, "function": ["benchmarking", "agent-framework", "llm-inference", "rag", "speech-recognition", "testing"], "health_score": 99, "homepage": "https://www.taubench.com", "language": "Python", "license": "MIT", "license_family": "permissive", "maturity": "active", "member_repos": ["sierra-research/tau2-bench", "sierra-research/tau-bench"], "name": "tau-bench", "platform": ["python", "cli", "cross-platform"], "pushed_at": "2026-08-18T17:07:54+00:00", "repo": "sierra-research/tau2-bench", "stars": 1882, "tags": ["llm-benchmark", "tool-use", "agent-evaluation", "conversational-agents", "voice-agents", "leaderboard", "pass-k-metric", "customer-service-domains", "ai-agents", "natural-language-processing"], "topics": ["benchmark", "llm", "ai", "language-model-agent", "conversational-agents"], "urls": [{"kind": "homepage", "url": "https://www.taubench.com"}], "use_cases": ["benchmark my llm agent on tool calling and customer service tasks", "evaluate how reliably an agent follows domain policies", "compare models on multi-turn conversational agent tasks", "test voice agents on real-time full-duplex conversations", "evaluate rag pipelines on knowledge-intensive banking tasks", "measure pass^k reliability of agent trajectories", "reproduce published agent benchmark scores"], "what_it_is": "τ-Bench (tau2-bench) is a Python benchmark for evaluating LLM agents on tool-agent-user interaction in real-world domains like retail, airline, telecom, and banking. It scores agents on conversations with simulated users, tool calls, knowledge retrieval, and policy adherence, including voice full-duplex evaluation, with a live public leaderboard.", "when_to_avoid": ["you need a general-purpose agent framework for building production agents rather than evaluating them", "your domain is unrelated to the provided retail/airline/telecom/banking scenarios and you cannot author custom tasks", "you need cheap, fast evals - simulated user LLM calls add cost and latency"], "when_to_choose": ["you need standardized, verifiable evaluation of tool-using conversational agents", "you want to compare your model against a public leaderboard", "you need both text and voice agent evaluation with realistic simulated users"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/repos/sierra-research/tau2-bench", "repo": "sierra-research/tau2-bench", "role": "main", "score": 79}, {"path": "/repos/sierra-research/tau-bench", "repo": "sierra-research/tau-bench", "role": "mirror", "score": 56}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:05:48.658890+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-30T03:13:44.976342+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "d185613c5b8293d161350c1cb8a1a02824c58aed1926aff33995d09d2ca54724", "fetched_at": "2026-08-28T04:05:48.658890+00:00", "kind": "readme", "missing": false, "url": "https://github.com/sierra-research/tau2-bench"}, {"content_hash": "d9d1f969d5d76ac461281917212b7ffa2455b52e2fed64e5f7ba90782987441e", "fetched_at": "2026-08-29T10:53:06.274282+00:00", "kind": "homepage", "missing": false, "url": "https://www.taubench.com"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:05:48.658890+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-30T03:13:44.976342+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "d185613c5b8293d161350c1cb8a1a02824c58aed1926aff33995d09d2ca54724", "fetched_at": "2026-08-28T04:05:48.658890+00:00", "kind": "readme", "missing": false, "url": "https://github.com/sierra-research/tau2-bench"}, {"content_hash": "d9d1f969d5d76ac461281917212b7ffa2455b52e2fed64e5f7ba90782987441e", "fetched_at": "2026-08-29T10:53:06.274282+00:00", "kind": "homepage", "missing": false, "url": "https://www.taubench.com"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-30T03:13:44.976342+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "d185613c5b8293d161350c1cb8a1a02824c58aed1926aff33995d09d2ca54724", "fetched_at": "2026-08-28T04:05:48.658890+00:00", "kind": "readme", "missing": false, "url": "https://github.com/sierra-research/tau2-bench"}, {"content_hash": "d9d1f969d5d76ac461281917212b7ffa2455b52e2fed64e5f7ba90782987441e", "fetched_at": "2026-08-29T10:53:06.274282+00:00", "kind": "homepage", "missing": false, "url": "https://www.taubench.com"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:05:48.658890+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:05:48.658890+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:05:48.658890+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-30T03:13:44.976342+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "d185613c5b8293d161350c1cb8a1a02824c58aed1926aff33995d09d2ca54724", "fetched_at": "2026-08-28T04:05:48.658890+00:00", "kind": "readme", "missing": false, "url": "https://github.com/sierra-research/tau2-bench"}, {"content_hash": "d9d1f969d5d76ac461281917212b7ffa2455b52e2fed64e5f7ba90782987441e", "fetched_at": "2026-08-29T10:53:06.274282+00:00", "kind": "homepage", "missing": false, "url": "https://www.taubench.com"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:05:48.658890+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:05:48.658890+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-30T03:13:44.976342+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "d185613c5b8293d161350c1cb8a1a02824c58aed1926aff33995d09d2ca54724", "fetched_at": "2026-08-28T04:05:48.658890+00:00", "kind": "readme", "missing": false, "url": "https://github.com/sierra-research/tau2-bench"}, {"content_hash": "d9d1f969d5d76ac461281917212b7ffa2455b52e2fed64e5f7ba90782987441e", "fetched_at": "2026-08-29T10:53:06.274282+00:00", "kind": "homepage", "missing": false, "url": "https://www.taubench.com"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:05:48.658890+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:05:48.658890+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:05:48.658890+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-30T03:13:44.976342+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "d185613c5b8293d161350c1cb8a1a02824c58aed1926aff33995d09d2ca54724", "fetched_at": "2026-08-28T04:05:48.658890+00:00", "kind": "readme", "missing": false, "url": "https://github.com/sierra-research/tau2-bench"}, {"content_hash": "d9d1f969d5d76ac461281917212b7ffa2455b52e2fed64e5f7ba90782987441e", "fetched_at": "2026-08-29T10:53:06.274282+00:00", "kind": "homepage", "missing": false, "url": "https://www.taubench.com"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:05:48.658890+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:05:48.658890+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-30T03:13:44.976342+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "d185613c5b8293d161350c1cb8a1a02824c58aed1926aff33995d09d2ca54724", "fetched_at": "2026-08-28T04:05:48.658890+00:00", "kind": "readme", "missing": false, "url": "https://github.com/sierra-research/tau2-bench"}, {"content_hash": "d9d1f969d5d76ac461281917212b7ffa2455b52e2fed64e5f7ba90782987441e", "fetched_at": "2026-08-29T10:53:06.274282+00:00", "kind": "homepage", "missing": false, "url": "https://www.taubench.com"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-30T03:13:44.976342+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "d185613c5b8293d161350c1cb8a1a02824c58aed1926aff33995d09d2ca54724", "fetched_at": "2026-08-28T04:05:48.658890+00:00", "kind": "readme", "missing": false, "url": "https://github.com/sierra-research/tau2-bench"}, {"content_hash": "d9d1f969d5d76ac461281917212b7ffa2455b52e2fed64e5f7ba90782987441e", "fetched_at": "2026-08-29T10:53:06.274282+00:00", "kind": "homepage", "missing": false, "url": "https://www.taubench.com"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-30T03:13:44.976342+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "d185613c5b8293d161350c1cb8a1a02824c58aed1926aff33995d09d2ca54724", "fetched_at": "2026-08-28T04:05:48.658890+00:00", "kind": "readme", "missing": false, "url": "https://github.com/sierra-research/tau2-bench"}, {"content_hash": "d9d1f969d5d76ac461281917212b7ffa2455b52e2fed64e5f7ba90782987441e", "fetched_at": "2026-08-29T10:53:06.274282+00:00", "kind": "homepage", "missing": false, "url": "https://www.taubench.com"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-30T03:13:44.976342+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "d185613c5b8293d161350c1cb8a1a02824c58aed1926aff33995d09d2ca54724", "fetched_at": "2026-08-28T04:05:48.658890+00:00", "kind": "readme", "missing": false, "url": "https://github.com/sierra-research/tau2-bench"}, {"content_hash": "d9d1f969d5d76ac461281917212b7ffa2455b52e2fed64e5f7ba90782987441e", "fetched_at": "2026-08-29T10:53:06.274282+00:00", "kind": "homepage", "missing": false, "url": "https://www.taubench.com"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 98, "longevity": 32, "rhythm": 82}, "computed_at": "2026-09-03T02:20:16.233290+00:00", "flags": [], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 450, "days_push": 15, "days_rel": 42, "gap_med": 83.0, "n_releases_24m": 5}, "score": 79, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}