{"adoption": {"forks": 81, "observed_at": "2026-08-28T04:03:19.063223+00:00", "stars": 1035}, "canonical_url": "https://ross.abutalabs.com/products/longmemeval", "card": {"archived": false, "artifact_type": "dataset", "description": "Benchmarking Chat Assistants on Long-Term Interactive Memory (ICLR 2025)", "domain": ["large-language-models", "artificial-intelligence", "chatbots", "testing"], "enriched": true, "function": ["benchmarking", "rag", "llm-inference", "chatbot"], "health_score": 69, "homepage": null, "language": "Python", "license": "MIT", "license_family": "permissive", "maturity": "active", "member_repos": ["xiaowu0162/LongMemEval"], "name": "xiaowu0162/LongMemEval", "platform": ["python", "cli"], "pushed_at": "2026-05-11T22:49:24+00:00", "repo": "xiaowu0162/LongMemEval", "stars": 1035, "tags": ["long-term-memory", "chat-assistants", "iclr-2025", "needle-in-a-haystack", "memory-benchmark", "multi-session-dialogue", "evaluation", "retrieval-augmented-generation", "natural-language-processing"], "topics": [], "urls": [], "use_cases": ["evaluate long-term memory of chat assistants", "benchmark llm memory across multiple sessions", "test temporal reasoning in conversational ai", "measure knowledge update handling in chatbots", "compare rag memory systems on long chat histories", "evaluate abstention when information is missing"], "what_it_is": "LongMemEval is a benchmark of 500 high-quality questions for evaluating the long-term memory abilities of chat assistants across five skills including information extraction, multi-session reasoning, knowledge updates, temporal reasoning, and abstention. The repository provides the dataset, evaluation code, and tooling for running chat systems against timestamped multi-session chat histories.", "when_to_avoid": ["you need a general-purpose LLM benchmark unrelated to memory", "you want a training dataset rather than an evaluation benchmark", "you need agentic long-term memory evaluation, for which LongMemEval-V2 is more appropriate"], "when_to_choose": ["you need a standardized benchmark for long-term conversational memory", "you are building or comparing memory-augmented chat assistants", "you want needle-in-a-haystack style evaluation over multi-session chat histories"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/longmemeval", "repo": "xiaowu0162/LongMemEval", "role": "main", "score": 58}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:03:19.063223+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-30T07:05:10.718689+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "90615d215ad99d39b5bd3dbb67f06d81aa935e5cdec81fa7010f0d87d313ef7c", "fetched_at": "2026-08-28T04:03:19.063223+00:00", "kind": "readme", "missing": false, "url": "https://github.com/xiaowu0162/LongMemEval"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:03:19.063223+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-30T07:05:10.718689+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "90615d215ad99d39b5bd3dbb67f06d81aa935e5cdec81fa7010f0d87d313ef7c", "fetched_at": "2026-08-28T04:03:19.063223+00:00", "kind": "readme", "missing": false, "url": "https://github.com/xiaowu0162/LongMemEval"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-30T07:05:10.718689+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "90615d215ad99d39b5bd3dbb67f06d81aa935e5cdec81fa7010f0d87d313ef7c", "fetched_at": "2026-08-28T04:03:19.063223+00:00", "kind": "readme", "missing": false, "url": "https://github.com/xiaowu0162/LongMemEval"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:03:19.063223+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:03:19.063223+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:03:19.063223+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-30T07:05:10.718689+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "90615d215ad99d39b5bd3dbb67f06d81aa935e5cdec81fa7010f0d87d313ef7c", "fetched_at": "2026-08-28T04:03:19.063223+00:00", "kind": "readme", "missing": false, "url": "https://github.com/xiaowu0162/LongMemEval"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:03:19.063223+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:03:19.063223+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-30T07:05:10.718689+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "90615d215ad99d39b5bd3dbb67f06d81aa935e5cdec81fa7010f0d87d313ef7c", "fetched_at": "2026-08-28T04:03:19.063223+00:00", "kind": "readme", "missing": false, "url": "https://github.com/xiaowu0162/LongMemEval"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:03:19.063223+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:03:19.063223+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:03:19.063223+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-30T07:05:10.718689+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "90615d215ad99d39b5bd3dbb67f06d81aa935e5cdec81fa7010f0d87d313ef7c", "fetched_at": "2026-08-28T04:03:19.063223+00:00", "kind": "readme", "missing": false, "url": "https://github.com/xiaowu0162/LongMemEval"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:03:19.063223+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:03:19.063223+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-30T07:05:10.718689+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "90615d215ad99d39b5bd3dbb67f06d81aa935e5cdec81fa7010f0d87d313ef7c", "fetched_at": "2026-08-28T04:03:19.063223+00:00", "kind": "readme", "missing": false, "url": "https://github.com/xiaowu0162/LongMemEval"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-30T07:05:10.718689+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "90615d215ad99d39b5bd3dbb67f06d81aa935e5cdec81fa7010f0d87d313ef7c", "fetched_at": "2026-08-28T04:03:19.063223+00:00", "kind": "readme", "missing": false, "url": "https://github.com/xiaowu0162/LongMemEval"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-30T07:05:10.718689+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "90615d215ad99d39b5bd3dbb67f06d81aa935e5cdec81fa7010f0d87d313ef7c", "fetched_at": "2026-08-28T04:03:19.063223+00:00", "kind": "readme", "missing": false, "url": "https://github.com/xiaowu0162/LongMemEval"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-30T07:05:10.718689+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "90615d215ad99d39b5bd3dbb67f06d81aa935e5cdec81fa7010f0d87d313ef7c", "fetched_at": "2026-08-28T04:03:19.063223+00:00", "kind": "readme", "missing": false, "url": "https://github.com/xiaowu0162/LongMemEval"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 81, "longevity": 49, "rhythm": 35}, "computed_at": "2026-09-03T02:20:16.233290+00:00", "flags": ["no_releases"], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 692, "days_push": 114, "days_rel": null, "gap_med": null, "n_releases_24m": 0}, "score": 58, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}