{"adoption": {"forks": 1137, "observed_at": "2026-08-28T04:10:59.961817+00:00", "stars": 12872}, "canonical_url": "https://ross.abutalabs.com/products/flashmla", "card": {"archived": false, "artifact_type": "library", "description": "FlashMLA: Efficient Multi-head Latent Attention Kernels", "domain": ["deep-learning", "large-language-models", "gpu-computing", "performance"], "enriched": true, "function": ["machine-learning", "llm-inference", "gpu-computing", "benchmarking"], "health_score": 77, "homepage": null, "language": "C++", "license": "MIT", "license_family": "permissive", "maturity": "active", "member_repos": ["deepseek-ai/FlashMLA"], "name": "deepseek-ai/FlashMLA", "platform": ["python", "cpp"], "pushed_at": "2026-07-28T06:18:26+00:00", "repo": "deepseek-ai/FlashMLA", "stars": 12872, "tags": ["attention-kernels", "cuda", "mla", "sparse-attention", "fp8", "flash-attention", "hopper", "blackwell", "gpu", "linux"], "topics": [], "urls": [], "use_cases": ["speed up MLA decoding for DeepSeek-style LLM inference", "run sparse attention with FP8 KV cache on Hopper GPUs", "benchmark attention kernel throughput on H800 or B200", "integrate optimized prefill attention kernels into an inference engine", "serve large language models with memory-efficient attention"], "what_it_is": "FlashMLA is DeepSeek's library of optimized CUDA attention kernels implementing Multi-head Latent Attention (MLA), including dense and token-level sparse attention for prefill and decoding stages with FP8 KV cache support. It powers DeepSeek-V3 and DeepSeek-V3.2 models and achieves up to 660 TFLOPS on H800 and 1460 TFLOPS on B200 GPUs.", "when_to_avoid": ["you use standard multi-head attention models without MLA support", "you run inference on non-NVIDIA hardware like AMD or Apple GPUs", "you need a turnkey inference server rather than low-level kernels"], "when_to_choose": ["you are serving DeepSeek-V3/V3.2 or other MLA-based models on NVIDIA Hopper or Blackwell GPUs", "you need maximum attention throughput for prefill or decoding in a custom inference stack", "you want FP8 KV cache sparse decoding kernels"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/flashmla", "repo": "deepseek-ai/FlashMLA", "role": "main", "score": 62}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:10:59.961817+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-29T17:13:43.354265+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "7f611d83d918262c66bc219e9c4aed2737acb3f589ea87b3a04746c4cfdd78b6", "fetched_at": "2026-08-28T04:10:59.961817+00:00", "kind": "readme", "missing": false, "url": "https://github.com/deepseek-ai/FlashMLA"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:10:59.961817+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-29T17:13:43.354265+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "7f611d83d918262c66bc219e9c4aed2737acb3f589ea87b3a04746c4cfdd78b6", "fetched_at": "2026-08-28T04:10:59.961817+00:00", "kind": "readme", "missing": false, "url": "https://github.com/deepseek-ai/FlashMLA"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-29T17:13:43.354265+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "7f611d83d918262c66bc219e9c4aed2737acb3f589ea87b3a04746c4cfdd78b6", "fetched_at": "2026-08-28T04:10:59.961817+00:00", "kind": "readme", "missing": false, "url": "https://github.com/deepseek-ai/FlashMLA"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:10:59.961817+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:10:59.961817+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:10:59.961817+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-29T17:13:43.354265+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "7f611d83d918262c66bc219e9c4aed2737acb3f589ea87b3a04746c4cfdd78b6", "fetched_at": "2026-08-28T04:10:59.961817+00:00", "kind": "readme", "missing": false, "url": "https://github.com/deepseek-ai/FlashMLA"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:10:59.961817+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:10:59.961817+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-29T17:13:43.354265+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "7f611d83d918262c66bc219e9c4aed2737acb3f589ea87b3a04746c4cfdd78b6", "fetched_at": "2026-08-28T04:10:59.961817+00:00", "kind": "readme", "missing": false, "url": "https://github.com/deepseek-ai/FlashMLA"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:10:59.961817+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:10:59.961817+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:10:59.961817+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-29T17:13:43.354265+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "7f611d83d918262c66bc219e9c4aed2737acb3f589ea87b3a04746c4cfdd78b6", "fetched_at": "2026-08-28T04:10:59.961817+00:00", "kind": "readme", "missing": false, "url": "https://github.com/deepseek-ai/FlashMLA"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:10:59.961817+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:10:59.961817+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-29T17:13:43.354265+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "7f611d83d918262c66bc219e9c4aed2737acb3f589ea87b3a04746c4cfdd78b6", "fetched_at": "2026-08-28T04:10:59.961817+00:00", "kind": "readme", "missing": false, "url": "https://github.com/deepseek-ai/FlashMLA"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-29T17:13:43.354265+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "7f611d83d918262c66bc219e9c4aed2737acb3f589ea87b3a04746c4cfdd78b6", "fetched_at": "2026-08-28T04:10:59.961817+00:00", "kind": "readme", "missing": false, "url": "https://github.com/deepseek-ai/FlashMLA"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-29T17:13:43.354265+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "7f611d83d918262c66bc219e9c4aed2737acb3f589ea87b3a04746c4cfdd78b6", "fetched_at": "2026-08-28T04:10:59.961817+00:00", "kind": "readme", "missing": false, "url": "https://github.com/deepseek-ai/FlashMLA"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-29T17:13:43.354265+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "7f611d83d918262c66bc219e9c4aed2737acb3f589ea87b3a04746c4cfdd78b6", "fetched_at": "2026-08-28T04:10:59.961817+00:00", "kind": "readme", "missing": false, "url": "https://github.com/deepseek-ai/FlashMLA"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 94, "longevity": 39, "rhythm": 35}, "computed_at": "2026-09-02T17:46:02.011165+00:00", "flags": ["no_releases"], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 558, "days_push": 36, "days_rel": null, "gap_med": null, "n_releases_24m": 0}, "score": 62, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}