{"adoption": {"forks": 3010, "observed_at": "2026-08-28T04:11:37.642717+00:00", "stars": 24787}, "canonical_url": "https://ross.abutalabs.com/products/flash-attention", "card": {"archived": false, "artifact_type": "library", "description": "Fast and memory-efficient exact attention", "domain": ["deep-learning", "large-language-models", "gpu-computing", "machine-learning"], "enriched": true, "function": ["machine-learning", "deep-learning", "gpu-computing", "benchmarking"], "health_score": 100, "homepage": null, "language": "Python", "license": "BSD-3-Clause", "license_family": "permissive", "maturity": "active", "member_repos": ["Dao-AILab/flash-attention"], "name": "Dao-AILab/flash-attention", "platform": ["python"], "pushed_at": "2026-08-26T08:47:00+00:00", "repo": "Dao-AILab/flash-attention", "stars": 24787, "tags": ["attention", "transformers", "cuda-kernels", "pytorch", "memory-efficiency", "flash-attention", "linux", "gpu", "cuda"], "topics": [], "urls": [], "use_cases": ["speed up transformer training on GPUs", "reduce memory usage of attention in LLM training", "run attention efficiently on H100 GPUs", "train large language models faster", "use exact attention instead of approximations", "benchmark attention kernel performance"], "what_it_is": "Official implementation of FlashAttention, FlashAttention-2, -3, and -4: fast and memory-efficient exact attention kernels for GPUs. It accelerates transformer training and inference on NVIDIA (and ROCm) hardware with IO-aware CUDA kernels.", "when_to_avoid": ["you are not using GPU-accelerated PyTorch with CUDA or ROCm", "you need Windows support or non-CUDA hardware", "you only need a high-level API and prefer frameworks that bundle attention kernels already"], "when_to_choose": ["you train or fine-tune transformers on NVIDIA GPUs and need faster, memory-efficient attention", "you need exact attention with long sequences that would otherwise OOM", "you target Hopper/Blackwell GPUs and want optimized FP16/BF16/FP8 kernels"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/flash-attention", "repo": "Dao-AILab/flash-attention", "role": "main", "score": 95}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:11:37.642717+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-29T16:56:09.518640+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "e0f876a0d696ff4414460f0139e2d33df19155b2bd5164cb660852a973074b1a", "fetched_at": "2026-08-28T04:11:37.642717+00:00", "kind": "readme", "missing": false, "url": "https://github.com/Dao-AILab/flash-attention"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:11:37.642717+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-29T16:56:09.518640+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "e0f876a0d696ff4414460f0139e2d33df19155b2bd5164cb660852a973074b1a", "fetched_at": "2026-08-28T04:11:37.642717+00:00", "kind": "readme", "missing": false, "url": "https://github.com/Dao-AILab/flash-attention"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-29T16:56:09.518640+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "e0f876a0d696ff4414460f0139e2d33df19155b2bd5164cb660852a973074b1a", "fetched_at": "2026-08-28T04:11:37.642717+00:00", "kind": "readme", "missing": false, "url": "https://github.com/Dao-AILab/flash-attention"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:11:37.642717+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:11:37.642717+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:11:37.642717+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-29T16:56:09.518640+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "e0f876a0d696ff4414460f0139e2d33df19155b2bd5164cb660852a973074b1a", "fetched_at": "2026-08-28T04:11:37.642717+00:00", "kind": "readme", "missing": false, "url": "https://github.com/Dao-AILab/flash-attention"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:11:37.642717+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:11:37.642717+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-29T16:56:09.518640+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "e0f876a0d696ff4414460f0139e2d33df19155b2bd5164cb660852a973074b1a", "fetched_at": "2026-08-28T04:11:37.642717+00:00", "kind": "readme", "missing": false, "url": "https://github.com/Dao-AILab/flash-attention"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:11:37.642717+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:11:37.642717+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:11:37.642717+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-29T16:56:09.518640+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "e0f876a0d696ff4414460f0139e2d33df19155b2bd5164cb660852a973074b1a", "fetched_at": "2026-08-28T04:11:37.642717+00:00", "kind": "readme", "missing": false, "url": "https://github.com/Dao-AILab/flash-attention"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:11:37.642717+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:11:37.642717+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-29T16:56:09.518640+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "e0f876a0d696ff4414460f0139e2d33df19155b2bd5164cb660852a973074b1a", "fetched_at": "2026-08-28T04:11:37.642717+00:00", "kind": "readme", "missing": false, "url": "https://github.com/Dao-AILab/flash-attention"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-29T16:56:09.518640+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "e0f876a0d696ff4414460f0139e2d33df19155b2bd5164cb660852a973074b1a", "fetched_at": "2026-08-28T04:11:37.642717+00:00", "kind": "readme", "missing": false, "url": "https://github.com/Dao-AILab/flash-attention"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-29T16:56:09.518640+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "e0f876a0d696ff4414460f0139e2d33df19155b2bd5164cb660852a973074b1a", "fetched_at": "2026-08-28T04:11:37.642717+00:00", "kind": "readme", "missing": false, "url": "https://github.com/Dao-AILab/flash-attention"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-29T16:56:09.518640+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "e0f876a0d696ff4414460f0139e2d33df19155b2bd5164cb660852a973074b1a", "fetched_at": "2026-08-28T04:11:37.642717+00:00", "kind": "readme", "missing": false, "url": "https://github.com/Dao-AILab/flash-attention"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 99, "longevity": 100, "rhythm": 87}, "computed_at": "2026-09-03T02:20:16.233290+00:00", "flags": [], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 1567, "days_push": 7, "days_rel": 84, "gap_med": 0, "n_releases_24m": 24}, "score": 95, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}