{"adoption": {"forks": 143, "observed_at": "2026-08-28T04:03:31.022434+00:00", "stars": 1082}, "canonical_url": "https://ross.abutalabs.com/products/fast-dllm", "card": {"archived": false, "artifact_type": "library", "description": "Official implementation of \"Fast-dLLM: Training-free Acceleration of Diffusion LLM by Enabling KV Cache and Parallel Decoding\"", "domain": ["large-language-models", "artificial-intelligence", "deep-learning", "gpu-computing", "autonomous-vehicles", "computer-vision"], "enriched": true, "function": ["llm-inference", "machine-learning", "deep-learning", "gpu-computing"], "health_score": 71, "homepage": "https://nvlabs.github.io/Fast-dLLM/", "language": "Python", "license": "Apache-2.0", "license_family": "permissive", "maturity": "active", "member_repos": ["NVlabs/Fast-dLLM"], "name": "NVlabs/Fast-dLLM", "platform": ["python"], "pushed_at": "2026-05-30T10:38:26+00:00", "repo": "NVlabs/Fast-dLLM", "stars": 1082, "tags": ["diffusion-llm", "kv-cache", "parallel-decoding", "block-diffusion", "speculative-decoding", "inference-acceleration", "vision-language-model", "research-code", "nvidia", "iclr-2026", "gpu", "linux"], "topics": [], "urls": [], "use_cases": ["accelerate diffusion llm inference", "speed up llada and dream text generation", "enable kv cache for bidirectional diffusion models", "parallel decode multiple tokens in diffusion language models", "convert autoregressive vlms to diffusion vlms", "efficient end-to-end autonomous driving with diffusion models"], "what_it_is": "NVIDIA's official implementation of Fast-dLLM, a family of training-free and fine-tuning-based acceleration techniques for diffusion-based large language models, vision-language models, and vision-language-action models. It enables KV cache reuse and confidence-aware parallel decoding to achieve up to 27.6x throughput improvement over standard diffusion LLM inference with minimal accuracy loss.", "when_to_avoid": ["you use standard autoregressive LLMs where conventional inference is already fast", "you need a production serving stack rather than research code", "you lack GPU hardware or work outside the supported model backbones"], "when_to_choose": ["you are running diffusion-based LLMs like LLaDA or Dream and need faster inference", "you want to reproduce ICLR 2026 research on diffusion LLM acceleration", "you need block-diffusion VLMs for multimodal or autonomous driving workloads", "you have NVIDIA GPUs and want training-free throughput gains"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/fast-dllm", "repo": "NVlabs/Fast-dLLM", "role": "main", "score": 57}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:03:31.022434+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-30T06:51:13.550174+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "bb764faf6ff640d54c8c5362ada0cef6a4780ef9160e3a1adedc2e6082d9c6af", "fetched_at": "2026-08-28T04:03:31.022434+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NVlabs/Fast-dLLM"}, {"content_hash": "78c746cb7c428a42c16e374dab04d55fa234852698f7c2518c67299fe57b336b", "fetched_at": "2026-08-29T12:53:33.731054+00:00", "kind": "homepage", "missing": false, "url": "https://nvlabs.github.io/Fast-dLLM/"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:03:31.022434+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-30T06:51:13.550174+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "bb764faf6ff640d54c8c5362ada0cef6a4780ef9160e3a1adedc2e6082d9c6af", "fetched_at": "2026-08-28T04:03:31.022434+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NVlabs/Fast-dLLM"}, {"content_hash": "78c746cb7c428a42c16e374dab04d55fa234852698f7c2518c67299fe57b336b", "fetched_at": "2026-08-29T12:53:33.731054+00:00", "kind": "homepage", "missing": false, "url": "https://nvlabs.github.io/Fast-dLLM/"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-30T06:51:13.550174+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "bb764faf6ff640d54c8c5362ada0cef6a4780ef9160e3a1adedc2e6082d9c6af", "fetched_at": "2026-08-28T04:03:31.022434+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NVlabs/Fast-dLLM"}, {"content_hash": "78c746cb7c428a42c16e374dab04d55fa234852698f7c2518c67299fe57b336b", "fetched_at": "2026-08-29T12:53:33.731054+00:00", "kind": "homepage", "missing": false, "url": "https://nvlabs.github.io/Fast-dLLM/"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:03:31.022434+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:03:31.022434+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:03:31.022434+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-30T06:51:13.550174+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "bb764faf6ff640d54c8c5362ada0cef6a4780ef9160e3a1adedc2e6082d9c6af", "fetched_at": "2026-08-28T04:03:31.022434+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NVlabs/Fast-dLLM"}, {"content_hash": "78c746cb7c428a42c16e374dab04d55fa234852698f7c2518c67299fe57b336b", "fetched_at": "2026-08-29T12:53:33.731054+00:00", "kind": "homepage", "missing": false, "url": "https://nvlabs.github.io/Fast-dLLM/"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:03:31.022434+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:03:31.022434+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-30T06:51:13.550174+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "bb764faf6ff640d54c8c5362ada0cef6a4780ef9160e3a1adedc2e6082d9c6af", "fetched_at": "2026-08-28T04:03:31.022434+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NVlabs/Fast-dLLM"}, {"content_hash": "78c746cb7c428a42c16e374dab04d55fa234852698f7c2518c67299fe57b336b", "fetched_at": "2026-08-29T12:53:33.731054+00:00", "kind": "homepage", "missing": false, "url": "https://nvlabs.github.io/Fast-dLLM/"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:03:31.022434+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:03:31.022434+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:03:31.022434+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-30T06:51:13.550174+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "bb764faf6ff640d54c8c5362ada0cef6a4780ef9160e3a1adedc2e6082d9c6af", "fetched_at": "2026-08-28T04:03:31.022434+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NVlabs/Fast-dLLM"}, {"content_hash": "78c746cb7c428a42c16e374dab04d55fa234852698f7c2518c67299fe57b336b", "fetched_at": "2026-08-29T12:53:33.731054+00:00", "kind": "homepage", "missing": false, "url": "https://nvlabs.github.io/Fast-dLLM/"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:03:31.022434+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:03:31.022434+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-30T06:51:13.550174+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "bb764faf6ff640d54c8c5362ada0cef6a4780ef9160e3a1adedc2e6082d9c6af", "fetched_at": "2026-08-28T04:03:31.022434+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NVlabs/Fast-dLLM"}, {"content_hash": "78c746cb7c428a42c16e374dab04d55fa234852698f7c2518c67299fe57b336b", "fetched_at": "2026-08-29T12:53:33.731054+00:00", "kind": "homepage", "missing": false, "url": "https://nvlabs.github.io/Fast-dLLM/"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-30T06:51:13.550174+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "bb764faf6ff640d54c8c5362ada0cef6a4780ef9160e3a1adedc2e6082d9c6af", "fetched_at": "2026-08-28T04:03:31.022434+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NVlabs/Fast-dLLM"}, {"content_hash": "78c746cb7c428a42c16e374dab04d55fa234852698f7c2518c67299fe57b336b", "fetched_at": "2026-08-29T12:53:33.731054+00:00", "kind": "homepage", "missing": false, "url": "https://nvlabs.github.io/Fast-dLLM/"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-30T06:51:13.550174+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "bb764faf6ff640d54c8c5362ada0cef6a4780ef9160e3a1adedc2e6082d9c6af", "fetched_at": "2026-08-28T04:03:31.022434+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NVlabs/Fast-dLLM"}, {"content_hash": "78c746cb7c428a42c16e374dab04d55fa234852698f7c2518c67299fe57b336b", "fetched_at": "2026-08-29T12:53:33.731054+00:00", "kind": "homepage", "missing": false, "url": "https://nvlabs.github.io/Fast-dLLM/"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-30T06:51:13.550174+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "bb764faf6ff640d54c8c5362ada0cef6a4780ef9160e3a1adedc2e6082d9c6af", "fetched_at": "2026-08-28T04:03:31.022434+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NVlabs/Fast-dLLM"}, {"content_hash": "78c746cb7c428a42c16e374dab04d55fa234852698f7c2518c67299fe57b336b", "fetched_at": "2026-08-29T12:53:33.731054+00:00", "kind": "homepage", "missing": false, "url": "https://nvlabs.github.io/Fast-dLLM/"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 85, "longevity": 33, "rhythm": 35}, "computed_at": "2026-09-02T17:46:02.011165+00:00", "flags": ["no_releases"], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 463, "days_push": 95, "days_rel": null, "gap_med": null, "n_releases_24m": 0}, "score": 57, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}