{"adoption": {"forks": 762, "observed_at": "2026-08-28T04:05:47.275266+00:00", "stars": 1880}, "canonical_url": "https://ross.abutalabs.com/products/redknot", "card": {"archived": false, "artifact_type": "library", "description": "Efficient Long-Context LLM Serving with Head-Aware KV Reuse and SegPagedAttention", "domain": ["large-language-models", "machine-learning", "gpu-computing"], "enriched": true, "function": ["llm-inference", "machine-learning", "gpu-computing"], "health_score": 79, "homepage": null, "language": "Python", "license": "Apache-2.0", "license_family": "permissive", "maturity": "active", "member_repos": ["rednote-machine-learning/RedKnot"], "name": "rednote-machine-learning/RedKnot", "platform": ["python"], "pushed_at": "2026-08-17T04:01:06+00:00", "repo": "rednote-machine-learning/RedKnot", "stars": 1880, "tags": ["kv-cache-reuse", "long-context", "attention-optimization", "sglang", "sparse-attention", "inference-acceleration", "gpu", "linux", "docker"], "topics": [], "urls": [], "use_cases": ["serve long-context LLMs with lower latency", "speed up prefill for large prompt inference", "reduce KV cache memory usage in LLM serving", "run Qwen3 or Llama models with sparse attention efficiently", "accelerate TTFT for RAG workloads with reusable prompt segments", "deploy LLM inference on limited GPU memory"], "what_it_is": "RedKnot is a long-context LLM inference acceleration library built on SGLang, using head-classified KV reuse, offline KV storage with RoPE relocation, sparse FFN, and a SegPagedAttention runtime. It reduces prefill FLOPs by 50-70% and speeds up TTFT 1.35x-3.2x with near-lossless accuracy.", "when_to_avoid": ["you need short-context inference where the overhead outweighs gains", "you require a fully stable release for models like DeepSeek-V4 or Qwen3.5 that are still being adapted", "you need Ascend NPU support that is still work in progress", "you need a simple inference stack without SGLang's complexity"], "when_to_choose": ["you serve long-context models and prefill latency is a bottleneck", "you already use SGLang and want drop-in attention acceleration", "your workloads have reusable prompt prefixes like RAG or system prompts", "you want near-lossless quality with large FLOPs savings"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/redknot", "repo": "rednote-machine-learning/RedKnot", "role": "main", "score": 58}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:05:47.275266+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-30T03:14:01.565091+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "c1a4c97879526221d08dad2ae64881b2e831da32d827fcce6fd473d7465d59cc", "fetched_at": "2026-08-28T04:05:47.275266+00:00", "kind": "readme", "missing": false, "url": "https://github.com/rednote-machine-learning/RedKnot"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:05:47.275266+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-30T03:14:01.565091+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "c1a4c97879526221d08dad2ae64881b2e831da32d827fcce6fd473d7465d59cc", "fetched_at": "2026-08-28T04:05:47.275266+00:00", "kind": "readme", "missing": false, "url": "https://github.com/rednote-machine-learning/RedKnot"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-30T03:14:01.565091+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "c1a4c97879526221d08dad2ae64881b2e831da32d827fcce6fd473d7465d59cc", "fetched_at": "2026-08-28T04:05:47.275266+00:00", "kind": "readme", "missing": false, "url": "https://github.com/rednote-machine-learning/RedKnot"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:05:47.275266+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:05:47.275266+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:05:47.275266+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-30T03:14:01.565091+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "c1a4c97879526221d08dad2ae64881b2e831da32d827fcce6fd473d7465d59cc", "fetched_at": "2026-08-28T04:05:47.275266+00:00", "kind": "readme", "missing": false, "url": "https://github.com/rednote-machine-learning/RedKnot"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:05:47.275266+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:05:47.275266+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-30T03:14:01.565091+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "c1a4c97879526221d08dad2ae64881b2e831da32d827fcce6fd473d7465d59cc", "fetched_at": "2026-08-28T04:05:47.275266+00:00", "kind": "readme", "missing": false, "url": "https://github.com/rednote-machine-learning/RedKnot"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:05:47.275266+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:05:47.275266+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:05:47.275266+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-30T03:14:01.565091+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "c1a4c97879526221d08dad2ae64881b2e831da32d827fcce6fd473d7465d59cc", "fetched_at": "2026-08-28T04:05:47.275266+00:00", "kind": "readme", "missing": false, "url": "https://github.com/rednote-machine-learning/RedKnot"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:05:47.275266+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:05:47.275266+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-30T03:14:01.565091+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "c1a4c97879526221d08dad2ae64881b2e831da32d827fcce6fd473d7465d59cc", "fetched_at": "2026-08-28T04:05:47.275266+00:00", "kind": "readme", "missing": false, "url": "https://github.com/rednote-machine-learning/RedKnot"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-30T03:14:01.565091+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "c1a4c97879526221d08dad2ae64881b2e831da32d827fcce6fd473d7465d59cc", "fetched_at": "2026-08-28T04:05:47.275266+00:00", "kind": "readme", "missing": false, "url": "https://github.com/rednote-machine-learning/RedKnot"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-30T03:14:01.565091+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "c1a4c97879526221d08dad2ae64881b2e831da32d827fcce6fd473d7465d59cc", "fetched_at": "2026-08-28T04:05:47.275266+00:00", "kind": "readme", "missing": false, "url": "https://github.com/rednote-machine-learning/RedKnot"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-30T03:14:01.565091+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "c1a4c97879526221d08dad2ae64881b2e831da32d827fcce6fd473d7465d59cc", "fetched_at": "2026-08-28T04:05:47.275266+00:00", "kind": "readme", "missing": false, "url": "https://github.com/rednote-machine-learning/RedKnot"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 98, "longevity": 6, "rhythm": 35}, "computed_at": "2026-09-02T17:46:02.011165+00:00", "flags": ["no_releases", "young"], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 90, "days_push": 16, "days_rel": null, "gap_med": null, "n_releases_24m": 0}, "score": 58, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}