{"adoption": {"forks": 728, "observed_at": "2026-08-28T04:10:11.863585+00:00", "stars": 8024}, "canonical_url": "https://ross.abutalabs.com/products/lmdeploy", "card": {"archived": false, "artifact_type": "library", "description": "LMDeploy is a toolkit for compressing, deploying, and serving LLMs.", "domain": ["large-language-models", "deep-learning", "machine-learning", "artificial-intelligence", "gpu-computing", "developer-tools"], "enriched": true, "function": ["llm-inference", "gpu-computing", "http-server", "api-framework", "chatbot"], "health_score": 100, "homepage": "https://lmdeploy.readthedocs.io/en/latest", "language": "Python", "license": "Apache-2.0", "license_family": "permissive", "maturity": "active", "member_repos": ["InternLM/lmdeploy"], "name": "InternLM/lmdeploy", "platform": ["python", "cli"], "pushed_at": "2026-08-26T12:01:03+00:00", "repo": "InternLM/lmdeploy", "stars": 8024, "tags": ["turbomind", "quantization", "model-serving", "cuda-kernels", "openai-compatible-api", "kv-cache", "inference-engine", "linux", "gpu", "docker"], "topics": ["cuda-kernels", "deepspeed", "fastertransformer", "llm-inference", "turbomind", "internlm", "llama", "llm", "codellama", "llama2", "llama3"], "urls": [], "use_cases": ["serve llm with openai compatible api", "quantize llama model to 4bit", "run deepseek v3 inference on gpu", "deploy chatbot backend for large language model", "speed up llm inference with turbomind", "compress and serve internlm models", "benchmark llm inference throughput"], "what_it_is": "LMDeploy is a toolkit for compressing, quantizing, deploying, and serving large language models, built around its high-performance TurboMind inference engine. It provides OpenAI-compatible serving APIs, CLI tools, and optimized CUDA kernels for fast LLM inference on NVIDIA GPUs.", "when_to_avoid": ["you need CPU-only or non-NVIDIA hardware inference", "you only want to fine-tune or train models rather than serve them", "you prefer a simpler pure-PyTorch stack without custom CUDA kernels"], "when_to_choose": ["you need high-throughput, low-latency LLM serving on NVIDIA GPUs", "you want built-in quantization (4bit, FP8, MXFP4) with a serving engine", "you deploy InternLM, Llama, Qwen, or DeepSeek models with an OpenAI-compatible API"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/lmdeploy", "repo": "InternLM/lmdeploy", "role": "main", "score": 95}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:10:11.863585+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-29T17:31:49.674397+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "57d0987acbf101c126e26d12c8c5ea4c9c2a1c686615b90a7091cce30efc10d6", "fetched_at": "2026-08-28T04:10:11.863585+00:00", "kind": "readme", "missing": false, "url": "https://github.com/InternLM/lmdeploy"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:10:11.863585+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-29T17:31:49.674397+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "57d0987acbf101c126e26d12c8c5ea4c9c2a1c686615b90a7091cce30efc10d6", "fetched_at": "2026-08-28T04:10:11.863585+00:00", "kind": "readme", "missing": false, "url": "https://github.com/InternLM/lmdeploy"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-29T17:31:49.674397+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "57d0987acbf101c126e26d12c8c5ea4c9c2a1c686615b90a7091cce30efc10d6", "fetched_at": "2026-08-28T04:10:11.863585+00:00", "kind": "readme", "missing": false, "url": "https://github.com/InternLM/lmdeploy"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:10:11.863585+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:10:11.863585+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:10:11.863585+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-29T17:31:49.674397+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "57d0987acbf101c126e26d12c8c5ea4c9c2a1c686615b90a7091cce30efc10d6", "fetched_at": "2026-08-28T04:10:11.863585+00:00", "kind": "readme", "missing": false, "url": "https://github.com/InternLM/lmdeploy"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:10:11.863585+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:10:11.863585+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-29T17:31:49.674397+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "57d0987acbf101c126e26d12c8c5ea4c9c2a1c686615b90a7091cce30efc10d6", "fetched_at": "2026-08-28T04:10:11.863585+00:00", "kind": "readme", "missing": false, "url": "https://github.com/InternLM/lmdeploy"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:10:11.863585+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:10:11.863585+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:10:11.863585+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-29T17:31:49.674397+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "57d0987acbf101c126e26d12c8c5ea4c9c2a1c686615b90a7091cce30efc10d6", "fetched_at": "2026-08-28T04:10:11.863585+00:00", "kind": "readme", "missing": false, "url": "https://github.com/InternLM/lmdeploy"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:10:11.863585+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:10:11.863585+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-29T17:31:49.674397+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "57d0987acbf101c126e26d12c8c5ea4c9c2a1c686615b90a7091cce30efc10d6", "fetched_at": "2026-08-28T04:10:11.863585+00:00", "kind": "readme", "missing": false, "url": "https://github.com/InternLM/lmdeploy"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-29T17:31:49.674397+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "57d0987acbf101c126e26d12c8c5ea4c9c2a1c686615b90a7091cce30efc10d6", "fetched_at": "2026-08-28T04:10:11.863585+00:00", "kind": "readme", "missing": false, "url": "https://github.com/InternLM/lmdeploy"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-29T17:31:49.674397+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "57d0987acbf101c126e26d12c8c5ea4c9c2a1c686615b90a7091cce30efc10d6", "fetched_at": "2026-08-28T04:10:11.863585+00:00", "kind": "readme", "missing": false, "url": "https://github.com/InternLM/lmdeploy"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-29T17:31:49.674397+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "57d0987acbf101c126e26d12c8c5ea4c9c2a1c686615b90a7091cce30efc10d6", "fetched_at": "2026-08-28T04:10:11.863585+00:00", "kind": "readme", "missing": false, "url": "https://github.com/InternLM/lmdeploy"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 99, "longevity": 83, "rhythm": 98}, "computed_at": "2026-09-02T17:46:02.011165+00:00", "flags": [], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 1175, "days_push": 7, "days_rel": 14, "gap_med": 20.0, "n_releases_24m": 33}, "score": 95, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}