{"adoption": {"forks": 552, "observed_at": "2026-08-28T04:08:07.081631+00:00", "stars": 3488}, "canonical_url": "https://ross.abutalabs.com/products/model-optimizer", "card": {"archived": false, "artifact_type": "library", "description": "A unified library of SOTA model optimization techniques like quantization, distillation, pruning, neural architecture search, speculative decoding, etc. It compresses deep learning models for downstream deployment frameworks like TensorRT-LLM, TensorRT, vLLM, etc. to optimize inference speed.", "domain": ["machine-learning", "deep-learning", "large-language-models", "gpu-computing", "developer-tools"], "enriched": true, "function": ["machine-learning", "deep-learning", "llm-inference", "llm-training", "gpu-computing", "sdk"], "health_score": 100, "homepage": "https://nvidia.github.io/Model-Optimizer/", "language": "Python", "license": "Apache-2.0", "license_family": "permissive", "maturity": "active", "member_repos": ["NVIDIA/Model-Optimizer"], "name": "NVIDIA/Model-Optimizer", "platform": ["python", "windows"], "pushed_at": "2026-08-26T22:43:18+00:00", "repo": "NVIDIA/Model-Optimizer", "stars": 3488, "tags": ["quantization", "model-compression", "pruning", "distillation", "neural-architecture-search", "speculative-decoding", "sparsity", "tensorrt-llm", "vllm", "onnx", "pytorch", "huggingface", "nvidia", "linux", "gpu"], "topics": [], "urls": [], "use_cases": ["quantize an LLM to FP8 or INT4 for faster inference", "compress a Hugging Face model for TensorRT-LLM deployment", "apply quantization-aware training to recover accuracy after quantization", "prune a deep learning model to reduce its size", "distill a large model into a smaller student model", "export a quantized checkpoint for vLLM or SGLang", "quantize ONNX models for onnxruntime deployment", "speed up LLM inference with speculative decoding"], "what_it_is": "NVIDIA Model Optimizer (ModelOpt) is a Python library of state-of-the-art model optimization techniques including quantization, pruning, distillation, NAS, speculative decoding, and sparsity. It takes Hugging Face, PyTorch, or ONNX models as input and exports optimized quantized checkpoints ready for deployment in inference frameworks like TensorRT-LLM, TensorRT, vLLM, and SGLang.", "when_to_avoid": ["you only target non-NVIDIA hardware without NVIDIA deployment frameworks", "you need a one-click inference server rather than a model optimization library", "your models are small enough that compression overhead is not worthwhile"], "when_to_choose": ["you need to compress or quantize PyTorch, Hugging Face, or ONNX models for NVIDIA GPU inference", "you deploy models with TensorRT-LLM, TensorRT, vLLM, or SGLang and want optimized checkpoints", "you want QAT, distillation, pruning, or NAS in a single unified API", "you need FP8/INT4/NVFP4 quantization for large language models"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/model-optimizer", "repo": "NVIDIA/Model-Optimizer", "role": "main", "score": 91}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:08:07.081631+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-29T18:36:05.731930+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "b09d0c013b6003c557acbee645a04870f3b4fdd930841072790c5b017f4159b5", "fetched_at": "2026-08-28T04:08:07.081631+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NVIDIA/Model-Optimizer"}, {"content_hash": "908d3062393a5bd6f85b6138cfc0e844c8c65376c3ecfd30ae4e18a45ecaf780", "fetched_at": "2026-08-29T09:30:08.015832+00:00", "kind": "homepage", "missing": false, "url": "https://nvidia.github.io/Model-Optimizer/"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:08:07.081631+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-29T18:36:05.731930+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "b09d0c013b6003c557acbee645a04870f3b4fdd930841072790c5b017f4159b5", "fetched_at": "2026-08-28T04:08:07.081631+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NVIDIA/Model-Optimizer"}, {"content_hash": "908d3062393a5bd6f85b6138cfc0e844c8c65376c3ecfd30ae4e18a45ecaf780", "fetched_at": "2026-08-29T09:30:08.015832+00:00", "kind": "homepage", "missing": false, "url": "https://nvidia.github.io/Model-Optimizer/"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-29T18:36:05.731930+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "b09d0c013b6003c557acbee645a04870f3b4fdd930841072790c5b017f4159b5", "fetched_at": "2026-08-28T04:08:07.081631+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NVIDIA/Model-Optimizer"}, {"content_hash": "908d3062393a5bd6f85b6138cfc0e844c8c65376c3ecfd30ae4e18a45ecaf780", "fetched_at": "2026-08-29T09:30:08.015832+00:00", "kind": "homepage", "missing": false, "url": "https://nvidia.github.io/Model-Optimizer/"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:08:07.081631+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:08:07.081631+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:08:07.081631+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-29T18:36:05.731930+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "b09d0c013b6003c557acbee645a04870f3b4fdd930841072790c5b017f4159b5", "fetched_at": "2026-08-28T04:08:07.081631+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NVIDIA/Model-Optimizer"}, {"content_hash": "908d3062393a5bd6f85b6138cfc0e844c8c65376c3ecfd30ae4e18a45ecaf780", "fetched_at": "2026-08-29T09:30:08.015832+00:00", "kind": "homepage", "missing": false, "url": "https://nvidia.github.io/Model-Optimizer/"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:08:07.081631+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:08:07.081631+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-29T18:36:05.731930+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "b09d0c013b6003c557acbee645a04870f3b4fdd930841072790c5b017f4159b5", "fetched_at": "2026-08-28T04:08:07.081631+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NVIDIA/Model-Optimizer"}, {"content_hash": "908d3062393a5bd6f85b6138cfc0e844c8c65376c3ecfd30ae4e18a45ecaf780", "fetched_at": "2026-08-29T09:30:08.015832+00:00", "kind": "homepage", "missing": false, "url": "https://nvidia.github.io/Model-Optimizer/"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:08:07.081631+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:08:07.081631+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:08:07.081631+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-29T18:36:05.731930+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "b09d0c013b6003c557acbee645a04870f3b4fdd930841072790c5b017f4159b5", "fetched_at": "2026-08-28T04:08:07.081631+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NVIDIA/Model-Optimizer"}, {"content_hash": "908d3062393a5bd6f85b6138cfc0e844c8c65376c3ecfd30ae4e18a45ecaf780", "fetched_at": "2026-08-29T09:30:08.015832+00:00", "kind": "homepage", "missing": false, "url": "https://nvidia.github.io/Model-Optimizer/"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:08:07.081631+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:08:07.081631+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-29T18:36:05.731930+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "b09d0c013b6003c557acbee645a04870f3b4fdd930841072790c5b017f4159b5", "fetched_at": "2026-08-28T04:08:07.081631+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NVIDIA/Model-Optimizer"}, {"content_hash": "908d3062393a5bd6f85b6138cfc0e844c8c65376c3ecfd30ae4e18a45ecaf780", "fetched_at": "2026-08-29T09:30:08.015832+00:00", "kind": "homepage", "missing": false, "url": "https://nvidia.github.io/Model-Optimizer/"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-29T18:36:05.731930+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "b09d0c013b6003c557acbee645a04870f3b4fdd930841072790c5b017f4159b5", "fetched_at": "2026-08-28T04:08:07.081631+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NVIDIA/Model-Optimizer"}, {"content_hash": "908d3062393a5bd6f85b6138cfc0e844c8c65376c3ecfd30ae4e18a45ecaf780", "fetched_at": "2026-08-29T09:30:08.015832+00:00", "kind": "homepage", "missing": false, "url": "https://nvidia.github.io/Model-Optimizer/"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-29T18:36:05.731930+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "b09d0c013b6003c557acbee645a04870f3b4fdd930841072790c5b017f4159b5", "fetched_at": "2026-08-28T04:08:07.081631+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NVIDIA/Model-Optimizer"}, {"content_hash": "908d3062393a5bd6f85b6138cfc0e844c8c65376c3ecfd30ae4e18a45ecaf780", "fetched_at": "2026-08-29T09:30:08.015832+00:00", "kind": "homepage", "missing": false, "url": "https://nvidia.github.io/Model-Optimizer/"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-29T18:36:05.731930+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "b09d0c013b6003c557acbee645a04870f3b4fdd930841072790c5b017f4159b5", "fetched_at": "2026-08-28T04:08:07.081631+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NVIDIA/Model-Optimizer"}, {"content_hash": "908d3062393a5bd6f85b6138cfc0e844c8c65376c3ecfd30ae4e18a45ecaf780", "fetched_at": "2026-08-29T09:30:08.015832+00:00", "kind": "homepage", "missing": false, "url": "https://nvidia.github.io/Model-Optimizer/"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 99, "longevity": 61, "rhythm": 98}, "computed_at": "2026-09-03T02:20:16.233290+00:00", "flags": [], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 862, "days_push": 7, "days_rel": 15, "gap_med": 28.0, "n_releases_24m": 21}, "score": 91, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}