{"adoption": {"forks": 112, "observed_at": "2026-09-03T02:15:14.996569+00:00", "stars": 1016}, "canonical_url": "https://ross.abutalabs.com/products/tutel", "card": {"archived": false, "artifact_type": "library", "description": "Tutel MoE: Optimized Mixture-of-Experts Library, Support GptOss/DeepSeek/Kimi-K2/Qwen3 using FP8/NVFP4/MXFP4", "domain": ["large-language-models", "deep-learning", "machine-learning", "gpu-computing"], "enriched": true, "function": ["llm-inference", "llm-training", "machine-learning", "gpu-computing"], "health_score": 92, "homepage": null, "language": "C", "license": "MIT", "license_family": "permissive", "maturity": "active", "member_repos": ["microsoft/Tutel"], "name": "microsoft/Tutel", "platform": ["python", "cloud"], "pushed_at": "2026-09-02T06:08:25+00:00", "repo": "microsoft/Tutel", "stars": 1016, "tags": ["mixture-of-experts", "moe", "pytorch", "quantization", "fp8", "nvfp4", "mxfp4", "deepseek", "distributed-training", "inference-optimization", "linux", "gpu", "docker"], "topics": ["pytorch", "moe", "mixture-of-experts", "deepseek", "llm"], "urls": [], "use_cases": ["run mixture-of-experts LLM inference with FP8 or FP4 quantization", "train MoE models efficiently on multi-GPU clusters", "serve DeepSeek or Kimi models on AMD MI300X GPUs", "speed up MoE expert routing in PyTorch", "fit large MoE models into limited GPU memory with NVFP4", "benchmark MoE inference throughput against vLLM or SGLang"], "what_it_is": "Tutel is Microsoft's optimized Mixture-of-Experts (MoE) library for efficient training and inference of large language models, featuring dynamic parallelism/sparsity switching and low-precision (FP8/NVFP4/MXFP4) inference for MoE models like DeepSeek, Kimi, GLM, Qwen3, and GPT-OSS. It runs on NVIDIA and AMD GPUs (A100, H100, MI300 series) and integrates with PyTorch.", "when_to_avoid": ["you need a general-purpose LLM serving stack with broad model support and tooling", "your models are dense (non-MoE) transformers", "you need a simple single-GPU inference solution without multi-GPU hardware"], "when_to_choose": ["you need optimized MoE training or inference on NVIDIA or AMD GPUs", "you want to run very large MoE models (DeepSeek, Kimi, GLM) with FP4/FP8 quantization on limited VRAM", "you need dynamic parallelism switching for MoE workloads in PyTorch"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/tutel", "repo": "microsoft/Tutel", "role": "main", "score": 79}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-09-03T02:15:14.996569+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-30T07:11:33.989106+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "0427742200e8eb3508616685513c32ce64e499a5db40e6057972d2dcb4cd5844", "fetched_at": "2026-09-03T02:15:14.996569+00:00", "kind": "readme", "missing": false, "url": "https://github.com/microsoft/Tutel"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-09-03T02:15:14.996569+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-30T07:11:33.989106+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "0427742200e8eb3508616685513c32ce64e499a5db40e6057972d2dcb4cd5844", "fetched_at": "2026-09-03T02:15:14.996569+00:00", "kind": "readme", "missing": false, "url": "https://github.com/microsoft/Tutel"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-30T07:11:33.989106+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "0427742200e8eb3508616685513c32ce64e499a5db40e6057972d2dcb4cd5844", "fetched_at": "2026-09-03T02:15:14.996569+00:00", "kind": "readme", "missing": false, "url": "https://github.com/microsoft/Tutel"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-09-03T02:15:14.996569+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-09-03T02:15:14.996569+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-09-03T02:15:14.996569+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-30T07:11:33.989106+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "0427742200e8eb3508616685513c32ce64e499a5db40e6057972d2dcb4cd5844", "fetched_at": "2026-09-03T02:15:14.996569+00:00", "kind": "readme", "missing": false, "url": "https://github.com/microsoft/Tutel"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-09-03T02:15:14.996569+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-09-03T02:15:14.996569+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-30T07:11:33.989106+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "0427742200e8eb3508616685513c32ce64e499a5db40e6057972d2dcb4cd5844", "fetched_at": "2026-09-03T02:15:14.996569+00:00", "kind": "readme", "missing": false, "url": "https://github.com/microsoft/Tutel"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-09-03T02:15:14.996569+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-09-03T02:15:14.996569+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-09-03T02:15:14.996569+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-30T07:11:33.989106+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "0427742200e8eb3508616685513c32ce64e499a5db40e6057972d2dcb4cd5844", "fetched_at": "2026-09-03T02:15:14.996569+00:00", "kind": "readme", "missing": false, "url": "https://github.com/microsoft/Tutel"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-09-03T02:15:14.996569+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-09-03T02:15:14.996569+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-30T07:11:33.989106+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "0427742200e8eb3508616685513c32ce64e499a5db40e6057972d2dcb4cd5844", "fetched_at": "2026-09-03T02:15:14.996569+00:00", "kind": "readme", "missing": false, "url": "https://github.com/microsoft/Tutel"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-30T07:11:33.989106+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "0427742200e8eb3508616685513c32ce64e499a5db40e6057972d2dcb4cd5844", "fetched_at": "2026-09-03T02:15:14.996569+00:00", "kind": "readme", "missing": false, "url": "https://github.com/microsoft/Tutel"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-30T07:11:33.989106+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "0427742200e8eb3508616685513c32ce64e499a5db40e6057972d2dcb4cd5844", "fetched_at": "2026-09-03T02:15:14.996569+00:00", "kind": "readme", "missing": false, "url": "https://github.com/microsoft/Tutel"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-30T07:11:33.989106+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "0427742200e8eb3508616685513c32ce64e499a5db40e6057972d2dcb4cd5844", "fetched_at": "2026-09-03T02:15:14.996569+00:00", "kind": "readme", "missing": false, "url": "https://github.com/microsoft/Tutel"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 100, "longevity": 100, "rhythm": 40}, "computed_at": "2026-09-03T02:20:16.233290+00:00", "flags": [], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 1853, "days_push": 0, "days_rel": 531, "gap_med": 27, "n_releases_24m": 2}, "score": 79, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 3, "stale_scrape": false}}