{"adoption": {"forks": 485, "observed_at": "2026-08-28T04:09:03.624006+00:00", "stars": 4945}, "canonical_url": "https://ross.abutalabs.com/products/fastllm", "card": {"archived": false, "artifact_type": "library", "description": "fastllm是后端无依赖的高性能大模型推理库。同时支持张量并行推理稠密模型和混合模式推理MOE模型，任意10G以上显卡即可推理满血DeepSeek。双路9004/9005服务器+单显卡部署DeepSeek满血满精度原版模型，单并发20tps；INT4量化模型单并发30tps，多并发可达60+。", "domain": ["large-language-models", "artificial-intelligence", "developer-tools"], "enriched": true, "function": ["llm-inference", "gpu-computing", "cli", "http-server", "chat-interface"], "health_score": 80, "homepage": null, "language": "C++", "license": "Apache-2.0", "license_family": "permissive", "maturity": "active", "member_repos": ["ztxz16/fastllm"], "name": "ztxz16/fastllm", "platform": ["windows", "python", "cpp", "cli"], "pushed_at": "2026-08-26T06:22:32+00:00", "repo": "ztxz16/fastllm", "stars": 4945, "tags": ["moe-inference", "tensor-parallelism", "cpu-gpu-hybrid", "fp8", "quantization", "deepseek", "openai-compatible-api", "self-contained-operators", "no-pytorch-dependency", "command-line", "linux", "gpu", "web-server", "android"], "topics": [], "urls": [], "use_cases": ["run DeepSeek R1 671B on a single GPU with limited VRAM", "serve an OpenAI-compatible API for local LLMs", "chat with Qwen or Llama models from the terminal", "deploy MOE models with CPU+GPU hybrid tensor parallelism", "run FP8 inference on older or domestic GPUs", "quantize models to INT4 or dynamic quantization for faster inference", "deploy LLM inference on AMD ROCm or Chinese domestic accelerators"], "what_it_is": "fastllm is a high-performance C++ LLM inference library with its own custom operators, requiring no PyTorch dependency. It supports tensor-parallel inference of dense models and hybrid CPU+GPU inference of MOE models, enabling full DeepSeek 671B inference on a single 10GB+ GPU.", "when_to_avoid": ["you need the broadest ecosystem integration and community tooling of vLLM or SGLang", "you require macOS or Apple Silicon support", "you need extensive fine-tuning or training features rather than inference only"], "when_to_choose": ["you need to run very large MOE models like DeepSeek 671B with limited GPU memory", "you want a PyTorch-free inference engine with broad GPU compatibility including older cards", "you need tensor parallelism across an arbitrary number of GPUs including odd counts", "you target domestic Chinese accelerators or AMD GPUs"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/fastllm", "repo": "ztxz16/fastllm", "role": "main", "score": 74}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:09:03.624006+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-29T18:17:54.046777+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "fd3d330cb89207bac94dd65a90b5b037b39fc3c252d9da026c647a1ce5cf6c05", "fetched_at": "2026-08-28T04:09:03.624006+00:00", "kind": "readme", "missing": false, "url": "https://github.com/ztxz16/fastllm"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:09:03.624006+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-29T18:17:54.046777+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "fd3d330cb89207bac94dd65a90b5b037b39fc3c252d9da026c647a1ce5cf6c05", "fetched_at": "2026-08-28T04:09:03.624006+00:00", "kind": "readme", "missing": false, "url": "https://github.com/ztxz16/fastllm"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-29T18:17:54.046777+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "fd3d330cb89207bac94dd65a90b5b037b39fc3c252d9da026c647a1ce5cf6c05", "fetched_at": "2026-08-28T04:09:03.624006+00:00", "kind": "readme", "missing": false, "url": "https://github.com/ztxz16/fastllm"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:09:03.624006+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:09:03.624006+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:09:03.624006+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-29T18:17:54.046777+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "fd3d330cb89207bac94dd65a90b5b037b39fc3c252d9da026c647a1ce5cf6c05", "fetched_at": "2026-08-28T04:09:03.624006+00:00", "kind": "readme", "missing": false, "url": "https://github.com/ztxz16/fastllm"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:09:03.624006+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:09:03.624006+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-29T18:17:54.046777+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "fd3d330cb89207bac94dd65a90b5b037b39fc3c252d9da026c647a1ce5cf6c05", "fetched_at": "2026-08-28T04:09:03.624006+00:00", "kind": "readme", "missing": false, "url": "https://github.com/ztxz16/fastllm"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:09:03.624006+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:09:03.624006+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:09:03.624006+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-29T18:17:54.046777+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "fd3d330cb89207bac94dd65a90b5b037b39fc3c252d9da026c647a1ce5cf6c05", "fetched_at": "2026-08-28T04:09:03.624006+00:00", "kind": "readme", "missing": false, "url": "https://github.com/ztxz16/fastllm"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:09:03.624006+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:09:03.624006+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-29T18:17:54.046777+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "fd3d330cb89207bac94dd65a90b5b037b39fc3c252d9da026c647a1ce5cf6c05", "fetched_at": "2026-08-28T04:09:03.624006+00:00", "kind": "readme", "missing": false, "url": "https://github.com/ztxz16/fastllm"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-29T18:17:54.046777+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "fd3d330cb89207bac94dd65a90b5b037b39fc3c252d9da026c647a1ce5cf6c05", "fetched_at": "2026-08-28T04:09:03.624006+00:00", "kind": "readme", "missing": false, "url": "https://github.com/ztxz16/fastllm"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-29T18:17:54.046777+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "fd3d330cb89207bac94dd65a90b5b037b39fc3c252d9da026c647a1ce5cf6c05", "fetched_at": "2026-08-28T04:09:03.624006+00:00", "kind": "readme", "missing": false, "url": "https://github.com/ztxz16/fastllm"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-29T18:17:54.046777+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "fd3d330cb89207bac94dd65a90b5b037b39fc3c252d9da026c647a1ce5cf6c05", "fetched_at": "2026-08-28T04:09:03.624006+00:00", "kind": "readme", "missing": false, "url": "https://github.com/ztxz16/fastllm"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 99, "longevity": 86, "rhythm": 35}, "computed_at": "2026-09-02T17:46:02.011165+00:00", "flags": ["no_releases"], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 1208, "days_push": 7, "days_rel": null, "gap_med": null, "n_releases_24m": 0}, "score": 74, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}