{"adoption": {"forks": 933, "observed_at": "2026-08-28T04:09:43.472025+00:00", "stars": 6447}, "canonical_url": "https://ross.abutalabs.com/products/fastertransformer", "card": {"archived": false, "artifact_type": "library", "description": "Transformer related optimization, including BERT, GPT", "domain": ["large-language-models", "deep-learning", "gpu-computing"], "enriched": true, "function": ["llm-inference", "deep-learning", "machine-learning", "gpu-computing"], "health_score": 20, "homepage": null, "language": "C++", "license": "Apache-2.0", "license_family": "permissive", "maturity": "maintenance", "member_repos": ["NVIDIA/FasterTransformer"], "name": "NVIDIA/FasterTransformer", "platform": ["cpp", "python"], "pushed_at": "2024-03-27T11:25:30+00:00", "repo": "NVIDIA/FasterTransformer", "stars": 6447, "tags": ["transformer", "bert", "gpt", "cuda", "tensor-cores", "inference-optimization", "tensor-parallelism", "pipeline-parallelism", "nvidia", "superseded-by-tensorrt-llm", "natural-language-processing", "linux", "gpu", "docker"], "topics": ["pytorch", "transformer", "gpt", "bert"], "urls": [], "use_cases": ["run fast BERT inference on NVIDIA GPUs", "serve GPT models with tensor parallelism", "speed up transformer encoder and decoder inference", "use FP16 or INT8 transformer inference with Tensor Cores", "benchmark optimized transformer decoding performance", "integrate optimized transformer kernels into PyTorch or TensorFlow"], "what_it_is": "NVIDIA's highly optimized C++/CUDA library for fast inference of Transformer-based models such as BERT, GPT, and encoder-decoder models, with TensorFlow, PyTorch, and Triton APIs. It leverages Tensor Cores with FP16/INT8/FP8 precision and supports tensor and pipeline parallelism, but development has moved to TensorRT-LLM.", "when_to_avoid": ["you are starting a new project - use TensorRT-LLM instead, since FasterTransformer is no longer developed", "you need the latest model support or FP8/Hopper optimizations", "you do not have NVIDIA Volta or newer GPUs"], "when_to_choose": ["you need maximum transformer inference throughput on NVIDIA GPUs and can use an older stack", "you want multi-GPU tensor or pipeline parallel inference for GPT-style models", "you need FP16/INT8 optimized BERT or GPT kernels in TensorFlow, PyTorch, or Triton"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/fastertransformer", "repo": "NVIDIA/FasterTransformer", "role": "main", "score": 23}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:09:43.472025+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-29T17:44:40.097132+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "8a7123d2da2c27171de95af89a448d7d394f7f7dc1b73dc6e9731c11c627a08d", "fetched_at": "2026-08-28T04:09:43.472025+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NVIDIA/FasterTransformer"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:09:43.472025+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-29T17:44:40.097132+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "8a7123d2da2c27171de95af89a448d7d394f7f7dc1b73dc6e9731c11c627a08d", "fetched_at": "2026-08-28T04:09:43.472025+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NVIDIA/FasterTransformer"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-29T17:44:40.097132+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "8a7123d2da2c27171de95af89a448d7d394f7f7dc1b73dc6e9731c11c627a08d", "fetched_at": "2026-08-28T04:09:43.472025+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NVIDIA/FasterTransformer"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:09:43.472025+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:09:43.472025+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:09:43.472025+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-29T17:44:40.097132+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "8a7123d2da2c27171de95af89a448d7d394f7f7dc1b73dc6e9731c11c627a08d", "fetched_at": "2026-08-28T04:09:43.472025+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NVIDIA/FasterTransformer"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:09:43.472025+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:09:43.472025+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-29T17:44:40.097132+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "8a7123d2da2c27171de95af89a448d7d394f7f7dc1b73dc6e9731c11c627a08d", "fetched_at": "2026-08-28T04:09:43.472025+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NVIDIA/FasterTransformer"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:09:43.472025+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:09:43.472025+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:09:43.472025+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-29T17:44:40.097132+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "8a7123d2da2c27171de95af89a448d7d394f7f7dc1b73dc6e9731c11c627a08d", "fetched_at": "2026-08-28T04:09:43.472025+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NVIDIA/FasterTransformer"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:09:43.472025+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:09:43.472025+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-29T17:44:40.097132+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "8a7123d2da2c27171de95af89a448d7d394f7f7dc1b73dc6e9731c11c627a08d", "fetched_at": "2026-08-28T04:09:43.472025+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NVIDIA/FasterTransformer"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-29T17:44:40.097132+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "8a7123d2da2c27171de95af89a448d7d394f7f7dc1b73dc6e9731c11c627a08d", "fetched_at": "2026-08-28T04:09:43.472025+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NVIDIA/FasterTransformer"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-29T17:44:40.097132+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "8a7123d2da2c27171de95af89a448d7d394f7f7dc1b73dc6e9731c11c627a08d", "fetched_at": "2026-08-28T04:09:43.472025+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NVIDIA/FasterTransformer"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-29T17:44:40.097132+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "8a7123d2da2c27171de95af89a448d7d394f7f7dc1b73dc6e9731c11c627a08d", "fetched_at": "2026-08-28T04:09:43.472025+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NVIDIA/FasterTransformer"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 0, "longevity": 100, "rhythm": 8}, "computed_at": "2026-09-03T02:20:16.233290+00:00", "flags": [], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 1979, "days_push": 889, "days_rel": null, "gap_med": null, "n_releases_24m": 0}, "score": 23, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}