{"adoption": {"forks": 486, "observed_at": "2026-08-28T04:08:58.850783+00:00", "stars": 4755}, "canonical_url": "https://ross.abutalabs.com/products/trlx", "card": {"archived": false, "artifact_type": "library", "description": "A repo for distributed training of language models with Reinforcement Learning via Human Feedback (RLHF)", "domain": ["large-language-models", "reinforcement-learning", "machine-learning", "deep-learning"], "enriched": true, "function": ["llm-training", "machine-learning", "reinforcement-learning"], "health_score": 21, "homepage": null, "language": "Python", "license": "MIT", "license_family": "permissive", "maturity": "maintenance", "member_repos": ["CarperAI/trlx"], "name": "CarperAI/trlx", "platform": ["python"], "pushed_at": "2024-01-08T20:07:19+00:00", "repo": "CarperAI/trlx", "stars": 4755, "tags": ["rlhf", "ppo", "ilql", "distributed-training", "pytorch", "huggingface", "nemo", "fine-tuning", "gpu", "linux"], "topics": ["machine-learning", "pytorch", "reinforcement-learning"], "urls": [], "use_cases": ["fine-tune an LLM with RLHF using a reward function", "train a 20B parameter language model with PPO", "apply ILQL to a reward-labeled dataset", "distributed RL fine-tuning of GPT-NeoX or Flan-T5", "align a language model with human preferences"], "what_it_is": "trlX is a distributed training framework for fine-tuning large language models with reinforcement learning from human feedback (RLHF), supporting PPO and ILQL algorithms. It provides Accelerate-backed trainers for models up to 20B parameters and NVIDIA NeMo-backed trainers for larger models.", "when_to_avoid": ["you only need supervised fine-tuning without reinforcement learning", "you want actively maintained tooling - the project is in maintenance mode and TRL is the more active successor", "you work outside the PyTorch/Hugging Face ecosystem"], "when_to_choose": ["you need distributed RLHF training for large causal or T5-based language models", "you want PPO or ILQL implementations on top of Hugging Face models", "you need to scale fine-tuning beyond 20B parameters with NeMo"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/trlx", "repo": "CarperAI/trlx", "role": "main", "score": 23}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:08:58.850783+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-29T18:18:53.522918+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "2873791643e937223d02dacbb747b10f9c54540900e29e5cc39a80775c8edb15", "fetched_at": "2026-08-28T04:08:58.850783+00:00", "kind": "readme", "missing": false, "url": "https://github.com/CarperAI/trlx"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:08:58.850783+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-29T18:18:53.522918+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "2873791643e937223d02dacbb747b10f9c54540900e29e5cc39a80775c8edb15", "fetched_at": "2026-08-28T04:08:58.850783+00:00", "kind": "readme", "missing": false, "url": "https://github.com/CarperAI/trlx"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-29T18:18:53.522918+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "2873791643e937223d02dacbb747b10f9c54540900e29e5cc39a80775c8edb15", "fetched_at": "2026-08-28T04:08:58.850783+00:00", "kind": "readme", "missing": false, "url": "https://github.com/CarperAI/trlx"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:08:58.850783+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:08:58.850783+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:08:58.850783+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-29T18:18:53.522918+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "2873791643e937223d02dacbb747b10f9c54540900e29e5cc39a80775c8edb15", "fetched_at": "2026-08-28T04:08:58.850783+00:00", "kind": "readme", "missing": false, "url": "https://github.com/CarperAI/trlx"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:08:58.850783+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:08:58.850783+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-29T18:18:53.522918+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "2873791643e937223d02dacbb747b10f9c54540900e29e5cc39a80775c8edb15", "fetched_at": "2026-08-28T04:08:58.850783+00:00", "kind": "readme", "missing": false, "url": "https://github.com/CarperAI/trlx"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:08:58.850783+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:08:58.850783+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:08:58.850783+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-29T18:18:53.522918+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "2873791643e937223d02dacbb747b10f9c54540900e29e5cc39a80775c8edb15", "fetched_at": "2026-08-28T04:08:58.850783+00:00", "kind": "readme", "missing": false, "url": "https://github.com/CarperAI/trlx"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:08:58.850783+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:08:58.850783+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-29T18:18:53.522918+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "2873791643e937223d02dacbb747b10f9c54540900e29e5cc39a80775c8edb15", "fetched_at": "2026-08-28T04:08:58.850783+00:00", "kind": "readme", "missing": false, "url": "https://github.com/CarperAI/trlx"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-29T18:18:53.522918+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "2873791643e937223d02dacbb747b10f9c54540900e29e5cc39a80775c8edb15", "fetched_at": "2026-08-28T04:08:58.850783+00:00", "kind": "readme", "missing": false, "url": "https://github.com/CarperAI/trlx"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-29T18:18:53.522918+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "2873791643e937223d02dacbb747b10f9c54540900e29e5cc39a80775c8edb15", "fetched_at": "2026-08-28T04:08:58.850783+00:00", "kind": "readme", "missing": false, "url": "https://github.com/CarperAI/trlx"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-29T18:18:53.522918+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "2873791643e937223d02dacbb747b10f9c54540900e29e5cc39a80775c8edb15", "fetched_at": "2026-08-28T04:08:58.850783+00:00", "kind": "readme", "missing": false, "url": "https://github.com/CarperAI/trlx"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 0, "longevity": 100, "rhythm": 8}, "computed_at": "2026-09-03T02:20:16.233290+00:00", "flags": [], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 1430, "days_push": 968, "days_rel": null, "gap_med": null, "n_releases_24m": 0}, "score": 23, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}