{"adoption": {"forks": 103, "observed_at": "2026-08-28T04:04:42.160985+00:00", "stars": 1429}, "canonical_url": "https://ross.abutalabs.com/products/moss-rlhf", "card": {"archived": false, "artifact_type": "library", "description": "Secrets of RLHF in Large Language Models Part I: PPO", "domain": ["large-language-models", "machine-learning", "deep-learning", "artificial-intelligence"], "enriched": true, "function": ["llm-training", "machine-learning", "deep-learning", "rag"], "health_score": 20, "homepage": null, "language": "Python", "license": "Apache-2.0", "license_family": "permissive", "maturity": "maintenance", "member_repos": ["OpenLMLab/MOSS-RLHF"], "name": "OpenLMLab/MOSS-RLHF", "platform": ["python"], "pushed_at": "2024-03-03T04:56:49+00:00", "repo": "OpenLMLab/MOSS-RLHF", "stars": 1429, "tags": ["rlhf", "ppo", "reward-model", "alignment", "ai-safety", "research-code", "llm-alignment", "gpu", "linux"], "topics": ["rlhf", "alignment", "ai-safety"], "urls": [], "use_cases": ["train an LLM with PPO-based RLHF", "train a reward model from human preference data", "reproduce RLHF alignment experiments from the paper", "download pretrained 7B reward and policy models", "use a cleaned HH-RLHF dataset with preference strength labels", "study reward model strength measurement for alignment research"], "what_it_is": "MOSS-RLHF is the open-source companion code for the paper 'Secrets of RLHF in Large Language Models Part I: PPO', providing implementations of PPO-based RLHF training and reward model training for large language models. It also releases 7B English and Chinese reward models, SFT and policy models, and a preference-strength-annotated HH-RLHF dataset.", "when_to_avoid": ["you need a production-ready, actively maintained RLHF training framework", "you want scalable multi-node training with the latest algorithm variants like GRPO or DPO out of the box", "you need commercial use of the released models (model weights are AGPL-3.0, data is CC BY-NC 4.0)", "you expect frequent updates - the last release was March 2024"], "when_to_choose": ["you want a research-grade reference implementation of PPO for LLM alignment", "you need reward model training code with an annotated preference dataset", "you are studying RLHF mechanics or writing an alignment paper", "you want 7B English/Chinese reward models to score LLM outputs"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/moss-rlhf", "repo": "OpenLMLab/MOSS-RLHF", "role": "main", "score": 29}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:04:42.160985+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-30T04:37:14.033122+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "ddd40fafaf8cba0c796007b1f44b0b9065a3f99b3a45d7bc85617f7db6c12edd", "fetched_at": "2026-08-28T04:04:42.160985+00:00", "kind": "readme", "missing": false, "url": "https://github.com/OpenLMLab/MOSS-RLHF"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:04:42.160985+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-30T04:37:14.033122+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "ddd40fafaf8cba0c796007b1f44b0b9065a3f99b3a45d7bc85617f7db6c12edd", "fetched_at": "2026-08-28T04:04:42.160985+00:00", "kind": "readme", "missing": false, "url": "https://github.com/OpenLMLab/MOSS-RLHF"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-30T04:37:14.033122+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "ddd40fafaf8cba0c796007b1f44b0b9065a3f99b3a45d7bc85617f7db6c12edd", "fetched_at": "2026-08-28T04:04:42.160985+00:00", "kind": "readme", "missing": false, "url": "https://github.com/OpenLMLab/MOSS-RLHF"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:04:42.160985+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:04:42.160985+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:04:42.160985+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-30T04:37:14.033122+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "ddd40fafaf8cba0c796007b1f44b0b9065a3f99b3a45d7bc85617f7db6c12edd", "fetched_at": "2026-08-28T04:04:42.160985+00:00", "kind": "readme", "missing": false, "url": "https://github.com/OpenLMLab/MOSS-RLHF"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:04:42.160985+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:04:42.160985+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-30T04:37:14.033122+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "ddd40fafaf8cba0c796007b1f44b0b9065a3f99b3a45d7bc85617f7db6c12edd", "fetched_at": "2026-08-28T04:04:42.160985+00:00", "kind": "readme", "missing": false, "url": "https://github.com/OpenLMLab/MOSS-RLHF"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:04:42.160985+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:04:42.160985+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:04:42.160985+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-30T04:37:14.033122+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "ddd40fafaf8cba0c796007b1f44b0b9065a3f99b3a45d7bc85617f7db6c12edd", "fetched_at": "2026-08-28T04:04:42.160985+00:00", "kind": "readme", "missing": false, "url": "https://github.com/OpenLMLab/MOSS-RLHF"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:04:42.160985+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:04:42.160985+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-30T04:37:14.033122+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "ddd40fafaf8cba0c796007b1f44b0b9065a3f99b3a45d7bc85617f7db6c12edd", "fetched_at": "2026-08-28T04:04:42.160985+00:00", "kind": "readme", "missing": false, "url": "https://github.com/OpenLMLab/MOSS-RLHF"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-30T04:37:14.033122+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "ddd40fafaf8cba0c796007b1f44b0b9065a3f99b3a45d7bc85617f7db6c12edd", "fetched_at": "2026-08-28T04:04:42.160985+00:00", "kind": "readme", "missing": false, "url": "https://github.com/OpenLMLab/MOSS-RLHF"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-30T04:37:14.033122+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "ddd40fafaf8cba0c796007b1f44b0b9065a3f99b3a45d7bc85617f7db6c12edd", "fetched_at": "2026-08-28T04:04:42.160985+00:00", "kind": "readme", "missing": false, "url": "https://github.com/OpenLMLab/MOSS-RLHF"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-30T04:37:14.033122+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "ddd40fafaf8cba0c796007b1f44b0b9065a3f99b3a45d7bc85617f7db6c12edd", "fetched_at": "2026-08-28T04:04:42.160985+00:00", "kind": "readme", "missing": false, "url": "https://github.com/OpenLMLab/MOSS-RLHF"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 0, "longevity": 82, "rhythm": 35}, "computed_at": "2026-09-02T17:46:02.011165+00:00", "flags": ["no_releases"], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 1155, "days_push": 913, "days_rel": null, "gap_med": null, "n_releases_24m": 0}, "score": 29, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}