{"adoption": {"forks": 99, "observed_at": "2026-08-28T04:05:51.082982+00:00", "stars": 1897}, "canonical_url": "https://ross.abutalabs.com/products/grpo-zero", "card": {"archived": false, "artifact_type": "library", "description": "Implementing DeepSeek R1's GRPO algorithm from scratch", "domain": ["large-language-models", "reinforcement-learning", "deep-learning", "machine-learning"], "enriched": true, "function": ["llm-training", "reinforcement-learning", "machine-learning"], "health_score": 30, "homepage": null, "language": "Python", "license": "Apache-2.0", "license_family": "permissive", "maturity": "active", "member_repos": ["policy-gradient/GRPO-Zero"], "name": "policy-gradient/GRPO-Zero", "platform": ["python"], "pushed_at": "2025-04-18T14:13:35+00:00", "repo": "policy-gradient/GRPO-Zero", "stars": 1897, "tags": ["grpo", "rlhf", "policy-gradient", "deepseek-r1", "pytorch", "from-scratch", "low-vram", "dapo", "gpu", "linux"], "topics": [], "urls": [], "use_cases": ["train an LLM with GRPO reinforcement learning from scratch", "run RLHF-style policy gradient training on a single consumer GPU", "learn how GRPO works by reading a minimal implementation", "fine-tune Qwen2.5 models on reasoning tasks with reinforcement learning", "train LLMs without transformers or vLLM dependencies", "experiment with DAPO improvements to GRPO"], "what_it_is": "A minimal from-scratch Python implementation of DeepSeek's GRPO (Group Relative Policy Optimization) algorithm for reinforcement learning training of large language models, depending only on PyTorch and tokenizers. It includes DAPO improvements like token-level policy gradient loss and KL divergence removal, and runs on a single 24-48GB GPU.", "when_to_avoid": ["you need a production-scale RLHF pipeline for large models", "you want built-in support for many model architectures or distributed training", "you need a turnkey fine-tuning framework with extensive configs"], "when_to_choose": ["you want a minimal, dependency-light GRPO implementation that fits in 24-48GB VRAM", "you want to understand or modify the GRPO algorithm internals", "you're training small models on verifiable-reward tasks like CountDown"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/grpo-zero", "repo": "policy-gradient/GRPO-Zero", "role": "main", "score": 27}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:05:51.082982+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-30T03:12:48.690969+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "8b2c5250a90ac5dca1706126fccc5dac4da1ff5959c350e9c04257a7971f946d", "fetched_at": "2026-08-28T04:05:51.082982+00:00", "kind": "readme", "missing": false, "url": "https://github.com/policy-gradient/GRPO-Zero"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:05:51.082982+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-30T03:12:48.690969+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "8b2c5250a90ac5dca1706126fccc5dac4da1ff5959c350e9c04257a7971f946d", "fetched_at": "2026-08-28T04:05:51.082982+00:00", "kind": "readme", "missing": false, "url": "https://github.com/policy-gradient/GRPO-Zero"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-30T03:12:48.690969+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "8b2c5250a90ac5dca1706126fccc5dac4da1ff5959c350e9c04257a7971f946d", "fetched_at": "2026-08-28T04:05:51.082982+00:00", "kind": "readme", "missing": false, "url": "https://github.com/policy-gradient/GRPO-Zero"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:05:51.082982+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:05:51.082982+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:05:51.082982+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-30T03:12:48.690969+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "8b2c5250a90ac5dca1706126fccc5dac4da1ff5959c350e9c04257a7971f946d", "fetched_at": "2026-08-28T04:05:51.082982+00:00", "kind": "readme", "missing": false, "url": "https://github.com/policy-gradient/GRPO-Zero"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:05:51.082982+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:05:51.082982+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-30T03:12:48.690969+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "8b2c5250a90ac5dca1706126fccc5dac4da1ff5959c350e9c04257a7971f946d", "fetched_at": "2026-08-28T04:05:51.082982+00:00", "kind": "readme", "missing": false, "url": "https://github.com/policy-gradient/GRPO-Zero"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:05:51.082982+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:05:51.082982+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:05:51.082982+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-30T03:12:48.690969+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "8b2c5250a90ac5dca1706126fccc5dac4da1ff5959c350e9c04257a7971f946d", "fetched_at": "2026-08-28T04:05:51.082982+00:00", "kind": "readme", "missing": false, "url": "https://github.com/policy-gradient/GRPO-Zero"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:05:51.082982+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:05:51.082982+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-30T03:12:48.690969+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "8b2c5250a90ac5dca1706126fccc5dac4da1ff5959c350e9c04257a7971f946d", "fetched_at": "2026-08-28T04:05:51.082982+00:00", "kind": "readme", "missing": false, "url": "https://github.com/policy-gradient/GRPO-Zero"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-30T03:12:48.690969+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "8b2c5250a90ac5dca1706126fccc5dac4da1ff5959c350e9c04257a7971f946d", "fetched_at": "2026-08-28T04:05:51.082982+00:00", "kind": "readme", "missing": false, "url": "https://github.com/policy-gradient/GRPO-Zero"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-30T03:12:48.690969+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "8b2c5250a90ac5dca1706126fccc5dac4da1ff5959c350e9c04257a7971f946d", "fetched_at": "2026-08-28T04:05:51.082982+00:00", "kind": "readme", "missing": false, "url": "https://github.com/policy-gradient/GRPO-Zero"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-30T03:12:48.690969+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "8b2c5250a90ac5dca1706126fccc5dac4da1ff5959c350e9c04257a7971f946d", "fetched_at": "2026-08-28T04:05:51.082982+00:00", "kind": "readme", "missing": false, "url": "https://github.com/policy-gradient/GRPO-Zero"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 17, "longevity": 36, "rhythm": 35}, "computed_at": "2026-09-02T17:46:02.011165+00:00", "flags": ["no_releases"], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 512, "days_push": 502, "days_rel": null, "gap_med": null, "n_releases_24m": 0}, "score": 27, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}