{"adoption": {"forks": 520, "observed_at": "2026-08-28T04:08:54.977090+00:00", "stars": 4609}, "canonical_url": "https://ross.abutalabs.com/products/self-instruct", "card": {"archived": false, "artifact_type": "library", "description": "Aligning pretrained language models with instruction data generated by themselves.", "domain": ["large-language-models", "artificial-intelligence", "machine-learning"], "enriched": true, "function": ["llm-training", "data-generation", "prompt-engineering", "machine-learning"], "health_score": 20, "homepage": null, "language": "Python", "license": "Apache-2.0", "license_family": "permissive", "maturity": "maintenance", "member_repos": ["yizhongw/self-instruct"], "name": "yizhongw/self-instruct", "platform": ["python", "cli"], "pushed_at": "2023-03-27T18:18:51+00:00", "repo": "yizhongw/self-instruct", "stars": 4609, "tags": ["instruction-tuning", "synthetic-data", "llm-alignment", "research-code", "gpt3", "natural-language-processing", "linux"], "topics": ["general-purpose-model", "language-model", "instruction-tuning"], "urls": [], "use_cases": ["generate synthetic instruction data for fine-tuning language models", "improve a model's instruction-following without manual annotation", "bootstrap a prompt dataset from a small seed set of tasks", "instruction-tune GPT-3 on model-generated data", "research bootstrapping methods for LLM alignment"], "what_it_is": "Self-Instruct is a framework and research codebase for aligning pretrained language models with instructions using data generated by the models themselves. It implements an iterative bootstrapping pipeline that generates, filters, and curates instruction data, and releases a 52K-instruction dataset for instruction-tuning.", "when_to_avoid": ["you need production-grade, actively maintained tooling", "you want to fine-tune open models with modern tooling rather than GPT-3 finetuning scripts", "you cannot tolerate noisy or biased synthetic data"], "when_to_choose": ["you need large-scale instruction-tuning data without human annotation", "you are researching self-generated or synthetic training data for LLMs", "you want to replicate the Self-Instruct paper pipeline"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/self-instruct", "repo": "yizhongw/self-instruct", "role": "main", "score": 31}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:08:54.977090+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-29T18:19:43.969167+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "beafa25e7be85dadbebd1e527c71f5edc415ba8594ac879b422c91f42eef88c3", "fetched_at": "2026-08-28T04:08:54.977090+00:00", "kind": "readme", "missing": false, "url": "https://github.com/yizhongw/self-instruct"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:08:54.977090+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-29T18:19:43.969167+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "beafa25e7be85dadbebd1e527c71f5edc415ba8594ac879b422c91f42eef88c3", "fetched_at": "2026-08-28T04:08:54.977090+00:00", "kind": "readme", "missing": false, "url": "https://github.com/yizhongw/self-instruct"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-29T18:19:43.969167+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "beafa25e7be85dadbebd1e527c71f5edc415ba8594ac879b422c91f42eef88c3", "fetched_at": "2026-08-28T04:08:54.977090+00:00", "kind": "readme", "missing": false, "url": "https://github.com/yizhongw/self-instruct"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:08:54.977090+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:08:54.977090+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:08:54.977090+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-29T18:19:43.969167+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "beafa25e7be85dadbebd1e527c71f5edc415ba8594ac879b422c91f42eef88c3", "fetched_at": "2026-08-28T04:08:54.977090+00:00", "kind": "readme", "missing": false, "url": "https://github.com/yizhongw/self-instruct"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:08:54.977090+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:08:54.977090+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-29T18:19:43.969167+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "beafa25e7be85dadbebd1e527c71f5edc415ba8594ac879b422c91f42eef88c3", "fetched_at": "2026-08-28T04:08:54.977090+00:00", "kind": "readme", "missing": false, "url": "https://github.com/yizhongw/self-instruct"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:08:54.977090+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:08:54.977090+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:08:54.977090+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-29T18:19:43.969167+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "beafa25e7be85dadbebd1e527c71f5edc415ba8594ac879b422c91f42eef88c3", "fetched_at": "2026-08-28T04:08:54.977090+00:00", "kind": "readme", "missing": false, "url": "https://github.com/yizhongw/self-instruct"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:08:54.977090+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:08:54.977090+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-29T18:19:43.969167+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "beafa25e7be85dadbebd1e527c71f5edc415ba8594ac879b422c91f42eef88c3", "fetched_at": "2026-08-28T04:08:54.977090+00:00", "kind": "readme", "missing": false, "url": "https://github.com/yizhongw/self-instruct"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-29T18:19:43.969167+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "beafa25e7be85dadbebd1e527c71f5edc415ba8594ac879b422c91f42eef88c3", "fetched_at": "2026-08-28T04:08:54.977090+00:00", "kind": "readme", "missing": false, "url": "https://github.com/yizhongw/self-instruct"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-29T18:19:43.969167+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "beafa25e7be85dadbebd1e527c71f5edc415ba8594ac879b422c91f42eef88c3", "fetched_at": "2026-08-28T04:08:54.977090+00:00", "kind": "readme", "missing": false, "url": "https://github.com/yizhongw/self-instruct"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-29T18:19:43.969167+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "beafa25e7be85dadbebd1e527c71f5edc415ba8594ac879b422c91f42eef88c3", "fetched_at": "2026-08-28T04:08:54.977090+00:00", "kind": "readme", "missing": false, "url": "https://github.com/yizhongw/self-instruct"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 0, "longevity": 96, "rhythm": 35}, "computed_at": "2026-09-03T02:20:16.233290+00:00", "flags": ["no_releases"], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 1352, "days_push": 1255, "days_rel": null, "gap_med": null, "n_releases_24m": 0}, "score": 31, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}