{"adoption": {"forks": 296, "observed_at": "2026-08-28T04:08:40.732789+00:00", "stars": 4267}, "canonical_url": "https://ross.abutalabs.com/products/mnbvc", "card": {"archived": false, "artifact_type": "dataset", "description": "MNBVC(Massive Never-ending BT Vast Chinese corpus)超大规模中文语料集。对标chatGPT训练的40T数据。MNBVC数据集不但包括主流文化，也包括各个小众文化甚至火星文的数据。MNBVC数据集包括新闻、作文、小说、书籍、杂志、论文、台词、帖子、wiki、古诗、歌词、商品介绍、笑话、糗事、聊天记录等一切形式的纯文本中文数据。", "domain": ["large-language-models", "machine-learning"], "enriched": true, "function": ["nlp", "machine-learning", "llm-training", "ocr", "data-science", "etl"], "health_score": 79, "homepage": null, "language": null, "license": "MIT", "license_family": "permissive", "maturity": "active", "member_repos": ["esbatmop/MNBVC"], "name": "esbatmop/MNBVC", "platform": ["python", "cross-platform"], "pushed_at": "2026-08-15T11:52:11+00:00", "repo": "esbatmop/MNBVC", "stars": 4267, "tags": ["chinese-corpus", "text-corpus", "llm-pretraining", "data-cleaning", "web-scraping", "multimodal", "open-dataset", "natural-language-processing", "data-engineering"], "topics": ["chinese", "chinese-language", "chinese-nlp", "chinese-simplified", "corpus-data", "nlp", "nlp-machine-learning"], "urls": [], "use_cases": ["download a large-scale Chinese corpus for LLM pretraining", "find diverse Chinese text data including niche internet slang", "get cleaned Chinese datasets from Hugging Face or ModelScope", "clean and deduplicate large Chinese text corpora", "extract text from PDFs and images for training data", "crawl GitHub repositories to build code training corpora"], "what_it_is": "MNBVC is a massive, continuously growing open-source Chinese text corpus aiming to rival the scale of data used to train ChatGPT, including news, novels, wiki, chat logs, poetry, and niche internet culture. The project also provides companion data-cleaning, OCR, deduplication, and code-repository crawling tools.", "when_to_avoid": ["you need curated, indexed, or copyright-cleared data since the project performs no copyright review", "you need small, well-annotated datasets for supervised tasks rather than raw pretraining text", "you work with non-Chinese languages primarily"], "when_to_choose": ["you need tens of terabytes of Chinese text for training or fine-tuning language models", "you want broad coverage of Chinese internet culture including minority communities", "you need companion tools for Chinese encoding detection, deduplication, and corpus cleaning"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/mnbvc", "repo": "esbatmop/MNBVC", "role": "main", "score": 75}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:08:40.732789+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-29T18:22:02.716688+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "9be53906a57d2761ff94e4a99d7903e0766a262a47ba07e99d38748ac917b997", "fetched_at": "2026-08-28T04:08:40.732789+00:00", "kind": "readme", "missing": false, "url": "https://github.com/esbatmop/MNBVC"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:08:40.732789+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-29T18:22:02.716688+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "9be53906a57d2761ff94e4a99d7903e0766a262a47ba07e99d38748ac917b997", "fetched_at": "2026-08-28T04:08:40.732789+00:00", "kind": "readme", "missing": false, "url": "https://github.com/esbatmop/MNBVC"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-29T18:22:02.716688+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "9be53906a57d2761ff94e4a99d7903e0766a262a47ba07e99d38748ac917b997", "fetched_at": "2026-08-28T04:08:40.732789+00:00", "kind": "readme", "missing": false, "url": "https://github.com/esbatmop/MNBVC"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:08:40.732789+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:08:40.732789+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:08:40.732789+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-29T18:22:02.716688+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "9be53906a57d2761ff94e4a99d7903e0766a262a47ba07e99d38748ac917b997", "fetched_at": "2026-08-28T04:08:40.732789+00:00", "kind": "readme", "missing": false, "url": "https://github.com/esbatmop/MNBVC"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:08:40.732789+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:08:40.732789+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-29T18:22:02.716688+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "9be53906a57d2761ff94e4a99d7903e0766a262a47ba07e99d38748ac917b997", "fetched_at": "2026-08-28T04:08:40.732789+00:00", "kind": "readme", "missing": false, "url": "https://github.com/esbatmop/MNBVC"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:08:40.732789+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:08:40.732789+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:08:40.732789+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-29T18:22:02.716688+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "9be53906a57d2761ff94e4a99d7903e0766a262a47ba07e99d38748ac917b997", "fetched_at": "2026-08-28T04:08:40.732789+00:00", "kind": "readme", "missing": false, "url": "https://github.com/esbatmop/MNBVC"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:08:40.732789+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:08:40.732789+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-29T18:22:02.716688+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "9be53906a57d2761ff94e4a99d7903e0766a262a47ba07e99d38748ac917b997", "fetched_at": "2026-08-28T04:08:40.732789+00:00", "kind": "readme", "missing": false, "url": "https://github.com/esbatmop/MNBVC"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-29T18:22:02.716688+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "9be53906a57d2761ff94e4a99d7903e0766a262a47ba07e99d38748ac917b997", "fetched_at": "2026-08-28T04:08:40.732789+00:00", "kind": "readme", "missing": false, "url": "https://github.com/esbatmop/MNBVC"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-29T18:22:02.716688+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "9be53906a57d2761ff94e4a99d7903e0766a262a47ba07e99d38748ac917b997", "fetched_at": "2026-08-28T04:08:40.732789+00:00", "kind": "readme", "missing": false, "url": "https://github.com/esbatmop/MNBVC"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-29T18:22:02.716688+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "9be53906a57d2761ff94e4a99d7903e0766a262a47ba07e99d38748ac917b997", "fetched_at": "2026-08-28T04:08:40.732789+00:00", "kind": "readme", "missing": false, "url": "https://github.com/esbatmop/MNBVC"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 97, "longevity": 95, "rhythm": 35}, "computed_at": "2026-09-02T17:46:02.011165+00:00", "flags": ["no_releases"], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 1341, "days_push": 18, "days_rel": null, "gap_med": null, "n_releases_24m": 0}, "score": 75, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}