{"adoption": {"forks": 134, "observed_at": "2026-08-28T04:03:33.400357+00:00", "stars": 1093}, "canonical_url": "https://ross.abutalabs.com/products/nlpdataset", "card": {"archived": false, "artifact_type": "dataset", "description": "记录本人整理的一些数据集", "domain": ["machine-learning", "data-science"], "enriched": true, "function": ["nlp", "machine-learning", "data-science"], "health_score": 20, "homepage": null, "language": null, "license": "Apache-2.0", "license_family": "permissive", "maturity": "stable", "member_repos": ["liucongg/NLPDataSet"], "name": "liucongg/NLPDataSet", "platform": [], "pushed_at": "2022-06-16T07:15:27+00:00", "repo": "liucongg/NLPDataSet", "stars": 1093, "tags": ["chinese-nlp", "named-entity-recognition", "ner-dataset", "bio-format", "text-summarization", "reading-comprehension", "text-similarity", "medical-nlp", "dataset-collection", "corpus", "natural-language-processing"], "topics": [], "urls": [], "use_cases": ["download chinese NER training data", "chinese medical named entity recognition dataset from electronic medical records", "chinese text similarity dataset like LCQMC", "chinese extractive reading comprehension QA corpus", "chinese abstractive summarization dataset", "unified BIO-format chinese ner corpus combining MSRA CLUENER CMeEE", "resume or finance domain NER dataset for model training"], "what_it_is": "A curated collection of Chinese NLP datasets gathered and cleaned by the author, covering named entity recognition (NER), text summarization, extractive reading comprehension (QA), and text similarity tasks. It consolidates 22 Chinese NER sources—spanning medical records, finance, e-commerce, social media, and news—into a unified BIO-tagged format.", "when_to_avoid": ["you need English or multilingual datasets - everything here is Chinese", "you require the latest official versions or exact per-source licensing - data was aggregated from third parties and last updated in mid-2022", "you need nested entity annotations preserved - nested entities were flattened during BIO conversion", "you need programmatic or API access - distribution is via a Baidu Netdisk download link"], "when_to_choose": ["you need many Chinese NER sources pre-cleaned into one consistent BIO-tagged format", "you want a single aggregation of Chinese NLP benchmarks across medical, finance, e-commerce, and social media domains", "you are training or benchmarking Chinese NER, QA, summarization, or similarity models and need ready-made data"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/nlpdataset", "repo": "liucongg/NLPDataSet", "role": "main", "score": 32}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:03:33.400357+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-30T06:48:57.367675+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "8cbdd754ce441d307395edd0c1b488d320308bea13a18996ac1eeea1738f98f8", "fetched_at": "2026-08-28T04:03:33.400357+00:00", "kind": "readme", "missing": false, "url": "https://github.com/liucongg/NLPDataSet"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:03:33.400357+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-30T06:48:57.367675+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "8cbdd754ce441d307395edd0c1b488d320308bea13a18996ac1eeea1738f98f8", "fetched_at": "2026-08-28T04:03:33.400357+00:00", "kind": "readme", "missing": false, "url": "https://github.com/liucongg/NLPDataSet"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-30T06:48:57.367675+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "8cbdd754ce441d307395edd0c1b488d320308bea13a18996ac1eeea1738f98f8", "fetched_at": "2026-08-28T04:03:33.400357+00:00", "kind": "readme", "missing": false, "url": "https://github.com/liucongg/NLPDataSet"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:03:33.400357+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:03:33.400357+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:03:33.400357+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-30T06:48:57.367675+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "8cbdd754ce441d307395edd0c1b488d320308bea13a18996ac1eeea1738f98f8", "fetched_at": "2026-08-28T04:03:33.400357+00:00", "kind": "readme", "missing": false, "url": "https://github.com/liucongg/NLPDataSet"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:03:33.400357+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:03:33.400357+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-30T06:48:57.367675+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "8cbdd754ce441d307395edd0c1b488d320308bea13a18996ac1eeea1738f98f8", "fetched_at": "2026-08-28T04:03:33.400357+00:00", "kind": "readme", "missing": false, "url": "https://github.com/liucongg/NLPDataSet"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:03:33.400357+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:03:33.400357+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:03:33.400357+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-30T06:48:57.367675+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "8cbdd754ce441d307395edd0c1b488d320308bea13a18996ac1eeea1738f98f8", "fetched_at": "2026-08-28T04:03:33.400357+00:00", "kind": "readme", "missing": false, "url": "https://github.com/liucongg/NLPDataSet"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:03:33.400357+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:03:33.400357+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-30T06:48:57.367675+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "8cbdd754ce441d307395edd0c1b488d320308bea13a18996ac1eeea1738f98f8", "fetched_at": "2026-08-28T04:03:33.400357+00:00", "kind": "readme", "missing": false, "url": "https://github.com/liucongg/NLPDataSet"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-30T06:48:57.367675+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "8cbdd754ce441d307395edd0c1b488d320308bea13a18996ac1eeea1738f98f8", "fetched_at": "2026-08-28T04:03:33.400357+00:00", "kind": "readme", "missing": false, "url": "https://github.com/liucongg/NLPDataSet"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-30T06:48:57.367675+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "8cbdd754ce441d307395edd0c1b488d320308bea13a18996ac1eeea1738f98f8", "fetched_at": "2026-08-28T04:03:33.400357+00:00", "kind": "readme", "missing": false, "url": "https://github.com/liucongg/NLPDataSet"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-30T06:48:57.367675+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "8cbdd754ce441d307395edd0c1b488d320308bea13a18996ac1eeea1738f98f8", "fetched_at": "2026-08-28T04:03:33.400357+00:00", "kind": "readme", "missing": false, "url": "https://github.com/liucongg/NLPDataSet"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 0, "longevity": 100, "rhythm": 35}, "computed_at": "2026-09-02T17:46:02.011165+00:00", "flags": ["no_releases"], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 1856, "days_push": 1539, "days_rel": null, "gap_med": null, "n_releases_24m": 0}, "score": 32, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}