{"adoption": {"forks": 983, "observed_at": "2026-08-28T04:09:34.100626+00:00", "stars": 5995}, "canonical_url": "https://ross.abutalabs.com/products/nlp-datasets", "card": {"archived": false, "artifact_type": "dataset", "description": "Alphabetical list of free/public domain datasets with text data for use in Natural Language Processing (NLP)", "domain": ["machine-learning", "data-science", "awesome-lists"], "enriched": true, "function": ["nlp", "data-science", "machine-learning"], "health_score": 20, "homepage": null, "language": null, "license": null, "license_family": "other", "maturity": "maintenance", "member_repos": ["niderhoff/nlp-datasets"], "name": "niderhoff/nlp-datasets", "platform": ["cross-platform"], "pushed_at": "2023-02-15T13:58:35+00:00", "repo": "niderhoff/nlp-datasets", "stars": 5995, "tags": ["text-corpus", "public-domain-data", "curated-list", "corpus", "text-data", "natural-language-processing"], "topics": [], "urls": [], "use_cases": ["find free text datasets for NLP projects", "locate public domain corpora for language model training", "find datasets for chatbot training", "source text data for sentiment analysis research", "find large text corpora for machine learning experiments", "discover annotated corpora and treebanks for linguistics research"], "what_it_is": "A curated alphabetical list of free and public-domain text datasets for Natural Language Processing. It catalogs raw unstructured text corpora along with links to annotated corpora and Treebank sources.", "when_to_avoid": ["you need a downloadable dataset package rather than a list of links", "you need curated, regularly updated datasets with guaranteed availability", "you need non-text multimodal datasets like images or audio"], "when_to_choose": ["you need a starting point to discover publicly available NLP text datasets", "you want free or public-domain corpora without licensing costs", "you need a variety of dataset types from reviews to web crawls to dialog corpora"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/nlp-datasets", "repo": "niderhoff/nlp-datasets", "role": "main", "score": 32}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:09:34.100626+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-29T17:49:50.633113+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "28078092463ff20e9c9f69112c871e6a7cd16a3d7d10583256de9d0062c5e0c5", "fetched_at": "2026-08-28T04:09:34.100626+00:00", "kind": "readme", "missing": false, "url": "https://github.com/niderhoff/nlp-datasets"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:09:34.100626+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-29T17:49:50.633113+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "28078092463ff20e9c9f69112c871e6a7cd16a3d7d10583256de9d0062c5e0c5", "fetched_at": "2026-08-28T04:09:34.100626+00:00", "kind": "readme", "missing": false, "url": "https://github.com/niderhoff/nlp-datasets"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-29T17:49:50.633113+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "28078092463ff20e9c9f69112c871e6a7cd16a3d7d10583256de9d0062c5e0c5", "fetched_at": "2026-08-28T04:09:34.100626+00:00", "kind": "readme", "missing": false, "url": "https://github.com/niderhoff/nlp-datasets"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:09:34.100626+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:09:34.100626+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:09:34.100626+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-29T17:49:50.633113+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "28078092463ff20e9c9f69112c871e6a7cd16a3d7d10583256de9d0062c5e0c5", "fetched_at": "2026-08-28T04:09:34.100626+00:00", "kind": "readme", "missing": false, "url": "https://github.com/niderhoff/nlp-datasets"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:09:34.100626+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:09:34.100626+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-29T17:49:50.633113+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "28078092463ff20e9c9f69112c871e6a7cd16a3d7d10583256de9d0062c5e0c5", "fetched_at": "2026-08-28T04:09:34.100626+00:00", "kind": "readme", "missing": false, "url": "https://github.com/niderhoff/nlp-datasets"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:09:34.100626+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:09:34.100626+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:09:34.100626+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-29T17:49:50.633113+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "28078092463ff20e9c9f69112c871e6a7cd16a3d7d10583256de9d0062c5e0c5", "fetched_at": "2026-08-28T04:09:34.100626+00:00", "kind": "readme", "missing": false, "url": "https://github.com/niderhoff/nlp-datasets"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:09:34.100626+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:09:34.100626+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-29T17:49:50.633113+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "28078092463ff20e9c9f69112c871e6a7cd16a3d7d10583256de9d0062c5e0c5", "fetched_at": "2026-08-28T04:09:34.100626+00:00", "kind": "readme", "missing": false, "url": "https://github.com/niderhoff/nlp-datasets"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-29T17:49:50.633113+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "28078092463ff20e9c9f69112c871e6a7cd16a3d7d10583256de9d0062c5e0c5", "fetched_at": "2026-08-28T04:09:34.100626+00:00", "kind": "readme", "missing": false, "url": "https://github.com/niderhoff/nlp-datasets"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-29T17:49:50.633113+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "28078092463ff20e9c9f69112c871e6a7cd16a3d7d10583256de9d0062c5e0c5", "fetched_at": "2026-08-28T04:09:34.100626+00:00", "kind": "readme", "missing": false, "url": "https://github.com/niderhoff/nlp-datasets"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-29T17:49:50.633113+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "28078092463ff20e9c9f69112c871e6a7cd16a3d7d10583256de9d0062c5e0c5", "fetched_at": "2026-08-28T04:09:34.100626+00:00", "kind": "readme", "missing": false, "url": "https://github.com/niderhoff/nlp-datasets"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 0, "longevity": 100, "rhythm": 35}, "computed_at": "2026-09-02T17:46:02.011165+00:00", "flags": ["no_releases", "no_license"], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 3814, "days_push": 1295, "days_rel": null, "gap_med": null, "n_releases_24m": 0}, "score": 32, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}