{"adoption": {"forks": 324, "observed_at": "2026-08-28T04:06:34.000095+00:00", "stars": 2284}, "canonical_url": "https://ross.abutalabs.com/products/lazynlp", "card": {"archived": false, "artifact_type": "library", "description": "Library to scrape and clean web pages to create massive datasets.", "domain": ["artificial-intelligence", "crawlers"], "enriched": true, "function": ["web-scraping", "nlp", "etl", "data-generation"], "health_score": 20, "homepage": null, "language": "Python", "license": null, "license_family": "other", "maturity": "abandoned", "member_repos": ["chiphuyen/lazynlp"], "name": "chiphuyen/lazynlp", "platform": ["python", "cli"], "pushed_at": "2020-11-11T12:16:30+00:00", "repo": "chiphuyen/lazynlp", "stars": 2284, "tags": ["web-crawling", "text-cleaning", "deduplication", "language-model-datasets", "corpus-building", "natural-language-processing", "data-engineering"], "topics": ["artificial-intelligence", "natural-language-processing", "nlp", "text-mining", "language-model", "python", "open", "data-science"], "urls": [], "use_cases": ["scrape web pages to build a large text corpus", "create a dataset larger than GPT-2's training data", "deduplicate URLs before crawling", "clean and normalize scraped web text", "collect text from Reddit, Gutenberg, and Wikipedia dumps", "build monolingual datasets for language model training"], "what_it_is": "A Python library for crawling, cleaning, and deduplicating web pages to build massive monolingual text datasets, suitable for training language models. It provides helpers for gathering URL lists from sources like Reddit, Gutenberg, and Wikipedia and processing them into large corpora.", "when_to_avoid": ["you need a maintained tool with active support", "you want a no-code scraping solution", "you need JavaScript-rendered page scraping", "you require a license for commercial use clarity"], "when_to_choose": ["you need to build a large monolingual text corpus from web pages", "you want to crawl and deduplicate URLs at scale in Python", "you're preparing pretraining data for language models"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/lazynlp", "repo": "chiphuyen/lazynlp", "role": "main", "score": 23}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:06:34.000095+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-30T02:41:05.888113+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "771b96ae9a89db083c7982c74f05c2775024943fd41cc5826b19efd93aa6f0e1", "fetched_at": "2026-08-28T04:06:34.000095+00:00", "kind": "readme", "missing": false, "url": "https://github.com/chiphuyen/lazynlp"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:06:34.000095+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-30T02:41:05.888113+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "771b96ae9a89db083c7982c74f05c2775024943fd41cc5826b19efd93aa6f0e1", "fetched_at": "2026-08-28T04:06:34.000095+00:00", "kind": "readme", "missing": false, "url": "https://github.com/chiphuyen/lazynlp"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-30T02:41:05.888113+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "771b96ae9a89db083c7982c74f05c2775024943fd41cc5826b19efd93aa6f0e1", "fetched_at": "2026-08-28T04:06:34.000095+00:00", "kind": "readme", "missing": false, "url": "https://github.com/chiphuyen/lazynlp"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:06:34.000095+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:06:34.000095+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:06:34.000095+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-30T02:41:05.888113+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "771b96ae9a89db083c7982c74f05c2775024943fd41cc5826b19efd93aa6f0e1", "fetched_at": "2026-08-28T04:06:34.000095+00:00", "kind": "readme", "missing": false, "url": "https://github.com/chiphuyen/lazynlp"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:06:34.000095+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:06:34.000095+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-30T02:41:05.888113+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "771b96ae9a89db083c7982c74f05c2775024943fd41cc5826b19efd93aa6f0e1", "fetched_at": "2026-08-28T04:06:34.000095+00:00", "kind": "readme", "missing": false, "url": "https://github.com/chiphuyen/lazynlp"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:06:34.000095+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:06:34.000095+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:06:34.000095+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-30T02:41:05.888113+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "771b96ae9a89db083c7982c74f05c2775024943fd41cc5826b19efd93aa6f0e1", "fetched_at": "2026-08-28T04:06:34.000095+00:00", "kind": "readme", "missing": false, "url": "https://github.com/chiphuyen/lazynlp"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:06:34.000095+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:06:34.000095+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-30T02:41:05.888113+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "771b96ae9a89db083c7982c74f05c2775024943fd41cc5826b19efd93aa6f0e1", "fetched_at": "2026-08-28T04:06:34.000095+00:00", "kind": "readme", "missing": false, "url": "https://github.com/chiphuyen/lazynlp"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-30T02:41:05.888113+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "771b96ae9a89db083c7982c74f05c2775024943fd41cc5826b19efd93aa6f0e1", "fetched_at": "2026-08-28T04:06:34.000095+00:00", "kind": "readme", "missing": false, "url": "https://github.com/chiphuyen/lazynlp"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-30T02:41:05.888113+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "771b96ae9a89db083c7982c74f05c2775024943fd41cc5826b19efd93aa6f0e1", "fetched_at": "2026-08-28T04:06:34.000095+00:00", "kind": "readme", "missing": false, "url": "https://github.com/chiphuyen/lazynlp"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-30T02:41:05.888113+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "771b96ae9a89db083c7982c74f05c2775024943fd41cc5826b19efd93aa6f0e1", "fetched_at": "2026-08-28T04:06:34.000095+00:00", "kind": "readme", "missing": false, "url": "https://github.com/chiphuyen/lazynlp"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 0, "longevity": 100, "rhythm": 8}, "computed_at": "2026-09-03T02:20:16.233290+00:00", "flags": ["no_license"], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 2744, "days_push": 2121, "days_rel": null, "gap_med": null, "n_releases_24m": 0}, "score": 23, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}