{"adoption": {"forks": 131, "observed_at": "2026-08-28T04:04:11.796386+00:00", "stars": 1270}, "canonical_url": "https://ross.abutalabs.com/products/deduplicate-text-datasets", "card": {"archived": true, "artifact_type": "library", "description": null, "domain": ["machine-learning", "large-language-models"], "enriched": true, "function": ["nlp", "data-science", "etl", "parser"], "health_score": 10, "homepage": null, "language": "Rust", "license": "Apache-2.0", "license_family": "permissive", "maturity": "maintenance", "member_repos": ["google-research/deduplicate-text-datasets"], "name": "google-research/deduplicate-text-datasets", "platform": ["windows", "rust", "python", "cli"], "pushed_at": "2024-07-30T21:44:50+00:00", "repo": "google-research/deduplicate-text-datasets", "stars": 1270, "tags": ["deduplication", "suffix-array", "text-corpus-cleaning", "research-code", "llm-training-data", "natural-language-processing", "data-engineering", "linux", "macos"], "topics": [], "urls": [], "use_cases": ["remove duplicate text from an LLM training corpus", "deduplicate a scraped web dataset like C4 before training", "find how often a sequence repeats in a large text dataset", "reduce model memorization by cleaning training data", "inspect duplicate clusters in a language modeling dataset", "speed up language model training by shrinking duplicated data"], "what_it_is": "A Rust implementation of ExactSubstr deduplication for language model training datasets, with Python scripts for running deduplication and inspecting results, from the ACL 2022 paper 'Deduplicating Training Data Makes Language Models Better'. It also ships deduplicated document clusters for C4, RealNews, LM1B, and Wiki-4B-en.", "when_to_avoid": ["you need fuzzy/semantic near-duplicate detection out of the box", "you want a polished production data pipeline rather than research code", "you only have a small machine and a web-scale dataset"], "when_to_choose": ["you need exact substring deduplication at scale for text corpora", "you want the research-validated method from the Google paper", "you have a large machine (many cores, lots of RAM) for web-scale datasets"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/deduplicate-text-datasets", "repo": "google-research/deduplicate-text-datasets", "role": "main", "score": 10}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:04:11.796386+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-30T05:03:26.316376+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "45dc6e284c142244943b2375d36ee1411756562d58467b8282247ef347ec9c45", "fetched_at": "2026-08-28T04:04:11.796386+00:00", "kind": "readme", "missing": false, "url": "https://github.com/google-research/deduplicate-text-datasets"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:04:11.796386+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-30T05:03:26.316376+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "45dc6e284c142244943b2375d36ee1411756562d58467b8282247ef347ec9c45", "fetched_at": "2026-08-28T04:04:11.796386+00:00", "kind": "readme", "missing": false, "url": "https://github.com/google-research/deduplicate-text-datasets"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-30T05:03:26.316376+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "45dc6e284c142244943b2375d36ee1411756562d58467b8282247ef347ec9c45", "fetched_at": "2026-08-28T04:04:11.796386+00:00", "kind": "readme", "missing": false, "url": "https://github.com/google-research/deduplicate-text-datasets"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:04:11.796386+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:04:11.796386+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:04:11.796386+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-30T05:03:26.316376+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "45dc6e284c142244943b2375d36ee1411756562d58467b8282247ef347ec9c45", "fetched_at": "2026-08-28T04:04:11.796386+00:00", "kind": "readme", "missing": false, "url": "https://github.com/google-research/deduplicate-text-datasets"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:04:11.796386+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:04:11.796386+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-30T05:03:26.316376+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "45dc6e284c142244943b2375d36ee1411756562d58467b8282247ef347ec9c45", "fetched_at": "2026-08-28T04:04:11.796386+00:00", "kind": "readme", "missing": false, "url": "https://github.com/google-research/deduplicate-text-datasets"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:04:11.796386+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:04:11.796386+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:04:11.796386+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-30T05:03:26.316376+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "45dc6e284c142244943b2375d36ee1411756562d58467b8282247ef347ec9c45", "fetched_at": "2026-08-28T04:04:11.796386+00:00", "kind": "readme", "missing": false, "url": "https://github.com/google-research/deduplicate-text-datasets"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:04:11.796386+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:04:11.796386+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-30T05:03:26.316376+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "45dc6e284c142244943b2375d36ee1411756562d58467b8282247ef347ec9c45", "fetched_at": "2026-08-28T04:04:11.796386+00:00", "kind": "readme", "missing": false, "url": "https://github.com/google-research/deduplicate-text-datasets"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-30T05:03:26.316376+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "45dc6e284c142244943b2375d36ee1411756562d58467b8282247ef347ec9c45", "fetched_at": "2026-08-28T04:04:11.796386+00:00", "kind": "readme", "missing": false, "url": "https://github.com/google-research/deduplicate-text-datasets"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-30T05:03:26.316376+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "45dc6e284c142244943b2375d36ee1411756562d58467b8282247ef347ec9c45", "fetched_at": "2026-08-28T04:04:11.796386+00:00", "kind": "readme", "missing": false, "url": "https://github.com/google-research/deduplicate-text-datasets"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-30T05:03:26.316376+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "45dc6e284c142244943b2375d36ee1411756562d58467b8282247ef347ec9c45", "fetched_at": "2026-08-28T04:04:11.796386+00:00", "kind": "readme", "missing": false, "url": "https://github.com/google-research/deduplicate-text-datasets"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 0, "longevity": 100, "rhythm": 35}, "computed_at": "2026-09-03T02:20:16.233290+00:00", "flags": ["no_releases", "archived"], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 1875, "days_push": 764, "days_rel": null, "gap_med": null, "n_releases_24m": 0}, "score": 10, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}