{"adoption": {"forks": 2843, "observed_at": "2026-08-28T04:10:59.377144+00:00", "stars": 12588}, "canonical_url": "https://ross.abutalabs.com/products/nsfw_data_scraper", "card": {"archived": false, "artifact_type": "dataset", "description": "Collection of scripts to aggregate image data for the purposes of training an NSFW Image Classifier", "domain": ["machine-learning", "computer-vision", "image-processing"], "enriched": true, "function": ["web-scraping", "image-processing", "machine-learning", "data-generation"], "health_score": 20, "homepage": null, "language": "Shell", "license": "MIT", "license_family": "permissive", "maturity": "maintenance", "member_repos": ["alex000kim/nsfw_data_scraper"], "name": "alex000kim/nsfw_data_scraper", "platform": [], "pushed_at": "2024-01-21T23:49:42+00:00", "repo": "alex000kim/nsfw_data_scraper", "stars": 12588, "tags": ["nsfw-classifier", "content-moderation", "image-classification", "dataset-collection", "shell-scripts", "deep-learning", "docker", "linux", "shell"], "topics": ["nsfw-classifier", "nsfw", "deep-learning", "content-moderation", "pornography", "machine-learning"], "urls": [], "use_cases": ["build a dataset for training an NSFW image classifier", "collect labeled images for content moderation model training", "scrape images from subreddits into train/test splits", "create a porn vs safe-for-work image classification dataset", "gather neutral and adult image categories for deep learning experiments"], "what_it_is": "A collection of shell scripts that automatically aggregate tens of thousands of images across five categories (porn, hentai, sexy, neutral, drawings) for training an NSFW image classifier. It runs inside Docker and produces train/test splits ready for CNN training with fastai.", "when_to_avoid": ["you need a clean, curated dataset - the authors warn the data is noisy", "you need a production-ready classifier rather than raw training data", "your environment cannot run Docker or long-running scraping jobs"], "when_to_choose": ["you need a large, pre-categorized NSFW image dataset for training a classifier", "you want an automated, Docker-based scraping pipeline with train/test splitting", "you are building content moderation tooling and need training data"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/nsfw_data_scraper", "repo": "alex000kim/nsfw_data_scraper", "role": "main", "score": 32}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:10:59.377144+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-29T17:13:56.760119+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "bbd52e7c591d4feeb9d2886030e4d8a79be5508feb849d49846895324b15811b", "fetched_at": "2026-08-28T04:10:59.377144+00:00", "kind": "readme", "missing": false, "url": "https://github.com/alex000kim/nsfw_data_scraper"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:10:59.377144+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-29T17:13:56.760119+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "bbd52e7c591d4feeb9d2886030e4d8a79be5508feb849d49846895324b15811b", "fetched_at": "2026-08-28T04:10:59.377144+00:00", "kind": "readme", "missing": false, "url": "https://github.com/alex000kim/nsfw_data_scraper"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-29T17:13:56.760119+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "bbd52e7c591d4feeb9d2886030e4d8a79be5508feb849d49846895324b15811b", "fetched_at": "2026-08-28T04:10:59.377144+00:00", "kind": "readme", "missing": false, "url": "https://github.com/alex000kim/nsfw_data_scraper"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:10:59.377144+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:10:59.377144+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:10:59.377144+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-29T17:13:56.760119+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "bbd52e7c591d4feeb9d2886030e4d8a79be5508feb849d49846895324b15811b", "fetched_at": "2026-08-28T04:10:59.377144+00:00", "kind": "readme", "missing": false, "url": "https://github.com/alex000kim/nsfw_data_scraper"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:10:59.377144+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:10:59.377144+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-29T17:13:56.760119+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "bbd52e7c591d4feeb9d2886030e4d8a79be5508feb849d49846895324b15811b", "fetched_at": "2026-08-28T04:10:59.377144+00:00", "kind": "readme", "missing": false, "url": "https://github.com/alex000kim/nsfw_data_scraper"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:10:59.377144+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:10:59.377144+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:10:59.377144+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-29T17:13:56.760119+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "bbd52e7c591d4feeb9d2886030e4d8a79be5508feb849d49846895324b15811b", "fetched_at": "2026-08-28T04:10:59.377144+00:00", "kind": "readme", "missing": false, "url": "https://github.com/alex000kim/nsfw_data_scraper"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:10:59.377144+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:10:59.377144+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-29T17:13:56.760119+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "bbd52e7c591d4feeb9d2886030e4d8a79be5508feb849d49846895324b15811b", "fetched_at": "2026-08-28T04:10:59.377144+00:00", "kind": "readme", "missing": false, "url": "https://github.com/alex000kim/nsfw_data_scraper"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-29T17:13:56.760119+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "bbd52e7c591d4feeb9d2886030e4d8a79be5508feb849d49846895324b15811b", "fetched_at": "2026-08-28T04:10:59.377144+00:00", "kind": "readme", "missing": false, "url": "https://github.com/alex000kim/nsfw_data_scraper"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-29T17:13:56.760119+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "bbd52e7c591d4feeb9d2886030e4d8a79be5508feb849d49846895324b15811b", "fetched_at": "2026-08-28T04:10:59.377144+00:00", "kind": "readme", "missing": false, "url": "https://github.com/alex000kim/nsfw_data_scraper"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-29T17:13:56.760119+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "bbd52e7c591d4feeb9d2886030e4d8a79be5508feb849d49846895324b15811b", "fetched_at": "2026-08-28T04:10:59.377144+00:00", "kind": "readme", "missing": false, "url": "https://github.com/alex000kim/nsfw_data_scraper"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 0, "longevity": 100, "rhythm": 35}, "computed_at": "2026-09-03T02:39:23.370411+00:00", "flags": ["no_releases"], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 2792, "days_push": 955, "days_rel": null, "gap_med": null, "n_releases_24m": 0}, "score": 32, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}