{"adoption": {"forks": 320, "observed_at": "2026-08-28T04:04:03.020865+00:00", "stars": 1225}, "canonical_url": "https://ross.abutalabs.com/products/scrapy-cluster", "card": {"archived": true, "artifact_type": "framework", "description": "This Scrapy project uses Redis and Kafka to create a distributed on demand scraping cluster.", "domain": ["crawlers", "big-data", "microservices"], "enriched": true, "function": ["web-scraping", "message-queue", "caching", "streaming", "etl"], "health_score": 10, "homepage": "http://scrapy-cluster.readthedocs.io/", "language": "Python", "license": "MIT", "license_family": "permissive", "maturity": "maintenance", "member_repos": ["istresearch/scrapy-cluster"], "name": "istresearch/scrapy-cluster", "platform": ["python", "self-hosted"], "pushed_at": "2023-11-07T12:16:25+00:00", "repo": "istresearch/scrapy-cluster", "stars": 1225, "tags": ["scrapy", "kafka", "redis", "distributed-crawling", "spider-cluster", "data-engineering", "automation", "docker", "linux"], "topics": ["python", "scrapy", "kafka", "redis", "scraping", "distributed"], "urls": [], "use_cases": ["crawl millions of pages across a cluster of scrapy workers", "distribute seed urls to multiple spider instances via redis", "submit scraping jobs and consume results through kafka topics", "scale web crawlers horizontally without downtime", "coordinate throttling of crawls across many machines", "run on-demand scraping of arbitrary websites"], "what_it_is": "Scrapy Cluster is a distributed web scraping framework built on Scrapy that uses Redis to coordinate crawl requests and Kafka as a data bus for submitting jobs and receiving results. It allows scaling many spider instances across machines with on-demand, dynamic crawling of arbitrary URLs.", "when_to_avoid": ["you only need a simple single-machine scraper", "you don't want to operate redis, kafka, and zookeeper infrastructure", "you need actively maintained software with recent development", "your project requires python 2.7-era scrapy compatibility only"], "when_to_choose": ["you need large-scale distributed crawling beyond a single scrapy instance", "you want kafka as an integration bus for crawl jobs and results", "you need dynamic, on-demand crawling of arbitrary urls", "you want to add or remove scraper nodes without losing data"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/scrapy-cluster", "repo": "istresearch/scrapy-cluster", "role": "main", "score": 10}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:04:03.020865+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-30T06:15:23.830595+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "fe2a76acc098811a4e7fb5dc8b637906968cc1e959e17269e414627d81e7bf05", "fetched_at": "2026-08-28T04:04:03.020865+00:00", "kind": "readme", "missing": false, "url": "https://github.com/istresearch/scrapy-cluster"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:04:03.020865+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-30T06:15:23.830595+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "fe2a76acc098811a4e7fb5dc8b637906968cc1e959e17269e414627d81e7bf05", "fetched_at": "2026-08-28T04:04:03.020865+00:00", "kind": "readme", "missing": false, "url": "https://github.com/istresearch/scrapy-cluster"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-30T06:15:23.830595+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "fe2a76acc098811a4e7fb5dc8b637906968cc1e959e17269e414627d81e7bf05", "fetched_at": "2026-08-28T04:04:03.020865+00:00", "kind": "readme", "missing": false, "url": "https://github.com/istresearch/scrapy-cluster"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:04:03.020865+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:04:03.020865+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:04:03.020865+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-30T06:15:23.830595+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "fe2a76acc098811a4e7fb5dc8b637906968cc1e959e17269e414627d81e7bf05", "fetched_at": "2026-08-28T04:04:03.020865+00:00", "kind": "readme", "missing": false, "url": "https://github.com/istresearch/scrapy-cluster"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:04:03.020865+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:04:03.020865+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-30T06:15:23.830595+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "fe2a76acc098811a4e7fb5dc8b637906968cc1e959e17269e414627d81e7bf05", "fetched_at": "2026-08-28T04:04:03.020865+00:00", "kind": "readme", "missing": false, "url": "https://github.com/istresearch/scrapy-cluster"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:04:03.020865+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:04:03.020865+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:04:03.020865+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-30T06:15:23.830595+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "fe2a76acc098811a4e7fb5dc8b637906968cc1e959e17269e414627d81e7bf05", "fetched_at": "2026-08-28T04:04:03.020865+00:00", "kind": "readme", "missing": false, "url": "https://github.com/istresearch/scrapy-cluster"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:04:03.020865+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:04:03.020865+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-30T06:15:23.830595+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "fe2a76acc098811a4e7fb5dc8b637906968cc1e959e17269e414627d81e7bf05", "fetched_at": "2026-08-28T04:04:03.020865+00:00", "kind": "readme", "missing": false, "url": "https://github.com/istresearch/scrapy-cluster"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-30T06:15:23.830595+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "fe2a76acc098811a4e7fb5dc8b637906968cc1e959e17269e414627d81e7bf05", "fetched_at": "2026-08-28T04:04:03.020865+00:00", "kind": "readme", "missing": false, "url": "https://github.com/istresearch/scrapy-cluster"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-30T06:15:23.830595+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "fe2a76acc098811a4e7fb5dc8b637906968cc1e959e17269e414627d81e7bf05", "fetched_at": "2026-08-28T04:04:03.020865+00:00", "kind": "readme", "missing": false, "url": "https://github.com/istresearch/scrapy-cluster"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-30T06:15:23.830595+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "fe2a76acc098811a4e7fb5dc8b637906968cc1e959e17269e414627d81e7bf05", "fetched_at": "2026-08-28T04:04:03.020865+00:00", "kind": "readme", "missing": false, "url": "https://github.com/istresearch/scrapy-cluster"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 0, "longevity": 100, "rhythm": 8}, "computed_at": "2026-09-03T02:20:16.233290+00:00", "flags": ["archived"], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 4159, "days_push": 1030, "days_rel": null, "gap_med": null, "n_releases_24m": 0}, "score": 10, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}