{"adoption": {"forks": 684, "observed_at": "2026-08-28T04:06:09.152177+00:00", "stars": 2050}, "canonical_url": "https://ross.abutalabs.com/products/docker-spark", "card": {"archived": false, "artifact_type": "infra-config", "description": "Apache Spark docker image ", "domain": ["big-data", "microservices"], "enriched": true, "function": ["container-runtime", "deployment", "developer-tools", "data-science", "etl"], "health_score": 67, "homepage": null, "language": "Shell", "license": null, "license_family": "other", "maturity": "active", "member_repos": ["big-data-europe/docker-spark"], "name": "big-data-europe/docker-spark", "platform": ["cross-platform"], "pushed_at": "2026-04-20T23:29:19+00:00", "repo": "big-data-europe/docker-spark", "stars": 2050, "tags": ["apache-spark", "docker-images", "spark-cluster", "docker-compose", "spark-master", "spark-worker", "hadoop", "standalone-cluster", "big-data-europe", "containerization", "data-engineering", "containers", "devops", "docker", "kubernetes", "linux"], "topics": ["spark-kubernetes", "kubernetes", "k8s-spark", "docker", "apache-spark"], "urls": [], "use_cases": ["spin up a standalone apache spark cluster with docker compose", "run spark master and worker containers for big data pipelines", "containerize spark applications in java scala or python", "deploy spark on kubernetes with docker images", "test spark jobs locally without installing spark", "integrate spark into a bde big data europe pipeline"], "what_it_is": "A set of Dockerfiles and Docker Compose configurations for building Apache Spark Docker images that run a standalone Spark cluster with one master and multiple workers. It supports many Spark/Hadoop/OpenJDK version combinations and lets users build Spark applications in Java, Scala, or Python to run on the cluster.", "when_to_avoid": ["you need a fully managed spark service like databricks or emr", "you require the newest spark version not yet covered by these images", "you prefer official apache spark binaries or spark kubernetes operator for production", "you need a license-cleared dependency since the repo has no explicit license"], "when_to_choose": ["you want a quick containerized spark cluster for development or testing", "you need specific spark hadoop and jdk version combinations as docker images", "you use docker compose or kubernetes to orchestrate spark master and workers", "you want to build spark apps against a consistent containerized environment"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/docker-spark", "repo": "big-data-europe/docker-spark", "role": "main", "score": 67}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:06:09.152177+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-30T02:57:35.543655+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "0233528fe714559040cd29a5a17c4d4701f4aa5d81fa3f5f3ff082aa0b8cf965", "fetched_at": "2026-08-28T04:06:09.152177+00:00", "kind": "readme", "missing": false, "url": "https://github.com/big-data-europe/docker-spark"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:06:09.152177+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-30T02:57:35.543655+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "0233528fe714559040cd29a5a17c4d4701f4aa5d81fa3f5f3ff082aa0b8cf965", "fetched_at": "2026-08-28T04:06:09.152177+00:00", "kind": "readme", "missing": false, "url": "https://github.com/big-data-europe/docker-spark"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-30T02:57:35.543655+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "0233528fe714559040cd29a5a17c4d4701f4aa5d81fa3f5f3ff082aa0b8cf965", "fetched_at": "2026-08-28T04:06:09.152177+00:00", "kind": "readme", "missing": false, "url": "https://github.com/big-data-europe/docker-spark"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:06:09.152177+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:06:09.152177+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:06:09.152177+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-30T02:57:35.543655+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "0233528fe714559040cd29a5a17c4d4701f4aa5d81fa3f5f3ff082aa0b8cf965", "fetched_at": "2026-08-28T04:06:09.152177+00:00", "kind": "readme", "missing": false, "url": "https://github.com/big-data-europe/docker-spark"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:06:09.152177+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:06:09.152177+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-30T02:57:35.543655+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "0233528fe714559040cd29a5a17c4d4701f4aa5d81fa3f5f3ff082aa0b8cf965", "fetched_at": "2026-08-28T04:06:09.152177+00:00", "kind": "readme", "missing": false, "url": "https://github.com/big-data-europe/docker-spark"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:06:09.152177+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:06:09.152177+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:06:09.152177+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-30T02:57:35.543655+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "0233528fe714559040cd29a5a17c4d4701f4aa5d81fa3f5f3ff082aa0b8cf965", "fetched_at": "2026-08-28T04:06:09.152177+00:00", "kind": "readme", "missing": false, "url": "https://github.com/big-data-europe/docker-spark"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:06:09.152177+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:06:09.152177+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-30T02:57:35.543655+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "0233528fe714559040cd29a5a17c4d4701f4aa5d81fa3f5f3ff082aa0b8cf965", "fetched_at": "2026-08-28T04:06:09.152177+00:00", "kind": "readme", "missing": false, "url": "https://github.com/big-data-europe/docker-spark"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-30T02:57:35.543655+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "0233528fe714559040cd29a5a17c4d4701f4aa5d81fa3f5f3ff082aa0b8cf965", "fetched_at": "2026-08-28T04:06:09.152177+00:00", "kind": "readme", "missing": false, "url": "https://github.com/big-data-europe/docker-spark"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-30T02:57:35.543655+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "0233528fe714559040cd29a5a17c4d4701f4aa5d81fa3f5f3ff082aa0b8cf965", "fetched_at": "2026-08-28T04:06:09.152177+00:00", "kind": "readme", "missing": false, "url": "https://github.com/big-data-europe/docker-spark"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-30T02:57:35.543655+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "0233528fe714559040cd29a5a17c4d4701f4aa5d81fa3f5f3ff082aa0b8cf965", "fetched_at": "2026-08-28T04:06:09.152177+00:00", "kind": "readme", "missing": false, "url": "https://github.com/big-data-europe/docker-spark"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 78, "longevity": 100, "rhythm": 35}, "computed_at": "2026-09-03T02:20:16.233290+00:00", "flags": ["no_releases", "no_license"], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 3982, "days_push": 135, "days_rel": null, "gap_med": null, "n_releases_24m": 0}, "score": 67, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}