{"adoption": {"forks": 135, "observed_at": "2026-08-28T04:04:42.952395+00:00", "stars": 1431}, "canonical_url": "https://ross.abutalabs.com/products/swelancer-benchmark", "card": {"archived": true, "artifact_type": "dataset", "description": "This repo contains the dataset and code for the paper \"SWE-Lancer: Can Frontier LLMs Earn $1 Million from Real-World Freelance Software Engineering?\"", "domain": ["large-language-models", "artificial-intelligence", "developer-tools", "testing"], "enriched": true, "function": ["benchmarking", "llm-inference", "machine-learning"], "health_score": 10, "homepage": null, "language": null, "license": null, "license_family": "other", "maturity": "maintenance", "member_repos": ["openai/SWELancer-Benchmark"], "name": "openai/SWELancer-Benchmark", "platform": ["python", "cross-platform"], "pushed_at": "2025-07-18T02:02:20+00:00", "repo": "openai/SWELancer-Benchmark", "stars": 1431, "tags": ["benchmark", "evaluation", "llm-evaluation", "software-engineering", "freelance-tasks", "openai", "swe-benchmark", "research", "docker"], "topics": [], "urls": [], "use_cases": ["evaluate llms on real-world software engineering tasks", "benchmark coding agents on freelance-style issues", "compare frontier models on paid software tasks", "run swe-lancer evaluations", "research llm coding capability", "measure ai performance on real github issues"], "what_it_is": "SWE-Lancer is a benchmark dataset and evaluation harness measuring whether frontier LLMs can complete real-world freelance software engineering tasks worth up to $1 million in aggregate. The codebase has been merged into OpenAI's preparedness repository, so this repo now serves as a pointer to the active project.", "when_to_avoid": ["you want an actively developed repo - use openai/preparedness instead", "you need a general-purpose coding assistant rather than an evaluation benchmark", "you lack the compute or API budget to run frontier model evaluations"], "when_to_choose": ["you need a rigorous benchmark of LLM software engineering ability on real paid tasks", "you are researching how frontier models handle realistic freelance coding work", "you want reproducible evaluation harnesses from OpenAI's preparedness project"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/swelancer-benchmark", "repo": "openai/SWELancer-Benchmark", "role": "main", "score": 10}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:04:42.952395+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-30T04:37:08.414647+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "dd642c1c6b58e458220dc17a452ce932c992a4a98fca8958f2d7ac6167e32dec", "fetched_at": "2026-08-28T04:04:42.952395+00:00", "kind": "readme", "missing": false, "url": "https://github.com/openai/SWELancer-Benchmark"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:04:42.952395+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-30T04:37:08.414647+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "dd642c1c6b58e458220dc17a452ce932c992a4a98fca8958f2d7ac6167e32dec", "fetched_at": "2026-08-28T04:04:42.952395+00:00", "kind": "readme", "missing": false, "url": "https://github.com/openai/SWELancer-Benchmark"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-30T04:37:08.414647+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "dd642c1c6b58e458220dc17a452ce932c992a4a98fca8958f2d7ac6167e32dec", "fetched_at": "2026-08-28T04:04:42.952395+00:00", "kind": "readme", "missing": false, "url": "https://github.com/openai/SWELancer-Benchmark"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:04:42.952395+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:04:42.952395+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:04:42.952395+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-30T04:37:08.414647+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "dd642c1c6b58e458220dc17a452ce932c992a4a98fca8958f2d7ac6167e32dec", "fetched_at": "2026-08-28T04:04:42.952395+00:00", "kind": "readme", "missing": false, "url": "https://github.com/openai/SWELancer-Benchmark"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:04:42.952395+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:04:42.952395+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-30T04:37:08.414647+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "dd642c1c6b58e458220dc17a452ce932c992a4a98fca8958f2d7ac6167e32dec", "fetched_at": "2026-08-28T04:04:42.952395+00:00", "kind": "readme", "missing": false, "url": "https://github.com/openai/SWELancer-Benchmark"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:04:42.952395+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:04:42.952395+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:04:42.952395+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-30T04:37:08.414647+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "dd642c1c6b58e458220dc17a452ce932c992a4a98fca8958f2d7ac6167e32dec", "fetched_at": "2026-08-28T04:04:42.952395+00:00", "kind": "readme", "missing": false, "url": "https://github.com/openai/SWELancer-Benchmark"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:04:42.952395+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:04:42.952395+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-30T04:37:08.414647+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "dd642c1c6b58e458220dc17a452ce932c992a4a98fca8958f2d7ac6167e32dec", "fetched_at": "2026-08-28T04:04:42.952395+00:00", "kind": "readme", "missing": false, "url": "https://github.com/openai/SWELancer-Benchmark"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-30T04:37:08.414647+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "dd642c1c6b58e458220dc17a452ce932c992a4a98fca8958f2d7ac6167e32dec", "fetched_at": "2026-08-28T04:04:42.952395+00:00", "kind": "readme", "missing": false, "url": "https://github.com/openai/SWELancer-Benchmark"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-30T04:37:08.414647+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "dd642c1c6b58e458220dc17a452ce932c992a4a98fca8958f2d7ac6167e32dec", "fetched_at": "2026-08-28T04:04:42.952395+00:00", "kind": "readme", "missing": false, "url": "https://github.com/openai/SWELancer-Benchmark"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-30T04:37:08.414647+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "dd642c1c6b58e458220dc17a452ce932c992a4a98fca8958f2d7ac6167e32dec", "fetched_at": "2026-08-28T04:04:42.952395+00:00", "kind": "readme", "missing": false, "url": "https://github.com/openai/SWELancer-Benchmark"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 32, "longevity": 40, "rhythm": 35}, "computed_at": "2026-09-03T02:20:16.233290+00:00", "flags": ["no_releases", "archived", "no_license"], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 561, "days_push": 412, "days_rel": null, "gap_med": null, "n_releases_24m": 0}, "score": 10, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}