{"slug":"swe-bench-pro","name":"SWE-Bench Pro","ownerSlug":"scale","artifactType":"Benchmark","availabilityKind":"public-sample","access":["Public / Open","Open source"],"summary":"A long-horizon software-engineering benchmark built from public and proprietary repositories.","description":"SWE-Bench Pro evaluates agents on realistic software-engineering problems. The published paper reports 1,865 problems from 41 actively maintained repositories; the public dataset and leaderboard are separately accessible.","domains":["coding","long-horizon"],"capabilities":{"long-horizon":"yes","code-execution":"yes","sandboxed":"yes","human-in-the-loop":"yes","expert-authored":"yes"},"taskCount":1865,"verifierType":"Repository test suites with human verification checkpoints","firstReleased":"2025-09-19","lastVerified":"2026-08-16","verificationStatus":"confirmed","confidence":"high","links":[{"label":"Research page","url":"https://labs.scale.com/papers/swe_bench_pro","kind":"paper"},{"label":"Public leaderboard","url":"https://labs.scale.com/leaderboard/swe_bench_pro_public","kind":"leaderboard"}],"sourceIds":["scale-swe-pro","scale-swe-public"],"relatedArtifactSlugs":["swe-bench-pro-private"],"canonicalUrl":"https://envindex.com/environments/swe-bench-pro","owner":{"slug":"scale","name":"Scale","website":"https://scale.com","tagline":"Data-labeling incumbent extending into agent evals and RL environments","description":"Scale AI is the original data-labeling powerhouse for AI labs and enterprises, now building RL environments and agent evaluation products under its agents and RL environments group. After Meta's 2025 investment and the departure of CEO Alexandr Wang, it lost some lab customers but continues to push into environments.","domains":["multi-domain","coding","data-labeling","rlhf"],"locations":["San Francisco","New York","Washington DC","London"],"founders":[{"name":"Alexandr Wang","xHandle":"alexandr_wang"},{"name":"Lucy Guo","xHandle":null}],"teamSize":"250+","funding":"$1.6B+ raised; Meta invested $14.3B at ~$29B valuation (June 2025)","raising":null,"founded":2016,"notable":"Publishes SWE-bench Pro benchmark"},"sources":[{"id":"scale-swe-pro","url":"https://labs.scale.com/papers/swe_bench_pro","title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","publisher":"Scale Labs","sourceType":"paper","publishedAt":"2025-09-19","accessedAt":"2026-08-16","verificationStatus":"confirmed"},{"id":"scale-swe-public","url":"https://labs.scale.com/leaderboard/swe_bench_pro_public","title":"SWE-Bench Pro  -  Public Dataset","publisher":"Scale Labs","sourceType":"official","accessedAt":"2026-08-16","verificationStatus":"confirmed"}]}