{"slug":"surge","name":"Surge","website":"https://surgehq.ai","tagline":"Bootstrapped human-data leader with a dedicated RL environments org","description":"Surge AI is the largest human data / RLHF vendor by revenue ($1.2B+ in 2024) serving OpenAI, Google, Anthropic, and Meta, and has created an internal organization dedicated to RL environments. It builds expert-graded evals and environment products such as EnterpriseBench.","domains":["multi-domain","data-labeling","rlhf"],"locations":["San Francisco","New York","Seattle"],"founders":[{"name":"Edwin Chen","xHandle":null}],"teamSize":"101-250","funding":"Bootstrapped; reported in talks to raise ~$1B at $25B+ valuation (2025)","raising":true,"founded":2020,"notable":"~$9.9M revenue per employee; CoreCraft benchmark","entityType":"Incumbent","canonicalUrl":"https://envindex.com/startup/surge","research":{"startupSlug":"surge","entityType":"Incumbent","lastResearched":"2026-08-16","lastVerified":"2026-08-16","verificationStatus":"company-reported","confidence":"high","sourceIds":["surge-products","surge-off-the-shelf","surge-corecraft","surge-benchmarks"]},"sources":[{"id":"surge-products","url":"https://surgehq.ai/products","title":"Products","publisher":"Surge AI","sourceType":"official","accessedAt":"2026-08-16","verificationStatus":"company-reported"},{"id":"surge-off-the-shelf","url":"https://surgehq.ai/ots","title":"Off-the-Shelf Training Data","publisher":"Surge AI","sourceType":"official","accessedAt":"2026-08-16","verificationStatus":"company-reported"},{"id":"surge-corecraft","url":"https://surgehq.ai/blog/rl-envs-real-world","title":"RL Environments and the Hierarchy of Agentic Capabilities","publisher":"Surge AI Research Team","sourceType":"official","publishedAt":"2025-11-03","accessedAt":"2026-08-16","verificationStatus":"company-reported"},{"id":"surge-benchmarks","url":"https://surgehq.ai/benchmarks","title":"Benchmarks","publisher":"Surge AI","sourceType":"official","accessedAt":"2026-08-16","verificationStatus":"company-reported"}],"capabilityCatalog":{"startupSlug":"surge","reviewedAt":"2026-08-16","status":"cataloged","items":[{"name":"RL Environments and Agents","kind":"RL Environment","summary":"Custom RL environments and verifier design for training and evaluating agentic models.","access":["Custom-built","Commercial","Request demo"],"sourceUrl":"/environments/surge-rl-environments-and-agents","sourceLabel":"EnvIndex catalog page","verificationStatus":"company-reported","lastVerified":"2026-08-16","artifactSlug":"surge-rl-environments-and-agents"},{"name":"Rubrics and Verifiers","kind":"Verifier","summary":"Custom rubric and verifier design for scoring complex model and agent behavior.","access":["Custom-built","Commercial","Request demo"],"sourceUrl":"/environments/surge-rubrics-and-verifiers","sourceLabel":"EnvIndex catalog page","verificationStatus":"company-reported","lastVerified":"2026-08-16","artifactSlug":"surge-rubrics-and-verifiers"},{"name":"RLHF","kind":"Dataset","summary":"Human preference and reward data for reinforcement learning from human feedback.","access":["Custom-built","Commercial","Request demo"],"sourceUrl":"/environments/surge-rlhf","sourceLabel":"EnvIndex catalog page","verificationStatus":"company-reported","lastVerified":"2026-08-16","artifactSlug":"surge-rlhf"},{"name":"Supervised Fine-Tuning Data","kind":"Dataset","summary":"Expert demonstrations for bootstrapping model capabilities, including computer and browser use.","access":["Custom-built","Commercial","Request demo"],"sourceUrl":"/environments/surge-sft","sourceLabel":"EnvIndex catalog page","verificationStatus":"company-reported","lastVerified":"2026-08-16","artifactSlug":"surge-sft"},{"name":"Human Evaluation","kind":"Evaluation Suite","summary":"Human evaluation programs for model quality, usefulness, safety, and subjective output characteristics.","access":["Custom-built","Commercial","Request demo"],"sourceUrl":"/environments/surge-human-evaluation","sourceLabel":"EnvIndex catalog page","verificationStatus":"company-reported","lastVerified":"2026-08-16","artifactSlug":"surge-human-evaluation"},{"name":"Expert Professional Domains","kind":"Dataset","summary":"Expert-authored data and judgment across professional, STEM, and humanities domains.","access":["Custom-built","Commercial","Request demo"],"sourceUrl":"/environments/surge-expert-professional-domains","sourceLabel":"EnvIndex catalog page","verificationStatus":"company-reported","lastVerified":"2026-08-16","artifactSlug":"surge-expert-professional-domains"},{"name":"Internationalization","kind":"Dataset","summary":"Language and culturally grounded training data across more than 70 reported languages.","access":["Custom-built","Commercial","Request demo"],"sourceUrl":"/environments/surge-internationalization","sourceLabel":"EnvIndex catalog page","verificationStatus":"company-reported","lastVerified":"2026-08-16","artifactSlug":"surge-internationalization"},{"name":"Multimodal Data","kind":"Dataset","summary":"Training and evaluation data spanning text, images, audio, and video.","access":["Custom-built","Commercial","Request demo"],"sourceUrl":"/environments/surge-multimodal-data","sourceLabel":"EnvIndex catalog page","verificationStatus":"company-reported","lastVerified":"2026-08-16","artifactSlug":"surge-multimodal-data"},{"name":"Off-The-Shelf Data Catalog","kind":"Environment Bundle","summary":"Pre-built commercial datasets and RL environments spanning coding, enterprise agents, STEM, tool use, and reasoning.","access":["Private catalog","Commercial","Off-the-shelf"],"sourceUrl":"/environments/surge-off-the-shelf-data","sourceLabel":"EnvIndex catalog page","verificationStatus":"company-reported","lastVerified":"2026-08-16","artifactSlug":"surge-off-the-shelf-data"},{"name":"CoreCraft","kind":"Benchmark","summary":"A simulated enterprise environment where agents complete customer-support and operational tasks inside a fictional PC retailer.","access":["Public / Open"],"sourceUrl":"/environments/surge-corecraft","sourceLabel":"EnvIndex catalog page","verificationStatus":"company-reported","lastVerified":"2026-08-16","artifactSlug":"surge-corecraft"},{"name":"HANDBOOK.md Agents","kind":"Benchmark","summary":"A long-context enterprise-agent benchmark built from unique RL environments with internal tools and external MCP servers.","access":["Public / Open"],"sourceUrl":"/environments/surge-handbook-agents","sourceLabel":"EnvIndex catalog page","verificationStatus":"company-reported","lastVerified":"2026-08-16","artifactSlug":"surge-handbook-agents"}]},"artifacts":[{"slug":"surge-rl-environments-and-agents","name":"RL Environments and Agents","ownerSlug":"surge","artifactType":"RL Environment","availabilityKind":"custom-capability","access":["Custom-built","Commercial","Request demo"],"summary":"Custom RL environments and verifier design for training and evaluating agentic models.","description":"Surge describes a capability for creating complex RL environments that challenge agentic models, together with verifiers that reward their behavior. This capability record does not imply a fixed public task count or a single packaged environment.","domains":["enterprise","multi-domain","long-horizon","tool-use"],"capabilities":{"tool-use":"yes","long-horizon":"yes","expert-authored":"yes","stateful":"unknown","multi-agent":"unknown","sandboxed":"unknown"},"verifierType":"Custom verifier design","lastVerified":"2026-08-16","verificationStatus":"company-reported","confidence":"high","links":[{"label":"Product page","url":"https://surgehq.ai/products","kind":"official"}],"sourceIds":["surge-products","surge-corecraft"],"relatedArtifactSlugs":["surge-corecraft","surge-handbook-agents"]},{"slug":"surge-rubrics-and-verifiers","name":"Rubrics and Verifiers","ownerSlug":"surge","artifactType":"Verifier","availabilityKind":"custom-capability","access":["Custom-built","Commercial","Request demo"],"summary":"Custom rubric and verifier design for scoring complex model and agent behavior.","description":"Surge describes designing grading systems that capture both successful behavior and deficiencies for outputs that cannot be evaluated with simple automatic checks.","domains":["alignment","multi-domain"],"capabilities":{"expert-authored":"yes","human-in-the-loop":"yes","tool-use":"unknown"},"verifierType":"Expert-designed rubrics and verifiers","lastVerified":"2026-08-16","verificationStatus":"company-reported","confidence":"high","links":[{"label":"Product page","url":"https://surgehq.ai/products","kind":"official"}],"sourceIds":["surge-products"]},{"slug":"surge-rlhf","name":"RLHF","ownerSlug":"surge","artifactType":"Dataset","availabilityKind":"custom-capability","access":["Custom-built","Commercial","Request demo"],"summary":"Human preference and reward data for reinforcement learning from human feedback.","description":"Surge reports generating preference and reward data intended to capture nuanced judgments for model alignment and post-training.","domains":["rlhf","alignment","multi-domain"],"capabilities":{"human-in-the-loop":"yes","expert-authored":"yes","multi-turn":"unknown"},"verifierType":"Human preference and reward signals","lastVerified":"2026-08-16","verificationStatus":"company-reported","confidence":"high","links":[{"label":"Product page","url":"https://surgehq.ai/products","kind":"official"}],"sourceIds":["surge-products"]},{"slug":"surge-sft","name":"Supervised Fine-Tuning Data","ownerSlug":"surge","artifactType":"Dataset","availabilityKind":"custom-capability","access":["Custom-built","Commercial","Request demo"],"summary":"Expert demonstrations for bootstrapping model capabilities, including computer and browser use.","description":"Surge describes supervised fine-tuning demonstrations that teach models foundational skills such as computer use, web navigation, and reasoning.","domains":["computer-use","browser","multi-domain"],"capabilities":{"computer-use":"yes","browser-use":"yes","expert-authored":"yes","human-in-the-loop":"yes"},"verifierType":"Unknown","lastVerified":"2026-08-16","verificationStatus":"company-reported","confidence":"high","links":[{"label":"Product page","url":"https://surgehq.ai/products","kind":"official"}],"sourceIds":["surge-products"]},{"slug":"surge-human-evaluation","name":"Human Evaluation","ownerSlug":"surge","artifactType":"Evaluation Suite","availabilityKind":"custom-capability","access":["Custom-built","Commercial","Request demo"],"summary":"Human evaluation programs for model quality, usefulness, safety, and subjective output characteristics.","description":"Surge describes human evaluation as a managed capability for assessing qualities that automatic evaluations and academic benchmarks may not capture reliably.","domains":["alignment","multi-domain"],"capabilities":{"human-in-the-loop":"yes","expert-authored":"yes"},"verifierType":"Human evaluation","lastVerified":"2026-08-16","verificationStatus":"company-reported","confidence":"high","links":[{"label":"Product page","url":"https://surgehq.ai/products","kind":"official"}],"sourceIds":["surge-products"]},{"slug":"surge-expert-professional-domains","name":"Expert Professional Domains","ownerSlug":"surge","artifactType":"Dataset","availabilityKind":"custom-capability","access":["Custom-built","Commercial","Request demo"],"summary":"Expert-authored data and judgment across professional, STEM, and humanities domains.","description":"Surge describes recruiting professionals and academics, including doctors, lawyers, investment bankers, mathematicians, and professors, to shape training and evaluation data.","domains":["finance","legal","medical","science","math","multi-domain"],"capabilities":{"expert-authored":"yes","human-in-the-loop":"yes"},"verifierType":"Varies by engagement","lastVerified":"2026-08-16","verificationStatus":"company-reported","confidence":"high","links":[{"label":"Product page","url":"https://surgehq.ai/products","kind":"official"}],"sourceIds":["surge-products"]},{"slug":"surge-internationalization","name":"Internationalization","ownerSlug":"surge","artifactType":"Dataset","availabilityKind":"custom-capability","access":["Custom-built","Commercial","Request demo"],"summary":"Language and culturally grounded training data across more than 70 reported languages.","description":"Surge reports internationalization work across more than 70 languages, with linguists designing data for grammar, idiom, and cultural context.","domains":["multi-domain"],"capabilities":{"expert-authored":"yes","human-in-the-loop":"yes"},"verifierType":"Expert linguistic review","lastVerified":"2026-08-16","verificationStatus":"company-reported","confidence":"high","links":[{"label":"Product page","url":"https://surgehq.ai/products","kind":"official"}],"sourceIds":["surge-products"]},{"slug":"surge-multimodal-data","name":"Multimodal Data","ownerSlug":"surge","artifactType":"Dataset","availabilityKind":"custom-capability","access":["Custom-built","Commercial","Request demo"],"summary":"Training and evaluation data spanning text, images, audio, and video.","description":"Surge describes multimodal data capabilities for model understanding and generation across images, audio, and video in addition to text.","domains":["multi-domain"],"capabilities":{"multimodal":"yes","human-in-the-loop":"yes","expert-authored":"unknown"},"verifierType":"Varies by engagement","lastVerified":"2026-08-16","verificationStatus":"company-reported","confidence":"high","links":[{"label":"Product page","url":"https://surgehq.ai/products","kind":"official"}],"sourceIds":["surge-products"]},{"slug":"surge-off-the-shelf-data","name":"Off-The-Shelf Data Catalog","ownerSlug":"surge","artifactType":"Environment Bundle","availabilityKind":"private-commercial","access":["Private catalog","Commercial","Off-the-shelf"],"summary":"Pre-built commercial datasets and RL environments spanning coding, enterprise agents, STEM, tool use, and reasoning.","description":"Surge describes a ready-to-use private commercial catalog containing expert-built training datasets, evaluation sets, and RL environments. Public pages expose selected categories and reported model results but not the full underlying inventory.","domains":["coding","enterprise","science","math","tool-use","multi-domain","long-horizon"],"capabilities":{"long-horizon":"yes","tool-use":"yes","expert-authored":"yes","sandboxed":"unknown","code-execution":"yes","file-manipulation":"yes"},"verifierType":"Includes hidden verifiers and benchmark-compatible grading; varies by catalog item","lastVerified":"2026-08-16","verificationStatus":"company-reported","confidence":"high","links":[{"label":"Request full catalog","url":"https://surgehq.ai/ots","kind":"access"}],"sourceIds":["surge-products","surge-off-the-shelf"],"relatedArtifactSlugs":["surge-rl-environments-and-agents"]},{"slug":"surge-corecraft","name":"CoreCraft","ownerSlug":"surge","artifactType":"Benchmark","availabilityKind":"public-sample","access":["Public / Open"],"summary":"A simulated enterprise environment where agents complete customer-support and operational tasks inside a fictional PC retailer.","description":"CoreCraft models a company with customers, orders, support tickets, records, and tools. Surge reports evaluating nine frontier models on 150 workplace tasks ranging from lookups to multi-step operational workflows.","domains":["enterprise","tool-use","long-horizon"],"capabilities":{"tool-use":"yes","long-horizon":"yes","multi-turn":"unknown","stateful":"yes","expert-authored":"yes","real-application-replica":"no"},"taskCount":150,"verifierType":"Task-specific environment verification; full implementation details Unknown","firstReleased":"2025-11-03","lastVerified":"2026-08-16","verificationStatus":"company-reported","confidence":"high","links":[{"label":"Research article","url":"https://surgehq.ai/blog/rl-envs-real-world","kind":"official"},{"label":"Benchmark page","url":"https://surgehq.ai/benchmarks","kind":"leaderboard"}],"sourceIds":["surge-corecraft","surge-benchmarks"],"relatedArtifactSlugs":["surge-rl-environments-and-agents","surge-handbook-agents"]},{"slug":"surge-handbook-agents","name":"HANDBOOK.md Agents","ownerSlug":"surge","artifactType":"Benchmark","availabilityKind":"public-sample","access":["Public / Open"],"summary":"A long-context enterprise-agent benchmark built from unique RL environments with internal tools and external MCP servers.","description":"HANDBOOK.md evaluates whether agents can follow a long company handbook while completing day-to-day professional tasks. Surge reports that each task is a unique RL environment across five enterprise domains.","domains":["enterprise","tool-use","long-horizon"],"capabilities":{"tool-use":"yes","mcp-support":"yes","long-horizon":"yes","expert-authored":"unknown","stateful":"unknown"},"verifierType":"Unknown","lastVerified":"2026-08-16","verificationStatus":"company-reported","confidence":"high","links":[{"label":"Benchmark page","url":"https://surgehq.ai/benchmarks","kind":"leaderboard"}],"sourceIds":["surge-benchmarks"],"relatedArtifactSlugs":["surge-corecraft","surge-rl-environments-and-agents"]}]}