{"slug":"handshake","name":"Handshake","website":"https://joinhandshake.com/research/ai","tagline":"Career network turned human-data and RL environments provider via Handshake AI","description":"Handshake operates the largest early-career network in the US and launched Handshake AI, a human data labs business that recruits PhD-level experts to build evals and RL environments for frontier AI labs. It leverages its network of millions of students and experts for domain-specific data.","domains":["multi-domain","data-labeling","rlhf"],"locations":["San Francisco","New York","Bangalore","Berlin"],"founders":[{"name":"Garrett Lord","xHandle":null}],"teamSize":"250+","funding":"Series F, $200M (2022, ~$3.5B valuation)","raising":null,"founded":2014,"notable":"Known for its 'Gandalf the Grader' benchmark work","entityType":"Data + environments","canonicalUrl":"https://envindex.com/startup/handshake","research":{"startupSlug":"handshake","entityType":"Data + environments","lastResearched":"2026-08-16","lastVerified":"2026-08-16","verificationStatus":"confirmed","confidence":"high","sourceIds":["handshake-gandalf","handshake-gandalf-github","handshake-banker-dataset"]},"sources":[{"id":"handshake-gandalf","url":"https://joinhandshake.com/research/ai/gandalf-the-grader/","title":"Your verifier is probably the bottleneck. We built one that isn’t.","publisher":"Handshake AI Research","sourceType":"official","publishedAt":"2026-05-27","accessedAt":"2026-08-16","verificationStatus":"company-reported"},{"id":"handshake-gandalf-github","url":"https://github.com/Handshake-AI-Research/gandalf-the-grader","title":"Gandalf the Grader source repository","publisher":"Handshake AI Research","sourceType":"repository","accessedAt":"2026-08-16","verificationStatus":"confirmed"},{"id":"handshake-banker-dataset","url":"https://huggingface.co/datasets/handshake-ai-research/bankertoolbench","title":"BankerToolBench dataset","publisher":"Handshake AI Research","sourceType":"dataset","accessedAt":"2026-08-16","verificationStatus":"confirmed"}],"capabilityCatalog":{"startupSlug":"handshake","reviewedAt":"2026-08-16","status":"cataloged","items":[{"name":"Expert AI Training Data","kind":"Data service","summary":"Expert-created training and evaluation data from Handshake's professional network.","access":["Commercial"],"sourceUrl":"https://joinhandshake.com/research/ai/","sourceLabel":"AI Research","verificationStatus":"company-reported","lastVerified":"2026-08-16"},{"name":"Gandalf the Grader","kind":"Verifier","summary":"An open-source reactive agent-as-judge that inspects files and environment state while grading agent work.","access":["Public / Open","Open source"],"sourceUrl":"/environments/gandalf-the-grader","sourceLabel":"EnvIndex catalog page","verificationStatus":"confirmed","lastVerified":"2026-08-16","artifactSlug":"gandalf-the-grader"},{"name":"BankerToolBench","kind":"Benchmark","summary":"A banking-workflow benchmark used to evaluate agents and verifier behavior on artifact-heavy tasks.","access":["Public / Open"],"sourceUrl":"/environments/bankertoolbench","sourceLabel":"EnvIndex catalog page","verificationStatus":"confirmed","lastVerified":"2026-08-16","artifactSlug":"bankertoolbench"}]},"artifacts":[{"slug":"gandalf-the-grader","name":"Gandalf the Grader","ownerSlug":"handshake","artifactType":"Verifier","availabilityKind":"public-sample","access":["Public / Open","Open source"],"summary":"An open-source reactive agent-as-judge that inspects files and environment state while grading agent work.","description":"Gandalf runs inside a rollout environment and chooses what evidence to inspect using available tools. Handshake reports results on BankerVerifierBench and an internal productivity benchmark; those performance claims remain company-reported unless independently reproduced.","domains":["finance","enterprise","tool-use"],"capabilities":{"tool-use":"yes","stateful":"yes","file-manipulation":"yes","human-in-the-loop":"yes"},"verifierType":"Reactive agent-as-judge","firstReleased":"2026-05-27","lastVerified":"2026-08-16","verificationStatus":"confirmed","confidence":"high","links":[{"label":"Technical article","url":"https://joinhandshake.com/research/ai/gandalf-the-grader/","kind":"official"},{"label":"Source code","url":"https://github.com/Handshake-AI-Research/gandalf-the-grader","kind":"github"}],"sourceIds":["handshake-gandalf","handshake-gandalf-github"],"relatedArtifactSlugs":["bankertoolbench"]},{"slug":"bankertoolbench","name":"BankerToolBench","ownerSlug":"handshake","artifactType":"Benchmark","availabilityKind":"public-sample","access":["Public / Open"],"summary":"A banking-workflow benchmark used to evaluate agents and verifier behavior on artifact-heavy tasks.","description":"BankerToolBench contains banking-domain tasks whose outputs can include spreadsheets and other work products. Handshake uses a 21-task slice in its reported BankerVerifierBench meta-evaluation of grading systems.","domains":["finance","enterprise","tool-use"],"capabilities":{"tool-use":"yes","file-manipulation":"yes","expert-authored":"yes","long-horizon":"unknown"},"scenarioCount":21,"verifierType":"Rubric grading; compatible with agent-as-judge verification","lastVerified":"2026-08-16","verificationStatus":"confirmed","confidence":"medium","links":[{"label":"Dataset","url":"https://huggingface.co/datasets/handshake-ai-research/bankertoolbench","kind":"hugging-face"}],"sourceIds":["handshake-banker-dataset","handshake-gandalf"],"relatedArtifactSlugs":["gandalf-the-grader"]}]}