{"cssClassNames":"blog-page page basicpage summit-page","templateName":"blog-page","canonicalLink":"https://www.snowflake.com/en/blog/engineering/data-eng-bench-data-engineering-agent-benchmark/","robotsTags":["index","follow"],"allowedRenditionsWidth":["320","480","640","768","960","1200","1440","1920"],"description":"Discover data-eng-bench, an open-source data engineering benchmark for AI agents. Compare Claude Code, Codex, and Snowflake CoCo on dbt and SQL tasks.","language":"en","title":"A Data Engineering Benchmark for AI Agents","analyticsPageType":"homepage","analyticsCategory":"general","analyticsSubCategory":"","excludeFromAnalytics":false,":mappedPath":"/en/blog/engineering/data-eng-bench-data-engineering-agent-benchmark/",":type":"snowflake-site/components/structure/page",":items":{"root":{"columnCount":12,"columnClassNames":{"experiencefragment-banner":"aem-GridColumn aem-GridColumn--default--12","experiencefragment-sub-header":"aem-GridColumn aem-GridColumn--default--12","experiencefragment-pre-footer":"aem-GridColumn aem-GridColumn--default--12","experiencefragment-header":"aem-GridColumn aem-GridColumn--default--12","markup_editor-table":"aem-GridColumn aem-GridColumn--default--12","responsivegrid":"aem-GridColumn aem-GridColumn--default--12","experiencefragment-footer":"aem-GridColumn aem-GridColumn--default--12","markup_editor":"aem-GridColumn aem-GridColumn--default--12","container_47873732":"aem-GridColumn aem-GridColumn--default--12"},"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12",":items":{"experiencefragment-banner":{"id":"experiencefragment-ac4b798ffc","localizedFragmentVariationPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/pushdown-banner/pushdown-banner-blank/jcr:content","configured":true,":type":"snowflake-site/components/experiencefragment","xfModelPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/pushdown-banner/pushdown-banner-blank.xfmodel.json"},"experiencefragment-header":{"id":"experiencefragment-680fa64c02","localizedFragmentVariationPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/mega-nav-header/master/jcr:content","configured":true,":type":"snowflake-site/components/experiencefragment","xfModelPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/mega-nav-header/master.xfmodel.json","languageNavPath":"/content/snowflake-site/global/en/blog/engineering/data-eng-bench-data-engineering-agent-benchmark.languagenav.json","appliedCssClassNames":"snowflake-sticky-nav-host"},"experiencefragment-sub-header":{"id":"experiencefragment-4e28309300","localizedFragmentVariationPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/sub-navigation/engineering-blog-sub-nav/jcr:content","configured":true,":type":"snowflake-site/components/experiencefragment","xfModelPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/sub-navigation/engineering-blog-sub-nav.xfmodel.json"},"responsivegrid":{"columnCount":12,"columnClassNames":{"container_breadcrumb":"aem-GridColumn aem-GridColumn--default--12","container_main_content":"aem-GridColumn aem-GridColumn--default--12"},"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12",":items":{"container_breadcrumb":{"layout":"RESPONSIVE_GRID","columnCount":12,"columnClassNames":{"breadcrumb":"aem-GridColumn aem-GridColumn--default--12"},"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","id":"blog-page-breadcrumb-indentation",":type":"snowflake-site/components/container",":items":{"breadcrumb":{"id":"breadcrumb-b23b09dd5d","breadcrumbItems":[{"title":"Blog","path":"/en/blog/engineering/","active":false},{"title":"Data Engineering","path":"/en/blog/engineering/data-engineering/","active":false},{"title":"Introducing Data-eng-bench: Why You Need \"Data-Native\" Harnesses for Data Engineering","path":"/en/blog/engineering/data-eng-bench-data-engineering-agent-benchmark/","active":false}],":type":"snowflake-site/components/blog/breadcrumb"}},":itemsOrder":["breadcrumb"],"appliedCssClassNames":"snowflake-container"},"container_main_content":{"layout":"RESPONSIVE_GRID","columnCount":12,"columnClassNames":{"flexible_column_container":"aem-GridColumn aem-GridColumn--default--12","related_content":"aem-GridColumn aem-GridColumn--default--12"},"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","id":"main-content",":type":"snowflake-site/components/container",":items":{"flexible_column_container":{"id":"flexible-column-container-36a69e18bb","propertiesId":"snowflake-blog-template-main-container","type":"2-column-60-40","alignColumns":"top","containerMaxWidth":"extra-large","topPadding":"none","bottomPadding":"none","spaceBetween":"none","reverseOnMobile":true,"carouselOnMobile":false,"backgroundImageOption":"none","flexible_column_content_container_1":{"layout":"SIMPLE","id":"container-9d439111c6",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"container_hero":{"layout":"RESPONSIVE_GRID","columnCount":12,"columnClassNames":{"blog_hero":"aem-GridColumn aem-GridColumn--default--12"},"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","id":"container-2e75e6b207",":type":"snowflake-site/components/container",":items":{"blog_hero":{"id":"blog-hero-47ec6551fc","linkedInShareUrl":"https://www.linkedin.com/shareArticle?mini=true&url=https%3A%2F%2Fwww.snowflake.com%2Fcontent%2Fsnowflake-site%2Fglobal%2Fen%2Fblog%2Fengineering%2Fdata-eng-bench-data-engineering-agent-benchmark&title=Introducing+Data-eng-bench%3A+Why+You+Need+%22Data-Native%22+Harnesses+for+Data+Engineering","twitterShareUrl":"https://x.com/intent/post?url=https%3A%2F%2Fwww.snowflake.com%2Fcontent%2Fsnowflake-site%2Fglobal%2Fen%2Fblog%2Fengineering%2Fdata-eng-bench-data-engineering-agent-benchmark&text=Introducing+Data-eng-bench%3A+Why+You+Need+%22Data-Native%22+Harnesses+for+Data+Engineering","facebookShareUrl":"https://www.facebook.com/sharer/sharer.php?u=https%3A%2F%2Fwww.snowflake.com%2Fcontent%2Fsnowflake-site%2Fglobal%2Fen%2Fblog%2Fengineering%2Fdata-eng-bench-data-engineering-agent-benchmark","showClaude":true,"showChatGpt":true,"authors":[{"authorImage":{"id":"image-9311e288fa","height":"800","lazyEnabled":true,"src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--249da901-4810-48b7-ab40-99208c5e3b73/default-author-image.png?quality=85&preferwebp=true","alt":"Snowflake AI Research","isLcpImage":false,"width":"800",":type":"snowflake-site/components/image"},"authorCta":{"id":"button-438739d40c","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/blog/authors/snowflake-ai-research/"},"linkTargetContentType":"DOCUMENT_LEARN",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_INTERNAL","text":"Snowflake AI Research"}}],"image":{"id":"image-961d41c3aa","height":"720","lazyEnabled":true,"src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--85d72614-270b-4273-9e68-6bb4ddc3a612/sf-eng-blog-ml-3.png?quality=85&preferwebp=true","isLcpImage":false,"width":"1680",":type":"snowflake-site/components/image"},"timeToRead":"10","publicationDate":"AUG 06, 2026","tag":{"tagText":"Data Engineering","tagColor":"#29B5E8"},"title":{"lines":["Introducing Data-eng-bench: Why You Need \"Data-Native\" Harnesses for Data Engineering"],"type":"heading2",":type":"snowflake-site/components/title-v2"},":type":"snowflake-site/components/blog/blog-hero"}},":itemsOrder":["blog_hero"]},"responsivegrid_content":{"columnCount":12,"columnClassNames":{"image_1517420975":"aem-GridColumn aem-GridColumn--default--12","image":"aem-GridColumn aem-GridColumn--default--12","blog_text_814042905":"aem-GridColumn aem-GridColumn--default--12","blog_text_1891894734":"aem-GridColumn aem-GridColumn--default--12","blog_text":"aem-GridColumn aem-GridColumn--default--12","image_1424448934":"aem-GridColumn aem-GridColumn--default--12","blog_text_624248501":"aem-GridColumn aem-GridColumn--default--12","blog_text_737427679":"aem-GridColumn aem-GridColumn--default--12"},"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","appliedCssClassNames":"snowflake-layout-container-inner-padding-small",":items":{"blog_text":{"id":"blog-text-22953c1de1","text":"\u003Cp\u003EAs AI agents move from writing individual functions to owning end-to-end workflows, data teams face a harsh reality: general-purpose coding agents, regardless of their proficiency in Python or SQL, can struggle with production-grade data engineering. They often fail to complete tasks or incur high costs. This is a demanding test of an agent's ability to navigate a large warehouse, reason about business logic and handle edge cases. This is precisely the kind of work that has historically been difficult to measure.\u003C/p\u003E\r\n\u003Cp\u003ETo measure this capability, we're open sourcing data-eng-bench, a benchmark for repository-level data engineering created in joint work with \u003Ca rel=\"noopener noreferrer\" target=\"_blank\" href=\"https://bespokelabs.ai/\"\u003EBespoke Labs\u003C/a\u003E. Tasks in data-eng-bench hand an agent a live dbt project connected to an enterprise-scale data warehouse and ask it to build and fix real data pipelines. Resulting dbt models are then evaluated against the business rules and edge cases a working pipeline must satisfy. Relative to \u003Ca rel=\"noopener noreferrer\" target=\"_blank\" href=\"https://github.com/dbt-labs/ade-bench\"\u003EADE-Bench\u003C/a\u003E, one of few other open benchmarks in this domain, data-eng-bench offers higher scale (103 tasks versus 63 tasks), robust tests that check invariants of the output data pipeline, and more complex task specifications.\u003C/p\u003E\r\n\u003Cp\u003EOur testing on data-eng-bench reveals a clear divide between generic agent harnesses, including Claude Code and OpenAI Codex, and the data-native agent harness \u003Ca rel=\"noopener noreferrer\" target=\"_blank\" href=\"https://docs.snowflake.com/en/user-guide/cortex-code/cortex-code\"\u003ESnowflake CoCo\u003C/a\u003E. Specifically, CoCo leverages its understanding of the data platform to consistently offer higher quality (that is, task completion rate) while incurring significantly lower cost. In more detail:\u003C/p\u003E\r\n\u003Cul\u003E\r\n\u003Cli\u003E\u003Cb\u003EThe harness matters for quality; the effect varies by model:\u003C/b\u003E Holding the harness fixed at CoCo, Pass@1 varies drastically across models: from 73.8% with Opus 5 to 64.1% with GPT 5.6 Sol to 56.6% with Sonnet 5, a 17-point spread. \u003Ci\u003EThe impact of the harness on quality depends on the model:\u003C/i\u003E Opus 5 performs the best with CoCo, dropping by ~4pp in Pass@1 with Claude Code; Sonnet 5 performs equally well with both CoCo and Claude Code; GPT 5.6 Sol performs the best with CoCo, dropping by 3.6pp with Codex.\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EThe harness matters significantly for cost efficiency:\u003C/b\u003E CoCo achieves higher quality at lower cost than other harnesses: With Opus 5, CoCo reports a 4pp higher Pass@1 at 3.9x lower cost than Claude Code. With Sonnet 5, CoCo reports the same Pass@1 at 2.3x lower cost than Claude Code. With GPT 5.6 Sol, CoCo reports a 3.6pp higher Pass@1 than Codex with Codex incurring 1.5x the cost of CoCo. CoCo completes tasks with 1.5x fewer tool operations and 2.2x fewer agent steps than Claude Code. Task solving patterns show that CoCo adopts a more efficient exploration and validation strategy, requiring 1.7x fewer SQL queries and 1.2x fewer file reads during these phases. CoCo also stays on scope in 2.2x more instances, skipping unnecessary DuckDB cross-validation that Claude Code performs by default.\u003C/li\u003E\r\n\u003C/ul\u003E\r\n\u003Cp\u003EWhile these results indicate that frontier agents have made big strides in tackling data engineering tasks, there is still headroom for further improvement. The strongest configuration we benchmarked, Snowflake CoCo with Opus 5, successfully solves 73.8% of tasks on the first attempt (average Pass@1 across 3 trials), but only 64.1% of tasks pass on all three runs (Pass^3). Sonnet 5 reports Pass^3 at 40.8% and GPT 5.6 Sol reports Pass^3 at &lt;56% across harnesses.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"image":{"id":"image-f0e079104d","height":"1267","lazyEnabled":true,"src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--e1443c2c-1528-442a-830a-42f145760cdd/figure-1.-introducing-data-eng-bench--why-you-need-data-native-harnesses-for-data-engineering.png?quality=85&preferwebp=true","alt":"Figure 1. Quality (calculated as the mean Pass@1 rate across 3 independent trials for each task) versus cost per trial by harness and model on data-eng-bench; up-and-left is better.","isLcpImage":true,"width":"2048","title":"Figure 1. Quality (calculated as the mean Pass@1 rate across 3 independent trials for each task) versus cost per trial by harness and model on data-eng-bench; up-and-left is better.",":type":"snowflake-site/components/image"},"blog_text_814042905":{"id":"blog-text-8c850a2e82","text":"\u003Cp\u003EIn the rest of this blog, we provide an overview of the data-eng-bench benchmark, report how frontier agents perform on it in terms of quality, token and cost efficiency, and what it means for teams adopting coding agents for data engineering tasks.\u003C/p\u003E\r\n\u003Cp\u003E\u003Ca href=\"https://github.com/Snowflake-Labs/data-eng-bench\" target=\"_blank\" rel=\"noopener noreferrer\"\u003E\u003Ci\u003EGet the benchmark\u003C/i\u003E\u003C/a\u003E.\u003C/p\u003E\r\n\u003Ch2\u003EThe data-eng-bench benchmark\u003C/h2\u003E\r\n\u003Cp\u003EWe designed data-eng-bench to mirror how enterprise data engineering teams actually operate: many source systems, layered staging-to-mart development and a shared project that must be maintained by many users. Here is a breakdown:\u003C/p\u003E\r\n\u003Cp\u003E\u003Cb\u003EOne shared data warehouse:\u003C/b\u003E Every task runs against a single, persistent retail data warehouse — 579 source tables across 19 schemas, roughly 8,000 columns in total — spanning orders, finance, procurement, marketing, inventory and more. That's broader than prior data engineering benchmarks, and each task requires navigating the same schema with different requirements.\u003C/p\u003E\r\n\u003Cp\u003E\u003Cb\u003E103 tasks of two variants:\u003C/b\u003E Each task gives the agent an instruction in natural language, a starting dbt project, and the data warehouse, then asks it to produce or correct models that materialize the right tables. The two variants are as follows:\u003C/p\u003E\r\n\u003Cul\u003E\r\n\u003Cli\u003E\u003Cb\u003EBuild (84 tasks):\u003C/b\u003E Author new models and keep the existing pipeline running. Build tasks can be further divided into (a) \u003Cb\u003EGreenfield: Scaffold a brand new dbt project\u003C/b\u003E from an empty state; and (b) \u003Cb\u003EBrownfield: Add new models into an existing multi-layer project\u003C/b\u003E while reusing and preserving dozens of production models.\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EFix (19 tasks):\u003C/b\u003E Diagnose and repair a subtly broken production model.\u003C/li\u003E\r\n\u003C/ul\u003E\r\n\u003Cp\u003E\u003Cb\u003EReal dbt mechanics, not free-form SQL:\u003C/b\u003E Agents work through dbt primitives: source declarations (45 tasks), reusable macros (13 tasks) and per-model materializations across staging, intermediate and mart layers. And 82% of gold solutions wire models together through explicit \u003Ccode\u003Eref\u003C/code\u003E dependencies, averaging roughly nine \u003Ccode\u003Eref\u003C/code\u003E calls each, which the agent must resolve into a coherent, compilable DAG. Those DAGs are meaningfully large: a median of four models per solution, and up to 42 for the biggest pipelines.\u003C/p\u003E\r\n\u003Cp\u003E\u003Cb\u003EDiverse, high-difficulty business rules:\u003C/b\u003E The tasks live in domains where data engineering is proven difficult — finance (ledger reconciliation, revenue recognition, multi-currency settlement), inventory (LIFO/FIFO costing, turnover, stockout risk), marketing (multi-touch attribution, campaign ROI) and customer analytics (RFM segmentation, churn, lifetime value). The business rules depend on the order and timing of events, accumulated or allocated values across many interdependent tables, and edge cases that should be handled exactly as defined.\u003C/p\u003E\r\n\u003Cp\u003E\u003Cb\u003EGraded by what the pipeline does, not what it looks like:\u003C/b\u003E Each task ships with a hidden verifier suite of 10–50 assertions that materializes the agent's models and interrogates the resulting tables directly. Those assertions encode the invariants a correct solution must satisfy (for example, output grain and column contracts, formula-level correctness, edge-case handling, idempotency across re-runs) and for the hardest tasks, independently recompute the expected result in Python.\u003C/p\u003E\r\n\u003Cp\u003E\u003Cb\u003EScoring is two-fold:\u003C/b\u003E At the test level, assertions provide partial credit (the fraction that pass). At the task level, a task counts as resolved \u003Ci\u003Eonly if every assertion passes\u003C/i\u003E — so a pipeline that's directionally right but wrong at the edges earns no task-level credit. This lets us separate &quot;can generate a plausible model&quot; from &quot;got the whole pipeline correct&quot; and pinpoint exactly where solutions fail.\u003C/p\u003E\r\n\u003Ch2\u003EMeasuring agent quality and cost\u003C/h2\u003E\r\n\u003Cp\u003EWe evaluate a combination of harnesses and models on data-eng-bench, isolating how much of an agent's performance comes from the scaffold versus the underlying model.\u003C/p\u003E\r\n\u003Cp\u003EWe test three models spanning proprietary families — Opus 5 (Anthropic), Sonnet 5 (Anthropic) and GPT 5.6 Sol (OpenAI), run with several harnesses that differ in context management, planning and tool interfaces: Snowflake CoCo, Claude Code and Codex. We recommend Code mode (cortex --mode code via CLI) while using CoCo on dbt tasks for quality/cost balance. For every combination, we report two quality metrics: Pass@1 and Pass^3 rates. Pass@1 and Pass^3 capture complementary dimensions of quality, measuring the capability ceiling and consistency across runs respectively. We also report the average cost per trial (in $) and the average number of total tokens (across input, cache and output) per trial.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"blog_text_1891894734":{"id":"blog-text-0ba3a40ef2","text":"\u003Ctable\u003E\r\n\u003Cthead\u003E\u003Ctr\u003E\u003Cth\u003EHarness\u003C/th\u003E\r\n\u003Cth\u003EModel\u003C/th\u003E\r\n\u003Cth\u003EPass@1\u003C/th\u003E\r\n\u003Cth\u003EPass^3\u003C/th\u003E\r\n\u003Cth\u003ECost per trial ($)\u003C/th\u003E\r\n\u003Cth\u003ECost multiplier\u003C/th\u003E\r\n\u003Cth\u003ETotal tokens per trial\u003C/th\u003E\r\n\u003C/tr\u003E\u003C/thead\u003E\u003Ctbody\u003E\u003Ctr\u003E\u003Ctd\u003ESnowflake CoCo (Code)\u003C/td\u003E\r\n\u003Ctd\u003EOpus 5\u003C/td\u003E\r\n\u003Ctd\u003E\u003Cb\u003E73.8%\u003C/b\u003E\u003C/td\u003E\r\n\u003Ctd\u003E\u003Cb\u003E64.1%\u003C/b\u003E\u003C/td\u003E\r\n\u003Ctd\u003E0.756\u003C/td\u003E\r\n\u003Ctd\u003E1\u003C/td\u003E\r\n\u003Ctd\u003E1,070,515\u003C/td\u003E\r\n\u003C/tr\u003E\u003Ctr\u003E\u003Ctd\u003E&nbsp;\u003C/td\u003E\r\n\u003Ctd\u003ESonnet 5\u003C/td\u003E\r\n\u003Ctd\u003E56.6%\u003C/td\u003E\r\n\u003Ctd\u003E40.8%\u003C/td\u003E\r\n\u003Ctd\u003E0.660\u003C/td\u003E\r\n\u003Ctd\u003E1\u003C/td\u003E\r\n\u003Ctd\u003E2,879,678\u003C/td\u003E\r\n\u003C/tr\u003E\u003Ctr\u003E\u003Ctd\u003E&nbsp;\u003C/td\u003E\r\n\u003Ctd\u003EGPT 5.6 Sol\u003C/td\u003E\r\n\u003Ctd\u003E64.1%\u003C/td\u003E\r\n\u003Ctd\u003E55.3%\u003C/td\u003E\r\n\u003Ctd\u003E0.358\u003C/td\u003E\r\n\u003Ctd\u003E1\u003C/td\u003E\r\n\u003Ctd\u003E436,236\u003C/td\u003E\r\n\u003C/tr\u003E\u003Ctr\u003E\u003Ctd\u003EClaude Code\u003C/td\u003E\r\n\u003Ctd\u003EOpus 5\u003C/td\u003E\r\n\u003Ctd\u003E69.6%\u003C/td\u003E\r\n\u003Ctd\u003E60.2%\u003C/td\u003E\r\n\u003Ctd\u003E2.959\u003C/td\u003E\r\n\u003Ctd\u003E3.914\u003C/td\u003E\r\n\u003Ctd\u003E4,810,868\u003C/td\u003E\r\n\u003C/tr\u003E\u003Ctr\u003E\u003Ctd\u003E&nbsp;\u003C/td\u003E\r\n\u003Ctd\u003ESonnet 5\u003C/td\u003E\r\n\u003Ctd\u003E56.6%\u003C/td\u003E\r\n\u003Ctd\u003E40.8%\u003C/td\u003E\r\n\u003Ctd\u003E1.530\u003C/td\u003E\r\n\u003Ctd\u003E2.318\u003C/td\u003E\r\n\u003Ctd\u003E7,914,293\u003C/td\u003E\r\n\u003C/tr\u003E\u003Ctr\u003E\u003Ctd\u003ECodex\u003C/td\u003E\r\n\u003Ctd\u003EGPT 5.6 Sol\u003C/td\u003E\r\n\u003Ctd\u003E60.5%\u003C/td\u003E\r\n\u003Ctd\u003E49.5%\u003C/td\u003E\r\n\u003Ctd\u003E0.538\u003C/td\u003E\r\n\u003Ctd\u003E1.503\u003C/td\u003E\r\n\u003Ctd\u003E812,306\u003C/td\u003E\r\n\u003C/tr\u003E\u003C/tbody\u003E\u003C/table\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"blog_text_624248501":{"id":"blog-text-cb1203e371","text":"\u003Cp\u003E\u003Ci\u003E\u003Cb\u003ETab 1.\u003C/b\u003E Strict, task-level resolve rates (%). Pass@1 = mean single-attempt pass rate (averaged across three attempts); Pass^3 = resolved if all three attempts succeed. Cost multipliers are set to 1 for each combination of CoCo with the 3 models (Opus 5, Sonnet 5, GPT 5.6 Sol); the cost multiplier for other harnesses in conjunction with the corresponding model is compared against this baseline. For example, Claude Code with Opus 5 is 3.9x more costly than CoCo with Opus 5.\u003C/i\u003E\u003C/p\u003E\r\n\u003Cp\u003EA key takeaway is that the harness matters significantly for quality and cost efficiency. On data-eng-bench, CoCo consistently achieves higher quality at lower cost than other harnesses:\u003C/p\u003E\r\n\u003Cul\u003E\r\n\u003Cli\u003EWith Opus 5, CoCo reports a 4pp higher Pass@1 at 3.9x lower cost than Claude Code.\u003C/li\u003E\r\n\u003Cli\u003EWith Sonnet 5, CoCo reports the same Pass@1 at 3x lower cost than Claude Code.\u003C/li\u003E\r\n\u003Cli\u003EWith GPT 5.6 Sol, CoCo reports a 3.6pp higher Pass@1 at 1.5x lower cost than Codex.\u003C/li\u003E\r\n\u003C/ul\u003E\r\n\u003Cp\u003EWe now delve into why CoCo is more cost efficient than Claude Code.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"image_1517420975":{"id":"image-d37d1d6854","height":"694","lazyEnabled":true,"src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--38d60883-4527-4195-8cdb-ce5ae749ce11/figure-2.-introducing-data-eng-bench--why-you-need-data-native-harnesses-for-data-engineering.png?quality=85&preferwebp=true","alt":"Figure 2. Avg. number of steps taken by the Opus 5 agent under CoCo and Claude Code per trial. ","isLcpImage":false,"width":"2048","title":"Figure 2. Avg. number of steps taken by the Opus 5 agent under CoCo and Claude Code per trial.",":type":"snowflake-site/components/image"},"image_1424448934":{"id":"image-8cac251dba","height":"1193","lazyEnabled":true,"src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--756369d2-575b-4763-b35d-b1f666196618/figure-3.-introducing-data-eng-bench--why-you-need-data-native-harnesses-for-data-engineering.png?quality=85&preferwebp=true","alt":"Figure 3. Avg. number of tool operations by Opus 5 under CoCo and Claude Code across various phases of solving a task from data-eng-bench.","isLcpImage":false,"width":"2048","title":"Figure 3. Avg. number of tool operations by Opus 5 under CoCo and Claude Code across various phases of solving a task from data-eng-bench.",":type":"snowflake-site/components/image"},"blog_text_737427679":{"id":"blog-text-377096f923","text":"\u003Cp\u003EWith Opus 5, CoCo requires 1.5x fewer tool operations and 2.2x fewer agent steps than Claude Code to solve a task on average. The agent spends most of its time on the predevelopment exploration and postdevelopment validation phases, under both harnesses. Overall, CoCo issues 1.7x fewer SQL queries, concentrates file reads to the exploration phase and produces 1.9x fewer file writes. The tool call patterns in different phases on task execution reveal the strategies adopted by Opus 5 under the two harnesses:\u003C/p\u003E\r\n\u003Col\u003E\r\n\u003Cli\u003E\u003Cb\u003EPhase 1 — Predevelopment setup and exploration:\u003C/b\u003E CoCo requires 1.3x fewer tool operations before writing or editing the first SQL file. Both agents spend similar effort reading existing models and project config (6.6 vs 7.1 file reads). The divergence is in SQL: CoCo uses 2x fewer queries for schema exploration and source data profiling before writing.\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EPhase 2 — dbt model development:\u003C/b\u003E Both agents primarily write SQL files (4.4 vs 5.6 file writes). Claude Code additionally reads more files during this phase — consulting existing staging models and macros mid-draft — while CoCo completes all reading in Phase 1 and writes without look-back.\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EPhase 3 — Build, validate and iterate on Snowflake:\u003C/b\u003E CoCo requires 1.7x fewer tool operations, with the gap driven primarily by Claude Code issuing 1.5x more SQL queries to verify the built models and 3x more shell setup commands around each build iteration.\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EPhase 4 — Cross-validation with DuckDB:\u003C/b\u003E Interestingly, Claude Code performs DuckDB cross-dialect validation in \u003Cb\u003E2.25x\u003C/b\u003E more trials than CoCo. This step is unnecessary given the backend is specified as Snowflake and evaluates the agent's ability to avoid irrelevant details in the task instructions.\u003C/li\u003E\r\n\u003C/ol\u003E\r\n\u003Cp\u003EThese differences reflect two distinct strategies. CoCo follows a plan-then-execute approach — it front-loads exploration, writes directly, verifies once and stops. Claude Code follows an explore-and-refine approach — it interleaves reading with writing throughout development, wraps each phase with additional verification passes and treats cross-dialect validation as a default rather than an optional step.\u003C/p\u003E\r\n\u003Cp\u003E\u003Cb\u003EGet started\u003C/b\u003E\u003C/p\u003E\r\n\u003Cp\u003EData-eng-bench is open source. Whether you build agents, harnesses or the models underneath them, it's a realistic, hard-to-saturate testbed for measuring autonomous data engineering.\u003C/p\u003E\r\n\u003Cul\u003E\r\n\u003Cli\u003E\u003Cb\u003EExplore the benchmark:\u003C/b\u003E \u003Ca href=\"https://github.com/Snowflake-Labs/data-eng-bench\" target=\"_blank\" rel=\"noopener noreferrer\"\u003Ehere\u003C/a\u003E\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EContribute:\u003C/b\u003E add tasks, harnesses or model results and tell us what you find.\u003C/li\u003E\r\n\u003C/ul\u003E\r\n\u003Cp\u003EWe're excited to see how far agents can go on the work that quietly powers every analytics stack. Try data-eng-bench, run your own model–harness combinations and share your results!\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"}},":itemsOrder":["blog_text","image","blog_text_814042905","blog_text_1891894734","blog_text_624248501","image_1517420975","image_1424448934","blog_text_737427679"],":type":"wcm/foundation/components/responsivegrid"},"responsivegrid_premium_content_banner":{"columnCount":12,"columnClassNames":{},"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","appliedCssClassNames":"snowflake-responsive-component-top-padding-medium",":items":{},":itemsOrder":[],":type":"wcm/foundation/components/responsivegrid"},"container_author_chip":{"layout":"RESPONSIVE_GRID","columnCount":12,"columnClassNames":{"author_chip":"aem-GridColumn aem-GridColumn--default--12"},"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","id":"container-d3ab1721a0",":type":"snowflake-site/components/container",":items":{"author_chip":{"id":"author-chip-5afe2e9206","title":{"id":"title","type":"heading2","lines":["Learn more about the authors"],":type":"snowflake-site/components/title-v2"},"authors":[{"authorImage":{"id":"image-9311e288fa","height":"800","lazyEnabled":true,"src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--249da901-4810-48b7-ab40-99208c5e3b73/default-author-image.png?quality=85&preferwebp=true","alt":"Snowflake AI Research","isLcpImage":false,"width":"800",":type":"snowflake-site/components/image"},"authorCta":{"id":"button-438739d40c","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/blog/authors/snowflake-ai-research/"},"linkTargetContentType":"DOCUMENT_LEARN",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_INTERNAL","text":"Snowflake AI Research"}}],":type":"snowflake-site/components/blog/author-chip"}},":itemsOrder":["author_chip"],"appliedCssClassNames":"snowflake-responsive-component-top-padding-medium"}},":itemsOrder":["container_hero","responsivegrid_content","responsivegrid_premium_content_banner","container_author_chip"]},"flexible_column_content_container_2":{"layout":"SIMPLE","id":"container-78117ef983",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"blog_table_of_content":{"id":"blog-table-of-content-731b57e25b",":type":"snowflake-site/components/blog/blog-table-of-content","tableOfContents":[]}},":itemsOrder":["blog_table_of_content"]},":type":"snowflake-site/components/flexible-column-container","isActiveTOC":false,"isBlogPage":true},"related_content":{"id":"related-content-56574239f1","relatedContent":[],":type":"snowflake-site/components/blog/related-content","isBlogPage":true}},":itemsOrder":["flexible_column_container","related_content"],"appliedCssClassNames":"snowflake-container"}},":itemsOrder":["container_breadcrumb","container_main_content"],":type":"wcm/foundation/components/responsivegrid"},"container_47873732":{"additionalClasses":"section--blog-newsletter","layout":"RESPONSIVE_GRID","columnCount":12,"columnClassNames":{"flexible_column_cont":"aem-GridColumn aem-GridColumn--default--12"},"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","id":"container-f007059622",":type":"snowflake-site/components/container",":items":{"flexible_column_cont":{"id":"flexible-column-container-33bf978ce2","type":"1-column","alignColumns":"top","containerMaxWidth":"extra-large","topPadding":"small","bottomPadding":"none","spaceBetween":"small","reverseOnMobile":false,"carouselOnMobile":false,"propertiesCSSClasses":"section--blog-newsletter","backgroundImageOption":"none","flexible_column_content_container_1":{"layout":"SIMPLE","id":"container-2811806ec2",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"marketo_v2":{"id":"marketo-v2-0bd73e5134","marketoForm":{"hidden":null,"formId":"3320","successUrl":null,"edit":false,"script":null,"values":null},"title":{"id":"title","type":"heading3","lines":["Subscribe to our blog newsletter","Get the best, coolest and latest delivered to your inbox each week"],":type":"snowflake-site/components/title-v2"},"munchkinId":"252-RFO-227","serverInstance":"252-RFO-227.mktoweb.com","marketoConfigured":true,"formConfigured":true,":type":"snowflake-site/components/form/marketo-v2"},"text":{"id":"text-663084d000","additionalClasses":"newsletter-disclaimer","text":"\u003Cp\u003EBy submitting this form, I understand Snowflake will process my personal information in accordance with their Privacy Notice.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/text"}},":itemsOrder":["marketo_v2","text"]},":type":"snowflake-site/components/flexible-column-container","isActiveTOC":false,"isBlogPage":true}},":itemsOrder":["flexible_column_cont"],"appliedCssClassNames":"snowflake-container"},"experiencefragment-pre-footer":{"id":"experiencefragment-74addab0cb","localizedFragmentVariationPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/get-started-pre-footer/get-started-pre-footer/jcr:content","configured":true,":type":"snowflake-site/components/experiencefragment","xfModelPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/get-started-pre-footer/get-started-pre-footer.xfmodel.json"},"markup_editor":{"id":"markup-editor-9bffe2a5f7","title":"Page CSS","cssContent":"@media screen and (min-width:768px){.snowflake-blog-author-chip-wrapper{justify-content:flex-start}.snowflake-blog-related-content-on-blog-page{max-width:1408px;margin-left:auto;margin-right:auto}.snowflake-text{font-family:Lato,sans-serif;font-weight:400;font-size:16px;line-height:24px}}.section--blog-newsletter{max-width:none;width:100%;padding-left:0;padding-right:0;margin-left:0;margin-right:0;margin-bottom:0}.section--blog-newsletter .mktoField{background-color:transparent !important}.section--blog-newsletter\u003E.container{padding-left:0;padding-right:0}@media screen and (min-width:768px){.section--blog-newsletter\u003E.container{padding-left:0;padding-right:0}}.newsletter-disclaimer p{font-size:14px !important}.section--blog-newsletter .snowflake-marketo-form-container{margin-bottom:24px;background-color:#f6f9fa;gap:48px;box-shadow:none}.section--blog-newsletter .snowflake-title p.snowflake-title-line:first-child{font-family:Texta;font-size:24px;line-height:26px;font-weight:700;margin-bottom:4px}.section--blog-newsletter .snowflake-title p.snowflake-title-line{text-transform:none;font-family:\"Lato\",sans-serif;font-size:16px;line-height:24px;font-weight:normal}@media screen and (min-width:1024px){.section--blog-newsletter .snowflake-marketo-form-container{display:flex;justify-content:center}.section--blog-newsletter .snowflake-title .snowflake-title-line{text-align:left}.section--blog-newsletter .snowflake-marketo-form .mktoFormRow:has(\u003E input[type=\"hidden\"]){flex-grow:0}.section--blog-newsletter .snowflake-marketo-form{display:flex;width:50% !important}.section--blog-newsletter .snowflake-marketo-form .mktoButtonRow{flex-grow:0;width:auto !important;margin-left:0;margin-right:0}.section--blog-newsletter .snowflake-marketo-form .mktoFormRow{flex-grow:1}.section--blog-newsletter\u003E.container{padding-left:0;padding-right:0}.section--blog-newsletter .snowflake-marketo-form-title{width:50%;margin-bottom:0 !important}.section--blog-newsletter .center .snowflake-title{align-items:flex-start}}.snowflake-sub-navigation a.snowflake-sub-navigation-primary-link{width:auto !important}.snowflake-blog-hero{align-items:stretch !important}",":type":"snowflake-site/components/markup-editor","isGSAPEnabled":false},"markup_editor-table":{"id":"markup-editor-7f1ed3bed0","title":"Table Styling CSS","cssContent":"#snowflake-blog-template-main-container table{width:100%;background-color:var(--ui-background-01);border-collapse:collapse;border:2px solid var(--ui-background-09);font-family:'Lato',sans-serif;color:var(--ui-background-09)}#snowflake-blog-template-main-container table thead{background-color:var(--ui-01)}#snowflake-blog-template-main-container table th,#snowflake-blog-template-main-container table td{border:2px solid var(--ui-background-09);padding:var(--spacing-01)}",":type":"snowflake-site/components/markup-editor","isGSAPEnabled":false},"experiencefragment-footer":{"id":"experiencefragment-9825917b71","localizedFragmentVariationPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/footer/master/jcr:content","configured":true,":type":"snowflake-site/components/experiencefragment","xfModelPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/footer/master.xfmodel.json"}},":itemsOrder":["experiencefragment-banner","experiencefragment-header","experiencefragment-sub-header","responsivegrid","container_47873732","experiencefragment-pre-footer","markup_editor","markup_editor-table","experiencefragment-footer"],":type":"wcm/foundation/components/responsivegrid"}},":itemsOrder":["root"],":hierarchyType":"page",":path":"/content/snowflake-site/global/en/blog/engineering/data-eng-bench-data-engineering-agent-benchmark","analyticsContentTags":["snowflake-site:taxonomy/blog/engineering-blog/data-engineering"],"analyticsEnabled":true,"isPasswordProtected":false,"coveoConfig":{"pipeline":"snowflake.com","searchHub":"snowflake.com","organizationId":"snowflakecomputingproduction8neljofn","apiKey":"xx335921a6-2a0a-40f2-a167-e390b4766c3d"},"analyticsDebugMode":false,"analyticsData":{"excludeFromAnalytics":false,"subCategory":"","pageType":"homepage","templateName":"blog-page","siteName":"snowflake","pageUrl":"/content/snowflake-site/global/en/blog/engineering/data-eng-bench-data-engineering-agent-benchmark","language":"en","category":"general","pageName":"Introducing Data-eng-bench: Why You Need \"Data-Native\" Harnesses for Data Engineering","contentTags":["snowflake-site:taxonomy/blog/engineering-blog/data-engineering"]},"locale":"en"}
  