{"templateName":"blog-page","cssClassNames":"blog-page page basicpage summit-page","allowedRenditionsWidth":["320","480","640","768","960","1200","1440","1920"],"language":"en","description":"Discover how to solve business logic interoperability in a multi-engine lakehouse. Learn to manage Apache Ossie (incubating), Iceberg Views, and semantic layers.","title":"Managing Business Logic Interoperability in Multi-Engine Lakehouses","analyticsPageType":"homepage","analyticsCategory":"general","analyticsSubCategory":"","excludeFromAnalytics":false,":mappedPath":"/en/blog/engineering/lakehouse-business-logic-interoperability/",":type":"snowflake-site/components/structure/page",":items":{"root":{"columnCount":12,"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","columnClassNames":{"experiencefragment-banner":"aem-GridColumn aem-GridColumn--default--12","experiencefragment-sub-header":"aem-GridColumn aem-GridColumn--default--12","experiencefragment-pre-footer":"aem-GridColumn aem-GridColumn--default--12","experiencefragment-header":"aem-GridColumn aem-GridColumn--default--12","markup_editor-table":"aem-GridColumn aem-GridColumn--default--12","responsivegrid":"aem-GridColumn aem-GridColumn--default--12","experiencefragment-footer":"aem-GridColumn aem-GridColumn--default--12","markup_editor":"aem-GridColumn aem-GridColumn--default--12","container_47873732":"aem-GridColumn aem-GridColumn--default--12"},":items":{"experiencefragment-banner":{"id":"experiencefragment-ad381149fd","localizedFragmentVariationPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/pushdown-banner/pushdown-banner-blank/jcr:content","configured":true,":type":"snowflake-site/components/experiencefragment","xfModelPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/pushdown-banner/pushdown-banner-blank.xfmodel.json"},"experiencefragment-header":{"id":"experiencefragment-3b902db8d9","localizedFragmentVariationPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/mega-nav-header/master/jcr:content","configured":true,":type":"snowflake-site/components/experiencefragment","xfModelPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/mega-nav-header/master.xfmodel.json","languageNavPath":"/content/snowflake-site/global/en/blog/engineering/lakehouse-business-logic-interoperability.languagenav.json","appliedCssClassNames":"snowflake-sticky-nav-host"},"experiencefragment-sub-header":{"id":"experiencefragment-a260942184","localizedFragmentVariationPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/sub-navigation/engineering-blog-sub-nav/jcr:content","configured":true,":type":"snowflake-site/components/experiencefragment","xfModelPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/sub-navigation/engineering-blog-sub-nav.xfmodel.json"},"responsivegrid":{"columnCount":12,"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","columnClassNames":{"container_breadcrumb":"aem-GridColumn aem-GridColumn--default--12","container_main_content":"aem-GridColumn aem-GridColumn--default--12"},":items":{"container_breadcrumb":{"layout":"RESPONSIVE_GRID","columnCount":12,"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","columnClassNames":{"breadcrumb":"aem-GridColumn aem-GridColumn--default--12"},"id":"blog-page-breadcrumb-indentation",":type":"snowflake-site/components/container",":items":{"breadcrumb":{"id":"breadcrumb-ea353aaf33","breadcrumbItems":[{"title":"Blog","path":"/en/blog/engineering/","active":false},{"title":"Data Engineering","path":"/en/blog/engineering/data-engineering/","active":false},{"title":"Managing Business Logic Interoperability in Multi-Engine Lakehouses","path":"/en/blog/engineering/lakehouse-business-logic-interoperability/","active":false}],":type":"snowflake-site/components/blog/breadcrumb"}},":itemsOrder":["breadcrumb"],"appliedCssClassNames":"snowflake-container"},"container_main_content":{"layout":"RESPONSIVE_GRID","columnCount":12,"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","columnClassNames":{"flexible_column_container":"aem-GridColumn aem-GridColumn--default--12","related_content":"aem-GridColumn aem-GridColumn--default--12"},"id":"main-content",":type":"snowflake-site/components/container",":items":{"flexible_column_container":{"id":"flexible-column-container-c6fe4796b1","propertiesId":"snowflake-blog-template-main-container","type":"2-column-60-40","alignColumns":"top","containerMaxWidth":"extra-large","topPadding":"none","bottomPadding":"none","spaceBetween":"none","reverseOnMobile":true,"carouselOnMobile":false,"backgroundImageOption":"none","flexible_column_content_container_1":{"layout":"SIMPLE","id":"container-f27c9e2c8d",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"container_hero":{"layout":"RESPONSIVE_GRID","columnCount":12,"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","columnClassNames":{"blog_hero":"aem-GridColumn aem-GridColumn--default--12"},"id":"container-6e595961f9",":type":"snowflake-site/components/container",":items":{"blog_hero":{"id":"blog-hero-3f4de1b284","linkedInShareUrl":"https://www.linkedin.com/shareArticle?mini=true&url=https%3A%2F%2Fwww.snowflake.com%2Fcontent%2Fsnowflake-site%2Fglobal%2Fen%2Fblog%2Fengineering%2Flakehouse-business-logic-interoperability&title=Managing+Business+Logic+Interoperability+in+Multi-Engine+Lakehouses","twitterShareUrl":"https://x.com/intent/post?url=https%3A%2F%2Fwww.snowflake.com%2Fcontent%2Fsnowflake-site%2Fglobal%2Fen%2Fblog%2Fengineering%2Flakehouse-business-logic-interoperability&text=Managing+Business+Logic+Interoperability+in+Multi-Engine+Lakehouses","facebookShareUrl":"https://www.facebook.com/sharer/sharer.php?u=https%3A%2F%2Fwww.snowflake.com%2Fcontent%2Fsnowflake-site%2Fglobal%2Fen%2Fblog%2Fengineering%2Flakehouse-business-logic-interoperability","showClaude":true,"showChatGpt":true,"authors":[{"authorImage":{"id":"image-26881491b0","height":"854","src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--d5dae03d-5733-42e0-813e-f92748305643/jason-hughes.jpg?preferwebp=true&quality=85","lazyEnabled":true,"width":"854",":type":"snowflake-site/components/image"},"authorCta":{"id":"button-07345160b8","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/blog/authors/jason-hughes/"},"linkTargetContentType":"DOCUMENT_LEARN",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_INTERNAL","text":"Jason Hughes"}},{"authorImage":{"id":"image-656465c8ad","height":"800","src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--85428841-cdd8-4cae-b3d0-4003575f4937/jim-lebonitte.png?preferwebp=true&quality=85","lazyEnabled":true,"width":"800",":type":"snowflake-site/components/image"},"authorCta":{"id":"button-66741ef508","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/blog/authors/jim-lebonitte/"},"linkTargetContentType":"DOCUMENT_LEARN",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_INTERNAL","text":"Jim Lebonitte"}}],"image":{"id":"image-6c7f0aaacb","height":"720","src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--cc092a50-6db2-40a7-b106-c4713a0b5246/sf-eng-blog-ml-1.png?preferwebp=true&quality=85","lazyEnabled":true,"width":"1680",":type":"snowflake-site/components/image"},"timeToRead":"28","publicationDate":"JUL 28, 2026","tag":{"tagText":"Data Engineering","tagColor":"#29B5E8"},"title":{"lines":["Managing Business Logic Interoperability in Multi-Engine Lakehouses"],"type":"heading2",":type":"snowflake-site/components/title-v2"},":type":"snowflake-site/components/blog/blog-hero"}},":itemsOrder":["blog_hero"]},"responsivegrid_content":{"columnCount":12,"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","columnClassNames":{"image_copy":"aem-GridColumn aem-GridColumn--default--12","image":"aem-GridColumn aem-GridColumn--default--12","blog_text_277633130_":"aem-GridColumn aem-GridColumn--default--12","blog_text_copy_copy_":"aem-GridColumn aem-GridColumn--default--12","blog_title_copy":"aem-GridColumn aem-GridColumn--default--12","blog_title_1262794393":"aem-GridColumn aem-GridColumn--default--12","blog_text_copy_copy__198540594":"aem-GridColumn aem-GridColumn--default--12","blog_title":"aem-GridColumn aem-GridColumn--default--12","blog_text_copy_copy__1799410961":"aem-GridColumn aem-GridColumn--default--12","markup_editor":"aem-GridColumn aem-GridColumn--default--12","blog_text_copy_copy__1311716111":"aem-GridColumn aem-GridColumn--default--12","blog_title_126279439":"aem-GridColumn aem-GridColumn--default--12","code_snippet":"aem-GridColumn aem-GridColumn--default--12","blog_text_copy_copy":"aem-GridColumn aem-GridColumn--default--12","blog_text_copy":"aem-GridColumn aem-GridColumn--default--12","blog_text":"aem-GridColumn aem-GridColumn--default--12","blog_text_277633130":"aem-GridColumn aem-GridColumn--default--12"},"appliedCssClassNames":"snowflake-layout-container-inner-padding-small",":items":{"blog_text_copy":{"id":"blog-text-157141e24d","text":"\u003Cp\u003EA CFO gets different revenue figures from two dashboards, even though both pointed at the same Iceberg table. This isn't always because the data drifted, but rather, because the two dashboards point to different engines. &quot;Recognized revenue&quot; was defined in a SQL view in each engine, and the definitions quietly drifted or contained a difference in how each engine handles nulls in a CASE expression. No error, no alert. Just two different answers from the same source of truth, and no clear owner for the reconciliation.\u003C/p\u003E\r\n\u003Cp\u003EThe series breaks down multi-engine lakehouse architectures into three interoperability dimensions: data, business logic and governance. Solving for interoperability requires ecosystem solutions that are vendor-neutral and provide customers with optionality at each of these dimensions. Apache Iceberg™ for data interoperability (and some minor efforts to address business logic interoperability) and Apache Polaris™ for governance interoperability have emerged as widely adopted open source projects with diverse communities. For business logic interoperability holistically, however, no open source project has reached the same level of adoption as those two, though progress is being made with \u003Ca href=\"https://ossie.apache.org/\" target=\"_blank\" rel=\"noopener noreferrer\"\u003EApache Ossie (incubating)\u003C/a\u003E, \u003Ca href=\"https://www.snowflake.com/en/blog/apache-ossie-open-semantic-interchange-incubator/\" target=\"_blank\" rel=\"noopener noreferrer\"\u003Eformerly known as Open Semantic Interchange (OSI)\u003C/a\u003E, which has seen vendor participation more than double in the past year.\u003C/p\u003E\r\n\u003Cp\u003EWhile business logic doesn't carry the same stakes as governance interoperability, where gaps can mean compliance violations and data breaches, it's still where organizations are most likely to get caught off-guard as they plan their lakehouse and AI deployments. These gaps are mostly not acceptable in an enterprise environment. The good news is that viable workarounds exist today, and for most organizations, those workarounds are enough to architect around the gaps while the ecosystem catches up.\u003C/p\u003E\r\n\u003Cp\u003EWhat a multi-engine lakehouse is really doing is unbundling a data warehouse. The data interoperability deep dive covers the storage layer of that unbundling, which is in reasonable shape. But a data warehouse/data processing system was never just storage; it was also business logic entities like views, user-defined functions (UDFs), stored procedures and metric definitions that turned raw tables into business answers.\u003C/p\u003E\r\n\u003Cp\u003EOnce you decouple compute engines from a previously tightly coupled compute engine and storage engine, all that business logic needs to either live somewhere securely accessible and engine-agnostic or be replicated independently in every engine that accesses the data. This overview provides a deep dive into existing mechanisms for managing business logic in multi-engine lakehouse environments and practical guidance for how to architect with today's solutions as you look to the future when interoperable business logic is fully solved.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"blog_title_copy":{"id":"blog-title-d317ed0cd2","propertiesId":"where-we-are-today","type":"heading2","lines":["Where we are today"],":type":"snowflake-site/components/blog/blog-title"},"blog_text":{"id":"blog-text-989d863f73","additionalClasses":"inline","text":"\u003Cp\u003EThere is no complete, widely adopted standard for sharing business logic across engines in production. Not for lack of attempts or progress: Iceberg Views, Iceberg UDFs, Substrait and Ossie all exist in various stages of maturity. None have achieved the cross-engine adoption needed to be a reliable architectural dependency today. For anyone making production architecture decisions right now, the practical effect is the same as having no standard at all, though the seeds being planted should factor into long-term planning.\u003C/p\u003E\r\n\u003Cp\u003EIt's worth framing the scope of this problem accurately. The majority of query volume in most data platforms hits tables directly: \u003Ccode\u003ESELECT\u003C/code\u003E against Iceberg tables where data interoperability (covered in the \u003Ca rel=\"noopener noreferrer\" target=\"_blank\" href=\"http://www.snowflake.com/en/blog/engineering/lakehouse-data-interoperability-guide\"\u003Edata interoperability deep dive\u003C/a\u003E) is sufficient. Business logic interoperability matters specifically for the workloads that route through or include views, UDFs, metric definitions and stored procedures. While that may be a smaller share of total query volume, it's where the highest-stakes business answers live. The CFO's revenue dashboard, the regulatory report, the executive KPI that drives a board decision: Those are the workloads that encode business logic, and those are the ones affected by the gaps described in this post.\u003C/p\u003E\r\n\u003Cp\u003EThe consequence is straightforward. Any metric, view, UDF or transformation logic defined in one engine must be manually redefined in every other engine that accesses the same data. There's no synchronization mechanism and no feedback loop that tells you when definitions have drifted. This isn't a hypothetical problem. Companies like Airbnb and LinkedIn have built entire internal frameworks to address it: \u003Ca rel=\"noopener noreferrer\" target=\"_blank\" href=\"https://medium.com/airbnb-engineering/how-airbnb-achieved-metric-consistency-at-scale-f23cc53dea70\"\u003EMinerva\u003C/a\u003E at Airbnb for metric consistency and \u003Ca rel=\"noopener noreferrer\" target=\"_blank\" href=\"https://github.com/linkedin/coral\"\u003ECoral\u003C/a\u003E at LinkedIn for cross-engine SQL translation. But those organizations have large dedicated engineering teams that can sustain these systems indefinitely. For most organizations, that's not a realistic path; the ongoing maintenance burden alone tends to outweigh the initial build cost.\u003C/p\u003E\r\n\u003Cp\u003EFor governance interoperability, tools like Immuta and Privacera exist to centralize policy definitions and push them to multiple engines. No equivalent exists for business logic. Semantic layer platforms like Cube and AtScale come closest: You define metrics once, and they generate engine-specific SQL at query time. But consumers must query through the tool for the consistency guarantee to hold; anything that bypasses it doesn't benefit, and the &quot;sit-in-the-middle&quot; architecture runs into the same scaling and single-point-of-failure problems that have historically limited data virtualization, even when they are distributed systems. Transformation tools like dbt and SQLMesh take a different approach. They generate per-engine SQL at build time from a single model definition, which is closer to &quot;define once, deploy to many,&quot; but they only cover the transformations managed through the tool. We cover both in detail later in this post.\u003C/p\u003E\r\n\u003Cp\u003EThe problem gets worse in decentralized organizations where different business units build their own logic independently on shared data sets, each internally consistent but unaware when their definitions diverge from each other.\u003C/p\u003E\r\n\u003Cp\u003EThe pragmatic answer for today: For high-stakes business logic (for example, finance, regulatory and executive dashboards), either physicalize the logic via a pipeline that computes the result of the logic and writes the results of that computation to a table so all engines inherently get the same answer, or centralize authoritative definitions into a single, enterprise-ready engine capable of handling highly concurrent workloads. Multi-engine reads on shared Iceberg tables are reasonable for ad hoc analysis; multi-engine definitions of &quot;recognized revenue&quot; are not, unless the maintenance burden is worth it for a given definition. This isn't a retreat from the lakehouse vision. Your data is still in Iceberg, portable and open. You're just being deliberate about where authoritative logic lives while the ecosystem catches up.\u003C/p\u003E\r\n\u003Cp\u003EHere's a summary table of what we'll go through in this blog:\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"blog_text_277633130_":{"id":"blog-text-7e1bef5975","text":"\u003Ctable\u003E\r\n\u003Cthead\u003E\u003Ctr\u003E\u003Cth style=\"min-width: 160px;\"\u003EMechanism\u003C/th\u003E\r\n\u003Cth\u003EWhat it covers\u003C/th\u003E\r\n\u003Cth\u003EState today\u003C/th\u003E\r\n\u003Cth\u003EPragmatic guidance\u003C/th\u003E\r\n\u003C/tr\u003E\u003C/thead\u003E\u003Ctbody\u003E\u003Ctr\u003E\u003Ctd\u003E\u003Cb\u003EIceberg Views\u003C/b\u003E\u003C/td\u003E\r\n\u003Ctd\u003ENamed SQL queries stored in the catalog; supports multiple dialect representations\u003C/td\u003E\r\n\u003Ctd\u003ESpec live; Spark/Trino ship single-dialect support; multi-dialect authoring limited to Java API; no official transpilation stance or tool\u003C/td\u003E\r\n\u003Ctd\u003EUse as a catalog-level registry; don't rely on for cross-engine consistency without verification; physicalize or designate a single authoritative engine for high-stakes logic\u003C/td\u003E\r\n\u003C/tr\u003E\u003Ctr\u003E\u003Ctd\u003E\u003Cb\u003EIceberg UDFs\u003C/b\u003E\u003C/td\u003E\r\n\u003Ctd\u003ESQL UDF metadata format in the catalog; same architecture as views\u003C/td\u003E\r\n\u003Ctd\u003ESpec published; no major engine has shipped support\u003C/td\u003E\r\n\u003Ctd\u003ETreat as engine-locked for now; route critical UDF workloads through a single engine\u003C/td\u003E\r\n\u003C/tr\u003E\u003Ctr\u003E\u003Ctd\u003E\u003Cb\u003EApache Ossie (incubating)\u003C/b\u003E\u003C/td\u003E\r\n\u003Ctd\u003EMetric and dimension definitions; YAML-based, vendor-neutral; AI context support\u003C/td\u003E\r\n\u003Ctd\u003E50+ orgs in working group; spec live under Apache 2.0; converters in progress; incubation underway at the Apache Software Foundation\u003C/td\u003E\r\n\u003Ctd\u003EMost credible path for metric consistency long-term; in the meantime, centralize authoritative definitions in a single engine or semantic layer\u003C/td\u003E\r\n\u003C/tr\u003E\u003Ctr\u003E\u003Ctd\u003E\u003Cb\u003Edbt Semantic Models / SQLMesh\u003C/b\u003E\u003C/td\u003E\r\n\u003Ctd\u003EMetric and transformation definitions as code; generate per-engine SQL at build time\u003C/td\u003E\r\n\u003Ctd\u003EMature; widely deployed\u003C/td\u003E\r\n\u003Ctd\u003EUseful for managing consistency within transformations they own; doesn't address behavioral divergence at query time; secondary to architectural mitigations\u003C/td\u003E\r\n\u003C/tr\u003E\u003Ctr\u003E\u003Ctd\u003E\u003Cb\u003Esqlglot\u003C/b\u003E\u003C/td\u003E\r\n\u003Ctd\u003ESQL transpilation across 30+ dialects; handles null ordering, type casting, function semantics\u003C/td\u003E\r\n\u003Ctd\u003EMature open source; actively maintained\u003C/td\u003E\r\n\u003Ctd\u003EUse for migrations; covers a lot of semantic ground, but not all; note that you take on operations and accountability\u003C/td\u003E\r\n\u003C/tr\u003E\u003Ctr\u003E\u003Ctd\u003E\u003Cb\u003ESubstrait\u003C/b\u003E\u003C/td\u003E\r\n\u003Ctd\u003ECross-language spec for serializing relational algebra\u003C/td\u003E\r\n\u003Ctd\u003EDuckDB, Arrow, DataFusion support; not much support outside of those\u003C/td\u003E\r\n\u003Ctd\u003ERight long-term concept; the project isn't very mature from an adoption perspective, so don't treat it as a production dependency until the engines you use actually adopt it\u003C/td\u003E\r\n\u003C/tr\u003E\u003Ctr\u003E\u003Ctd\u003E\u003Cb\u003EStored procedures\u003C/b\u003E\u003C/td\u003E\r\n\u003Ctd\u003EEngine-proprietary procedural languages; no shared abstraction exists\u003C/td\u003E\r\n\u003Ctd\u003ENo community spec; no portability mechanism; no clear path\u003C/td\u003E\r\n\u003Ctd\u003EAccept as engine-locked; route those workloads through a single engine; refactor to declarative SQL or externalize the logic to a shared tool\u003C/td\u003E\r\n\u003C/tr\u003E\u003C/tbody\u003E\u003C/table\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"blog_title":{"id":"blog-title-30a6327ba4","propertiesId":"what-exists-but-doesnt-fully-solve-the-business-logic-problem","type":"heading2","lines":["What exists but doesn't fully solve the business logic problem"],":type":"snowflake-site/components/blog/blog-title"},"blog_text_277633130":{"id":"blog-text-3a53bd946a","additionalClasses":"inline","text":"\u003Cp\u003EBefore evaluating the tools and specs the community has built, it's worth establishing why cross-engine business logic consistency is hard for the workloads that depend on it, which is certainly not all workloads. The behavioral differences across SQL engines are the root cause; everything that follows in this section is an attempt to work around them.\u003C/p\u003E\r\n\u003Ch3\u003ECross-engine behavioral differences\u003C/h3\u003E\r\n\u003Cp\u003EThe most fundamental challenge for business logic interoperability isn't that engines use different SQL syntax; syntax differences are well-understood and tools like sqlglot or custom tooling can translate between them. The harder problem is that engines with identical syntax produce different results because they implement SQL semantics differently. These aren't bugs; each engine is internally consistent. The problem is that &quot;correct&quot; is engine-specific, and two engines can both be right by their own standards while producing different answers from the same query on the same data. These divergences exist regardless of the underlying table format; they're properties of the compute engines, not the storage layer.\u003C/p\u003E\r\n\u003Cp\u003EA few concrete examples:\u003C/p\u003E\r\n\u003Cul\u003E\r\n\u003Cli\u003E\u003Cb\u003E\u003Ccode\u003EORDER BY\u003C/code\u003E with null values:\u003C/b\u003E \u003Ca rel=\"noopener noreferrer\" target=\"_blank\" href=\"https://spark.apache.org/docs/latest/sql-ref-syntax-qry-select-orderby.html\"\u003ESpark defaults to nulls first for ascending order\u003C/a\u003E; \u003Ca rel=\"noopener noreferrer\" target=\"_blank\" href=\"https://duckdb.org/docs/current/sql/query_syntax/orderby\"\u003EDuckDB\u003C/a\u003E and Trino default to nulls last. A query that sorts users by \u003Ccode\u003Elast_login_date\u003C/code\u003E ascending to find inactive accounts will place users who have never logged in (null) at the top on Spark and at the bottom on DuckDB or Trino, changing which users appear in the first page of results.\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EType casting:\u003C/b\u003E \u003Ca rel=\"noopener noreferrer\" target=\"_blank\" href=\"https://spark.apache.org/docs/latest/sql-ref-ansi-compliance.html\"\u003ESpark's behavior depends on whether ANSI mode is enabled\u003C/a\u003E; with ANSI mode off (the historical default; Spark 3.x defaults to ANSI mode off; Spark 4.0 changed this default, though many production deployments remain on 3.x), \u003Ccode\u003ECAST('a' AS INT)\u003C/code\u003E silently returns null, while \u003Ca rel=\"noopener noreferrer\" target=\"_blank\" href=\"https://duckdb.org/docs/current/sql/expressions/cast\"\u003EDuckDB\u003C/a\u003E and \u003Ca rel=\"noopener noreferrer\" target=\"_blank\" href=\"https://trino.io/docs/current/functions/conversion.html\"\u003ETrino\u003C/a\u003E throw an error on invalid casts by default (both provide \u003Ccode\u003ETRY_CAST\u003C/code\u003E as the null-returning alternative). A pipeline that casts user input and filters on the result will silently include rows with null values on Spark (ANSI off) that would have been excluded on DuckDB or Trino.\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003ETimestamp handling:\u003C/b\u003E The timezone that an engine assumes when resolving timestamp values adds another layer. How timezone resolution interacts with session configuration and how daily or weekly aggregations behave across DST boundaries all vary.\u003C/li\u003E\r\n\u003C/ul\u003E\r\n\u003Cp\u003EThese differences tend to surface in production on edge cases rather than in testing on clean data.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"image":{"id":"image-8d50c8b3b1","height":"1125","src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--c4f0f664-cd4c-437d-9ab0-28c0cdc69b84/business-logic-fig1.png?preferwebp=true&quality=85","alt":"Figure 1: An example where the same query against the same table produces different results, depending on what engine executed the query.","lazyEnabled":true,"width":"1808","title":"Figure 1: An example where the same query against the same table produces different results, depending on what engine executed the query.",":type":"snowflake-site/components/image"},"blog_text_copy_copy__1799410961":{"id":"blog-text-c40795000e","additionalClasses":"inline","text":"\u003Cp\u003E&nbsp;\u003C/p\u003E\r\n\u003Cp\u003ESQL is where most of the business logic portability discussion focuses, but it's not the only language in play. UDFs can be written in Python, Java or Scala with no cross-engine portability mechanism; stored procedures depend on engine-proprietary procedural languages with no equivalents elsewhere. Both are covered in dedicated sections later in this post.\u003C/p\u003E\r\n\u003Cp\u003EThe \u003Ca href=\"http://www.snowflake.com/en/blog/engineering/lakehouse-data-governance-interoperability\" target=\"_blank\" rel=\"noopener noreferrer\"\u003Egovernance deep dive\u003C/a\u003E covers fine-grained access control (FGAC) enforcement approaches in detail. The primitive to carry forward from there is that SQL behavioral divergence can affect who sees what data, not just whether a dashboard number is right. One consequence that extends beyond analytics: FGAC policies (row filters, column masks) are typically SQL expressions. For simple predicates on standard operators (\u003Ccode\u003E&lt;\u003C/code\u003E, \u003Ccode\u003E&gt;=\u003C/code\u003E, \u003Ccode\u003E!=\u003C/code\u003E), all major engines execute them the same way. But filters involving type casting, null handling or function calls enter the territory where engines can produce different results, meaning the security boundary itself varies by engine. A row filter that evaluates differently on Engine B than Engine A means users may see different data depending on which engine they query through, with no error to flag the discrepancy.\u003C/p\u003E\r\n\u003Ch3\u003EIceberg views\u003C/h3\u003E\r\n\u003Cp\u003EWith that behavioral variance as context, let's look at what the Iceberg community has built to share business logic across engines. The \u003Ca href=\"https://iceberg.apache.org/view-spec/\" target=\"_blank\" rel=\"noopener noreferrer\"\u003EIceberg view spec\u003C/a\u003E defines a metadata object for storing named SQL queries in the catalog alongside tables. Each view version contains one or more \u003Cb\u003Erepresentations\u003C/b\u003E: a \u003Ccode\u003E{type: &quot;sql&quot;, sql: &quot;...&quot;, dialect: &quot;...&quot;}\u003C/code\u003E tuple that records the SQL text and the dialect it was written in. A view can store multiple representations for different dialects, so in principle an engine can look for a representation matching its own dialect. The view is versioned, discoverable via the REST Catalog API and appears as a table-like object to downstream consumers.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"code_snippet":{"id":"code-snippet-6f71c51562","language":"javascript","codeSnippet":"// simplified — the full spec includes schema, default catalog/namespace, and other fields\r\n{\r\n  \"version-id\": 2,\r\n  \"representations\": [\r\n    { \"type\": \"sql\", \"sql\": \"SELECT date_add(order_date, 30) ...\", \"dialect\": \"spark\" },\r\n    { \"type\": \"sql\", \"sql\": \"SELECT date_add('day', 30, order_date) ...\", \"dialect\": \"trino\" }\r\n  ],\r\n  ...\r\n}","multiLine":true,":type":"snowflake-site/components/code-snippet"},"blog_text_copy_copy":{"id":"blog-text-6b565eeed9","text":"\u003Cp\u003EIceberg views are a real step forward from having no shared view mechanism at all. It also stops well short of solving the problem. The spec is a metadata container: It stores what you give it, with no translation layer, no behavioral normalization and no mechanism to verify that multiple dialect representations actually produce the same results. If you store a Spark SQL representation and a Trino SQL representation of the same view, keeping them semantically equivalent as both the underlying data and the business logic evolve is entirely your responsibility.\u003C/p\u003E\r\n\u003Cp\u003EThe deeper issue, covered in detail in the cross-engine behavioral differences section above, is that even identical SQL text can produce different results across engines. The behavioral divergences described there (null ordering, type casting, timestamp handling) all apply directly to Iceberg views. A view stored in one dialect and executed by another engine is subject to every one of those differences.\u003C/p\u003E\r\n\u003Cp\u003EThe community is actively developing an \u003Ca href=\"https://github.com/apache/iceberg/issues/10043\" target=\"_blank\" rel=\"noopener noreferrer\"\u003EIceberg materialized view spec\u003C/a\u003E with the same metadata-container architecture, specifying a SQL dialect, as well as adding a data dimension to the problem. The materialized output is a physical Iceberg table, so you inherit both the business logic and data interoperability issues.\u003C/p\u003E\r\n\u003Cp\u003EThe community is actively working on closing these gaps. The Iceberg view spec already supports storing multiple dialect representations per view, but today that's practically only accessible via the Java ViewBuilder API. Engines like Spark and Trino only write a single dialect through SQL. Making multi-dialect authoring easier via PyIceberg and engine-level SQL support is an \u003Ca href=\"https://github.com/apache/iceberg/issues/12675\" target=\"_blank\" rel=\"noopener noreferrer\"\u003Eongoing discussion\u003C/a\u003E, but the conversation in the community has evolved. The community has also recently merged a structured expressions spec that represents value expressions as structured signatures rather than dialect-specific SQL text, which is covered in the &quot;\u003Ca href=\"#open-problems-and-active-work\"\u003EOpen problems and active work\u003C/a\u003E&quot; section below.\u003C/p\u003E\r\n\u003Ch5\u003EPragmatic guidance\u003C/h5\u003E\r\n\u003Cp\u003EIceberg views are worth using as a catalog-level registry for view definitions; having view metadata in the catalog rather than scattered across engine-specific catalogs is a genuine organizational improvement. But don't rely on them for cross-engine consistency without additional discipline. If a view encodes high-stakes business logic, either: (a) physicalize it as a table so all engines read the same precomputed result; or (b) designate one engine as the authoritative source for that view, and treat the Iceberg view as documentation rather than an executable contract across engines. If you do maintain multiple dialect representations, build cross-engine regression tests that compare results on representative data and run them on a schedule; the drift is silent and gradual.\u003C/p\u003E\r\n\u003Ch3\u003EIceberg UDFs\u003C/h3\u003E\r\n\u003Cp\u003EThe Iceberg community has also specified a \u003Ca href=\"https://iceberg.apache.org/udf-spec/\" target=\"_blank\" rel=\"noopener noreferrer\"\u003ESQL UDF metadata format\u003C/a\u003E that follows the same architectural pattern as views: Store the function body, its SQL dialect, input/output types and versioning metadata in the catalog. Like views, UDFs can store multiple representations for different dialects. The spec is published, but no major engine has shipped support yet; this is earlier in its lifecycle than Iceberg views.\u003C/p\u003E\r\n\u003Cp\u003EThe more structural challenge is that the Iceberg UDF spec only covers SQL UDFs. In practice, many UDFs that encode critical business logic are written in Python, Java or Scala. SQL UDFs have the same dialect divergence problem as SQL views: The function body is SQL text, and the same behavioral differences (type casting, null handling, function semantics) apply. Non-SQL UDFs add another dimension. Python UDFs have relatively broad support across engines (Snowpark, Spark, Trino, DuckDB all support them), though the registration mechanisms and runtime environments differ. But other languages are less portable. Scala UDFs, for example, have no equivalent in Trino or DuckDB. For non-SQL UDFs to become portable via the Iceberg spec, the spec would first need to support non-SQL languages as representation types, and the community would need to converge on a common invocation contract across engines. That's a harder problem than SQL dialect translation, though in many cases the core function logic is already portable; it's the wiring around it (type declarations, serialization, available libraries) that isn't.\u003C/p\u003E\r\n\u003Ch5\u003EPragmatic guidance\u003C/h5\u003E\r\n\u003Cp\u003ETreat UDFs as engine-locked for now. If a UDF encodes critical business logic, that's one of the stronger arguments for routing that workload through a single engine rather than attempting cross-engine portability. Where necessary, physicalize the logic via a pipeline that writes results to a table so all engines inherently get the same answer.\u003C/p\u003E\r\n\u003Ch3\u003ESemantic layer and Apache Ossie (incubating)\u003C/h3\u003E\r\n\u003Cp\u003EBefore covering \u003Ca href=\"https://ossie.apache.org/\" target=\"_blank\" rel=\"noopener noreferrer\"\u003EOssie\u003C/a\u003E specifically, it's worth disambiguating a term: Semantic layer. Semantic layer means different things to different people. Some use it to mean a metric definitions layer. Some use it to mean a BI modeling layer (LookML, Cube, AtScale). Some use it to mean the full stack of business logic on top of raw tables. For this post, the functional pieces that a &quot;semantic layer&quot; typically refers to are covered individually: View-based transformations are covered in the Iceberg views section above; access policies are covered in the \u003Ca href=\"http://www.snowflake.com/en/blog/engineering/lakehouse-data-governance-interoperability\" target=\"_blank\" rel=\"noopener noreferrer\"\u003Egovernance deep dive\u003C/a\u003E; and metric and dimension definitions are covered below. What matters is the specific interoperability problem for each, not the umbrella term.\u003C/p\u003E\r\n\u003Cp\u003EOssie is the most significant initiative addressing semantic layer definition portability. \u003Ca href=\"https://www.snowflake.com/en/news/press-releases/snowflake-salesforce-dbt-labs-and-more-revolutionize-data-readiness-for-ai-with-open-semantic-interchange-initiative/\" target=\"_blank\" rel=\"noopener noreferrer\"\u003EWith contributions since its inception from Snowflake, alongside Salesforce, dbt Labs and others\u003C/a\u003E in 2025, the \u003Ca href=\"https://ossie.apache.org/#members\" target=\"_blank\" rel=\"noopener noreferrer\"\u003Eworking group\u003C/a\u003E has grown to include more than 50 organizations: Databricks, Cube, AtScale, Dremio, Starburst, Collibra, Oracle and many others have joined since founding. The breadth of participation signals genuine industry alignment, not a single-vendor initiative.\u003C/p\u003E\r\n\u003Cp\u003EThe \u003Ca href=\"https://github.com/apache/ossie/blob/main/core-spec/spec.md\" target=\"_blank\" rel=\"noopener noreferrer\"\u003Espec is live\u003C/a\u003E under Apache 2.0 licensing: a declarative YAML-based standard that defines semantic models, data sets, metrics, dimensions, relationships and AI context in a vendor-neutral format. Metric expressions support dialect annotations, so a metric can carry its calculation in ANSI SQL (or other dialects) alongside the semantic metadata. The initiative has a \u003Ca href=\"https://github.com/apache/ossie/blob/main/ROADMAP.md\" target=\"_blank\" rel=\"noopener noreferrer\"\u003Epublic roadmap\u003C/a\u003E with active working groups developing metric semantics (aggregation, grain, cumulative metrics), catalog integration (including an \u003Ca href=\"https://github.com/apache/ossie/pull/94\" target=\"_blank\" rel=\"noopener noreferrer\"\u003EApache Polaris converter\u003C/a\u003E) and ontology/conceptual interoperability. Future roadmap efforts cover a semantic query language and reference engine, and AI-native context for grounded query generation. Converters for Snowflake, dbt, Apache Polaris and Salesforce exist. Snowflake, for one, already offers bidirectional support: the \u003Ca href=\"https://docs.snowflake.com/en/sql-reference/stored-procedures/system_create_semantic_view_from_ossie_yaml\" target=\"_blank\" rel=\"noopener noreferrer\"\u003ESYSTEM$CREATE_SEMANTIC_VIEW_FROM_OSSIE_YAML\u003C/a\u003E stored procedure (in public preview at time of publication) builds a native semantic view from an Ossie model, and the \u003Ca href=\"https://docs.snowflake.com/en/sql-reference/functions/system_read_ossie_yaml_from_semantic_view\" target=\"_blank\" rel=\"noopener noreferrer\"\u003ESYSTEM$READ_OSSIE_YAML_FROM_SEMANTIC_VIEW\u003C/a\u003E function (public preview at publication time) exports an existing semantic view back to Ossie YAML, so a definition can round-trip between the open spec and a governed engine object. The initiative has already \u003Ca href=\"https://lists.apache.org/thread/fjkqj5s23pxx946sxbfqc7218mc9k4or\" target=\"_blank\" rel=\"noopener noreferrer\"\u003Etransitioned to foundation-led governance\u003C/a\u003E as an incubating project in the Apache Software Foundation.\u003C/p\u003E\r\n\u003Cp\u003EWhat Ossie covers directly addresses the CFO-gets-different-numbers scenario from the current state assessment: If &quot;recognized revenue&quot; is defined once in an Ossie model and consumed by every tool in the stack, the definition doesn't drift. That's the highest-value slice of business logic interoperability.\u003C/p\u003E\r\n\u003Cp\u003EWhat Ossie does not cover today: UDFs, stored procedures, procedural logic and transformation logic. Metrics are important, arguably the highest-value category, but they're one category. Organizations that adopt Ossie will have metric consistency across tools but still face the drift problem for everything else discussed in this post. The scope limitation is deliberate and appropriate for a first initiative; it's a step in the right direction that should be taken into account when planning your architecture.\u003C/p\u003E\r\n\u003Ch5\u003EPragmatic guidance\u003C/h5\u003E\r\n\u003Cp\u003EOssie defines the standard, but a standard still has to be implemented and governed somewhere, and that requirement holds, no matter whose tools you use: Metric and dimension definitions need to live as first-class, governed objects, defined once at the source, so an AI agent asked for &quot;ARR&quot; resolves against a single authoritative definition instead of grabbing whichever revenue field it lands on, and every consumer, BI tool or agent, inherits that one definition rather than reimplementing it. You can build this yourself, but it means standing up and maintaining a governed definition store that every engine and tool resolves against, and keeping it in sync as definitions change. If you'd rather not, Snowflake provides this capability out of the box through Horizon Context. Definitions live as native, governed semantic objects and because Snowflake semantic views are easily round-tripped to and from Ossie YAML (see stored procedure and function above), adopting it now doesn't wall you off from the open standard as it matures.\u003C/p\u003E\r\n\u003Cp\u003EFor organizations operating across multiple engines or prioritizing cross-vendor portability long-term, Apache Ossie (incubating) is the most promising path and worth tracking closely given the breadth of industry participation.\u003C/p\u003E\r\n\u003Cp\u003EIn the meantime, the pragmatic approach remains the same as the broader guidance in this post: Centralize authoritative metric definitions in a single engine or semantic layer, and treat other consumers as downstream, or physicalize the logic. dbt and SQLMesh can also help here by managing metric definitions as code and generating per-engine SQL from a single source, though they're limited to the transformations they manage and don't solve behavioral divergence. For organizations evaluating semantic layer platforms (Cube, AtScale and others in the Ossie working group), watch for native Ossie spec support as a signal of long-term portability.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"blog_title_1262794393":{"id":"blog-title-86eb276f09","propertiesId":"open-problems-and-active-work","type":"heading2","lines":["Open problems and active work"],":type":"snowflake-site/components/blog/blog-title"},"blog_text_copy_copy__1311716111":{"id":"blog-text-f84ded6282","text":"\u003Cp\u003EThe previous section covered what exists today and where workarounds are needed. This section covers the problems that don't have shipped solutions yet: Some have active community work with real momentum (transpilation, intermediate representations); some are recognized gaps with no clear path forward (stored procedures); and some are lower-priority known issues that will likely be the last to get addressed (the remaining long tail).\u003C/p\u003E\r\n\u003Ch3\u003ETranspilation and intermediate representations\u003C/h3\u003E\r\n\u003Cp\u003EThe long-term solution to SQL behavioral divergence is likely some form of canonical representation. Instead of maintaining N copies of the same logic in N dialects, store one version and adapt it at query time or build time. Two approaches are being pursued: syntactic transpilation (translate SQL text between dialects) and intermediate representations (store logic in an engine-agnostic format that gets compiled to engine-specific SQL).\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"image_copy":{"id":"image-b7f619a9e7","height":"776","src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--c5e4c313-3c91-446f-84b6-9b449d0363fe/business-logic-fig2.png?preferwebp=true&quality=85","alt":"Figure 2: A visual representation of the concept of the single intermediate representation along with adapters for engines.","lazyEnabled":true,"width":"1545","title":"Figure 2: A visual representation of the concept of the single intermediate representation along with adapters for engines.",":type":"snowflake-site/components/image"},"blog_text_copy_copy__198540594":{"id":"blog-text-2091c5e162","additionalClasses":"inline","text":"\u003Cp\u003E&nbsp;\u003C/p\u003E\r\n\u003Cp\u003E\u003Ca href=\"https://github.com/tobymao/sqlglot\" target=\"_blank\" rel=\"noopener noreferrer\"\u003Esqlglot\u003C/a\u003E is an open source Python SQL parser and transpiler with support for \u003Ca href=\"https://github.com/tobymao/sqlglot/blob/main/sqlglot/dialects/__init__.py\" target=\"_blank\" rel=\"noopener noreferrer\"\u003E30+ dialects\u003C/a\u003E. It handles function name differences, quoting conventions, type mappings and many syntax transformations, and it \u003Ca href=\"https://github.com/tobymao/sqlglot#unsupported-errors\" target=\"_blank\" rel=\"noopener noreferrer\"\u003Eemits warnings\u003C/a\u003E rather than silently producing wrong SQL when it can't translate cleanly. sqlglot goes beyond syntax in several important cases:\u003C/p\u003E\r\n\u003Cul\u003E\r\n\u003Cli\u003ETranspiling \u003Ccode\u003EORDER BY col ASC\u003C/code\u003E from Spark to DuckDB correctly adds an explicit \u003Ccode\u003ENULLS FIRST\u003C/code\u003E clause to preserve Spark's null ordering default\u003C/li\u003E\r\n\u003Cli\u003E\u003Ccode\u003ECAST('a' AS INT)\u003C/code\u003E from Spark (where invalid casts return null by default) becomes \u003Ccode\u003ETRY_CAST\u003C/code\u003E on DuckDB (where \u003Ccode\u003ECAST\u003C/code\u003E throws)\u003C/li\u003E\r\n\u003Cli\u003E\u003Ccode\u003ECONCAT\u003C/code\u003E calls get rewritten to preserve each engine's null propagation behavior\u003C/li\u003E\r\n\u003C/ul\u003E\r\n\u003Cp\u003EThese are genuine semantic translations, not just keyword swaps. That said, there are categories of divergence that no syntactic transpiler can address: decimal precision behavior (the same division on the same types produces different precision across engines), division-by-zero semantics (null vs. infinity vs. error) and implicit-type coercion rules that are embedded in engine runtimes rather than expressed in SQL text. For migrations, build-time SQL generation and reducing the manual work needed to stand up multi-dialect pipelines, sqlglot is a useful tool in that space. But it doesn't eliminate the need to test cross-engine equivalence for critical business logic. For high-stakes business logic, transpilation is a tool on the path to a good architecture, not a substitute for one. Authoritative definitions still need to live somewhere deliberate: a single engine or physicalized as tables.\u003C/p\u003E\r\n\u003Cp\u003ELinkedIn's \u003Ca href=\"https://github.com/linkedin/coral\" target=\"_blank\" rel=\"noopener noreferrer\"\u003ECoral\u003C/a\u003E takes a different approach. It is a Java-based SQL translation and rewrite engine that uses an intermediate representation (based on Apache Calcite's relational algebra) to translate between HiveQL, Spark SQL and Trino dialects. LinkedIn operates it in production across its data platform, demonstrating that IR-based translation works at large data volumes for a specific set of dialects, particularly when maintained by a large engineering organization. The limitation, however, is that the supported dialect set is narrow, and extending it requires deep familiarity with Calcite's framework. While Coral is open source, it's not a turnkey solution for most teams.\u003C/p\u003E\r\n\u003Cp\u003EThe \u003Ca href=\"https://substrait.io/\" target=\"_blank\" rel=\"noopener noreferrer\"\u003ESubstrait\u003C/a\u003E project is working toward a broader version of this concept: a cross-language specification for serializing relational algebra that any engine could produce or consume. If business logic were stored as Substrait plans rather than SQL text, dialect differences would be handled by engine-specific adapters rather than text translation. In practice, adoption has been uneven. \u003Ca href=\"https://substrait.io/community/powered_by/\" target=\"_blank\" rel=\"noopener noreferrer\"\u003EApache Arrow's Acero and DataFusion\u003C/a\u003E (as a producer and consumer), and Ibis (as a producer) have native Substrait support, but the major query engines that dominate Iceberg lakehouse deployments don't have native Substrait integration. Until the engines you actually use support it, Substrait is a promising architecture rather than a production dependency.\u003C/p\u003E\r\n\u003Cp\u003EWithin Iceberg itself, the recently merged expressions spec (Iceberg PR \u003Ca href=\"https://github.com/apache/iceberg/pull/16652\" target=\"_blank\" rel=\"noopener noreferrer\"\u003E#16652\u003C/a\u003E) is a narrower take on the same idea. Instead of raw SQL text, it represents expressions as standardized expression signatures. The structure is engine-agnostic, but the spec deliberately stops there. It doesn't say how the defined expressions actually behave, leaving that to each engine. It standardizes the shape of an expression, not its implementation behavior. That makes it foundational plumbing (scalar expressions, not full queries) that view and UDF definitions could eventually build on, but not a fix for the behavioral divergence described above.\u003C/p\u003E\r\n\u003Cp\u003EA fourth, semantic-layer-level variant is on Ossie's roadmap: a reference engine plus an Ossie-to-SQL compiler and a cross-implementation conformance suite, so a single semantic model definition could be compiled to each engine's dialect with consistent interpretation. It is scoped to Ossie-modeled semantic queries (not arbitrary SQL, views or UDFs) and is on the roadmap rather than shipped, but it applies the same canonical-representation idea at the semantic layer.\u003C/p\u003E\r\n\u003Cp\u003EThere's also an accountability problem that applies to all transpilation and IR approaches. In a single-engine world, when a query produces the wrong answer, there's one vendor to call. In a multi-engine world with a translation layer in between, a wrong result could be the fault of the source engine, the target engine or the translation layer. Debugging cross-engine result differences becomes a cross-vendor support problem: slow to resolve, ambiguous in ownership and costly when it surfaces in a downstream business decision. The \u003Ca href=\"http://www.snowflake.com/en/blog/engineering/lakehouse-data-governance-interoperability\" target=\"_blank\" rel=\"noopener noreferrer\"\u003Egovernance deep dive\u003C/a\u003E covers why this accountability gap is particularly consequential for security policies.\u003C/p\u003E\r\n\u003Ch5\u003EPragmatic guidance\u003C/h5\u003E\r\n\u003Cp\u003Esqlglot covers more ground than most teams expect: Use it for migrations, build-time SQL generation and as a first pass when standing up multi-dialect pipelines. For the divergences it handles (null ordering, type casting, null propagation in functions), it genuinely reduces the manual work. For the divergences it can't (precision, division-by-zero, implicit coercion), cross-engine regression tests on representative data are the backstop. Watch Substrait adoption across the specific engines in your architecture; it's a sound long-term concept but not yet a dependency you can rely on. For critical business logic, the architectural guidance from the current state assessment still applies: Centralize authoritative definitions in a single engine or physicalize the results rather than depending entirely on a translation layer.\u003C/p\u003E\r\n\u003Ch3\u003EStored procedures and procedural logic\u003C/h3\u003E\r\n\u003Cp\u003EUnlike views, UDFs and metrics, stored procedures don't have an Iceberg spec equivalent, a community proposal or even a widely discussed path forward. This isn't an oversight; it's a reflection of how deeply engine-coupled stored procedures are. Views and UDFs are at least SQL expressions that can be stored and potentially translated. Stored procedures are full programs: loops, conditionals, exception handling, transaction control, cursor management, dynamic SQL and calls to engine-specific system functions, all written in engine-proprietary procedural languages (PL/pgSQL, Snowflake Scripting, Spark's programmatic APIs). There's no shared abstraction because there's no shared surface area to abstract over.\u003C/p\u003E\r\n\u003Cp\u003EIn a single-engine world, stored procedures are where critical multi-step business workflows live: end-of-month financial close processes, data quality enforcement routines, complex ETL orchestration, conditional transformation logic that can't be expressed as a single SQL statement. In a multi-engine world, these are entirely engine-locked with no portability path. Transpilation tools like sqlglot operate on SQL expressions, not procedural control flow. Intermediate representations (IRs), like Substrait, model relational algebra, not imperative logic. Even if every other category of business logic described in this post achieved full cross-engine portability tomorrow, stored procedures would remain tied to the engine they were written for.\u003C/p\u003E\r\n\u003Cp\u003EStored procedures are also where complex governance logic sometimes lives: multi-step anonymization workflows, conditional access decisions that go beyond what a single-row filter expression can handle, audit logging routines, to name a few. When that logic is engine-locked, the compliance implications extend beyond analytics consistency into security posture; the \u003Ca href=\"http://www.snowflake.com/en/blog/engineering/lakehouse-data-governance-interoperability\" target=\"_blank\" rel=\"noopener noreferrer\"\u003Egovernance deep dive\u003C/a\u003E covers this in detail.\u003C/p\u003E\r\n\u003Ch5\u003EPragmatic guidance\u003C/h5\u003E\r\n\u003Cp\u003EAccept that stored procedures are engine-locked and plan accordingly. If a critical business workflow lives in a stored procedure, that's one of the strongest arguments for routing that workload through a single engine. Where possible, refactor procedural logic into declarative SQL (which has a portability path, even if an imperfect one) or externalize the orchestration layer to tools like Airflow, Dagster or Prefect, which coordinate workflow steps without depending on engine-specific procedural languages.\u003C/p\u003E\r\n\u003Ch3\u003EThe remaining long tail\u003C/h3\u003E\r\n\u003Cp\u003EEven if views, UDFs, metrics and stored procedures all had cross-engine solutions tomorrow, a set of lower-priority business logic categories would remain engine-specific. Triggers and event-driven logic have no portability mechanism and no community proposal addressing them. Scheduled tasks and job orchestration defined inside an engine (as opposed to external orchestrators like Airflow) are tied to that engine's scheduler semantics. Materialized view refresh logic and maintenance schedules are engine-specific by definition. Session state and context functions used in business logic (\u003Ccode\u003ECURRENT_USER()\u003C/code\u003E, \u003Ccode\u003ECURRENT_ROLE()\u003C/code\u003E, session variables) return engine-specific values with no cross-engine equivalence guarantee, which matters when those functions appear in access control policies or audit logic. None of these are typically the source of the &quot;two dashboards, different numbers&quot; problem, which is why they're lower priority than the categories covered earlier. But they are worth acknowledging: Any architecture assessment that only accounts for views, UDFs and metrics will undercount the total surface area of engine-coupled business logic.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"blog_title_126279439":{"id":"blog-title-142667de70","propertiesId":"diagnostic-questions","type":"heading2","lines":["Diagnostic questions"],":type":"snowflake-site/components/blog/blog-title"},"blog_text_copy_copy_":{"id":"blog-text-b8316890e2","text":"\u003Cp\u003EIf you're evaluating or already operating a multi-engine lakehouse, these are the questions worth asking before you're in too deep and discover the answers the hard way:\u003C/p\u003E\r\n\u003Col\u003E\r\n\u003Cli\u003E\u003Cb\u003EFor any business logic you can't physicalize as an output table\u003C/b\u003E (active users, recognized revenue, churn), how do you detect when definitions diverge across engines, and what's your reconciliation process?\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EWhich engine's business logic definition is the explicit, documented system of record\u003C/b\u003E for high-stakes metrics (finance, regulatory reporting, executive dashboards) — and is that decision enforced architecturally or just assumed?\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EWhen two engines produce different results from the same Iceberg table\u003C/b\u003E, what is your escalation path? If neither engine vendor can reproduce the issue independently and a translation layer sits in between, have you mapped that cross-vendor accountability chain before you're in production — and do you know who owns the debugging when an Iceberg view produces a wrong result due to dialect differences?\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EWhen you need to update a business logic definition\u003C/b\u003E (for example, how &quot;churn&quot; or &quot;MRR&quot; is calculated), what is your process for deploying that change consistently across all engines at the same time — and what's your rollback plan if one engine fails validation after another has already gone live?\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EHave you audited the SQL behavioral differences between your engines\u003C/b\u003E, including null ordering, type casting behavior (ANSI mode, TRY_CAST availability) and timestamp resolution?\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EHow does each engine in your architecture handle timestamp storage, timezone resolution and DST transitions\u003C/b\u003E — and have you specifically tested daily and weekly aggregations that cross a DST boundary on a global data set?\u003C/li\u003E\r\n\u003C/ol\u003E\r\n\u003Ch2\u003EConclusion\u003C/h2\u003E\r\n\u003Cp\u003EBusiness logic interoperability is the least mature of the three pillars, but it's not a dead end. Most query volume in a typical data platform hits tables directly, where data interoperability is sufficient; the gaps described in this post affect a smaller but disproportionately high-stakes slice of workloads. The building blocks are in progress: Iceberg views and UDFs provide catalog-level metadata containers, Apache Ossie (incubating) is converging on metric definition portability with broad industry participation, and transpilation tools cover a good amount of ground.\u003C/p\u003E\r\n\u003Cp\u003EThe problem is that none of these are sufficient on their own today, and the pragmatic path remains architectural: Physicalize high-stakes logic as tables; keep authoritative logic definitions in a single engine, where the cons of physicalization outweigh the pros; and treat cross-engine logic consistency as something you verify rather than assume. The data is portable, but the business logic isn't yet. Treat that as a known constraint, and architect around it deliberately.\u003C/p\u003E\r\n\u003Cp\u003ETo learn more, check out \u003Ca href=\"http://www.snowflake.com/en/blog/engineering/multi-engine-lakehouse-interoperability\" target=\"_blank\" rel=\"noopener noreferrer\"\u003EAn Architect's Guide to Multi-Engine Lakehouses: What's Solved and What Isn't\u003C/a\u003E, which is a brief summary of all three pillars, and the other deep investigations into \u003Ca href=\"http://www.snowflake.com/en/blog/engineering/lakehouse-data-interoperability-guide\" target=\"_blank\" rel=\"noopener noreferrer\"\u003Edata interoperability\u003C/a\u003E and \u003Ca href=\"http://www.snowflake.com/en/blog/engineering/lakehouse-data-governance-interoperability\" target=\"_blank\" rel=\"noopener noreferrer\"\u003Egovernance interoperability\u003C/a\u003E.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"markup_editor":{"id":"markup-editor-c8f5fb7856","title":"inline","cssContent":".inline:not(pre)\u003Ecode[class*=language-]{color:#272822;background:#f6f9fa}",":type":"snowflake-site/components/markup-editor","isGSAPEnabled":false}},":itemsOrder":["blog_text_copy","blog_title_copy","blog_text","blog_text_277633130_","blog_title","blog_text_277633130","image","blog_text_copy_copy__1799410961","code_snippet","blog_text_copy_copy","blog_title_1262794393","blog_text_copy_copy__1311716111","image_copy","blog_text_copy_copy__198540594","blog_title_126279439","blog_text_copy_copy_","markup_editor"],":type":"wcm/foundation/components/responsivegrid"},"responsivegrid_premium_content_banner":{"columnCount":12,"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","columnClassNames":{},"appliedCssClassNames":"snowflake-responsive-component-top-padding-medium",":items":{},":itemsOrder":[],":type":"wcm/foundation/components/responsivegrid"},"container_author_chip":{"layout":"RESPONSIVE_GRID","columnCount":12,"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","columnClassNames":{"author_chip":"aem-GridColumn aem-GridColumn--default--12"},"id":"container-f6647b0abf",":type":"snowflake-site/components/container",":items":{"author_chip":{"id":"author-chip-097a916612","title":{"id":"title","type":"heading2","lines":["Learn more about the authors"],":type":"snowflake-site/components/title-v2"},"authors":[{"authorImage":{"id":"image-26881491b0","height":"854","src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--d5dae03d-5733-42e0-813e-f92748305643/jason-hughes.jpg?preferwebp=true&quality=85","lazyEnabled":true,"width":"854",":type":"snowflake-site/components/image"},"authorCta":{"id":"button-07345160b8","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/blog/authors/jason-hughes/"},"linkTargetContentType":"DOCUMENT_LEARN",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_INTERNAL","text":"Jason Hughes"},"authorTitle":"Principal Data Platform Architect, Applied Field Engineering"},{"authorImage":{"id":"image-656465c8ad","height":"800","src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--85428841-cdd8-4cae-b3d0-4003575f4937/jim-lebonitte.png?preferwebp=true&quality=85","lazyEnabled":true,"width":"800",":type":"snowflake-site/components/image"},"authorCta":{"id":"button-66741ef508","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/blog/authors/jim-lebonitte/"},"linkTargetContentType":"DOCUMENT_LEARN",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_INTERNAL","text":"Jim Lebonitte"},"authorTitle":"Director, GTM Platform & Architecture AFE"}],":type":"snowflake-site/components/blog/author-chip"}},":itemsOrder":["author_chip"],"appliedCssClassNames":"snowflake-responsive-component-top-padding-medium"}},":itemsOrder":["container_hero","responsivegrid_content","responsivegrid_premium_content_banner","container_author_chip"]},"flexible_column_content_container_2":{"layout":"SIMPLE","id":"container-fdce2e191d",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"blog_table_of_content":{"id":"blog-table-of-content-3d7d24c173",":type":"snowflake-site/components/blog/blog-table-of-content","tableOfContents":[{"headingText":"Where we are today","level":"h2","anchorId":"#where-we-are-today","hierarchicalChildrenStructure":[]},{"headingText":"What exists but doesn't fully solve the business logic problem","level":"h2","anchorId":"#what-exists-but-doesnt-fully-solve-the-business-logic-problem","hierarchicalChildrenStructure":[]},{"headingText":"Open problems and active work","level":"h2","anchorId":"#open-problems-and-active-work","hierarchicalChildrenStructure":[]},{"headingText":"Diagnostic questions","level":"h2","anchorId":"#diagnostic-questions","hierarchicalChildrenStructure":[]}]}},":itemsOrder":["blog_table_of_content"]},":type":"snowflake-site/components/flexible-column-container","isBlogPage":true,"isActiveTOC":true},"related_content":{"id":"related-content-3b5392e27a","relatedContent":[],":type":"snowflake-site/components/blog/related-content","isBlogPage":true}},":itemsOrder":["flexible_column_container","related_content"],"appliedCssClassNames":"snowflake-container"}},":itemsOrder":["container_breadcrumb","container_main_content"],":type":"wcm/foundation/components/responsivegrid"},"container_47873732":{"additionalClasses":"section--blog-newsletter","layout":"RESPONSIVE_GRID","columnCount":12,"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","columnClassNames":{"flexible_column_cont":"aem-GridColumn aem-GridColumn--default--12"},"id":"container-c899b1d69f",":type":"snowflake-site/components/container",":items":{"flexible_column_cont":{"id":"flexible-column-container-bc5417dece","type":"1-column","alignColumns":"top","containerMaxWidth":"extra-large","topPadding":"small","bottomPadding":"none","spaceBetween":"small","reverseOnMobile":false,"carouselOnMobile":false,"propertiesCSSClasses":"section--blog-newsletter","backgroundImageOption":"none","flexible_column_content_container_1":{"layout":"SIMPLE","id":"container-6ae6c8c1a1",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"marketo_v2":{"id":"marketo-v2-fb0f27ef31","marketoForm":{"hidden":null,"formId":"3320","edit":false,"successUrl":null,"script":null,"values":null},"title":{"id":"title","type":"heading3","lines":["Subscribe to our blog newsletter","Get the best, coolest and latest delivered to your inbox each week"],":type":"snowflake-site/components/title-v2"},"munchkinId":"252-RFO-227","serverInstance":"252-RFO-227.mktoweb.com","marketoConfigured":true,"formConfigured":true,":type":"snowflake-site/components/form/marketo-v2"},"text":{"id":"text-fac62baaad","additionalClasses":"newsletter-disclaimer","text":"\u003Cp\u003EBy submitting this form, I understand Snowflake will process my personal information in accordance with their Privacy Notice.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/text"}},":itemsOrder":["marketo_v2","text"]},":type":"snowflake-site/components/flexible-column-container","isBlogPage":true,"isActiveTOC":true}},":itemsOrder":["flexible_column_cont"],"appliedCssClassNames":"snowflake-container"},"experiencefragment-pre-footer":{"id":"experiencefragment-534be7208d","localizedFragmentVariationPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/get-started-pre-footer/get-started-pre-footer/jcr:content","configured":true,":type":"snowflake-site/components/experiencefragment","xfModelPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/get-started-pre-footer/get-started-pre-footer.xfmodel.json"},"markup_editor":{"id":"markup-editor-a2428317d3","title":"Page CSS","cssContent":"@media screen and (min-width:768px){.snowflake-blog-author-chip-wrapper{justify-content:flex-start}.snowflake-blog-related-content-on-blog-page{max-width:1408px;margin-left:auto;margin-right:auto}.snowflake-text{font-family:Lato,sans-serif;font-weight:400;font-size:16px;line-height:24px}}.section--blog-newsletter{max-width:none;width:100%;padding-left:0;padding-right:0;margin-left:0;margin-right:0;margin-bottom:0}.section--blog-newsletter .mktoField{background-color:transparent !important}.section--blog-newsletter\u003E.container{padding-left:0;padding-right:0}@media screen and (min-width:768px){.section--blog-newsletter\u003E.container{padding-left:0;padding-right:0}}.newsletter-disclaimer p{font-size:14px !important}.section--blog-newsletter .snowflake-marketo-form-container{margin-bottom:24px;background-color:#f6f9fa;gap:48px;box-shadow:none}.section--blog-newsletter .snowflake-title p.snowflake-title-line:first-child{font-family:Texta;font-size:24px;line-height:26px;font-weight:700;margin-bottom:4px}.section--blog-newsletter .snowflake-title p.snowflake-title-line{text-transform:none;font-family:\"Lato\",sans-serif;font-size:16px;line-height:24px;font-weight:normal}@media screen and (min-width:1024px){.section--blog-newsletter .snowflake-marketo-form-container{display:flex;justify-content:center}.section--blog-newsletter .snowflake-title .snowflake-title-line{text-align:left}.section--blog-newsletter .snowflake-marketo-form .mktoFormRow:has(\u003E input[type=\"hidden\"]){flex-grow:0}.section--blog-newsletter .snowflake-marketo-form{display:flex;width:50% !important}.section--blog-newsletter .snowflake-marketo-form .mktoButtonRow{flex-grow:0;width:auto !important;margin-left:0;margin-right:0}.section--blog-newsletter .snowflake-marketo-form .mktoFormRow{flex-grow:1}.section--blog-newsletter\u003E.container{padding-left:0;padding-right:0}.section--blog-newsletter .snowflake-marketo-form-title{width:50%;margin-bottom:0 !important}.section--blog-newsletter .center .snowflake-title{align-items:flex-start}}.snowflake-sub-navigation a.snowflake-sub-navigation-primary-link{width:auto !important}.snowflake-blog-hero{align-items:stretch !important}",":type":"snowflake-site/components/markup-editor","isGSAPEnabled":false},"markup_editor-table":{"id":"markup-editor-c8b21b140d","title":"Table Styling CSS","cssContent":"#snowflake-blog-template-main-container table{width:100%;background-color:var(--ui-background-01);border-collapse:collapse;border:2px solid var(--ui-background-09);font-family:'Lato',sans-serif;color:var(--ui-background-09)}#snowflake-blog-template-main-container table thead{background-color:var(--ui-01)}#snowflake-blog-template-main-container table th,#snowflake-blog-template-main-container table td{border:2px solid var(--ui-background-09);padding:var(--spacing-01)}",":type":"snowflake-site/components/markup-editor","isGSAPEnabled":false},"experiencefragment-footer":{"id":"experiencefragment-04729138a2","localizedFragmentVariationPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/footer/master/jcr:content","configured":true,":type":"snowflake-site/components/experiencefragment","xfModelPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/footer/master.xfmodel.json"}},":itemsOrder":["experiencefragment-banner","experiencefragment-header","experiencefragment-sub-header","responsivegrid","container_47873732","experiencefragment-pre-footer","markup_editor","markup_editor-table","experiencefragment-footer"],":type":"wcm/foundation/components/responsivegrid"}},":itemsOrder":["root"],":hierarchyType":"page",":path":"/content/snowflake-site/global/en/blog/engineering/lakehouse-business-logic-interoperability","analyticsContentTags":["snowflake-site:taxonomy/blog/engineering-blog/data-engineering"],"analyticsEnabled":true,"coveoConfig":{"pipeline":"snowflake.com","apiKey":"xx335921a6-2a0a-40f2-a167-e390b4766c3d","organizationId":"snowflakecomputingproduction8neljofn","searchHub":"snowflake.com"},"analyticsDebugMode":false,"analyticsData":{"excludeFromAnalytics":false,"subCategory":"","pageType":"homepage","templateName":"blog-page","siteName":"snowflake","pageUrl":"/content/snowflake-site/global/en/blog/engineering/lakehouse-business-logic-interoperability","language":"en","category":"general","pageName":"Managing Business Logic Interoperability in Multi-Engine Lakehouses","contentTags":["snowflake-site:taxonomy/blog/engineering-blog/data-engineering"]},"isPasswordProtected":false,"locale":"en"}
  