{"templateName":"blog-page","cssClassNames":"blog-page page basicpage summit-page","allowedRenditionsWidth":["320","480","640","768","960","1200","1440","1920"],"description":"Discover how to achieve data interoperability in a multi-engine lakehouse. Learn about Apache Iceberg v3, REST Catalog compliance, and best practices.","language":"en","title":"Operationalizing Data Interoperability in Multi-Engine Lakehouses","analyticsPageType":"homepage","analyticsCategory":"general","analyticsSubCategory":"","excludeFromAnalytics":false,":items":{"root":{"columnCount":12,"columnClassNames":{"experiencefragment-banner":"aem-GridColumn aem-GridColumn--default--12","experiencefragment-sub-header":"aem-GridColumn aem-GridColumn--default--12","experiencefragment-pre-footer":"aem-GridColumn aem-GridColumn--default--12","experiencefragment-header":"aem-GridColumn aem-GridColumn--default--12","markup_editor-table":"aem-GridColumn aem-GridColumn--default--12","responsivegrid":"aem-GridColumn aem-GridColumn--default--12","experiencefragment-footer":"aem-GridColumn aem-GridColumn--default--12","markup_editor":"aem-GridColumn aem-GridColumn--default--12","container_47873732":"aem-GridColumn aem-GridColumn--default--12"},"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12",":items":{"experiencefragment-banner":{"id":"experiencefragment-7c0eda7a9b","localizedFragmentVariationPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/pushdown-banner/pushdown-banner-blank/jcr:content","configured":true,":type":"snowflake-site/components/experiencefragment","xfModelPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/pushdown-banner/pushdown-banner-blank.xfmodel.json?callerPage=/content/snowflake-site/global/en/blog/engineering/lakehouse-data-interoperability-guide"},"experiencefragment-header":{"id":"experiencefragment-e1f95c6a8b","localizedFragmentVariationPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/mega-nav-header/master/jcr:content","configured":true,":type":"snowflake-site/components/experiencefragment","xfModelPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/mega-nav-header/master.xfmodel.json?callerPage=/content/snowflake-site/global/en/blog/engineering/lakehouse-data-interoperability-guide","appliedCssClassNames":"snowflake-sticky-nav-host"},"experiencefragment-sub-header":{"id":"experiencefragment-6692b60cde","localizedFragmentVariationPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/sub-navigation/engineering-blog-sub-nav/jcr:content","configured":true,":type":"snowflake-site/components/experiencefragment","xfModelPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/sub-navigation/engineering-blog-sub-nav.xfmodel.json?callerPage=/content/snowflake-site/global/en/blog/engineering/lakehouse-data-interoperability-guide"},"responsivegrid":{"columnCount":12,"columnClassNames":{"container_breadcrumb":"aem-GridColumn aem-GridColumn--default--12","container_main_content":"aem-GridColumn aem-GridColumn--default--12"},"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12",":items":{"container_breadcrumb":{"layout":"RESPONSIVE_GRID","columnCount":12,"columnClassNames":{"breadcrumb":"aem-GridColumn aem-GridColumn--default--12"},"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","id":"blog-page-breadcrumb-indentation",":items":{"breadcrumb":{"id":"breadcrumb-771aa7cb2f","breadcrumbItems":[{"title":"Blog","path":"/en/blog/engineering/","active":false},{"title":"Data Engineering","path":"/en/blog/engineering/data-engineering/","active":false},{"title":"Operationalizing Data Interoperability in Multi-Engine Lakehouses","path":"/en/blog/engineering/lakehouse-data-interoperability-guide/","active":false}],":type":"snowflake-site/components/blog/breadcrumb"}},":itemsOrder":["breadcrumb"],"appliedCssClassNames":"snowflake-container",":type":"snowflake-site/components/container"},"container_main_content":{"layout":"RESPONSIVE_GRID","columnCount":12,"columnClassNames":{"flexible_column_container":"aem-GridColumn aem-GridColumn--default--12","related_content":"aem-GridColumn aem-GridColumn--default--12"},"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","id":"main-content",":items":{"flexible_column_container":{"id":"flexible-column-container-436d5705a3","propertiesId":"snowflake-blog-template-main-container","type":"2-column-60-40","alignColumns":"top","containerMaxWidth":"extra-large","topPadding":"none","bottomPadding":"none","spaceBetween":"none","reverseOnMobile":true,"carouselOnMobile":false,"backgroundImageOption":"none","flexible_column_content_container_1":{"layout":"SIMPLE","id":"container-346a9f2e30",":items":{"container_hero":{"layout":"RESPONSIVE_GRID","columnCount":12,"columnClassNames":{"blog_hero":"aem-GridColumn aem-GridColumn--default--12"},"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","id":"container-5ce365f46d",":items":{"blog_hero":{"id":"blog-hero-b42b3da55f","linkedInShareUrl":"https://www.linkedin.com/shareArticle?mini=true&url=https%3A%2F%2Fwww.snowflake.com%2Fcontent%2Fsnowflake-site%2Fglobal%2Fen%2Fblog%2Fengineering%2Flakehouse-data-interoperability-guide&title=Operationalizing+Data+Interoperability+in+Multi-Engine+Lakehouses","twitterShareUrl":"https://x.com/intent/post?url=https%3A%2F%2Fwww.snowflake.com%2Fcontent%2Fsnowflake-site%2Fglobal%2Fen%2Fblog%2Fengineering%2Flakehouse-data-interoperability-guide&text=Operationalizing+Data+Interoperability+in+Multi-Engine+Lakehouses","facebookShareUrl":"https://www.facebook.com/sharer/sharer.php?u=https%3A%2F%2Fwww.snowflake.com%2Fcontent%2Fsnowflake-site%2Fglobal%2Fen%2Fblog%2Fengineering%2Flakehouse-data-interoperability-guide","showClaude":true,"showChatGpt":true,"authors":[{"authorImage":{"id":"image-8e36e5b7c2","height":"854","src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--d5dae03d-5733-42e0-813e-f92748305643/jason-hughes.jpg?quality=85&preferwebp=true","lazyEnabled":true,"width":"854",":type":"snowflake-site/components/image"},"authorCta":{"id":"button-fb8fa794b5","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/blog/authors/jason-hughes/"},"linkTargetContentType":"DOCUMENT_LEARN",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_INTERNAL","text":"Jason Hughes"}},{"authorImage":{"id":"image-b92d9f0cb4","height":"800","src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--85428841-cdd8-4cae-b3d0-4003575f4937/jim-lebonitte.png?quality=85&preferwebp=true","lazyEnabled":true,"width":"800",":type":"snowflake-site/components/image"},"authorCta":{"id":"button-c30620369e","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/blog/authors/jim-lebonitte/"},"linkTargetContentType":"DOCUMENT_LEARN",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_INTERNAL","text":"Jim Lebonitte"}}],"image":{"id":"image-36139a92e7","height":"720","src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--cc092a50-6db2-40a7-b106-c4713a0b5246/sf-eng-blog-ml-1.png?quality=85&preferwebp=true","lazyEnabled":true,"width":"1680",":type":"snowflake-site/components/image"},"timeToRead":"40","publicationDate":"JUL 28, 2026","tag":{"tagText":"Data Engineering","tagColor":"#29B5E8"},"title":{"lines":["Operationalizing Data Interoperability in Multi-Engine Lakehouses"],"type":"heading2",":type":"snowflake-site/components/title-v2"},":type":"snowflake-site/components/blog/blog-hero"}},":itemsOrder":["blog_hero"],":type":"snowflake-site/components/container"},"responsivegrid_content":{"columnCount":12,"columnClassNames":{"image_copy":"aem-GridColumn aem-GridColumn--default--12","image":"aem-GridColumn aem-GridColumn--default--12","image_copy_copy":"aem-GridColumn aem-GridColumn--default--12","blog_title_copy":"aem-GridColumn aem-GridColumn--default--12","blog_text_copy_447586789":"aem-GridColumn aem-GridColumn--default--12","blog_text_814893266":"aem-GridColumn aem-GridColumn--default--12","image_copy_701714592":"aem-GridColumn aem-GridColumn--default--12","blog_text_copy_905954621":"aem-GridColumn aem-GridColumn--default--12","blog_text_814893266_":"aem-GridColumn aem-GridColumn--default--12","blog_title":"aem-GridColumn aem-GridColumn--default--12","blog_title_copy_copy":"aem-GridColumn aem-GridColumn--default--12","blog_text_copy_628289373":"aem-GridColumn aem-GridColumn--default--12","blog_text_copy":"aem-GridColumn aem-GridColumn--default--12","blog_text":"aem-GridColumn aem-GridColumn--default--12","blog_text_copy_1844728901":"aem-GridColumn aem-GridColumn--default--12","blog_title_copy_copy_848238316":"aem-GridColumn aem-GridColumn--default--12","blog_text_copy_1511145138":"aem-GridColumn aem-GridColumn--default--12","blog_text_copy_1759338821":"aem-GridColumn aem-GridColumn--default--12","blog_title_copy_copy_280634654":"aem-GridColumn aem-GridColumn--default--12"},"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","appliedCssClassNames":"snowflake-layout-container-inner-padding-small",":items":{"blog_text_814893266_":{"id":"blog-text-44f3d909dc","text":"\u003Cp\u003EIn discussions about multi-engine data architectures, the question that comes up most often is also the simplest: Can we have one copy of data that multiple engines efficiently and reliably read and write to without copying? For the first time in the long history of data architecture, the honest answer is &quot;mostly yes.&quot; Apache Iceberg™ and the REST Catalog spec, along with their broad adoption, solved something that was genuinely hard five years ago. Most major engines and platforms participate as good citizens in this data architecture today.\u003C/p\u003E\r\n\u003Cp\u003EBut the more interesting question (and the one most architecture reviews skip) is: What does &quot;mostly yes&quot; mean, practically, when you have a specific set of engines, catalogs and workloads in production? The structural problem is solved, but some operational problems remain. Making data interoperability reliable at scale, with decentralized write pipelines and AI/ML workloads in the mix, requires (1) a concrete understanding of where the industry currently stands, (2) an assessment of trade-offs in the architectural assessment phase and (3) making decisions based on the current state and trade-offs. That's what this post covers.\u003C/p\u003E\r\n\u003Cp\u003EIf you haven't already, read the summary post, \u003Ca href=\"http://www.snowflake.com/en/blog/engineering/multi-engine-lakehouse-interoperability\" target=\"_blank\" rel=\"noopener noreferrer\"\u003EAn Architect's Guide to Interoperability\u003C/a\u003E, which offers a brief summary of all three interoperability dimensions (data, governance and business logic interoperability) and starts to identify the gaps that exist around effective multi-engine lakehouse architectures. This post, in particular, looks comprehensively at data interoperability. It explores the concepts and guidance introduced in the summary post, plus a full analysis of workarounds and a status check on the ecosystem's progress toward that ideal vision.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"blog_title":{"id":"blog-title-8907089d29","propertiesId":"current-state-assessment","type":"heading2","lines":["Current state assessment"],":type":"snowflake-site/components/blog/blog-title"},"blog_text_814893266":{"id":"blog-text-3731b85f74","text":"\u003Cp\u003EOf the three interoperability dimensions this series covers, data is the furthest along — and to a significant degree. Apache Iceberg, paired with the \u003Cb\u003EIceberg REST Catalog (IRC)\u003C/b\u003E as the access layer, has fundamentally changed how we approach data architecture. Now, you have the ability to store data once and have multiple engines read it from a shared catalog. That's meaningful and wasn't broadly true in practice five years ago.\u003C/p\u003E\r\n\u003Cp\u003EThe benefits of this community-led achievement are concrete:\u003C/p\u003E\r\n\u003Cul\u003E\r\n\u003Cli\u003E\u003Cb\u003ENo more data duplication:\u003C/b\u003E No data drift, no copy pipelines, no redundant storage costs.\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EFreedom to use specialized engines for different workloads:\u003C/b\u003E Flexibility to meet one's needs, rather than forcing everything through one tool.\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EInsurance against the future:\u003C/b\u003E When your data is in a widely adopted open format, new engines, vendor acquisitions, pricing changes and better tools all become compute decisions, not data migrations.\u003C/li\u003E\r\n\u003C/ul\u003E\r\n\u003Cp\u003EMost major engines can participate today via the REST Catalog spec. That breadth of interoperability didn't exist at this scale before Iceberg.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"image_copy_copy":{"id":"image-b866cd37ab","height":"1175","src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--7bf47a50-b7aa-42bd-9215-add2b6d0b1b7/data-fig1.png?quality=85&preferwebp=true","alt":"Figure 1: From duplicated data per platform (top) to a single shared data layer that multiple platforms read from via Iceberg (bottom).","lazyEnabled":true,"width":"1570","title":"Figure 1: From duplicated data per platform (top) to a single shared data layer that multiple platforms read from via Iceberg (bottom).",":type":"snowflake-site/components/image"},"blog_text_copy":{"id":"blog-text-ecc418cb0e","text":"\u003Cp\u003E&nbsp;\u003C/p\u003E\r\n\u003Cp\u003EThat said, there are some important caveats worth understanding and considering. For starters, engine support for Iceberg features is uneven. Spec versions are named bundles of capabilities — a way of grouping features and managing breaking changes across the ecosystem — but support isn't always similarly bundled. Not every vendor that claims v2 support, for instance, actually supports every feature in v2 (for example, equality deletes, nan_value_counts in statistics, the data types fixed(L) and uuid), and v3 is likewise not all-or-nothing across the ecosystem. Sometimes this matters, sometimes it doesn't. Performance consistency requires active engineering discipline because Iceberg doesn't enforce file hygiene; decentralized write pipelines will produce fragmented, inconsistently configured tables that degrade quietly at scale. Disaster recovery has possible solutions today but no clean standardized path until the community ships the relative-paths proposal for v4.\u003C/p\u003E\r\n\u003Cp\u003EThere are also genuinely unsolved problems. AI/ML workload requirements (for example, wide tables, vector similarity search and point lookups) are not well-served by the current state of the Iceberg plus Parquet combination. The community knows this; proposals are in flight, but the format landscape (Lance, specialized vector databases, improvements to Parquet for wide columns) is still not settled.\u003C/p\u003E\r\n\u003Cp\u003EFor standard analytics workloads where you control the write side, multi-engine data access is achievable today with engineering discipline. For AI/ML-heavy workloads, specialized access patterns or decentralized write environments, the gaps are meaningful enough to factor into architecture decisions before you're six months in.\u003C/p\u003E\r\n\u003Cp\u003ELet's dive into a detailed assessment and practical guide.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"blog_title_copy":{"id":"blog-title-487c0e63fe","propertiesId":"what-works-well-today","type":"heading2","lines":["What works well today"],":type":"snowflake-site/components/blog/blog-title"},"blog_text_copy_1759338821":{"id":"blog-text-a461d68145","text":"\u003Ch3\u003EThe REST Catalog foundation\u003C/h3\u003E\r\n\u003Cp\u003EBefore the Iceberg REST Catalog spec, multi-engine data access required a separate integration for every engine-catalog pair. Every engine that wanted to read or write to Iceberg tables needed its own integration with every catalog and storage backend it might encounter (for example, Glue, Hive Metastore and Nessie), each with its own authentication model and catalog Java Archive (JAR) file. For a data platform running three engines against two storage backends, that's six integrations to build, maintain and debug independently.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"image_copy":{"id":"image-c36c12ef25","height":"1611","src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--a4f91db8-d9f6-4010-a9f5-ac500c152e07/data-fig2.png?quality=85&preferwebp=true","alt":"Figure 2: Before the introduction of the Iceberg REST Catalog, connecting five engines and five storage backends created 25 integration points, whereas with IRC, there are only 10.","lazyEnabled":true,"width":"3258","title":"Figure 2: Before the introduction of the Iceberg REST Catalog, connecting five engines and five storage backends created 25 integration points, whereas with IRC, there are only 10.",":type":"snowflake-site/components/image"},"blog_text_copy_628289373":{"id":"blog-text-cade0a6e53","text":"\u003Cp\u003E&nbsp;\u003C/p\u003E\r\n\u003Cp\u003E\u003Cb\u003EThe Iceberg REST Catalog (IRC) APIs collapse that integration space\u003C/b\u003E. Each engine integrates once with the REST Catalog API: a standardized set of endpoints and request-and-response payloads for catalog operations such as listing namespaces, loading table metadata and committing updates. Each catalog implementation exposes the same API surface. An engine that &quot;speaks IRC&quot; can reach any conformant catalog without modification, and any conformant catalog can serve any IRC-compliant engine. However, the Amazon S3 compatibility analogy is useful here: Storage providers frequently claim &quot;S3-compatible&quot; APIs without supporting every S3 endpoint, and the same dynamic applies to IRC. The spec has many endpoints, with new ones added as the standard evolves, and support for each varies across engines and catalogs. Over time, however, we believe the ecosystem should converge, and support will be more thorough, but if you expect to rely on a specific IRC capability (particularly relevant for governance APIs, which we cover \u003Ca rel=\"noopener noreferrer\" target=\"_blank\" href=\"http://www.snowflake.com/en/blog/engineering/lakehouse-data-governance-interoperability\"\u003Ehere\u003C/a\u003E), it's important to verify support timelines across your engines and catalogs before committing to that dependency.\u003C/p\u003E\r\n\u003Cp\u003EApache Polaris, the open source catalog project that Snowflake co-founded, donated to the Apache Software Foundation (ASF) and is now a \u003Ca rel=\"noopener noreferrer\" target=\"_blank\" href=\"https://www.snowflake.com/en/blog/apache-polaris-top-level-project/\"\u003ETop-Level Project at the ASF\u003C/a\u003E, implements the IRC spec, as do other major catalog implementations. Most major compute engines (for example, Apache Spark, Apache Flink, Trino, Snowflake, DuckDB and others) have IRC clients. Because Iceberg decouples data from any specific compute engine and IRC standardizes the interface between engines and catalogs, you can add or replace engines without touching the data layer or reconfiguring storage. Part of what makes this work cleanly is IRC's \u003Cb\u003Ecredential vending model\u003C/b\u003E: The catalog issues short-lived, scoped storage tokens to engines on demand, so engines never hold long-lived storage access independently. This foundation is what makes &quot;structural interoperability&quot; a fair characterization, not just an empty marketing claim. The plumbing to connect any conformant engine to any conformant catalog genuinely exists today. The caveat is that &quot;conformant&quot; is doing real work in that sentence. IRC compliance across the ecosystem is uneven in ways that matter, which we'll cover in a subsequent section in this deep dive.\u003C/p\u003E\r\n\u003Ch3\u003EData type coverage\u003C/h3\u003E\r\n\u003Cp\u003EFor standard analytics workloads, Iceberg's \u003Ca rel=\"noopener noreferrer\" target=\"_blank\" href=\"https://iceberg.apache.org/spec/#schemas-and-data-types\"\u003Etype system\u003C/a\u003E has been largely comprehensive since v2. The full set of primitive and nested types covers the vast majority of relational analytics use cases.\u003C/p\u003E\r\n\u003Cp\u003E\u003Ca rel=\"noopener noreferrer\" target=\"_blank\" href=\"https://www.snowflake.com/en/blog/apache-iceberg-v3-table-spec-oss-shared-success/\"\u003EIceberg v3\u003C/a\u003E closed the most significant remaining gaps. The headline addition was a native, efficient \u003Cb\u003E\u003Ccode\u003EVARIANT\u003C/code\u003E data type\u003C/b\u003E for schema-variable semi-structured data that Snowflake led and heavily contributed to. Semi-structured and nested structures can now be stored and queried natively without shoehorning them into a string column, having to define a strict schema or preshredding to a fixed schema at write time. v3 also added nanosecond-precision timestamps, geography and geometry types for spatial workloads, along with \u003Cb\u003Edeletion vectors\u003C/b\u003E, a more efficient mechanism for row-level deletes. Row-lineage support rounds out the main additions, enabling each row to carry a stable identifier across updates, greatly improving CDC use cases and efficiency. For workloads beyond standard analytics (for example, AI/ML embeddings, high-dimensional search and unstructured data) the picture is different, and covered in the sections ahead.\u003C/p\u003E\r\n\u003Ch3\u003ESchema evolution and type promotion\u003C/h3\u003E\r\n\u003Cp\u003EIceberg brings \u003Ca rel=\"noopener noreferrer\" target=\"_blank\" href=\"https://iceberg.apache.org/docs/latest/evolution/\"\u003Emetadata-only schema evolution\u003C/a\u003E to the data lake: Adding, renaming, dropping or reordering columns requires no data rewrite. In a multi-engine environment, this reduces the coordination problem that would emerge when every team had to synchronize pipeline cutovers around almost every schema change. Schema changes are cheap; the risk of corrupting data during migration is eliminated; and downstream consumers aren't racing against a data rewrite in progress.\u003C/p\u003E\r\n\u003Cp\u003EThe second concrete benefit is that Iceberg brings \u003Cb\u003Edefined, enforced type promotion rules\u003C/b\u003E to the data lake, something that didn't exist before. In a raw object-storage data lake, how a type change was handled depended entirely on which engine touched the data next, and different engines had different opinions. Iceberg specifies exactly which type changes are safe (for example, int to long, float to double and decimal precision widening) and enforces them at the catalog level, regardless of which engine is writing. Incompatible changes are rejected outright rather than silently producing wrong results downstream. In a multi-engine environment where multiple teams are writing to shared tables with different tools, this consistency matters.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"blog_title_copy_copy":{"id":"blog-title-5b9192d6a2","propertiesId":"what-requires-work-but-has-solutions","type":"heading2","lines":["What requires work but has solutions"],":type":"snowflake-site/components/blog/blog-title"},"blog_text_copy_905954621":{"id":"blog-text-67762767c7","text":"\u003Ch3\u003EFormat and engine compatibility\u003C/h3\u003E\r\n\u003Cp\u003EThe S3 compatibility dynamic from the REST Catalog section applies here too: &quot;Supports Iceberg v2&quot; or &quot;supports Iceberg v3&quot; on a release page rarely means what you'd expect. Spec version, feature completeness within that version and catalog compliance all vary independently across engines, and the gaps surface at query time rather than setup time. This piece explores v2 gaps, v3 support and how to plan for and mitigate inconsistent support across engines.\u003C/p\u003E\r\n\u003Ch4\u003EIceberg v2 spec gaps that persist\u003C/h4\u003E\r\n\u003Cp\u003EIceberg spec compliance is feature-by-feature, not all-or-nothing. Some v2 features still have meaningful gaps across engines today. Equality deletes are one concrete example: They're part of the v2 spec, but many engines still don't support writing them, and some don't support reading them. The reason read-side support is limited is the computational overhead. Evaluating equality delete files can require a join-like operation at scan time against every data file that is part of the table at that time, which is expensive at scale. However, in the situation of high-throughput streaming CDC writes, equality deletes are very efficient for the writer because they don't need to do any reading of any data, which is why they exist in the first place. The broader community is now moving toward \u003Cb\u003Edeletion vectors\u003C/b\u003E (added in v3) and eventual index support targeted for v4 as the preferred path for streaming workload row-level deletes. However, this plan may have some problems since the indexes are currently planned to be asynchronously updated, not as part of the write commit, which means that the indexes can't be relied on to be an up-to-date picture of the table state. There have been discussions around making index updates synchronous (that is, as part of the write commit to the table, you'll also update the index structure, all as one atomic commit), but there hasn't been an appetite to go that route, at time of writing.\u003C/p\u003E\r\n\u003Ch5\u003EPragmatic guidance\u003C/h5\u003E\r\n\u003Cul\u003E\r\n\u003Cli\u003ETest equality delete support (both read and write) explicitly with your specific engine versions before relying on it in production. Also, assess whether your write engines can be configured to avoid producing equality deletes in the first place. For example, Flink currently writes equality deletes in CDC mode, but not in append mode, depending on configuration. If a write engine can't be configured away from equality deletes and your read engines don't handle them well, that's a signal to reconsider the write engine choice. If switching engines isn't an option, there are a few workaround patterns that let you keep both engines while isolating the problem:\u003Cul\u003E\r\n\u003Cli\u003E\u003Cb\u003EChangelog + MERGE INTO\u003C/b\u003E: Have Flink append to a changelog table (not in CDC mode), then periodically run \u003Ccode\u003EMERGE INTO\u003C/code\u003E via Snowflake or Spark to apply changes to a separate consumption table. The consumption table never has equality deletes.\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EStaging + MERGE INTO via Spark\u003C/b\u003E: Have Flink write in CDC mode to a staging table, then periodically run \u003Ccode\u003EMERGE INTO\u003C/code\u003E via Spark to apply changes to a consumption table that your read engine points at.\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EPeriodic rewrite + pause\u003C/b\u003E: Pause Flink writing, run \u003Ccode\u003Erewrite_data_files\u003C/code\u003E in Spark with a delete file threshold of 1 to compact away equality deletes, then refresh the Iceberg table in your read engine and unpause Flink. Simple but requires a write pause window.\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EInfrequent Flink commits + timed rewrite\u003C/b\u003E: Configure Flink to commit infrequently (for example, every 30-60 minutes), and run \u003Ccode\u003Erewrite_data_files\u003C/code\u003E in Spark with \u003Ccode\u003Euse-starting-sequence-number=false\u003C/code\u003E and a delete file threshold of 1 between commits, then refresh your read engine before the next Flink commit lands. This avoids a pause window but introduces unpredictable data freshness; the rewrite may not always complete in time.\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003ETwo-branch continuous conversion with PK index\u003C/b\u003E: Use Iceberg's branch support to separate write and read workloads entirely. Flink writes equality deletes to a \u003Ccode\u003Ereal-time\u003C/code\u003E branch; a background process continuously converts them to positional deletes via PK index lookup and commits clean snapshots to a separate \u003Ccode\u003Emain\u003C/code\u003E branch that read-engines point at exclusively. Compactions run on the real-time branch rather than main, making them conflict-free. This is the most operationally clean approach for high-delete-rate workloads: No write pauses, no data freshness tradeoffs, and reads always see a clean snapshot. The trade-off is implementation complexity: You're building and operating the conversion process yourself.\u003C/li\u003E\r\n\u003C/ul\u003E\r\n\u003C/li\u003E\r\n\u003Cli\u003EFor tables with active row-level-delete workloads, evaluate whether deletion vectors are supported by your engines and, if so, migrate to using them over equality deletes. Snowflake supports \u003Ca href=\"https://docs.snowflake.com/en/user-guide/tables-iceberg-manage#use-row-level-deletes\" target=\"_blank\" rel=\"noopener noreferrer\"\u003Epositional deletes for v2 Iceberg tables and deletion vectors for v3\u003C/a\u003E across both Snowflake-managed and externally managed tables. Watch the v4 spec progress on index support (there is an active sync on the topic in this \u003Ca href=\"https://docs.google.com/document/d/1N6a2IOzC6Qsqv7NBqHKesees4N6WF49YUSIX2FrF7S0/edit?pli=1&amp;tab=t.0#heading=h.hs6r9d26w1y2\" target=\"_blank\" rel=\"noopener noreferrer\"\u003Ecommunity Google doc\u003C/a\u003E with the design and notes from community sync on the topic, but the Iceberg mailing list is the best place to stay up to date). Once it lands and gains engine adoption, it will further reinforce the deprecation path for equality deletes. However, it's possible that even if/when that happens, there could still be use cases that aren't solved for and need equality deletes. For example, unless the index is maintained synchronously and the index maintenance is required, the CDC writer can't rely on it for completeness to avoid writing equality deletes at scale. In this potential situation, equality deletes are still likely required for a nonnegligible amount of use cases.\u003C/li\u003E\r\n\u003C/ul\u003E\r\n\u003Ch4\u003Ev3: Patchwork support and the one-way door\u003C/h4\u003E\r\n\u003Cp\u003EFor v3 specifically, the right question isn't &quot;does this engine support v3?&quot; It's &quot;does this engine support the specific v3 features my workload requires?&quot; VARIANT type support, deletion vector read and write, row lineage, nanosecond timestamps are all independently variable across engines, and Spark, Trino, Flink and others have varying levels of v3 support today. &quot;Supports v3&quot; in release notes often means the baseline spec is handled, not that every feature is implemented. Note that there is one strict requirement for being able to claim writing support for v3 tables, and that is writing row lineage. The \u003Ca href=\"https://iceberg.apache.org/spec/\" target=\"_blank\" rel=\"noopener noreferrer\"\u003EApache Iceberg spec\u003C/a\u003E defines the upgrade path from v2 to v3 but specifies no downgrade operation. Thus, this is the one-way door. Once a table's \u003Ccode\u003Eformat-version\u003C/code\u003E is set to \u003Ccode\u003E3\u003C/code\u003E in its metadata, any engine interacting with that table must handle v3 at minimum. Verify the specific features you need in each engine's documentation, not the marketing claim, because, again, there is no way to go back.\u003C/p\u003E\r\n\u003Ch5\u003EPragmatic guidance\u003C/h5\u003E\r\n\u003Cul\u003E\r\n\u003Cli\u003EBefore upgrading any table to v3, map the specific features you need against the documented v3 support in every engine and catalog that touches that table, including both readers and writers.\u003C/li\u003E\r\n\u003Cli\u003EUpgrade tables incrementally: Start with the ones that genuinely require a v3 feature (VARIANT is a common driver), and leave remaining tables at v2 until your full engine set is ready.\u003C/li\u003E\r\n\u003Cli\u003ERemember: The \u003Ca href=\"https://iceberg.apache.org/spec/\" target=\"_blank\" rel=\"noopener noreferrer\"\u003EApache Iceberg spec\u003C/a\u003E defines an upgrade path to v3 but \u003Cb\u003Eno downgrade operation\u003C/b\u003E. Additionally, in the Java implementation, downgrading is \u003Ca href=\"https://github.com/apache/iceberg/blob/9b64b317c6f91de8bdaf4bc6cd9aab15945dbd37/core/src/main/java/org/apache/iceberg/TableMetadata.java#L1077-L1081\" target=\"_blank\" rel=\"noopener noreferrer\"\u003Eexplicitly prohibited\u003C/a\u003E. This makes it critical to do testing that is representative of production workloads prior to upgrading tables to v3 in production.\u003C/li\u003E\r\n\u003C/ul\u003E\r\n\u003Ch3\u003EEvaluating IRC implementations to avoid being locked out of the ecosystem\u003C/h3\u003E\r\n\u003Cp\u003EThe concept of &quot;bidirectional IRC compliance&quot; is best understood from the catalog's perspective. \u003Cb\u003EInbound\u003C/b\u003E is an engine connecting to your catalog via the IRC API to read or write your tables (that is, the catalog-as-server path). \u003Cb\u003EOutbound\u003C/b\u003E is your catalog doing federation to another catalog via IRC (that is, the catalog-as-client path, where your catalog reaches out to a remote catalog to pull in its tables). Some platforms support one direction but not the other, and the marketing claim &quot;supports Iceberg REST Catalog APIs&quot; washes right over that nuance.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"image":{"id":"image-5b3c114b28","height":"2594","src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--6f9bcbdd-9597-4bcd-8038-1c74a7cc7e2f/data-fig3.png?quality=85&preferwebp=true","alt":"Figure 3: Diagram showing the inbound and outbound definitions, from the perspective of Platform 1.","lazyEnabled":true,"width":"2276","title":"Figure 3: Diagram showing the inbound and outbound definitions, from the perspective of Platform 1.",":type":"snowflake-site/components/image"},"blog_text_copy_1844728901":{"id":"blog-text-3e530a52fa","text":"\u003Cp\u003E&nbsp;\u003C/p\u003E\r\n\u003Cp\u003EThis matters because a common multi-platform architecture involves a primary catalog that federates tables from several remote catalogs. If the remote catalog doesn't support outbound federation correctly, or the primary catalog's federation client doesn't implement the full IRC spec, you'll discover the gap at query time rather than at architecture time. As noted in \u003Ca rel=\"noopener noreferrer\" target=\"_blank\" href=\"https://www.snowflake.com/en/blog/snowflake-commitment-iceberg-interoperability/\"\u003ESnowflake's analysis of IRC compliance across the ecosystem\u003C/a\u003E, stated and actual IRC support can diverge. The IRC spec does not require generation or consumption of vended credentials. Not supporting both producing and consuming vended credentials can also lead to \u003Ca rel=\"noopener noreferrer\" target=\"_blank\" href=\"https://www.snowflake.com/en/blog/bidirectional-interoperability-snowflake-horizon-databricks/\"\u003Eproblems with management of long-lived credentials\u003C/a\u003E. Claiming compliance on a feature matrix is not the same as passing the full spec against your actual workloads.\u003C/p\u003E\r\n\u003Cp\u003EThe compliance gap also has a forward-looking dimension that's easy to miss at architecture time. The IRC spec is actively evolving. The scan planning API, for instance, is a recent addition that enables server-side query planning, pushing filter and projection pushdown into the catalog rather than the engine. Future additions will go further: Read restrictions and access control enforcement at the catalog layer are on the roadmap, which would allow the catalog to govern what data an engine can see without relying on the engine to enforce it.\u003C/p\u003E\r\n\u003Cp\u003EIf your platform doesn't support full bidirectional IRC compliance today, it also can't adopt these capabilities as they land. You're locked out of the ecosystem's trajectory, not just its current state. This becomes particularly consequential in the governance space, where catalog-enforced read restrictions would eliminate a whole class of engine-trust problems.\u003C/p\u003E\r\n\u003Ch4\u003EPragmatic guidance\u003C/h4\u003E\r\n\u003Cul\u003E\r\n\u003Cli\u003ETest both inbound and outbound IRC compliance in your own environment with your own data and access patterns before committing to a catalog and engine architecture. Unfortunately, you can't rely on vendor marketing claims or documented compatibility matrices. Specifically, test with the IRC API endpoints you plan to use (including newer endpoints like the scan planning APIs and the transactions/commit endpoint) since these are the areas where support gaps are most common.\u003C/li\u003E\r\n\u003Cli\u003EIRC compliance is one dimension of your catalog selection decision, but weigh it against your other requirements (such as performance, governance capabilities, ecosystem support and operational overhead), based on what actually matters for your organization and workloads.\u003C/li\u003E\r\n\u003C/ul\u003E\r\n\u003Ch3\u003EWrite-side consistency\u003C/h3\u003E\r\n\u003Cp\u003EIceberg is deliberately unopinionated about physical layout: file sizes, row group configuration, column statistics coverage, shredding strategy. That flexibility is useful, but in a multi-engine environment it means write pipelines can silently produce tables that perform well for one engine and poorly for another, with no clear messages to signal a problem. These are Parquet-layer and engine-configuration problems that apply equally to any table format built on Parquet, including Delta Lake and Hudi.\u003C/p\u003E\r\n\u003Ch4\u003EFile hygiene\u003C/h4\u003E\r\n\u003Cp\u003EAs introduced above, the lack of enforcement means different write pipelines operating against the same table can make different decisions about file sizes, row group sizes and statistics coverage, and what's optimal for one engine's read path can be suboptimal for another's. Without a central authority enforcing write discipline, tables can degrade gradually and silently: Query performance deteriorates, maintenance costs climb, and the degradation is hard to attribute because it doesn't emit error messages. In a centralized organization with Iceberg expertise and controlled write pipelines, this is manageable overhead. In a decentralized organization where many teams write to shared tables with different tools and varying levels of Iceberg expertise, this can quickly become a real operational liability.\u003C/p\u003E\r\n\u003Cp\u003EFor organizations where Snowflake is the primary write path, \u003Ca rel=\"noopener noreferrer\" target=\"_blank\" href=\"https://docs.snowflake.com/en/user-guide/tables-iceberg#use-snowflake-as-the-catalog\"\u003ESnowflake-managed Iceberg tables\u003C/a\u003E offer a pragmatic approach that addresses much of this. Snowflake takes an opinionated stance on write hygiene by default: \u003Ca rel=\"noopener noreferrer\" target=\"_blank\" href=\"https://docs.snowflake.com/en/user-guide/tables-iceberg-manage#set-a-target-file-size\"\u003EAdaptive file sizing\u003C/a\u003E (\u003Ccode\u003ETARGET_FILE_SIZE = AUTO\u003C/code\u003E) tunes file sizes based on table characteristics rather than leaving the decision to individual pipelines; column statistics are written for all columns by default; \u003Ca rel=\"noopener noreferrer\" target=\"_blank\" href=\"https://docs.snowflake.com/en/user-guide/tables-iceberg-manage#table-optimization-for-snowflake-managed-iceberg-tables\"\u003Eand automatic data compaction, manifest compaction and snapshot expiry\u003C/a\u003E run continuously without requiring teams to schedule or manage them. The scope caveat is something to remember though: These defaults apply to Snowflake's write path. External engines writing to the same table through Horizon Catalog operate independently and don't inherit Snowflake's file-hygiene defaults — though they benefit from Snowflake's ongoing table maintenance operations on the storage side. For tables with genuinely mixed write workloads, the problem described above remains.\u003C/p\u003E\r\n\u003Ch5\u003EPragmatic guidance\u003C/h5\u003E\r\n\u003Cul\u003E\r\n\u003Cli\u003EFor centralized organizations, establish explicit write pipeline standards (for example, target file size ranges, row group size and stat coverage) and enforce them in code review or pipeline templates rather than relying on individual contributors to know the defaults.\u003C/li\u003E\r\n\u003Cli\u003EFor decentralized organizations, the more opinionated the platform, the lower the surface area for misconfiguration. Platforms that enforce sensible defaults by design significantly reduce this risk compared to giving every team direct control over write configuration.\u003C/li\u003E\r\n\u003Cli\u003EIn both cases, run regular compaction jobs on frequently written tables to normalize file sizes, and monitor file size distributions as part of routine table health checks.\u003C/li\u003E\r\n\u003C/ul\u003E\r\n\u003Ch4\u003EColumn statistics defaults\u003C/h4\u003E\r\n\u003Cp\u003EA concrete and frequently overlooked example of write-side consistency is column statistics. Iceberg writes per-column min/max bounds, null counts and value counts into manifest metadata (that is, the statistics that query engines use to skip files that can't match a filter). The Iceberg table property \u003Ccode\u003Ewrite.metadata.metrics.max-inferred-column-defaults\u003C/code\u003E controls how many columns receive full statistics; \u003Ca rel=\"noopener noreferrer\" target=\"_blank\" href=\"https://iceberg.apache.org/docs/latest/configuration/#write-properties\"\u003Eits default is 100\u003C/a\u003E. For a pipeline using this default, columns beyond position 100 get no column-level statistics in Iceberg metadata at all: no min/max bounds, no null count, no value counts. The only statistics available are in the Parquet file footers, which is not as efficient as Iceberg metadata statistics for column-level predicate pushdown. A query that adds a filter on column 101, as compared to a filter on column 100, can degrade significantly without any error or warning, and the degradation only surfaces at scale when a user changes a filter predicate.\u003C/p\u003E\r\n\u003Cp\u003ESnowflake defaults to writing full statistics for all columns because in the vast majority of cases, the query-performance benefits of having stats outweigh the costs of metadata write overhead. To be fair, there are situations where the cons outweigh the pros here, but those are generally few and far between. This is a sensible default that not all engines share.\u003C/p\u003E\r\n\u003Ch5\u003EPragmatic guidance\u003C/h5\u003E\r\n\u003Cul\u003E\r\n\u003Cli\u003EFor centralized organizations, audit your table creation and write pipelines for the \u003Ccode\u003Ewrite.metadata.metrics.max-inferred-column-defaults\u003C/code\u003E setting. For tables with more than 100 columns where you filter beyond position 100, explicitly set this property to ensure full statistics are written for your actual filter columns. This is a table-level property; set it at table creation time and include it in any table provisioning standards or templates.\u003C/li\u003E\r\n\u003Cli\u003EFor decentralized organizations, ensure that this setting is addressed by any group creating tables or writing pipelines, whether by documentation, process or technical framework.\u003C/li\u003E\r\n\u003C/ul\u003E\r\n\u003Ch4\u003EVARIANT and shredding\u003C/h4\u003E\r\n\u003Cp\u003EVARIANT columns introduce a related but distinct problem. Iceberg v3's \u003Ca rel=\"noopener noreferrer\" target=\"_blank\" href=\"https://www.snowflake.com/en/engineering-blog/apache-iceberg-v3-variant-type/\"\u003EVARIANT type\u003C/a\u003E can be physically stored in Parquet with or without \u003Cb\u003Eshredding\u003C/b\u003E (also known as subcolumnarization), which is the process of extracting sub-fields from the JSON structure into separate Parquet columns so that engines can benefit from columnar scan optimization on specific paths. Without shredding, querying any sub-field requires deserializing the full blob for every row. With shredding, a query on a specific path reads only that path's column. However, shredding support varies across engines, and the degree to which engines shred (how deeply they recurse into nested structures) is not standardized. Moreover, Iceberg's table metadata doesn't record the shredding strategy that was used when writing. A table written by an engine that shreds aggressively will have different physical characteristics than one written by an engine that doesn't shred at all. A user using a query engine reading both might expect them to perform equally but instead will see degraded performance with no obvious explanation.\u003C/p\u003E\r\n\u003Cp\u003EWhat makes this situation particularly difficult is that the right shredding strategy is highly variable: It depends on the shape of your specific data and your query access patterns, not a static per-table rule. For simple, well-understood schemas with predictable access patterns, shredding decisions are straightforward. In production at scale (evolving schemas, mixed access patterns and/or multiple teams writing to the same tables) getting shredding right is a genuinely hard problem and ultimately presents a set of trade-offs. There are discussions ongoing in the Iceberg community around a \u003Ca rel=\"noopener noreferrer\" target=\"_blank\" href=\"https://lists.apache.org/thread/f8l27lr55nlrdg4qwno29r796d4v2p8l\"\u003Eshredding algorithm/definition\u003C/a\u003E.\u003C/p\u003E\r\n\u003Cp\u003ESnowflake has \u003Ca rel=\"noopener noreferrer\" target=\"_blank\" href=\"https://dl.acm.org/doi/10.1145/2882903.2903741\"\u003Espent a decade\u003C/a\u003E iterating on its approach to semi-structured data and shredding/subcolumnarization. That experience shows in its influence in \u003Ca rel=\"noopener noreferrer\" target=\"_blank\" href=\"https://www.snowflake.com/en/blog/engineering/apache-iceberg-v3-variant-type/\"\u003Eshaping the community design\u003C/a\u003E for VARIANT in Iceberg v3, as well as the rubber-meets-the-road \u003Ca rel=\"noopener noreferrer\" target=\"_blank\" href=\"https://www.snowflake.com/en/blog/engineering/snowflake-iceberg-v3-variant-performance/\"\u003Eperformance on VARIANT in Iceberg v3\u003C/a\u003E. The rest of the ecosystem is earlier on that curve.\u003C/p\u003E\r\n\u003Ch5\u003EPragmatic guidance\u003C/h5\u003E\r\n\u003Cul\u003E\r\n\u003Cli\u003EIf VARIANT is a primary access pattern in your multi-engine architecture, test cross-engine query performance with representative data before production. Specifically, test queries written in each engine against data written by every other engine.\u003C/li\u003E\r\n\u003Cli\u003EWhere possible, standardize on a single write engine for VARIANT-heavy tables to eliminate shredding variability on the write side that impacts the read side.\u003C/li\u003E\r\n\u003Cli\u003EFor mixed-write scenarios, profile query performance per engine combination and set expectations accordingly.\u003C/li\u003E\r\n\u003C/ul\u003E\r\n\u003Ch4\u003EEncoding and compression\u003C/h4\u003E\r\n\u003Cp\u003EEncoding and compression add a further layer of variability. Not all engines support all Parquet encoding algorithms, and the Iceberg spec doesn't mandate which encoding to use for a table. Engines tend to write what they know, which means tables in a multi-engine environment can have physically heterogeneous files: some written with one engine's preferred encoding, some with another's. This creates read-side inefficiencies when an engine encounters an encoding it handles suboptimally. One example of active standardization work: \u003Ca rel=\"noopener noreferrer\" target=\"_blank\" href=\"https://github.com/apache/parquet-format/issues/533\"\u003EALP (adaptive lossless floating-point) encoding\u003C/a\u003E for floating-point columns (contributed to the Parquet community by Snowflake engineer \u003Ca rel=\"noopener noreferrer\" target=\"_blank\" href=\"https://www.snowflake.com/en/blog/authors/prateek-gaur/\"\u003EPrateek Gaur\u003C/a\u003E). Once ratified and adopted across engines, it benefits the entire ecosystem.\u003C/p\u003E\r\n\u003Ch5\u003EPragmatic guidance\u003C/h5\u003E\r\n\u003Cul\u003E\r\n\u003Cli\u003EDefault to widely supported encodings for tables shared across multiple engines.\u003C/li\u003E\r\n\u003Cli\u003EAvoid experimental or engine-specific encodings in tables where the pros of multi-engine access outweigh the cons of degraded performance from not leveraging these advanced encodings.\u003C/li\u003E\r\n\u003Cli\u003EWatch the ALP vote progress for floating-point-heavy workloads.\u003C/li\u003E\r\n\u003C/ul\u003E\r\n\u003Ch3\u003EStorage bucket management and lifecycle\u003C/h3\u003E\r\n\u003Cp\u003EOwning the storage buckets means owning everything in them, including Iceberg's own metadata. This is a materially different operational model than data warehouses, and it catches teams off-guard. With Hive, metadata lived in the catalog; the data files were blobs the catalog pointed at. With Iceberg, the metadata layer (that is, manifest files, manifest lists, table metadata JSON) lives in S3 alongside your data files. That co-location is part of what makes Iceberg fast and portable. It also means that with customer-managed storage, you're on the hook for bucket-level concerns (for example, lifecycle policies, security configuration and access controls) regardless of who manages the Iceberg metadata itself.\u003C/p\u003E\r\n\u003Cp\u003EThere's an important distinction between \u003Ci\u003Eowning\u003C/i\u003E the storage and \u003Ci\u003Emanaging\u003C/i\u003E the Iceberg metadata operations. Even on customer-managed storage, some vendors will manage the metadata for you — running compaction, snapshot expiration, orphan cleanup and manifest rewriting as part of their service. Snowflake does this for Snowflake-managed Iceberg tables on customer-managed storage. But no vendor can override your bucket-level lifecycle policies or security settings (nor would you generally want them to). That responsibility stays with whoever owns the bucket. There is the option to go with vendor-managed storage but still store the data in Iceberg format, so other engines can access. Snowflake is one vendor that offers that in the form of \u003Ca rel=\"noopener noreferrer\" target=\"_blank\" href=\"https://docs.snowflake.com/en/user-guide/tables-iceberg-internal-storage\"\u003ESnowflake Storage for Apache Iceberg\u003C/a\u003E™. With that offering, you're off the ownership hook entirely: Snowflake owns the storage and manages the metadata, eliminating both layers of operational surface, while maintaining full read/write interoperability for other engines.\u003C/p\u003E\r\n\u003Cp\u003EWithout regular \u003Ca rel=\"noopener noreferrer\" target=\"_blank\" href=\"https://iceberg.apache.org/docs/latest/maintenance/\"\u003Etable maintenance\u003C/a\u003E (that is, orphan file cleanup, snapshot expiration, manifest rewriting and compaction) tables degrade silently, and storage costs compound. Orphan files accumulate from failed writes; old snapshots hold references to precompaction files (so you're paying double); manifest metadata bloats until query planning slows down or maintenance jobs OOM. The Slack engineering team described in its \u003Ca rel=\"noopener noreferrer\" target=\"_blank\" href=\"https://www.youtube.com/watch?v=2Qi8WLgKOv0\"\u003EIceberg Summit 2025 talk\u003C/a\u003E how a dev table receiving streaming writes every five minutes for six months, without maintenance, accumulated 12.5 TB of manifest files alone (pure metadata), and orphan file accumulation was &quot;more prevalent than expected&quot; across both batch and streaming workloads. If you're managing maintenance yourself (that is, not using a vendor that handles it for you), it needs to start on day one, not when performance degrades.\u003C/p\u003E\r\n\u003Cp\u003EThe most acute footgun in this space is \u003Cb\u003ES3 lifecycle policies applied naively\u003C/b\u003E. With Iceberg, metadata lives in the bucket alongside data, and lifecycle policies don't distinguish between a Parquet data file and a \u003Ccode\u003Emetadata.json\u003C/code\u003E. S3 lifecycle policies must be scoped to not delete any file Iceberg still references, and data deletion at the storage layer should always be driven by Iceberg's own snapshot expiration or orphan file cleanup processes, not bucket-level policies acting independently. Storage class lifecycle policies are fine to use, but only use tiers that still provide synchronous responses/access. Calculate the query performance degradations (and therefore cost) you're going to hit, and see if the pros outweigh the cons for the table and access patterns.\u003C/p\u003E\r\n\u003Ch4\u003EPragmatic guidance\u003C/h4\u003E\r\n\u003Cul\u003E\r\n\u003Cli\u003EFix S3 lifecycle policies before anything else. Scope bucket-level deletion policies to not delete files Iceberg still references. Let Iceberg manage its own lifecycle through snapshot expiration. If you're using versioned S3 buckets, only hard-delete after Iceberg has expired the relevant snapshots.\u003C/li\u003E\r\n\u003Cli\u003EStart maintenance on Day 1. That is, orphan cleanup, snapshot expiration, manifest rewriting and compaction should run from the first write. Catching up on months of accumulated debt is dramatically more expensive than running it routinely. If you go with a vendor to manage this, these are usually on by default, so it's not something you have to worry about.\u003C/li\u003E\r\n\u003Cli\u003EFigure out your compaction and snapshot expiration strategy together. Compaction alone doesn't reclaim storage; old snapshots still reference precompaction files until expired.\u003C/li\u003E\r\n\u003C/ul\u003E\r\n\u003Ch3\u003EPhysics hasn't changed: Cross-region access and disaster recovery\u003C/h3\u003E\r\n\u003Cp\u003EIceberg solves a lot of problems, but it doesn't repeal the laws of networking. Data stored in one region still has latency and egress costs when accessed from another. Replication still requires copying bytes across physical distance. And Iceberg's own metadata model introduces a specific structural problem that makes both cross-region access and disaster recovery harder than they were with simpler formats.\u003C/p\u003E\r\n\u003Cp\u003EThat structural problem is \u003Cb\u003Eabsolute paths\u003C/b\u003E. Every file reference in Iceberg metadata (that is, manifest files, manifest lists, table metadata JSON) stores the full path: \u003Ccode\u003Es3://my-bucket/warehouse/my_table/data/part-00001.parquet\u003C/code\u003E, not a relative \u003Ccode\u003Edata/part-00001.parquet\u003C/code\u003E. If you need the same table queryable from a different location (whether for cross-region/cross-cloud environments or for business continuity and disaster recovery) you can't simply replicate the storage contents (which most cloud object storages provide out-of-the-box). The data files copy fine, but every metadata file still points to the original bucket. The replicated table is not queryable as-is without additional tooling to rewrite the paths.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"image_copy_701714592":{"id":"image-4441a79aa5","height":"1523","src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--7800f8b9-5435-434b-a714-1a32ea9bbd4b/data-fig4.png?quality=85&preferwebp=true","alt":"Figure 4: A visual depiction of why binary replication of Iceberg tables doesn’t solve cross-region replication.","lazyEnabled":true,"width":"2909","title":"Figure 4: A visual depiction of why binary replication of Iceberg tables doesn’t solve cross-region replication.",":type":"snowflake-site/components/image"},"blog_text_copy_447586789":{"id":"blog-text-de8b2f7e8c","text":"\u003Cp\u003E&nbsp;\u003C/p\u003E\r\n\u003Cp\u003EAccording to \u003Ca rel=\"noopener noreferrer\" target=\"_blank\" href=\"https://lists.apache.org/thread/dqlgo3rmz1tnshx0bgk2cv4qm1h06k2n\"\u003Ethe vote thread on the dev mailing list\u003C/a\u003E, the v4 spec will include a feature to support \u003Ca rel=\"noopener noreferrer\" target=\"_blank\" href=\"https://github.com/apache/iceberg/pull/15630\"\u003Erelative paths\u003C/a\u003E, which would eliminate this problem entirely by storing paths relative to the table's base location. It's actually a topic that's been \u003Ca rel=\"noopener noreferrer\" target=\"_blank\" href=\"https://github.com/apache/iceberg/issues/3142\"\u003Ediscussed for a while in the Iceberg community\u003C/a\u003E, so we're glad to see it making real progress. Once v4 is ratified and adopted, you'll be able to replicate a table and simply point the catalog at the new base path. But v4 ratification, implementation across engines and broad adoption is likely at least one to three quarters out (possibly longer depending on which vendors and engines you rely on), as of this writing. It's the right long-term answer, but it isn't the answer that's available today.\u003C/p\u003E\r\n\u003Cp\u003EThe options for keeping tables queryable across regions today:\u003C/p\u003E\r\n\u003Cul\u003E\r\n\u003Cli\u003E\u003Cb\u003ESnowflake:\u003C/b\u003E \u003Ca rel=\"noopener noreferrer\" target=\"_blank\" href=\"https://docs.snowflake.com/en/user-guide/tables-iceberg-replication\"\u003EProvides multiple replication options depending on how the Iceberg table is managed.\u003C/a\u003E For Snowflake-managed Iceberg tables, replication covers both BCDR and cross-region data sharing (via listing auto-fulfillment). Data and metadata stay consistent without custom tooling. For externally managed Iceberg tables on customer-managed storage, Snowflake supports replication for cross-region data sharing, though not BCDR in that configuration.\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EAWS S3 Tables:\u003C/b\u003E Provide built-in replication within the AWS ecosystem, but AWS only; they don't help for cross-cloud scenarios.\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EEngine-driven replication:\u003C/b\u003E Use a compute engine in the destination region to periodically run SQL statements (\u003Ccode\u003EINSERT INTO ... SELECT\u003C/code\u003E or \u003Ccode\u003EMERGE INTO\u003C/code\u003E) to materialize a local copy from the source table. This gives co-located reads in both regions at the cost of replication latency and the operational overhead of maintaining the sync pipeline.\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EBuild and maintain custom tooling:\u003C/b\u003E An initial copy with \u003Ccode\u003Eaws s3 cp\u003C/code\u003E (or other cloud equivalents) is straightforward; efficiently replicating ongoing changes (i.e., tracking new files, keeping metadata consistent, handling in-flight transactions) is significantly harder and becomes a piece of infrastructure to maintain indefinitely.\u003C/li\u003E\r\n\u003C/ul\u003E\r\n\u003Cp\u003EWith that shared context, let's consider the two specific use cases more deeply.\u003C/p\u003E\r\n\u003Ch4\u003ECross-region and cross-cloud access\u003C/h4\u003E\r\n\u003Cp\u003EThe strong recommendation for any Iceberg deployment is co-location: Compute engines should run in the same cloud region as the storage they're querying. Cross-region and cross-cloud reads work. Nothing technically prevents them, but they come with real consequences:\u003C/p\u003E\r\n\u003Cul\u003E\r\n\u003Cli\u003EEgress costs on every query (outside of local engine caching queried data on disk, but that's hard to model confidently and reliably)\u003C/li\u003E\r\n\u003Cli\u003ELatency that degrades query performance in ways that are hard to diagnose because they look like slow engines rather than slow networks\u003C/li\u003E\r\n\u003Cli\u003EIncreased compute costs because queries that would be fast with local storage spend time waiting on network I/O\u003C/li\u003E\r\n\u003Cli\u003EAn expanded security surface, since data is now traversing inter-region and potentially inter-cloud network paths rather than staying within a cloud-region's trust boundary\u003C/li\u003E\r\n\u003C/ul\u003E\r\n\u003Cp\u003EIn practice, many organizations end up in cross-region or cross-cloud situations regardless: a legacy warehouse in one region, a new Iceberg deployment in another; an acquisition that brought a different cloud footprint; an analytics team in Europe with data stored in the U.S. We see this somewhat often: A company is using two platforms in different regions without realizing it, because data is simply being copied between them. Iceberg as a project doesn't create these problems, but it doesn't solve them either. If your architecture requires cross-region or cross-cloud data access, you need to account for how you're going to solve this problem rather than assume the open format will simply absorb them.\u003C/p\u003E\r\n\u003Ch5\u003EPragmatic guidance\u003C/h5\u003E\r\n\u003Cul\u003E\r\n\u003Cli\u003EAudit your engine-to-storage topology before deploying at scale. If engines and storage are in different regions, quantify the egress cost and latency impact with representative queries on representative data volumes before committing to the architecture.\u003C/li\u003E\r\n\u003Cli\u003EWhere possible, co-locate compute with storage. This is often the single highest-leverage configuration decision for query performance (and in the cloud, therefore cost) on large tables.\u003C/li\u003E\r\n\u003Cli\u003EWhere co-location isn't achievable, use one of the replication approaches above to maintain a local copy, rather than accepting persistent cross-region reads at scale.\u003C/li\u003E\r\n\u003C/ul\u003E\r\n\u003Ch4\u003EBusiness continuity and disaster recovery (BCDR)\u003C/h4\u003E\r\n\u003Cp\u003EBCDR requires a cross-region or cross-cloud setup almost by definition. A DR copy in the same region as your primary doesn't protect against a regional failure. The absolute paths problem described above is what makes this specifically hard for Iceberg: Standard object replication doesn't produce a queryable replica without path rewriting. Until v4 relative paths land, you need one of the solutions above to maintain a DR copy that's actually usable in a failover scenario. Also, since there are multiple levels to BCDR, if you're trying to protect against cloud failure, multi-region but single-cloud solutions don't cut it.\u003C/p\u003E\r\n\u003Ch5\u003EPragmatic guidance\u003C/h5\u003E\r\n\u003Cul\u003E\r\n\u003Cli\u003EStart by assessing which tables and workloads actually require BCDR. Not everything does, and the answer shapes the solution.\u003C/li\u003E\r\n\u003Cli\u003EFor tables that do require BCDR, the build-versus-buy decision usually comes down to the ongoing maintenance cost. The initial copy is easy, but maintaining a consistent, queryable replica through ongoing writes, compactions and schema changes is a sustained engineering investment.\u003C/li\u003E\r\n\u003Cli\u003EEvaluate the options based on the most common attributes of your tables and workloads to define a default. Then, for any outlier tables or workloads, assess the options for each table/workload and see which approach is best for it.\u003C/li\u003E\r\n\u003Cli\u003EIf you're using Snowflake Storage for Apache Iceberg and Snowflake-managed Iceberg tables, this is handled out of the box. For externally managed Iceberg tables on customer-managed storage with a near-term BCDR requirement, evaluate the engineering cost of maintaining a custom replication solution against the alternatives.\u003C/li\u003E\r\n\u003C/ul\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"blog_title_copy_copy_280634654":{"id":"blog-title-4dd0e1a10c","propertiesId":"open-problems-and-active-work","type":"heading2","lines":["Open problems and active work"],":type":"snowflake-site/components/blog/blog-title"},"blog_text_copy_1511145138":{"id":"blog-text-d8b74e7db4","text":"\u003Ch3\u003EAI/ML workloads\u003C/h3\u003E\r\n\u003Cp\u003EAI/ML workloads push on the combination of Iceberg and Parquet in ways that standard analytics don't. The gaps here are real, acknowledged by the community, and actively being worked on, but not many solutions have shipped yet. This is genuinely &quot;watch this space&quot; territory.\u003C/p\u003E\r\n\u003Cp\u003EThe gaps fall into two kinds of issues. Some are access-pattern problems on otherwise structured data: wide feature tables and vector similarity search, where the file and index formats are the limiting factor. Others are data-type problems: unstructured and multimodal data that the table format doesn't model at all. There are a few new formats that have been created to address these problems, the main ones being Lance, Vortex and Nimble. Lance is generally more popular than the other two, so it's the one we'll discuss here.\u003C/p\u003E\r\n\u003Cp\u003ELance is a purpose-built lakehouse format for AI/ML, structured as a stack of interoperating specs rather than a single format: a file format (the analog to Parquet), a table format (the analog to Iceberg), plus its own index and catalog specs. It doesn't line up against just Parquet or just Iceberg. Depending on which layers you adopt, Lance can stand as an alternative to Iceberg or slot in as a complement to it. The complementary path is concrete: Iceberg's \u003Ca href=\"https://iceberg.apache.org/blog/apache-iceberg-file-format-api/\" target=\"_blank\" rel=\"noopener noreferrer\"\u003Epluggable File Format API\u003C/a\u003E shipped in 1.11.0, decoupling table metadata from the physical file layout, and the Iceberg team explicitly names Lance (and Vortex) as formats it's meant to accommodate. A working Lance implementation on that API is still nascent. At the catalog layer you can already run both today: Apache Polaris manages Lance tables alongside Iceberg tables through its Generic Table API (not as an Iceberg file format, but as a separate table format coexisting in the same catalog). It's early, and how this shakes out isn't settled. The capabilities that make it attractive span Lance's file, index and table layers:\u003C/p\u003E\r\n\u003Cul\u003E\r\n\u003Cli\u003EEfficient vector similarity search\u003C/li\u003E\r\n\u003Cli\u003EFast random access by row ID\u003C/li\u003E\r\n\u003Cli\u003ENative support for the column-append pattern that AI/ML feature pipelines need\u003C/li\u003E\r\n\u003C/ul\u003E\r\n\u003Ch4\u003EWide tables\u003C/h4\u003E\r\n\u003Cp\u003EAI/ML feature engineering routinely produces tables with hundreds or thousands of columns (for example, feature stores, embedding tables and training data sets where each feature is a column). Parquet was designed for wide reads across many rows of relatively few columns, not as much for narrow reads across a few rows of many columns. At scale, wide Iceberg tables stored in Parquet hit concrete performance problems: Parquet metadata bloats and becomes expensive to parse, row groups become small because even a modest number of rows reaches the size threshold when spread across thousands of columns, compression suffers, and adding new columns requires rewriting every data file.\u003C/p\u003E\r\n\u003Cp\u003EThe Iceberg and Parquet communities are actively exploring this problem space. A \u003Ca href=\"https://lists.apache.org/thread/h0941sdq9jwrb6sj0pjfjjxov8tx7ov9\" target=\"_blank\" rel=\"noopener noreferrer\"\u003Emailing list thread on &quot;Wide tables in V4&quot;\u003C/a\u003E proposed splitting wide tables across multiple physical files (column families), with early benchmarks (\u003Ca href=\"https://github.com/apache/iceberg/pull/13306\" target=\"_blank\" rel=\"noopener noreferrer\"\u003EPR #13306\u003C/a\u003E) showing meaningful read and write improvements for 10,000-column tables. There's also a parallel effort in the Parquet community to replace the Thrift footer with FlatBuffers to improve parse performance for wide schemas. Whether the solution lands in Parquet, Iceberg or both is still being worked out, and either path is realistically a v4-or-later timeline.\u003C/p\u003E\r\n\u003Cp\u003ELance addresses this at the file and table layers with column-append that doesn't rewrite existing files (see the format note above).\u003C/p\u003E\r\n\u003Ch4\u003EVector search and index structures\u003C/h4\u003E\r\n\u003Cp\u003EThe single-copy interoperability promise doesn't extend to vector search workloads yet for broad ecosystem support. Vectors can be stored in Iceberg today, but storage isn't the problem. Efficient similarity search at scale requires dedicated index structures that Iceberg has no standard for. A \u003Ca href=\"https://github.com/apache/iceberg/issues/12636\" target=\"_blank\" rel=\"noopener noreferrer\"\u003Ecommunity proposal\u003C/a\u003E was opened in 2025 but was \u003Ca href=\"https://github.com/apache/iceberg/issues/12636\" target=\"_blank\" rel=\"noopener noreferrer\"\u003Eclosed as stale in January 2026\u003C/a\u003E without reaching consensus. That said, there is \u003Ca href=\"https://docs.google.com/document/d/1N6a2IOzC6Qsqv7NBqHKesees4N6WF49YUSIX2FrF7S0/edit?pli=1&amp;tab=t.0#heading=h.hs6r9d26w1y2\" target=\"_blank\" rel=\"noopener noreferrer\"\u003Erenewed discussion in the Iceberg community\u003C/a\u003E to support vector indexes. The design space is genuinely hard (cross-file indexes, no agreement on format, nontrivial interaction with Iceberg's file-level abstraction), and the community hasn't converged on an approach yet.\u003C/p\u003E\r\n\u003Cp\u003EWithout native index support, vector similarity search on Iceberg tables requires a full scan or a brute-force approach that doesn't scale. The workarounds today are purpose-built vector databases for the search workload (with a pointer table in Iceberg for the structured metadata), or formats like Lance that were designed from the ground up for this access pattern.\u003C/p\u003E\r\n\u003Ch4\u003EUnstructured and multimodal data\u003C/h4\u003E\r\n\u003Cp\u003EIceberg is a table format: It manages structured, columnar data in Parquet files (or, less frequently, in ORC and Avro files). Images, video, audio, PDFs and other unstructured data don't fit that model natively. There's no Iceberg-standard way to store a JPEG inside an Iceberg table, and forcing binary blobs into Parquet columns defeats the purpose of columnar storage (no pruning, no statistics, no compression benefit).\u003C/p\u003E\r\n\u003Cp\u003EA usable pattern today is a pointer table. Store the raw files in object storage in their native format, and maintain a separate Iceberg table with a column pointing to each file's URI along with metadata columns (file type, creation timestamp, associated entity IDs and extracted features). This works: It's queryable, it benefits from Iceberg's schema evolution and partition pruning on the metadata, and it lets you join structured and unstructured data through the pointer. Many production systems use this pattern successfully.\u003C/p\u003E\r\n\u003Cp\u003EThe limitation is that the pointer table and the files it references are not transactionally linked:\u003C/p\u003E\r\n\u003Cul\u003E\r\n\u003Cli\u003EDeleting a row from the metadata table doesn't delete the underlying file\u003C/li\u003E\r\n\u003Cli\u003EMoving files in storage doesn't update the pointers\u003C/li\u003E\r\n\u003Cli\u003ESchema changes to the metadata don't propagate to the files\u003C/li\u003E\r\n\u003C/ul\u003E\r\n\u003Cp\u003EYou're managing two independent systems, and keeping them consistent is operational overhead that grows with data volume.\u003C/p\u003E\r\n\u003Cp\u003EAs multimodal AI workloads grow (for example, RAG pipelines over document corpora, video analytics and image classification at scale) this gap will become an ever-widening chasm. There was an \u003Ca href=\"https://github.com/apache/iceberg/issues/859\" target=\"_blank\" rel=\"noopener noreferrer\"\u003Eearly proposal\u003C/a\u003E for an Iceberg unstructured data module, but it hasn't progressed beyond discussion before the issue was closed as stale in 2024. As of writing, there's no active proposal or accepted design for native unstructured data support in Iceberg, and it's unclear whether the right answer is extending Iceberg or using a purpose-built file format (Lance already supports native blob storage with lazy loading) alongside it.\u003C/p\u003E\r\n\u003Ch4\u003EWhere this nets out\u003C/h4\u003E\r\n\u003Cp\u003EFor workloads where vector search and point lookups are the primary access pattern, Lance has real performance advantages today that Parquet can't match without the index and wide-table work described above.\u003C/p\u003E\r\n\u003Cp\u003EThe biggest differentiating factor between Lance and Parquet is ecosystem maturity. Parquet has broad engine support, a large community and years of production-hardening across analytics workloads. Lance as a file format has a smaller ecosystem: fewer engines and less operational tooling. For mixed workloads that combine analytics and AI/ML, Parquet remains the right default. For specialized AI/ML workloads where vector search and point lookups dominate and the performance delta is significant, Lance is worth evaluating. But understand you're trading ecosystem breadth for access pattern optimization. This trade-off space is genuinely not settled, and the community's progress on wide tables and index support will narrow the gap over time.\u003C/p\u003E\r\n\u003Cp\u003EBuilding on that File Format API, there's active work in the Iceberg and Lance communities to bring the Lance file format to Iceberg (for example, the foundations for future support and a POC for support). At the same time, there are discussions in the Iceberg and Parquet communities about trying to bring the benefits of Lance to Parquet. At this point, it is not yet clear how all this will shake out.\u003C/p\u003E\r\n\u003Ch5\u003EPragmatic guidance\u003C/h5\u003E\r\n\u003Cul\u003E\r\n\u003Cli\u003EDefault to Iceberg and Parquet for workloads that mix analytics and AI/ML for now.\u003C/li\u003E\r\n\u003Cli\u003EReach for a purpose-built format like Lance only when a specialized access pattern dominates (vector search, point lookups, column-append, or native blob/unstructured) and the performance gap justifies the ecosystem trade-off.\u003C/li\u003E\r\n\u003Cli\u003EFor vector search specifically: If it's critical, accept the data-copy trade-off today (keep a copy in a platform with native vector search) rather than forcing Iceberg and Parquet to serve it; or, if avoiding copies is the overriding priority, store exclusively in a format that supports it natively and accept the ecosystem trade-offs.\u003C/li\u003E\r\n\u003Cli\u003EFor unstructured and multimodal data: Use the pointer-table pattern as the default for now, and own the consistency problem explicitly (reconciliation, orphan cleanup, cascading deletes, lifecycle rules) since the table and files aren't transactionally linked.\u003C/li\u003E\r\n\u003Cli\u003EWhere both are needed, use a unified catalog like Apache Polaris to manage Iceberg and Lance side by side rather than standardizing on one format for everything.\u003C/li\u003E\r\n\u003Cli\u003EDon't architect around native Iceberg support that doesn't exist yet (vector indexes, unstructured). Treat it as watch-this-space and reassess as the File Format API, wide-table and index work progresses.\u003C/li\u003E\r\n\u003C/ul\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"blog_title_copy_copy_848238316":{"id":"blog-title-3da4184b62","propertiesId":"diagnostic-questions","type":"heading2","lines":["Diagnostic questions"],":type":"snowflake-site/components/blog/blog-title"},"blog_text":{"id":"blog-text-44131f6285","text":"\u003Cp\u003EIf you're evaluating or already operating a multi-engine Iceberg architecture, these are the questions worth asking before you're in too deep and finding out the hard way:\u003C/p\u003E\r\n\u003Col\u003E\r\n\u003Cli\u003E\u003Cb\u003EHave you tested/validated bidirectional IRC compliance\u003C/b\u003E — both inbound (engines connecting to your catalog) and outbound (your catalog federating to remote catalogs) — with the specific endpoints your workloads require? Not marketing claims; actual tests with your data and access patterns.\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EHave you mapped v2 and v3 feature support per engine against your workload requirements?\u003C/b\u003E Specifically, do all your read engines handle the delete mechanisms your write engines produce? Do all engines that touch a v3 table support the v3 features that table uses?\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EIs all compute co-located in the same cloud region as storage?\u003C/b\u003E If not, have you quantified the egress cost and latency impact, and do you have an explicit strategy for keeping cross-region copies in sync?\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EDo you have a BCDR plan for the Iceberg tables that need one?\u003C/b\u003E Have you verified that your replication approach produces queryable replicas, given the absolute paths constraint?\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EAre your write pipelines centralized or decentralized?\u003C/b\u003E If decentralized, how do you enforce file size targets, column statistics coverage, VARIANT shredding consistency and encoding choices across teams and tools?\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EFor tables with VARIANT columns accessed by multiple engines\u003C/b\u003E, have you tested cross-engine query performance with representative data? Do you know which engine wrote the data and what level of shredding it did?\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EDo any of your tables have more than 100 columns?\u003C/b\u003E If so, have you verified that column statistics are being written for the columns you actually filter on, or are queries silently scanning more files than necessary?\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EDo any of your write engines produce equality deletes?\u003C/b\u003E If so, can all your read engines handle them efficiently, or do you have a conversion or isolation strategy in place?\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EDo you have governance controls preventing storage lifecycle policies from corrupting tables?\u003C/b\u003E Specifically, have you verified that S3 (or equivalent) lifecycle rules can't delete any file (metadata or data) that Iceberg still references? Deleting a metadata file can make a table unreadable; deleting a data file that manifest files still point to can produce query errors on reads that hit that file, and by some definitions, corrupt the table.\u003C/li\u003E\r\n\u003C/ol\u003E\r\n\u003Ch2\u003EConclusion\u003C/h2\u003E\r\n\u003Cp\u003EData interoperability is the most mature of the three pillars, but &quot;most mature&quot; isn't the same as &quot;done.&quot; The gap between what works for standard analytics today and what's needed for AI/ML workloads, decentralized write environments and cross-region/cross-cloud situations is real enough to need a plan before you're six months in and paying the price for learning the hard way.\u003C/p\u003E\r\n\u003Cp\u003ETo learn more, check out \u003Ca href=\"http://www.snowflake.com/en/blog/engineering/multi-engine-lakehouse-interoperability\" target=\"_blank\" rel=\"noopener noreferrer\"\u003EAn Architect's Guide to Multi-Engine Lakehouses: What's Solved and What Isn't\u003C/a\u003E, which is a brief summary of all three pillars, and the other deep investigations into \u003Ca href=\"http://www.snowflake.com/en/blog/engineering/lakehouse-business-logic-interoperability\" target=\"_blank\" rel=\"noopener noreferrer\"\u003Ebusiness logic interoperability\u003C/a\u003E and \u003Ca href=\"http://www.snowflake.com/en/blog/engineering/lakehouse-data-governance-interoperability\" target=\"_blank\" rel=\"noopener noreferrer\"\u003Egovernance interoperability\u003C/a\u003E.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"}},":itemsOrder":["blog_text_814893266_","blog_title","blog_text_814893266","image_copy_copy","blog_text_copy","blog_title_copy","blog_text_copy_1759338821","image_copy","blog_text_copy_628289373","blog_title_copy_copy","blog_text_copy_905954621","image","blog_text_copy_1844728901","image_copy_701714592","blog_text_copy_447586789","blog_title_copy_copy_280634654","blog_text_copy_1511145138","blog_title_copy_copy_848238316","blog_text"],":type":"wcm/foundation/components/responsivegrid"},"responsivegrid_premium_content_banner":{"columnCount":12,"columnClassNames":{},"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","appliedCssClassNames":"snowflake-responsive-component-top-padding-medium",":items":{},":itemsOrder":[],":type":"wcm/foundation/components/responsivegrid"},"container_author_chip":{"layout":"RESPONSIVE_GRID","columnCount":12,"columnClassNames":{"author_chip":"aem-GridColumn aem-GridColumn--default--12"},"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","id":"container-be93b60db7",":items":{"author_chip":{"id":"author-chip-fb60402191","title":{"id":"title","type":"heading2","lines":["Learn more about the authors"],":type":"snowflake-site/components/title-v2"},"authors":[{"authorImage":{"id":"image-8e36e5b7c2","height":"854","src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--d5dae03d-5733-42e0-813e-f92748305643/jason-hughes.jpg?quality=85&preferwebp=true","lazyEnabled":true,"width":"854",":type":"snowflake-site/components/image"},"authorCta":{"id":"button-fb8fa794b5","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/blog/authors/jason-hughes/"},"linkTargetContentType":"DOCUMENT_LEARN",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_INTERNAL","text":"Jason Hughes"},"authorTitle":"Principal Data Platform Architect, Applied Field Engineering"},{"authorImage":{"id":"image-b92d9f0cb4","height":"800","src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--85428841-cdd8-4cae-b3d0-4003575f4937/jim-lebonitte.png?quality=85&preferwebp=true","lazyEnabled":true,"width":"800",":type":"snowflake-site/components/image"},"authorCta":{"id":"button-c30620369e","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/blog/authors/jim-lebonitte/"},"linkTargetContentType":"DOCUMENT_LEARN",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_INTERNAL","text":"Jim Lebonitte"},"authorTitle":"Director, GTM Platform & Architecture AFE"}],":type":"snowflake-site/components/blog/author-chip"}},":itemsOrder":["author_chip"],"appliedCssClassNames":"snowflake-responsive-component-top-padding-medium",":type":"snowflake-site/components/container"}},":itemsOrder":["container_hero","responsivegrid_content","responsivegrid_premium_content_banner","container_author_chip"],":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container"},"flexible_column_content_container_2":{"layout":"SIMPLE","id":"container-21143f8b4a",":items":{"blog_table_of_content":{"id":"blog-table-of-content-e1dd806e7d","tableOfContents":[{"headingText":"Current state assessment","level":"h2","anchorId":"#current-state-assessment","hierarchicalChildrenStructure":[]},{"headingText":"What works well today","level":"h2","anchorId":"#what-works-well-today","hierarchicalChildrenStructure":[]},{"headingText":"What requires work but has solutions","level":"h2","anchorId":"#what-requires-work-but-has-solutions","hierarchicalChildrenStructure":[]},{"headingText":"Open problems and active work","level":"h2","anchorId":"#open-problems-and-active-work","hierarchicalChildrenStructure":[]},{"headingText":"Diagnostic questions","level":"h2","anchorId":"#diagnostic-questions","hierarchicalChildrenStructure":[]}],":type":"snowflake-site/components/blog/blog-table-of-content"}},":itemsOrder":["blog_table_of_content"],":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container"},":type":"snowflake-site/components/flexible-column-container","isBlogPage":true,"isActiveTOC":true},"related_content":{"id":"related-content-08e3980403","relatedContent":[],":type":"snowflake-site/components/blog/related-content","isBlogPage":true}},":itemsOrder":["flexible_column_container","related_content"],"appliedCssClassNames":"snowflake-container",":type":"snowflake-site/components/container"}},":itemsOrder":["container_breadcrumb","container_main_content"],":type":"wcm/foundation/components/responsivegrid"},"container_47873732":{"additionalClasses":"section--blog-newsletter","layout":"RESPONSIVE_GRID","columnCount":12,"columnClassNames":{"flexible_column_cont":"aem-GridColumn aem-GridColumn--default--12"},"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","id":"container-83783a9d70",":items":{"flexible_column_cont":{"id":"flexible-column-container-2c155337f6","type":"1-column","alignColumns":"top","containerMaxWidth":"extra-large","topPadding":"small","bottomPadding":"none","spaceBetween":"small","reverseOnMobile":false,"carouselOnMobile":false,"propertiesCSSClasses":"section--blog-newsletter","backgroundImageOption":"none","flexible_column_content_container_1":{"layout":"SIMPLE","id":"container-095f316478",":items":{"marketo_v2":{"id":"marketo-v2-ac476cfb1e","marketoForm":{"hidden":null,"successUrl":null,"edit":false,"formId":"3320","script":null,"values":null},"title":{"id":"title","type":"heading3","lines":["Subscribe to our blog newsletter","Get the best, coolest and latest delivered to your inbox each week"],":type":"snowflake-site/components/title-v2"},"munchkinId":"252-RFO-227","serverInstance":"252-RFO-227.mktoweb.com","formConfigured":true,"marketoConfigured":true,":type":"snowflake-site/components/form/marketo-v2"},"text":{"id":"text-3fa1f0e21d","additionalClasses":"newsletter-disclaimer","text":"\u003Cp\u003EBy submitting this form, I understand Snowflake will process my personal information in accordance with their Privacy Notice.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/text"}},":itemsOrder":["marketo_v2","text"],":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container"},":type":"snowflake-site/components/flexible-column-container","isBlogPage":true,"isActiveTOC":true}},":itemsOrder":["flexible_column_cont"],"appliedCssClassNames":"snowflake-container",":type":"snowflake-site/components/container"},"experiencefragment-pre-footer":{"id":"experiencefragment-329fb5a459","localizedFragmentVariationPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/get-started-pre-footer/get-started-pre-footer/jcr:content","configured":true,":type":"snowflake-site/components/experiencefragment","xfModelPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/get-started-pre-footer/get-started-pre-footer.xfmodel.json?callerPage=/content/snowflake-site/global/en/blog/engineering/lakehouse-data-interoperability-guide"},"markup_editor":{"id":"markup-editor-ed0babdb47","title":"Page CSS","cssContent":"@media screen and (min-width:768px){.snowflake-blog-author-chip-wrapper{justify-content:flex-start}.snowflake-blog-related-content-on-blog-page{max-width:1408px;margin-left:auto;margin-right:auto}.snowflake-text{font-family:Lato,sans-serif;font-weight:400;font-size:16px;line-height:24px}}.section--blog-newsletter{max-width:none;width:100%;padding-left:0;padding-right:0;margin-left:0;margin-right:0;margin-bottom:0}.section--blog-newsletter .mktoField{background-color:transparent !important}.section--blog-newsletter\u003E.container{padding-left:0;padding-right:0}@media screen and (min-width:768px){.section--blog-newsletter\u003E.container{padding-left:0;padding-right:0}}.newsletter-disclaimer p{font-size:14px !important}.section--blog-newsletter .snowflake-marketo-form-container{margin-bottom:24px;background-color:#f6f9fa;gap:48px;box-shadow:none}.section--blog-newsletter .snowflake-title p.snowflake-title-line:first-child{font-family:Texta;font-size:24px;line-height:26px;font-weight:700;margin-bottom:4px}.section--blog-newsletter .snowflake-title p.snowflake-title-line{text-transform:none;font-family:\"Lato\",sans-serif;font-size:16px;line-height:24px;font-weight:normal}@media screen and (min-width:1024px){.section--blog-newsletter .snowflake-marketo-form-container{display:flex;justify-content:center}.section--blog-newsletter .snowflake-title .snowflake-title-line{text-align:left}.section--blog-newsletter .snowflake-marketo-form .mktoFormRow:has(\u003E input[type=\"hidden\"]){flex-grow:0}.section--blog-newsletter .snowflake-marketo-form{display:flex;width:50% !important}.section--blog-newsletter .snowflake-marketo-form .mktoButtonRow{flex-grow:0;width:auto !important;margin-left:0;margin-right:0}.section--blog-newsletter .snowflake-marketo-form .mktoFormRow{flex-grow:1}.section--blog-newsletter\u003E.container{padding-left:0;padding-right:0}.section--blog-newsletter .snowflake-marketo-form-title{width:50%;margin-bottom:0 !important}.section--blog-newsletter .center .snowflake-title{align-items:flex-start}}.snowflake-sub-navigation a.snowflake-sub-navigation-primary-link{width:auto !important}.snowflake-blog-hero{align-items:stretch !important}",":type":"snowflake-site/components/markup-editor","isGSAPEnabled":false},"markup_editor-table":{"id":"markup-editor-56bfd85449","title":"Table Styling CSS","cssContent":"#snowflake-blog-template-main-container table{width:100%;background-color:var(--ui-background-01);border-collapse:collapse;border:2px solid var(--ui-background-09);font-family:'Lato',sans-serif;color:var(--ui-background-09)}#snowflake-blog-template-main-container table thead{background-color:var(--ui-01)}#snowflake-blog-template-main-container table th,#snowflake-blog-template-main-container table td{border:2px solid var(--ui-background-09);padding:var(--spacing-01)}",":type":"snowflake-site/components/markup-editor","isGSAPEnabled":false},"experiencefragment-footer":{"id":"experiencefragment-4a65318187","localizedFragmentVariationPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/footer/master/jcr:content","configured":true,":type":"snowflake-site/components/experiencefragment","xfModelPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/footer/master.xfmodel.json?callerPage=/content/snowflake-site/global/en/blog/engineering/lakehouse-data-interoperability-guide"}},":itemsOrder":["experiencefragment-banner","experiencefragment-header","experiencefragment-sub-header","responsivegrid","container_47873732","experiencefragment-pre-footer","markup_editor","markup_editor-table","experiencefragment-footer"],":type":"wcm/foundation/components/responsivegrid"}},":itemsOrder":["root"],":mappedPath":"/en/blog/engineering/lakehouse-data-interoperability-guide/",":type":"snowflake-site/components/structure/page",":hierarchyType":"page",":path":"/content/snowflake-site/global/en/blog/engineering/lakehouse-data-interoperability-guide","isPasswordProtected":false,"analyticsContentTags":["snowflake-site:taxonomy/blog/engineering-blog/data-engineering"],"analyticsEnabled":true,"coveoConfig":{"searchHub":"snowflake.com","pipeline":"snowflake.com","apiKey":"xx335921a6-2a0a-40f2-a167-e390b4766c3d","organizationId":"snowflakecomputingproduction8neljofn"},"analyticsDebugMode":false,"analyticsData":{"excludeFromAnalytics":false,"subCategory":"","pageType":"homepage","templateName":"blog-page","siteName":"snowflake","pageUrl":"/content/snowflake-site/global/en/blog/engineering/lakehouse-data-interoperability-guide","language":"en","category":"general","pageName":"Operationalizing Data Interoperability in Multi-Engine Lakehouses","contentTags":["snowflake-site:taxonomy/blog/engineering-blog/data-engineering"]},"locale":"en"}
  