{"templateName":"base-page-template54","cssClassNames":"page basicpage summit-page","canonicalLink":"https://www.snowflake.com/en/artificial-intelligence/machine-learning/model-training/data-augmentation/","robotsTags":[],"allowedRenditionsWidth":["320","480","640","768","960","1200","1440","1920"],"language":"en","description":"Data augmentation expands training data with useful variation to improve model coverage and reliability. Explore key methods, use cases and best practices.","title":"Data Augmentation: Methods and Best Practices | Snowflake","analyticsPageType":"homepage","analyticsCategory":"general","analyticsSubCategory":"","excludeFromAnalytics":false,":mappedPath":"/en/artificial-intelligence/machine-learning/model-training/data-augmentation/",":type":"snowflake-site/components/structure/page",":items":{"root":{"columnCount":12,"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","columnClassNames":{"markup_editor_928258845":"aem-GridColumn aem-GridColumn--default--12","experiencefragment-banner":"aem-GridColumn aem-GridColumn--default--12","experiencefragment-header":"aem-GridColumn aem-GridColumn--default--12","responsivegrid":"aem-GridColumn aem-GridColumn--default--12","markup_editor_597730182":"aem-GridColumn aem-GridColumn--default--12","experiencefragment-footer":"aem-GridColumn aem-GridColumn--default--12","experiencefragment":"aem-GridColumn aem-GridColumn--default--12","modal_container":"aem-GridColumn aem-GridColumn--default--12","markup_editor":"aem-GridColumn aem-GridColumn--default--12"},":items":{"experiencefragment-banner":{"id":"experiencefragment-23ca4d3247","localizedFragmentVariationPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/pushdown-banner/master/jcr:content","configured":true,"xfModelPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/pushdown-banner/master.xfmodel.json",":type":"snowflake-site/components/experiencefragment"},"experiencefragment-header":{"id":"experiencefragment-212a884236","localizedFragmentVariationPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/mega-nav-header/master/jcr:content","configured":true,"languageNavPath":"/content/snowflake-site/global/en/artificial-intelligence/machine-learning/model-training/data-augmentation.languagenav.json","xfModelPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/mega-nav-header/master.xfmodel.json",":type":"snowflake-site/components/experiencefragment"},"responsivegrid":{"columnCount":12,"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","columnClassNames":{"flexible_column_cont_939100716":"aem-GridColumn aem-GridColumn--default--12","flexible_column_cont":"aem-GridColumn aem-GridColumn--default--12","flexible_column_cont_1158003461":"aem-GridColumn aem-GridColumn--default--12","flexible_column_cont_663228916":"aem-GridColumn aem-GridColumn--default--12","flexible_column_cont_912630531":"aem-GridColumn aem-GridColumn--default--12","flexible_column_cont_1398138236":"aem-GridColumn aem-GridColumn--default--12","flexible_column_cont_1786318617":"aem-GridColumn aem-GridColumn--default--12","flexible_column_cont_1467213961":"aem-GridColumn aem-GridColumn--default--12","flexible_column_cont_1377146023":"aem-GridColumn aem-GridColumn--default--12"},":items":{"flexible_column_cont":{"id":"flexible-column-container-d205ff0304","propertiesId":"hub-hero-breadcrumbs","type":"1-column","alignColumns":"top","containerMaxWidth":"extra-large","topPadding":"small","bottomPadding":"none","spaceBetween":"small","reverseOnMobile":false,"carouselOnMobile":false,"propertiesCSSClasses":"page-section","backgroundImageOption":"none","flexible_column_content_container_1":{"layout":"SIMPLE","id":"container-d13891ca0b",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"breadcrumb":{"id":"breadcrumb-8bd8191674","items":[{"id":"breadcrumb-8bd8191674-item-f710b50fc9","link":{"valid":true,"url":"/en/artificial-intelligence/machine-learning/"},"active":false,"current":false,"title":"Machine Learning",":type":"snowflake-site/components/structure/page","appliedCssClassNames":"summit-page"},{"id":"breadcrumb-8bd8191674-item-7b414e79ce","link":{"valid":true,"url":"/en/artificial-intelligence/machine-learning/model-training/"},"active":false,"current":false,"title":"ML Model Training",":type":"snowflake-site/components/structure/page","appliedCssClassNames":"summit-page"},{"id":"breadcrumb-8bd8191674-item-37b3f092d5","link":{"valid":true,"url":"/en/artificial-intelligence/machine-learning/model-training/data-augmentation/"},"active":true,"current":true,"title":"Data Augmentation",":type":"snowflake-site/components/structure/page","appliedCssClassNames":"summit-page"}],":type":"snowflake-site/components/breadcrumb"}},":itemsOrder":["breadcrumb"]},"isBlogPage":false,"isActiveTOC":false,":type":"snowflake-site/components/flexible-column-container","appliedCssClassNames":"snowflake-flexible-column-container-gray-10-bg"},"flexible_column_cont_939100716":{"id":"flexible-column-container-111486b18e","propertiesId":"hub-hero","type":"2-column-even","alignColumns":"center","containerMaxWidth":"extra-large","topPadding":"extra-small","bottomPadding":"extra-small","spaceBetween":"small","reverseOnMobile":false,"carouselOnMobile":false,"propertiesCSSClasses":"page-section","backgroundImageOption":"none","flexible_column_content_container_1":{"layout":"SIMPLE","id":"container-078d597e35",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"title_v2":{"id":"title-v2-27fe69f40b","additionalClasses":"hub-hero__headline","type":"heading1","lines":["Data Augmentation: How to Expand Training Data Without Distorting the Signal"],":type":"snowflake-site/components/title-v2","appliedCssClassNames":"left-alignment"},"text":{"id":"text-97d2bf359e","additionalClasses":"hub-hero__subheadline","text":"\u003Cp\u003EMore training data doesn’t automatically produce a better model. Learn how data augmentation introduces useful variation, improves coverage and helps models perform more reliably in real-world conditions.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-size-regular text-color-text-05"},"container":{"additionalClasses":"hub-hero__authors","layout":"RESPONSIVE_GRID","columnCount":12,"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","columnClassNames":{"content_chip_copy":"aem-GridColumn aem-GridColumn--default--12","content_chip":"aem-GridColumn aem-GridColumn--default--12"},"id":"container-a3029191e2",":type":"snowflake-site/components/container",":items":{"content_chip":{"id":"content-chip-5c4ced4ef8","cta":{"id":"cta","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/blog/authors/Laurie-Macpherson/"},"linkTargetContentType":"GENERIC",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_INTERNAL","text":"Read bio"},"image":{"id":"image","height":"800","isLcpImage":true,"src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--a7e9fdff-6f08-4edc-9cd1-e213bb234aaa/laurie-macpherson.jpg?preferwebp=true&quality=85","lazyEnabled":true,"alt":"Laurie MacPherson","width":"800",":type":"snowflake-site/components/image"},"headline":{"id":"title","type":"heading5","lines":["Laurie MacPherson","Technical Writer, Snowflake"],":type":"snowflake-site/components/title-v2"},":type":"snowflake-site/components/content-chip"},"content_chip_copy":{"id":"content-chip-43d0e7c0b1","cta":{"id":"cta","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/blog/authors/david-gaule/"},"linkTargetContentType":"GENERIC",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_INTERNAL","text":"Read bio"},"image":{"id":"image","height":"512","isLcpImage":false,"src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--fd9454ea-3b59-4d19-95be-534774cbd226/david.jpg?preferwebp=true&quality=85","lazyEnabled":true,"alt":"David Gaule","width":"512",":type":"snowflake-site/components/image"},"headline":{"id":"title","type":"heading5","lines":["David Gaule","Technical Editor, Snowflake"],":type":"snowflake-site/components/title-v2"},":type":"snowflake-site/components/content-chip"}},":itemsOrder":["content_chip","content_chip_copy"],"appliedCssClassNames":"snowflake-responsive-container-inner-padding-small"}},":itemsOrder":["title_v2","text","container"],"appliedCssClassNames":"snowflake-responsive-container-inner-padding-extra-small"},"flexible_column_content_container_2":{"additionalClasses":"hub-hero__video-column","layout":"SIMPLE","id":"container-33fde76c6e",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"youtube":{"id":"embed-721cf2171e","youtubeVideoId":"-HWNc-Hd90U","layout":"responsive","youtubeAspectRatio":"56.25","youtubeAutoPlay":false,"youtubeLoop":false,"youtubeMute":false,"youtubePlaysInline":false,"youtubeRel":false,"embeddableResourceType":"core/wcm/components/embed/v1/embed/embeddable/youtube","type":"EMBEDDABLE",":type":"snowflake-site/components/youtube"}},":itemsOrder":["youtube"]},"isBlogPage":false,"isActiveTOC":false,":type":"snowflake-site/components/flexible-column-container","appliedCssClassNames":"snowflake-flexible-column-container-gray-10-bg"},"flexible_column_cont_1398138236":{"id":"flexible-column-container-a57b8a100e","propertiesId":"hub-hero-related-topics","type":"1-column","alignColumns":"top","containerMaxWidth":"extra-large","topPadding":"extra-small","bottomPadding":"medium","spaceBetween":"small","reverseOnMobile":false,"carouselOnMobile":false,"propertiesCSSClasses":"page-section","backgroundImageOption":"none","flexible_column_content_container_1":{"additionalClasses":"related-topics-outer-container border-top","layout":"SIMPLE","id":"container-249046863d",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"text_894059747":{"id":"text-ef8b54e1da","additionalClasses":"seo-hub-hero__related-topic-label","text":"\u003Cp\u003EML Model Training Topics:\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-size-regular text-color-text-05"},"text":{"id":"text-702e15a41b","additionalClasses":"related-topics ","text":"\u003Cul\u003E\r\n\u003Cli\u003E\u003Ca href=\"https://www.snowflake.com/en/artificial-intelligence/machine-learning/model-training/gradient-descent/\"\u003EGradient Descent\u003C/a\u003E\u003C/li\u003E\r\n\u003C/ul\u003E\r\n","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-size-small"}},":itemsOrder":["text_894059747","text"]},"isBlogPage":false,"isActiveTOC":false,":type":"snowflake-site/components/flexible-column-container","appliedCssClassNames":"snowflake-flexible-column-container-gray-10-bg"},"flexible_column_cont_663228916":{"id":"flexible-column-container-12f6223bcd","propertiesId":"hub-body","type":"2-column-60-40","alignColumns":"top","containerMaxWidth":"extra-large","topPadding":"medium","bottomPadding":"medium","spaceBetween":"small","reverseOnMobile":false,"carouselOnMobile":false,"propertiesCSSClasses":"page-section","backgroundImageOption":"none","flexible_column_content_container_1":{"additionalClasses":"longform-content","layout":"SIMPLE","id":"hub-body-content",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"callout__0":{"id":"text-318f233c32","additionalClasses":"callout callout--general","text":"\u003Cp\u003E\u003Cstrong\u003EDATA AUGMENTATION DEFINED\u003C/strong\u003E\u003C/p\u003E\n\u003Cp\u003EData augmentation is the practice of enriching a training data set with modified or newly generated examples that preserve the information relevant to the model’s task.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-size-regular text-color-text-05"},"text__0":{"id":"text-b43d648b28","text":"\u003Cp\u003EAs foundation models become more powerful — and easier for organizations to access — the real competitive advantage is shifting toward data: the data used to train models, adapt them to specific tasks and evaluate how well they perform. That shift is putting fresh attention on a familiar \u003Ca href=\"https://www.snowflake.com/en/artificial-intelligence/machine-learning/\"\u003Emachine learning\u003C/a\u003E technique: data augmentation.\u003C/p\u003E\n\u003Cp\u003EData augmentation helps models experience more of the variety they’re likely to encounter once they’re deployed. It isn’t just about making a data set bigger. The goal is to make it more representative, resilient and useful.\u003C/p\u003E\n\u003Cp\u003EToday, augmentation can mean anything from simple geometric transformations to entirely new examples generated by \u003Ca href=\"https://www.snowflake.com/en/artificial-intelligence/generative-ai/diffusion-model/\"\u003Ediffusion models\u003C/a\u003E or \u003Ca href=\"https://www.snowflake.com/en/fundamentals/large-language-model/\"\u003Elarge language models\u003C/a\u003E (LLMs). Practitioners typically turn to these techniques when the original data set lacks enough variation, misses important classes or doesn’t capture the edge cases that show up in the real world. The challenge is choosing methods that expand the model’s experience without blurring the signal it needs to learn.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"},"title_what-is-data-augmentation":{"id":"title-v2-90293ff979","additionalClasses":"anchor-title anchor-title--what-is-data-augmentation","type":"heading2","lines":["What is data augmentation?"],":type":"snowflake-site/components/title-v2"},"text_what-is-data-augmentation_0":{"id":"text-e265a39143","text":"\u003Cp\u003EData augmentation creates additional training examples by transforming, recombining or generating data for a particular training task. Practitioners use it to add variation, improve coverage of underrepresented cases and reduce the model’s dependence on patterns that may not hold outside the training data.\u003C/p\u003E\n\u003Cp\u003ETraditional augmentation operates directly on observed data. An image pipeline might crop, rotate or adjust the brightness of a source image, or a language pipeline might mask tokens or produce a paraphrase. In each case, the transformation should be label-preserving, meaning the altered example may look or sound different, but it should retain the property represented by the target label.\u003C/p\u003E\n\u003Ch3\u003EData augmentation vs. synthetic data\u003C/h3\u003E\n\u003Cp\u003EThe distinction between data augmentation and \u003Ca href=\"https://www.snowflake.com/en/fundamentals/synthetic-data/\"\u003Esynthetic data\u003C/a\u003E is useful, although modern generative methods make it less absolute than it once was. Traditional augmentation modifies or recombines real examples. Synthetic data is artificially generated and may simulate people, events, transactions or environments without corresponding one-to-one source records. Generative augmentation sits at the overlap, typically using diffusion models or LLMs to produce additional examples for a specific training task.\u003C/p\u003E\n\u003Ch3\u003EData augmentation vs. feature engineering\u003C/h3\u003E\n\u003Cp\u003EData augmentation also differs from \u003Ca href=\"https://www.snowflake.com/en/artificial-intelligence/machine-learning/feature-engineering/\"\u003Efeature engineering\u003C/a\u003E. Data augmentation creates additional training examples, while feature engineering changes how each example is represented to the model — for example, feature engineering might convert a customer’s sign-up date into account age or summarize many transactions as total spend over the past 30 days.\u003C/p\u003E\n\u003Cp\u003E\u003Cem\u003EHear from Tuhin Ghosh, Head of Data Science for the Platform Product Group at Coinbase, as he shares how Snowflake ML capabilities are simplifying the way Coinbase delivers machine learning at scale:\u003C/em\u003E\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"},"yt_what-is-data-augmentation_0":{"id":"embed-6aac4bdf67","youtubeVideoId":"Eh33vfAP30U","layout":"responsive","youtubeAspectRatio":"56.25","youtubeAutoPlay":false,"youtubeLoop":false,"youtubeMute":false,"youtubePlaysInline":false,"youtubeRel":false,"embeddableResourceType":"core/wcm/components/embed/v1/embed/embeddable/youtube","type":"EMBEDDABLE",":type":"snowflake-site/components/youtube"},"title_how-data-augmentation-improves-model-training":{"id":"title-v2-68a0dda659","additionalClasses":"anchor-title anchor-title--how-data-augmentation-improves-model-training","type":"heading2","lines":["How data augmentation improves model training"],":type":"snowflake-site/components/title-v2"},"text_how-data-augmentation-improves-model-training_0":{"id":"text-f46c6b6fdd","text":"\u003Cp\u003EData augmentation changes the examples a model sees during training, which can improve generalization when the added variation reflects the task and expected operating conditions. It’s commonly used to reduce overfitting, expand coverage of underrepresented cases and make a model less dependent on incidental patterns in the original data set. Whether it works depends on what changes, what stays fixed and how those augmented examples enter the training process.\u003C/p\u003E\n\u003Ch3\u003EReducing overfitting\u003C/h3\u003E\n\u003Cp\u003EA model can overfit when it learns from every pattern available in its training data, including patterns that don’t generalize to new examples. A classifier trained mostly on daytime road images, for example, may associate brightness with a particular object class. Data augmentation introduces controlled variation so that these incidental characteristics become less reliable shortcuts.\u003C/p\u003E\n\u003Cp\u003EAugmentation often works alongside regularization, dropout and early stopping. Those methods constrain the model or training process directly, while augmentation changes the data from which the model learns.\u003C/p\u003E\n\u003Ch3\u003ELearning which variation should leave the prediction unchanged\u003C/h3\u003E\n\u003Cp\u003EData augmentation encourages the model to learn invariance, meaning that its prediction should remain stable under a defined transformation. In the daytime images example, when the model is trained on images at several brightness levels, the model must rely more heavily on characteristics that remain consistent across the variations.\u003C/p\u003E\n\u003Cp\u003EThe appropriate invariance depends on the task. Rotation may preserve the label when classifying flowers photographed from arbitrary angles, yet invalidate an example when orientation carries diagnostic, geographic or mechanical meaning. A synonym replacement may preserve the topic of a document while changing its sentiment, intent or factual meaning. Cropping can remove the object or context that supports the label.\u003C/p\u003E\n\u003Cp\u003EAugmentation policies need to reflect plausible variation in the target environment. A transformation is useful only when it changes an incidental characteristic without removing or altering the information the model needs to make the correct prediction.\u003C/p\u003E\n\u003Ch3\u003EApplying augmentation online or offline\u003C/h3\u003E\n\u003Cp\u003EPractitioners can create augmented examples in advance or generate them dynamically during training.\u003C/p\u003E\n\u003Cul\u003E\n\u003Cli\u003E\u003Cstrong\u003EOffline augmentation\u003C/strong\u003E creates and stores transformed examples in advance. This makes the resulting data easier to inspect, validate and reproduce, although it requires additional storage and limits the model to a fixed set of variants.\u003C/li\u003E\n\u003Cli\u003E\u003Cstrong\u003EOnline augmentation\u003C/strong\u003E samples transformations during training. The same source example may appear differently across epochs, exposing the model to a wider range of variation without storing every version. Reproducing the run requires tracking the random seed, transformation parameters and pipeline version.\u003C/li\u003E\n\u003C/ul\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"},"card_v2_how-data-augmentation-improves-model-training_0":{"id":"card-v2-767d7fbd8d","additionalClasses":"seo-customer","configurationStatus":{"configured":true,"message":""},"image":{"id":"image","height":"351","isLcpImage":false,"src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--c655902a-70fa-475f-aa6a-38d5b3ed553e/skai%25403x.png?preferwebp=true&quality=85","lazyEnabled":true,"alt":"Skai Logo","width":"624",":type":"snowflake-site/components/image"},"type":"content-card","title":{"id":"title","type":"heading4","lines":["Skai"],":type":"snowflake-site/components/title-v2"},"button":{"id":"button","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/customers/all-customers/case-study/skai/"},"linkTargetContentType":"DOCUMENT",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_INTERNAL","text":"Read the full case study"},"text":{"id":"text","text":"\u003Cp\u003ESkai uses Snowflake Cortex AI and Snowpark to help customers rapidly categorize products across major ecommerce platforms such as Amazon, Target and Walmart. With an LLM-powered categorization workflow built in Snowflake, Skai deployed a production-ready tool in just two days instead of an estimated month, categorized nearly 100,000 products in 30 days and achieved 99.98% category relevance after applying Snowpark-based validation filters. Results reported in the Skai case study as of Sept. 2024.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/text"},"layoutStyle":"horizontal",":type":"snowflake-site/components/card-v2"},"title_data-augmentation-methods-by-modality-and-task":{"id":"title-v2-087a4b5ff3","additionalClasses":"anchor-title anchor-title--data-augmentation-methods-by-modality-and-task","type":"heading2","lines":["Data augmentation methods by modality and task"],":type":"snowflake-site/components/title-v2"},"text_data-augmentation-methods-by-modality-and-task_0":{"id":"text-4d6d55f8e2","text":"\u003Cp\u003EAugmentation works differently across images, text, audio and tabular data because each modality preserves meaning in a different way. The method has to match both the structure of the data and the task the model is learning. The techniques below illustrate the common approaches and the constraints practitioners need to consider when applying them.\u003C/p\u003E\n\u003Ch3\u003EComputer vision\u003C/h3\u003E\n\u003Cp\u003E\u003Ca href=\"https://www.snowflake.com/en/artificial-intelligence/computer-vision/\"\u003EComputer vision\u003C/a\u003E augmentation changes how an image appears while preserving the object or scene the model is meant to recognize. The methods generally fall into four categories:\u003C/p\u003E\n\u003Cul\u003E\n\u003Cli\u003E\u003Cstrong\u003EGeometric transformations\u003C/strong\u003E reposition or reshape the image through operations such as rotation, translation, flipping, cropping and scaling.\u003C/li\u003E\n\u003Cli\u003E\u003Cstrong\u003EPhotometric transformations\u003C/strong\u003E change visual properties including brightness, contrast, color, saturation and exposure.\u003C/li\u003E\n\u003Cli\u003E\u003Cstrong\u003ENoise and occlusion techniques\u003C/strong\u003E introduce blur, Gaussian noise or partially obscure parts of an image, encouraging the model to rely less on individual pixels or small visual cues.\u003C/li\u003E\n\u003Cli\u003E\u003Cstrong\u003ESample-combination methods\u003C/strong\u003E create new training examples from multiple images. Mixup blends two images and their labels, while CutMix inserts a region from one image into another and adjusts the target accordingly.\u003C/li\u003E\n\u003C/ul\u003E\n\u003Cp\u003EThe transformation choice depends on what the model is predicting. Horizontal flipping may be appropriate for classifying many household objects, for example, because the object remains the same when its left and right sides are reversed. But the same transformation could change the meaning of road signs, reverse medically relevant laterality or misrepresent equipment whose components have different functions on each side. Cropping creates a similar trade-off: It can teach the model to recognize an object when framing changes, but an aggressive crop may remove the feature that makes the label relevant.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"},"callout_data-augmentation-methods-by-modality-and-task_0":{"id":"text-73f5c5d258","additionalClasses":"callout callout--warning","text":"\u003Cp\u003E\u003Cstrong\u003ECOMMON PITFALL\u003C/strong\u003E\u003C/p\u003E\n\u003Cp\u003EIt’s easy to assume that familiar transformations are always safe. But flipping, rotating or cropping an image can change spatial relationships or the annotations used for detection and segmentation.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-size-regular text-color-text-05"},"text_data-augmentation-methods-by-modality-and-task_1":{"id":"text-1825c3f4d2","text":"\u003Ch3\u003EText and natural language processing\u003C/h3\u003E\r\n\u003Cp\u003EText augmentation ranges from small edits to fully generated examples. The methods differ in how much of the original wording they change:\u003C/p\u003E\r\n\u003Cul\u003E\r\n\u003Cli\u003E\u003Cb\u003ELocal edits\u003C/b\u003E modify individual tokens through synonym replacement, insertion, deletion or masking.\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EBack-translation\u003C/b\u003E translates text into another language and back again, often producing different wording with similar meaning.\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EContextual replacement\u003C/b\u003E uses a language model to choose substitutions that fit the surrounding sentence more naturally than a static thesaurus.\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003ELLM-based augmentation\u003C/b\u003E can produce paraphrases, class-conditioned examples, counterexamples and rewrites that vary tone, structure or terminology.\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003ERetrieval-assisted generation\u003C/b\u003E uses retrieved domain-specific context to guide generation, helping constrain outputs when the task depends on specialized language or facts.\u003C/li\u003E\r\n\u003C/ul\u003E\r\n\u003Cp\u003EThe main risk is semantic drift. For tasks such as toxicity detection, intent classification or contract extraction, even a small wording change can move an example into a different class.\u003C/p\u003E\r\n\u003Cp\u003EFor this reason, text augmentation typically requires more than a quality check for grammar or fluency. Depending on the task, practitioners may use semantic-similarity measures, rule-based validation, model-assisted filtering or human review to confirm that the transformed example still supports the original label.\u003C/p\u003E\r\n\u003Ch3\u003EAudio and time-series data\u003C/h3\u003E\r\n\u003Cp\u003EAudio augmentation typically changes how a signal is recorded or heard while preserving its underlying content. These methods expose a model to differences in speakers, microphones and acoustic environments:\u003C/p\u003E\r\n\u003Cul\u003E\r\n\u003Cli\u003E\u003Cb\u003ETime stretching\u003C/b\u003E changes playback speed without necessarily changing pitch.\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EPitch shifting\u003C/b\u003E raises or lowers frequency.\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EVolume scaling\u003C/b\u003E adjusts loudness.\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003ETime shifting\u003C/b\u003E moves the audio forward or backward in time.\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EBackground-noise injection\u003C/b\u003E simulates different recording environments.\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EFrequency masking\u003C/b\u003E hides selected frequency bands in a spectrogram.\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003ETime masking\u003C/b\u003E removes short intervals from a spectrogram.\u003C/li\u003E\r\n\u003C/ul\u003E\r\n\u003Cp\u003EWhether these methods are appropriate depends on which part of the signal the model is supposed to recognize. A transformation can vary recording conditions or speaker characteristics when those properties are incidental to the task, but it should preserve any feature that determines the target.\u003C/p\u003E\r\n\u003Cp\u003ETime-series augmentation follows the same general principle, although the constraints are often stricter. Common techniques include:\u003C/p\u003E\r\n\u003Cul\u003E\r\n\u003Cli\u003E\u003Cb\u003EJittering\u003C/b\u003E, which adds small amounts of random noise\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EScaling\u003C/b\u003E, which changes the magnitude of values\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EWindow slicing\u003C/b\u003E, which trains on selected portions of a sequence\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003ETime warping\u003C/b\u003E, which stretches or compresses sections of a sequence\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003ESequence mixing\u003C/b\u003E, which combines information from multiple examples\u003C/li\u003E\r\n\u003C/ul\u003E\r\n\u003Cp\u003EThese methods can increase variation, but they must preserve temporal order, seasonality and relationships among channels. Even a small change to the data can make the sequence unrealistic, change when an event occurs or obscure the anomaly the model is supposed to detect.\u003C/p\u003E\r\n\u003Ch3\u003ETabular data\u003C/h3\u003E\r\n\u003Cp\u003EUnlike images or text, tabular data rarely has an obvious “safe” transformation. Each row represents a real entity or event whose values are related to one another. The goal isn't simply to create variation — it's to create new examples that remain statistically and logically plausible.\u003C/p\u003E\r\n\u003Cp\u003ECommon approaches include:\u003C/p\u003E\r\n\u003Cul\u003E\r\n\u003Cli\u003E\u003Cb\u003EFeature perturbation\u003C/b\u003E, which adjusts selected values within defined ranges\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EBootstrapping\u003C/b\u003E, which resamples observed records\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EInterpolation\u003C/b\u003E, which creates examples between nearby records\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003ESMOTE\u003C/b\u003E (synthetic minority over-sampling technique), which creates synthetic minority-class examples by interpolating between an observed minority example and selected minority-class neighbors in feature space\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EGenerative modeling\u003C/b\u003E, which attempts to reproduce conditional relationships across multiple fields\u003C/li\u003E\r\n\u003C/ul\u003E\r\n","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"},"title_generative-and-advanced-augmentation-methods":{"id":"title-v2-31bd7f4794","additionalClasses":"anchor-title anchor-title--generative-and-advanced-augmentation-methods","type":"heading2","lines":["Generative and advanced augmentation methods"],":type":"snowflake-site/components/title-v2"},"text_generative-and-advanced-augmentation-methods_0":{"id":"text-53da87d21a","text":"\u003Cp\u003EThe techniques covered above create variation by modifying, masking, resampling or recombining existing examples. Newer augmentation methods can go further, producing examples with different content, structure or combinations that aren't already represented in the source data.\u003C/p\u003E\r\n\u003Cp\u003EThese approaches include:\u003C/p\u003E\r\n\u003Cul\u003E\r\n\u003Cli\u003E\u003Cb\u003EGenerative augmentation\u003C/b\u003E, which uses models trained on existing data to create new examples. Diffusion models can generate or edit images, while LLMs can produce paraphrases, counterexamples and examples for underrepresented classes.\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EConditional generation\u003C/b\u003E, which gives practitioners more control over the output by specifying a class, attribute, scenario or other condition. This can help target gaps in the training set rather than generating examples indiscriminately.\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EAdversarial augmentation\u003C/b\u003E, which creates deliberately difficult examples that expose weaknesses near a model’s decision boundary. The goal is generally to expose the model to difficult, boundary-adjacent examples and improve robustness, including robustness to specified perturbations or attacks.\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EMultimodal augmentation\u003C/b\u003E, which generates or modifies related data together, such as an image and its caption, while attempting to preserve the relationship between them.\u003C/li\u003E\r\n\u003C/ul\u003E\r\n\u003Cp\u003EThese methods are useful when the missing variation can’t be expressed through a predefined change to an existing example. A team may need examples that introduce new combinations of conditions, represent a rare class or vary the semantic content rather than only the surface form.\u003C/p\u003E\r\n\u003Cp\u003EThe additional flexibility also increases the validation burden, however. Generated examples may look realistic or read naturally while carrying the wrong label, reproducing bias or introducing artifacts that the downstream model learns as shortcuts. For this reason, teams need to evaluate label consistency, diversity, domain validity and downstream performance, while also tracking the source data, model version and generation settings associated with each example.\u003C/p\u003E\r\n\u003Cp\u003EGenerative and advanced methods also need to be evaluated for distribution shift. Augmentation intentionally changes the mix of examples used for training, which can help when it adds variation the model is likely to encounter later, but it can also create a new mismatch if generated examples overrepresent unlikely conditions, reflect artifacts of the generator or differ systematically from real data. Teams should compare the augmented data with production-relevant distributions and test whether gains that appeared in training hold up in evaluation.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"},"callout_generative-and-advanced-augmentation-methods_0":{"id":"text-5c75f0d6cd","additionalClasses":"callout callout--tip","text":"\u003Cp\u003E\u003Cstrong\u003EQUICK TIP\u003C/strong\u003E\u003C/p\u003E\n\u003Cp\u003EGenerate examples to fill a defined coverage gap, such as a rare class, scenario or phrasing pattern. Unconstrained generation can add volume without adding useful diversity.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-size-regular text-color-text-05"},"title_avoid-common-data-augmentation-errors":{"id":"title-v2-50b000e51c","additionalClasses":"anchor-title anchor-title--avoid-common-data-augmentation-errors","type":"heading2","lines":["Avoid common data augmentation errors"],":type":"snowflake-site/components/title-v2"},"text_avoid-common-data-augmentation-errors_0":{"id":"text-d0652b6f98","text":"\u003Cp\u003ESeveral implementation errors can make an apparently sophisticated augmentation pipeline less reliable than the unaugmented baseline.\u003C/p\u003E\n\u003Cul\u003E\n\u003Cli\u003E\u003Cstrong\u003ELabel corruption:\u003C/strong\u003E A transformation changes the property represented by the target. This can happen visibly, as when cropping removes the labeled object, or semantically, as when a paraphrase reverses the meaning of a sentence.\u003C/li\u003E\n\u003Cli\u003E\u003Cstrong\u003EUnrealistic examples:\u003C/strong\u003E Perturbations and generated records may violate physical conditions, language conventions or business rules. A model can learn from these records even when they could never appear in production.\u003C/li\u003E\n\u003Cli\u003E\u003Cstrong\u003ETrain-validation leakage:\u003C/strong\u003E Variants derived from the same source example appear across the training and validation sets. Because they share content, the validation result can overstate generalization. Split source records before augmentation, then keep every derived variant in the same partition as its source.\u003C/li\u003E\n\u003Cli\u003E\u003Cstrong\u003EBias amplification:\u003C/strong\u003E A generator may reproduce patterns from its pretraining data or emphasize already common features. Oversampling can also multiply biased or mislabeled source examples.\u003C/li\u003E\n\u003Cli\u003E\u003Cstrong\u003EDistorted class priors:\u003C/strong\u003E Balancing classes during training may improve learning, but it changes the label frequencies observed by the model. Evaluation and calibration should account for the frequencies expected in production.\u003C/li\u003E\n\u003Cli\u003E\u003Cstrong\u003ERedundant examples:\u003C/strong\u003E Thousands of nearly identical variants add compute without adding meaningful coverage. Diversity checks can show whether the augmented data explores a useful region of the input space.\u003C/li\u003E\n\u003Cli\u003E\u003Cstrong\u003EUntracked pipeline state:\u003C/strong\u003E Seeds, transformation parameters, source identifiers, model versions and library versions affect the resulting data set. Without that metadata, teams may be unable to reproduce a promising run or investigate an unexpected result.\u003C/li\u003E\n\u003C/ul\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"},"title_run-reproducible-data-augmentation-pipelines-on-snowflake":{"id":"title-v2-fa5e87ca3f","additionalClasses":"anchor-title anchor-title--run-reproducible-data-augmentation-pipelines-on-snowflake","type":"heading2","lines":["Run reproducible data augmentation pipelines on Snowflake"],":type":"snowflake-site/components/title-v2"},"text_run-reproducible-data-augmentation-pipelines-on-snowflake_0":{"id":"text-22a3cc9bd7","text":"\u003Cp\u003EA reproducible augmentation workflow needs three things: access to the governed source data, a record of how that data was transformed and a way to connect the resulting training set to the model that used it. Snowflake supports the workflow across data preparation, metadata tracking and model management.\u003C/p\u003E\n\u003Cp\u003ESnowpark provides Python APIs for processing data in Snowflake, allowing teams to implement preparation and augmentation logic close to governed source data. The design will vary by data type: tabular techniques may run directly over DataFrames, while image, audio and generative workloads may require file-based processing, specialized libraries or external model services.\u003C/p\u003E\n\u003Cp\u003ETeams can preserve the connection between original and augmented data by recording stable identifiers and metadata such as the transformation type, parameter values, random seed and generator version. For online augmentation, the pipeline may store the configuration needed to recreate the process rather than every transient example.\u003C/p\u003E\n\u003Cp\u003E\u003Ca href=\"https://docs.snowflake.com/en/developer-guide/snowflake-ml/feature-store/overview\"\u003ESnowflake Feature Store\u003C/a\u003E supports the management and reuse of features used during training, while \u003Ca href=\"https://docs.snowflake.com/en/developer-guide/snowflake-ml/model-registry/overview\"\u003ESnowflake Model Registry\u003C/a\u003E can connect trained models with their associated metadata. Together, these capabilities give practitioners a clearer record of the source data, augmentation logic and model output, making it easier to compare strategies and reproduce a successful run.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"},"title_learning-from-the-right-variation":{"id":"title-v2-181e95c32f","additionalClasses":"anchor-title anchor-title--learning-from-the-right-variation","type":"heading2","lines":["Learning from the right variation"],":type":"snowflake-site/components/title-v2"},"text_learning-from-the-right-variation_0":{"id":"text-997ea65173","text":"\u003Cp\u003EData augmentation broadens the range of examples a model learns from, but more data alone doesn't produce a better model. The value comes from introducing variation that reflects the conditions a system is expected to encounter while preserving the signal the model is meant to learn. As generative techniques expand what’s possible, successful augmentation increasingly depends on careful evaluation, reproducible pipelines and a clear understanding of which variation belongs in the training data — and which does not.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"},"callout_learning-from-the-right-variation_0":{"id":"text-91b79504bb","additionalClasses":"callout callout--general","text":"\u003Cp\u003E\u003Cstrong\u003EKEY TAKEAWAY\u003C/strong\u003E\u003C/p\u003E\n\u003Cp\u003EMore training data only helps when it teaches the model something useful. Augmentation should broaden coverage without changing the meaning, label or relationships the model needs to learn.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-size-regular text-color-text-05"}},":itemsOrder":["callout__0","text__0","title_what-is-data-augmentation","text_what-is-data-augmentation_0","yt_what-is-data-augmentation_0","title_how-data-augmentation-improves-model-training","text_how-data-augmentation-improves-model-training_0","card_v2_how-data-augmentation-improves-model-training_0","title_data-augmentation-methods-by-modality-and-task","text_data-augmentation-methods-by-modality-and-task_0","callout_data-augmentation-methods-by-modality-and-task_0","text_data-augmentation-methods-by-modality-and-task_1","title_generative-and-advanced-augmentation-methods","text_generative-and-advanced-augmentation-methods_0","callout_generative-and-advanced-augmentation-methods_0","title_avoid-common-data-augmentation-errors","text_avoid-common-data-augmentation-errors_0","title_run-reproducible-data-augmentation-pipelines-on-snowflake","text_run-reproducible-data-augmentation-pipelines-on-snowflake_0","title_learning-from-the-right-variation","text_learning-from-the-right-variation_0","callout_learning-from-the-right-variation_0"],"appliedCssClassNames":"snowflake-responsive-container-inner-padding-medium"},"flexible_column_content_container_2":{"additionalClasses":"hub-sidebar","layout":"SIMPLE","id":"hub-body-aside",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"container":{"additionalClasses":"sticky-sidebar","layout":"RESPONSIVE_GRID","columnCount":12,"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","columnClassNames":{"text_943981956_copy_":"aem-GridColumn aem-GridColumn--default--12","text_copy":"aem-GridColumn aem-GridColumn--default--12"},"id":"container-3e18a31c31",":type":"snowflake-site/components/container",":items":{"text_943981956_copy_":{"id":"text-8ef20e5503","additionalClasses":"eyebrow-text","text":"\u003Cp\u003EIn This Guide\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-size-regular"},"text_copy":{"id":"text-fb33b5e3c7","additionalClasses":"page-toc","text":"\u003Cul\u003E\u003Cli data-anchor=\"what-is-data-augmentation\"\u003EWhat is data augmentation?\u003C/li\u003E\u003Cli data-anchor=\"how-data-augmentation-improves-model-training\"\u003EHow data augmentation improves model training\u003C/li\u003E\u003Cli data-anchor=\"data-augmentation-methods-by-modality-and-task\"\u003EData augmentation methods by modality and task\u003C/li\u003E\u003Cli data-anchor=\"generative-and-advanced-augmentation-methods\"\u003EGenerative and advanced augmentation methods\u003C/li\u003E\u003Cli data-anchor=\"avoid-common-data-augmentation-errors\"\u003EAvoid common data augmentation errors\u003C/li\u003E\u003Cli data-anchor=\"run-reproducible-data-augmentation-pipelines-on-snowflake\"\u003ERun reproducible data augmentation pipelines on Snowflake\u003C/li\u003E\u003Cli data-anchor=\"learning-from-the-right-variation\"\u003ELearning from the right variation\u003C/li\u003E\u003C/ul\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-size-small text-color-text-05"}},":itemsOrder":["text_943981956_copy_","text_copy"],"appliedCssClassNames":"snowflake-responsive-container-inner-padding-medium"}},":itemsOrder":["container"],"appliedCssClassNames":"snowflake-responsive-container-inner-padding-small"},"isBlogPage":false,"isActiveTOC":false,":type":"snowflake-site/components/flexible-column-container"},"flexible_column_cont_1786318617":{"id":"flexible-column-container-d3754a7e97","propertiesId":"hub-faq","type":"2-column-40-60","alignColumns":"top","containerMaxWidth":"extra-large","topPadding":"large","bottomPadding":"large","spaceBetween":"small","reverseOnMobile":false,"carouselOnMobile":false,"backgroundImageOption":"none","flexible_column_content_container_1":{"layout":"SIMPLE","id":"hub-faq-intro",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"title_v2_copy":{"id":"title-v2-86acd8ca8e","additionalClasses":"hub-faq__headline","type":"heading2","lines":["Frequently Asked Questions"],":type":"snowflake-site/components/title-v2","appliedCssClassNames":"left-alignment"},"text_copy":{"id":"text-20a31d7a8c","additionalClasses":"hub-faq__subheadline","text":"\u003Cp\u003EYour common questions about data augmentation, answered by Snowflake experts.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"}},":itemsOrder":["title_v2_copy","text_copy"],"appliedCssClassNames":"snowflake-responsive-container-inner-padding-extra-small"},"flexible_column_content_container_2":{"layout":"SIMPLE","id":"hub-faq-accordions",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"simple_snowflake_acc":{"id":"simple-snowflake-accordion-2e794dbdc1","additionalClasses":"seo-hub__faqs","showDivider":false,"accordionItemsList":[{"title":"Is data augmentation the same as synthetic data?","richText":"\u003Cp\u003ENo, traditional data augmentation transforms or recombines existing examples for use in model training. Synthetic data is a broader category of artificially generated data and may support testing, simulation, privacy or analysis in addition to augmentation. Generative augmentation sits at the overlap because it creates synthetic examples specifically to improve a training data set.\u003C/p\u003E"},{"title":"Does data augmentation always reduce overfitting?","richText":"\u003Cp\u003ENo, appropriate augmentation often reduces memorization and improves generalization, but unrealistic, mislabeled or overly aggressive transformations can reduce model performance. The effect should be measured against an unaugmented baseline.\u003C/p\u003E"},{"title":"Is data augmentation used during inference?","richText":"\u003Cp\u003EData augmentation is typically applied during training. Test-time augmentation is a separate method that runs inference over multiple transformed versions of an input and combines the predictions. It may improve performance in some tasks, but it adds inference cost and latency.\u003C/p\u003E"},{"title":"How do you know whether a data augmentation method works?","richText":"\u003Cp\u003ECompare the augmented training run with an unaugmented baseline under controlled conditions. Evaluate downstream performance on clean data, relevant slices and production-representative distribution shifts, then check calibration and consistency across repeated runs.\u003C/p\u003E"}],":type":"snowflake-site/components/simple-snowflake-accordion"}},":itemsOrder":["simple_snowflake_acc"]},"isBlogPage":false,"isActiveTOC":false,":type":"snowflake-site/components/flexible-column-container","appliedCssClassNames":"snowflake-flexible-column-container-gray-10-bg"},"flexible_column_cont_1467213961":{"id":"flexible-column-container-9ee29689a0","propertiesId":"hub-explore-resources-header","type":"1-column","alignColumns":"top","containerMaxWidth":"extra-large","topPadding":"large","bottomPadding":"none","spaceBetween":"small","reverseOnMobile":false,"carouselOnMobile":false,"backgroundImageOption":"none","flexible_column_content_container_1":{"layout":"SIMPLE","id":"container-fe70ab911e",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"title_v2":{"id":"title-v2-f56b793f48","additionalClasses":"hub-explore-resources-header__headline","type":"heading2","lines":["Explore AI Resources"],":type":"snowflake-site/components/title-v2","appliedCssClassNames":"left-alignment"}},":itemsOrder":["title_v2"]},"isBlogPage":false,"isActiveTOC":false,":type":"snowflake-site/components/flexible-column-container","appliedCssClassNames":"snowflake-flexible-column-container-gray-10-bg"},"flexible_column_cont_912630531":{"id":"flexible-column-container-ea407cb959","propertiesId":"hub-explore-resources-grid","type":"1-column","alignColumns":"top","containerMaxWidth":"extra-large","topPadding":"small","bottomPadding":"large","spaceBetween":"small","reverseOnMobile":false,"carouselOnMobile":false,"backgroundImageOption":"none","flexible_column_content_container_1":{"layout":"SIMPLE","id":"hub-explore-resources-grid-inner",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"resource_chip_0":{"id":"content-chip-76fd4a61cc","tagText":"GUIDE","tagColor":"#71D3DC","cta":{"id":"cta","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"https://www.snowflake.com/en/developers/guides/getting-started-with-snowpark-python-scikit/"},"linkTargetContentType":"GENERIC",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_EXTERNAL","text":"Read more"},"headline":{"id":"title-v2-cd8664143b","type":"heading5","lines":["Getting Started with Snowpark for Python with Scikit-learn"],":type":"snowflake-site/components/title-v2"},":type":"snowflake-site/components/content-chip","appliedCssClassNames":"snowflake-content-chip-white-bg"},"resource_chip_1":{"id":"content-chip-5298ca5890","tagText":"BLOG","tagColor":"#29B5E8","cta":{"id":"cta","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"https://www.snowflake.com/en/blog/engineering/llm-best-data-scientist-benchmark-evaluation/"},"linkTargetContentType":"GENERIC",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_EXTERNAL","text":"Read more"},"headline":{"id":"title","type":"heading5","lines":["Benchmarking LLMs on Writing Feature Engineering Code"],":type":"snowflake-site/components/title-v2"},":type":"snowflake-site/components/content-chip","appliedCssClassNames":"snowflake-content-chip-white-bg"},"resource_chip_2":{"id":"content-chip-7b7e5d096e","tagText":"BLOG","tagColor":"#29B5E8","cta":{"id":"cta","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"https://medium.com/snowflake/automatically-creating-features-for-ml-in-snowflake-fedb99b90cb4"},"linkTargetContentType":"GENERIC",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_EXTERNAL","text":"Read more"},"headline":{"id":"title","type":"heading5","lines":["Automatically Creating Features for ML in Snowflake"],":type":"snowflake-site/components/title-v2"},":type":"snowflake-site/components/content-chip","appliedCssClassNames":"snowflake-content-chip-white-bg"},"resource_chip_3":{"id":"content-chip-773a452c66","tagText":"GUIDE","tagColor":"#71D3DC","cta":{"id":"cta","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"https://www.snowflake.com/en/developers/guides/intro-to-feature-store/"},"linkTargetContentType":"GENERIC",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_EXTERNAL","text":"Read more"},"headline":{"id":"title-v2-e46029a0b3","type":"heading5","lines":["Introduction to Snowflake Feature Store with Snowflake Notebooks"],":type":"snowflake-site/components/title-v2"},":type":"snowflake-site/components/content-chip","appliedCssClassNames":"snowflake-content-chip-white-bg"}},":itemsOrder":["resource_chip_0","resource_chip_1","resource_chip_2","resource_chip_3"]},"isBlogPage":false,"isActiveTOC":false,":type":"snowflake-site/components/flexible-column-container","appliedCssClassNames":"snowflake-flexible-column-container-gray-10-bg"},"flexible_column_cont_1158003461":{"id":"flexible-column-container-ad0cb855f8","propertiesId":"hub-explore-topics-header","type":"1-column","alignColumns":"top","containerMaxWidth":"extra-large","topPadding":"large","bottomPadding":"none","spaceBetween":"small","reverseOnMobile":false,"carouselOnMobile":false,"backgroundImageOption":"none","flexible_column_content_container_1":{"layout":"SIMPLE","id":"container-68b5f18019",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"title_v2_copy_copy":{"id":"title-v2-6085cf3b5c","additionalClasses":"hub-explore-topics-header__headline","type":"heading2","lines":["Explore AI Topics"],":type":"snowflake-site/components/title-v2","appliedCssClassNames":"left-alignment"},"text_copy_copy":{"id":"text-5a6dfc14d2","additionalClasses":"hub-explore-topics-header__subheadline","text":"\u003Cp\u003EDeep dives into every aspect of artificial intelligence\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"}},":itemsOrder":["title_v2_copy_copy","text_copy_copy"],"appliedCssClassNames":"snowflake-responsive-container-inner-padding-extra-small"},"isBlogPage":false,"isActiveTOC":false,":type":"snowflake-site/components/flexible-column-container","appliedCssClassNames":"snowflake-flexible-column-container-white-bg"},"flexible_column_cont_1377146023":{"id":"flexible-column-container-a3e40b0271","propertiesId":"hub-explore-topics-grid","type":"3-column-even","alignColumns":"top","containerMaxWidth":"extra-large","topPadding":"small","bottomPadding":"large","spaceBetween":"small","reverseOnMobile":false,"carouselOnMobile":false,"backgroundImageOption":"none","flexible_column_content_container_1":{"layout":"SIMPLE","id":"container-a20379af78",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"topic_card_0":{"id":"card-v2-aac909b94b","configurationStatus":{"configured":true,"message":""},"type":"content-card","title":{"id":"title","type":"heading4","lines":["Feature Engineering"],":type":"snowflake-site/components/title-v2"},"button":{"id":"button","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/artificial-intelligence/machine-learning/feature-engineering/"},"linkTargetContentType":"GENERIC",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_INTERNAL","text":"Learn more"},"text":{"id":"text","text":"\u003Cp\u003EFeature engineering turns raw data into meaningful signals that drive model accuracy, generalization and production reliability.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text"},"layoutStyle":"vertical",":type":"snowflake-site/components/card-v2"}},":itemsOrder":["topic_card_0"]},"flexible_column_content_container_2":{"layout":"SIMPLE","id":"container-6f266b911e",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"topic_card_1":{"id":"card-v2-52752a44ee","configurationStatus":{"configured":true,"message":""},"type":"content-card","title":{"id":"title","type":"heading4","lines":["Machine Learning"],":type":"snowflake-site/components/title-v2"},"button":{"id":"button","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/artificial-intelligence/machine-learning/"},"linkTargetContentType":"GENERIC",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_INTERNAL","text":"Learn more"},"text":{"id":"text","text":"\u003Cp\u003EMachine learning helps organizations turn existing data into scalable models they can build, deploy and monitor with confidence.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text"},"layoutStyle":"vertical",":type":"snowflake-site/components/card-v2"}},":itemsOrder":["topic_card_1"]},"flexible_column_content_container_3":{"layout":"SIMPLE","id":"container-3e58abe7b6",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"topic_card_2":{"id":"card-v2-7189328c1b","configurationStatus":{"configured":true,"message":""},"type":"content-card","title":{"id":"title","type":"heading4","lines":["AI Engineering"],":type":"snowflake-site/components/title-v2"},"button":{"id":"button","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/artificial-intelligence/ai-engineering/"},"linkTargetContentType":"GENERIC",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_INTERNAL","text":"Learn more"},"text":{"id":"text","text":"\u003Cp\u003EAI engineering turns capable models into reliable production systems by combining data, orchestration, evaluation, and operational controls.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text"},"layoutStyle":"vertical",":type":"snowflake-site/components/card-v2"}},":itemsOrder":["topic_card_2"]},"isBlogPage":false,"isActiveTOC":false,":type":"snowflake-site/components/flexible-column-container","appliedCssClassNames":"snowflake-flexible-column-container-white-bg"}},":itemsOrder":["flexible_column_cont","flexible_column_cont_939100716","flexible_column_cont_1398138236","flexible_column_cont_663228916","flexible_column_cont_1786318617","flexible_column_cont_1467213961","flexible_column_cont_912630531","flexible_column_cont_1158003461","flexible_column_cont_1377146023"],":type":"wcm/foundation/components/responsivegrid"},"modal_container":{"layout":"SIMPLE","id":"container-2fb7472d59",":type":"snowflake-site/components/modal/modal-container",":items":{},":itemsOrder":[]},"markup_editor_928258845":{"id":"markup-editor-f9342f16b9","title":" ","cssContent":".snowflake-flexible-column-container-gray-10-bg\u003E.snowflake-flexible-column-container{background-color:var(--ui-background-05) !important}.text-size-regular:has(.seo-hub-hero__related-topic-label){display:flex;align-items:center}.hub-hero__headline span{text-transform:none !important}.hub-hero__subheadline p{max-width:70ch;margin-top:8px}.hub-hero__authors \u003E .container \u003E .cmp-container \u003E .aem-container{display:flex;flex-direction:row}.hub-hero__authors \u003E .container \u003E .cmp-container \u003E .aem-container \u003E div{width:auto !important;margin:24px 48px 0 0 !important}.hub-hero__authors .heading-5-v2{gap:var(--spacing-00)}.hub-hero__authors .snowflake-content-chip-button{display:none !important}.hub-hero__authors .snowflake-person-chip-content .body-2,.hub-hero__authors .snowflake-content-chip-content .snowflake-title-v2-line{font-size:16px !important;line-height:20px !important;font-family:\"Lato\",sans-serif !important;color:#000 !important;font-weight:600 !important}.hub-hero__authors .snowflake-person-chip-content .body-3,.hub-hero__authors .snowflake-content-chip-content .snowflake-title-v2-line:not(:first-child){font-weight:400 !important;color:var(--text-05) !important;font-size:16px !important}.hub-hero__authors .snowflake-image-container img{aspect-ratio:1 !important;border-radius:100%;overflow:hidden}.hub-hero__authors .snowflake-person-chip-avatar{width:56px;height:56px}.hub-hero__authors .snowflake-content-chip{align-items:center;display:inline-flex}.hub-hero__authors .snowflake-content-chip-image{max-width:56px;line-height:0;margin-right:var(--spacing-03)}.hub-hero__authors .snowflake-person-chip-inner-horizontal{gap:var(--spacing-03)}@media screen and (min-width:1367px){.hub-hero__headline .heading-1-v2{font-size:48px;line-height:44px}}#hub-explore-resources-grid-inner\u003E.container\u003E.cmp-container\u003E.aem-container{display:flex;flex-direction:row;flex-wrap:wrap;gap:24px}#hub-explore-resources-grid-inner\u003E.container\u003E.cmp-container\u003E.aem-container::before,#hub-explore-resources-grid-inner\u003E.container\u003E.cmp-container\u003E.aem-container::after{display:none}#hub-explore-resources-grid-inner\u003E.container\u003E.cmp-container\u003E.aem-container\u003Ediv{width:calc(25% - 18px)}.text-color-text-05 .snowflake-text h2,.text-color-text-05.cq-Editable-dom h2,.text-color-text-05 .snowflake-text h3,.text-color-text-05.cq-Editable-dom h3,.text-color-text-05 .snowflake-text h4,.text-color-text-05.cq-Editable-dom h4,.text-color-text-05 .snowflake-text h5,.text-color-text-05.cq-Editable-dom h5,.text-color-text-05 .snowflake-text h6,.text-color-text-05.cq-Editable-dom h6{color:#000 !important}",":type":"snowflake-site/components/markup-editor","isGSAPEnabled":false},"markup_editor_597730182":{"id":"markup-editor-8f09c65be1","title":" ","cssContent":".sf-copy-markdown [data-copy-md]{display:inline-flex;align-items:center;gap:6px;padding:8px 16px;font-family:'Texta',sans-serif;text-transform:uppercase;font-weight:800 !important;font-size:14px;font-weight:500;color:#11567f;background:#f0faff;border:1px solid #b8e6f9;border-radius:24px;cursor:pointer;transition:background .2s ease,border-color .2s ease,color .2s ease}.sf-copy-markdown [data-copy-md]:hover{background:#ddf3fc;border-color:#29b5e8}.sf-copy-markdown [data-copy-md][data-copied=\"1\"]{color:#0f7b3e;background:#ecfdf5;border-color:#6ee7a0;pointer-events:none}.sf-copy-markdown [data-copy-md] svg{flex-shrink:0}.longform-conten .snowflake-content-chip-white-bg .snowflake-content-chip{box-shadow:0 0 24px 4px rgba(0,0,0,.02),0 4px 8px 0 rgba(0,0,0,.04);flex-direction:row-reverse;align-items:center}.longform-conten .snowflake-content-chip-button{display:none}.longform-conten .snowflake-content-chip-image__inner{aspect-ratio:5 / 3;display:flex;justify-content:center;align-items:center;background-color:var(--ui-01);border-radius:4px}.longform-conten .snowflake-content-chip-image{margin-right:0;margin-left:48px}.longform-conten .snowflake-content-chip-image img{width:50%;border-radius:0 !important;object-fit:contain}.longform-content .black-blue-text-color .snowflake-title-v2-line:not(:first-child){font-size:14px !important;font-weight:400 !important;color:rgba(0,0,0,.6) !important;margin-top:8px !important}.page-toc ul li:first-child{padding-top:0 !important}.page-toc ul li:last-child{padding-bottom:0 !important}.seo-hub__top-bar \u003E .container \u003E .cmp-container \u003E .aem-container \u003E div:first-child{flex-grow:1}.sf-copy-markdown{margin-top:40px !important}.page-toc ul{margin-top:16px !important}.seo-hub__top-bar \u003E .container \u003E .cmp-container \u003E .aem-container{display:flex;justify-content:space-between}.sf-copy-markdown{}.callout.snowflake-text p:not(:first-child){margin-top:var(--spacing-01)}.callout \u003E span \u003E p:first-child \u003E strong,.callout \u003E span \u003E p:first-child \u003E b{text-transform:uppercase;font-family:'Texta',sans-serif;font-size:16px !important;color:var(--ui-01) !important}.seo-hub-hero__subheadline p{max-width:50ch}.tag-group ul{list-style-type:none;padding:0;margin:0;display:flex;flex-direction:row;row-gap:12px;column-gap:8px;align-items:center;flex-wrap:wrap}.tag-group ul li:first-child{flex-shrink:0}.tag-group ul li a{display:inline-block;padding:2px 12px;border-radius:48px;background-color:#ededed;color:#666;font-size:14px !important}@media screen and (min-width:1367px){.seo-hub-hero__headline span.snowflake-title-v2-line{font-size:56px !important}}.callout.snowflake-text p:not(:first-child){margin-top:var(--spacing-01)}.callout \u003E span \u003E p:first-child \u003E b{text-transform:uppercase;font-family:'Texta',sans-serif;font-size:16px !important;color:var(--ui-01) !important}.seo-hub-hero__subheadline p{max-width:80ch}#hero:has(.snowflake-youtube-lite) .seo-hub-hero__subheadline p{max-width:50ch}.tag-group ul{list-style-type:none;padding:0;margin:0;display:flex;flex-direction:row;row-gap:12px;column-gap:8px;align-items:center;flex-wrap:wrap}.tag-group ul li:first-child{width:100%;flex-shrink:0}.tag-group ul li a{display:inline-block;padding:2px 12px;border-radius:48px;background-color:#ededed;color:#666;font-size:14px !important}@media screen and (min-width:1367px){.seo-hub-hero__headline span.snowflake-title-v2-line{font-size:56px !important}}",":type":"snowflake-site/components/markup-editor","isGSAPEnabled":false},"markup_editor":{"id":"markup-editor-14f812c43b","title":" ","cssContent":"div.snowflake-breadcrumb a.snowflake-breadcrumb-item,.snowflake-breadcrumb div.snowflake-breadcrumb-item{text-transform:none;font-weight:500}.snowflake-breadcrumb svg{display:none !important}.snowflake-breadcrumb a:has(svg)::after{content:'/';margin:0 12px;color:#666}.hub-sidebar{padding:0 40px}.sticky-sidebar{max-width:340px;margin-left:auto}.page-toc ul{list-style-type:none;padding:0}.page-toc li{padding:8px 16px;border-left:4px solid var(--ui-01);cursor:pointer;transition:300ms ease all}.page-toc li:hover{color:var(--ui-01);border-color:#7fd3f1;transition:300ms ease all}.callout,.customer-card{background-color:#eef9fd;border-left:4px solid var(--ui-01);padding:24px 24px 24px 32px;border-radius:4px}.logo-container{max-width:180px}.longform-content li{margin-top:1rem !important}div.longform-content p{max-width:80ch}.bolder .snowflake-title-v2-line{font-weight:900 !important}.border-top\u003Ediv{border-top:1px solid #ccc;padding-top:48px}.related-topics ul{list-style-type:none;padding:0;margin:0;display:flex;gap:8px;flex-wrap:wrap}.related-topics li{display:inline-block;border:1px solid #ccc;padding:4px 12px;border-radius:24px}div.longform-content .snowflake-text h2,div.longform-content .snowflake-text .heading-2-v2,div.longform-content .snowflake-text h3,div.longform-content .snowflake-text .heading-3-v2,div.longform-content .snowflake-title-v2 .heading-3-v2,div.longform-content .snowflake-text h4,div.longform-content .snowflake-text .heading-4-v2,div.longform-content .snowflake-title-v2 .heading-4-v2,div.longform-content .snowflake-text h5,div.longform-content .snowflake-text .heading-5-v2,div.longform-content .snowflake-title-v2 .heading-5-v2,div.longform-content .snowflake-text h6,div.longform-content .snowflake-title-v2 .heading-6-v2,div.longform-content .snowflake-text .heading-6-v2{text-transform:none !important}div.longform-content .snowflake-text h2,div.longform-content .snowflake-text .heading-2-v2,div.longform-content .snowflake-text h3,div.longform-content .snowflake-text .heading-3-v2,div.longform-content .snowflake-text h4,div.longform-content .snowflake-text .heading-4-v2,div.longform-content .snowflake-text h5,div.longform-content .snowflake-text .heading-5-v2,div.longform-content .snowflake-text h6,div.longform-content .snowflake-text .heading-6-v2{margin-top:1.5rem !important;line-height:1.1 !important}div.longform-content .snowflake-text h3,div.longform-content .snowflake-text .heading-3-v2,div.longform-content .snowflake-title-v2 .heading-3-v2,div.longform-content .snowflake-text h4,div.longform-content .snowflake-text .heading-4-v2,div.longform-content .snowflake-title-v2 .heading-4-v2,div.longform-content .snowflake-text h5,div.longform-content .snowflake-text .heading-5-v2,div.longform-content .snowflake-title-v2 .heading-5-v2,div.longform-content .snowflake-text h6,div.longform-content .snowflake-text .heading-6-v2,div.longform-content .snowflake-title-v2 .heading-6-v2{font-family:Lato,sans-serif !important;font-weight:800 !important}div.longform-content .snowflake-text h2,div.longform-content .snowflake-text .heading-2-v2,div.longform-content .snowflake-title-v2 .heading-2-v2{text-transform:none !important;font-size:28px !important}div.longform-content .snowflake-text h3,div.longform-content .snowflake-text .heading-3-v2,div.longform-content .snowflake-title-v2 .heading-3-v2{font-size:22px !important}div.longform-content .snowflake-text h4,div.longform-content .snowflake-text .heading-4-v2,div.longform-content .snowflake-title-v2 .heading-4-v2{font-size:18px !important}div.longform-content .snowflake-text h5,div.longform-content .snowflake-text .heading-5-v2,div.longform-content .snowflake-title-v2 .heading-5-v2{font-size:16px !important}div.longform-content .snowflake-text h6,div.longform-content .snowflake-text .heading-6-v2,div.longform-content .snowflake-title-v2 .heading-6-v2{font-size:14px !important}@media screen and (min-width:992px){div.longform-content .snowflake-text h2,div.longform-content .snowflake-text .heading-2-v2,div.longform-content .snowflake-title-v2 .heading-2-v2{font-size:38px !important}div.longform-content .snowflake-text h3,div.longform-content .snowflake-text .heading-3-v2,div.longform-content .snowflake-title-v2 .heading-3-v2{font-size:26px !important}div.longform-content .snowflake-text h4,div.longform-content .snowflake-text .heading-4-v2,div.longform-content .snowflake-title-v2 .heading-4-v2{font-size:22px !important}div.longform-content .snowflake-text h5,div.longform-content .snowflake-text .heading-5-v2,div.longform-content .snowflake-title-v2 .heading-5-v2{font-size:18px !important}div.longform-content .snowflake-text h6,div.longform-content .snowflake-text .heading-6-v2,div.longform-content .snowflake-title-v2 .heading-6-v2{font-size:16px !important}}.sticky-sidebar .page-toc li.is-active{font-weight:600;color:var(--snow-blue,#29b5e8)}.sticky-sidebar .page-toc li[data-anchor]{cursor:pointer}.longform-content table{margin-top:24px;margin-bottom:24px;width:100%;background-color:var(--ui-background-01);border-collapse:collapse;border:2px solid var(--ui-background-09);font-family:'Lato',sans-serif;color:var(--ui-background-09)}.longform-content table thead{background-color:var(--ui-01)}.longform-content th,.longform-content td{min-width:120px;border:2px solid var(--ui-background-09);padding:var(--spacing-01)}.longform-content ol{margin-top:0 !important}.longform-content ol li{margin-bottom:1rem !important}.longform-content ul li{margin:0;padding:0 0 0 32px;position:relative}.longform-content ul{list-style-type:none}.longform-content ul li::before{content:\"\";display:block;border-radius:100%;background:#29b5e8;width:18px;height:18px;position:absolute;top:4px;left:0;border:5px solid #e5f2f7;box-sizing:border-box}.seo-customer.snowflake-card-v2-advanced-horizontal .snowflake-card-v2-advanced-image-container{max-width:200px}.seo-customer.snowflake-card-v2-advanced-horizontal .snowflake-card-v2-advanced-image-container img{object-fit:contain}.related-topics-outer-container \u003E .container \u003E .cmp-container \u003E .aem-container{display:flex;flex-direction:row}.related-topics-outer-container \u003E .container \u003E .cmp-container \u003E .aem-container \u003E div{width:auto !important;margin:0 !important}.related-topics-outer-container \u003E .container \u003E .cmp-container \u003E .aem-container \u003E div:first-child{margin-right:16px !important;flex-shrink:0}","jsContent":"(function(){var OFFSET=100;if(window.gsap&&window.ScrollTrigger){gsap.registerPlugin(ScrollTrigger);var sidebar=document.querySelector('.sticky-sidebar');var body=document.querySelector('.longform-content');if(sidebar&&body){ScrollTrigger.create({trigger:sidebar,start:'top 100px',endTrigger:body,end:'bottom bottom',pin:sidebar,pinSpacing:false});}}document.addEventListener('click',function(e){var li=e.target.closest('li[data-anchor]');if(!li)return;var slug=li.getAttribute('data-anchor');var heading=document.querySelector('.anchor-title--'+CSS.escape(slug));if(!heading)return;e.preventDefault();var top=heading.getBoundingClientRect().top+window.pageYOffset-OFFSET;window.scrollTo({top:top,behavior:'smooth'});history.replaceState(null,'','#'+slug);},false);var headings=document.querySelectorAll('[class*=\"anchor-title--\"]');if(headings.length&&'IntersectionObserver'in window){var io=new IntersectionObserver(function(entries){entries.forEach(function(entry){if(!entry.isIntersecting)return;var cls=Array.from(entry.target.classList).find(function(c){return c.indexOf('anchor-title--')===0;});if(!cls)return;var slug=cls.replace('anchor-title--','');document.querySelectorAll('li[data-anchor]').forEach(function(li){li.classList.toggle('is-active',li.getAttribute('data-anchor')===slug);});});},{rootMargin:'-20% 0px -70% 0px',threshold:0});headings.forEach(function(h){io.observe(h);});}})();",":type":"snowflake-site/components/markup-editor","isGSAPEnabled":true},"experiencefragment-footer":{"id":"experiencefragment-252776d0df","localizedFragmentVariationPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/footer/master/jcr:content","configured":true,"xfModelPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/footer/master.xfmodel.json",":type":"snowflake-site/components/experiencefragment"},"experiencefragment":{"id":"experiencefragment-ad3cb41350","localizedFragmentVariationPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/footer-legal-disclaimers/master/jcr:content","configured":true,"xfModelPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/footer-legal-disclaimers/master.xfmodel.json",":type":"snowflake-site/components/experiencefragment"}},":itemsOrder":["experiencefragment-banner","experiencefragment-header","responsivegrid","modal_container","markup_editor_928258845","markup_editor_597730182","markup_editor","experiencefragment-footer","experiencefragment"],":type":"wcm/foundation/components/responsivegrid"}},":itemsOrder":["root"],":hierarchyType":"page",":path":"/content/snowflake-site/global/en/artificial-intelligence/machine-learning/model-training/data-augmentation","analyticsContentTags":[],"analyticsEnabled":true,"coveoConfig":{"searchHub":"snowflake.com","pipeline":"snowflake.com","apiKey":"xx335921a6-2a0a-40f2-a167-e390b4766c3d","organizationId":"snowflakecomputingproduction8neljofn"},"analyticsDebugMode":false,"analyticsData":{"excludeFromAnalytics":false,"subCategory":"","pageType":"homepage","templateName":"base-page-template54","siteName":"snowflake","pageUrl":"/content/snowflake-site/global/en/artificial-intelligence/machine-learning/model-training/data-augmentation","language":"en","category":"general","pageName":"Data Augmentation: How to Expand Training Data Without Distorting the Signal","contentTags":[]},"isPasswordProtected":false,"locale":"en"}
  