{"templateName":"base-page-template54","cssClassNames":"page basicpage summit-page","canonicalLink":"https://www.snowflake.com/en/artificial-intelligence/machine-learning/feature-engineering/embeddings/","robotsTags":[],"allowedRenditionsWidth":["320","480","640","768","960","1200","1440","1920"],"description":"Learn how embeddings turn text, images and other data into vectors that power semantic search, RAG, recommendations and other AI applications.","language":"en","title":"Embeddings for Semantic Search and RAG | Snowflake","analyticsPageType":"homepage","analyticsCategory":"general","analyticsSubCategory":"","excludeFromAnalytics":false,":mappedPath":"/en/artificial-intelligence/machine-learning/feature-engineering/embeddings/",":type":"snowflake-site/components/structure/page",":path":"/content/snowflake-site/global/en/artificial-intelligence/machine-learning/feature-engineering/embeddings",":items":{"root":{"columnCount":12,"columnClassNames":{"markup_editor_928258845":"aem-GridColumn aem-GridColumn--default--12","experiencefragment-banner":"aem-GridColumn aem-GridColumn--default--12","experiencefragment-header":"aem-GridColumn aem-GridColumn--default--12","responsivegrid":"aem-GridColumn aem-GridColumn--default--12","markup_editor_597730182":"aem-GridColumn aem-GridColumn--default--12","experiencefragment-footer":"aem-GridColumn aem-GridColumn--default--12","experiencefragment":"aem-GridColumn aem-GridColumn--default--12","modal_container":"aem-GridColumn aem-GridColumn--default--12","markup_editor":"aem-GridColumn aem-GridColumn--default--12"},"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12",":items":{"experiencefragment-banner":{"id":"experiencefragment-141b886d11","localizedFragmentVariationPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/pushdown-banner/master/jcr:content","configured":true,":type":"snowflake-site/components/experiencefragment","xfModelPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/pushdown-banner/master.xfmodel.json"},"experiencefragment-header":{"id":"experiencefragment-eca7105750","localizedFragmentVariationPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/mega-nav-header/master/jcr:content","configured":true,":type":"snowflake-site/components/experiencefragment","xfModelPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/mega-nav-header/master.xfmodel.json","languageNavPath":"/content/snowflake-site/global/en/artificial-intelligence/machine-learning/feature-engineering/embeddings.languagenav.json"},"responsivegrid":{"columnCount":12,"columnClassNames":{"flexible_column_cont_939100716":"aem-GridColumn aem-GridColumn--default--12","flexible_column_cont":"aem-GridColumn aem-GridColumn--default--12","flexible_column_cont_1158003461":"aem-GridColumn aem-GridColumn--default--12","flexible_column_cont_663228916":"aem-GridColumn aem-GridColumn--default--12","flexible_column_cont_912630531":"aem-GridColumn aem-GridColumn--default--12","flexible_column_cont_1398138236":"aem-GridColumn aem-GridColumn--default--12","flexible_column_cont_1786318617":"aem-GridColumn aem-GridColumn--default--12","flexible_column_cont_1467213961":"aem-GridColumn aem-GridColumn--default--12","flexible_column_cont_1377146023":"aem-GridColumn aem-GridColumn--default--12"},"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12",":items":{"flexible_column_cont":{"id":"flexible-column-container-f49b393357","propertiesId":"hub-hero-breadcrumbs","type":"1-column","alignColumns":"top","containerMaxWidth":"extra-large","topPadding":"small","bottomPadding":"none","spaceBetween":"small","reverseOnMobile":false,"carouselOnMobile":false,"propertiesCSSClasses":"page-section","backgroundImageOption":"none","flexible_column_content_container_1":{"layout":"SIMPLE","id":"container-6d63da9880",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"breadcrumb":{"id":"breadcrumb-295c6fff1b","items":[{"id":"breadcrumb-295c6fff1b-item-f710b50fc9","link":{"valid":true,"url":"/en/artificial-intelligence/machine-learning/"},"active":false,"current":false,"title":"Machine Learning",":type":"snowflake-site/components/structure/page","appliedCssClassNames":"summit-page"},{"id":"breadcrumb-295c6fff1b-item-ebee0cb8da","link":{"valid":true,"url":"/en/artificial-intelligence/machine-learning/feature-engineering/"},"active":false,"current":false,"title":"Feature Engineering",":type":"snowflake-site/components/structure/page","appliedCssClassNames":"summit-page"},{"id":"breadcrumb-295c6fff1b-item-486a23f542","link":{"valid":true,"url":"/en/artificial-intelligence/machine-learning/feature-engineering/embeddings/"},"active":true,"current":true,"title":"Embeddings",":type":"snowflake-site/components/structure/page","appliedCssClassNames":"summit-page"}],":type":"snowflake-site/components/breadcrumb"}},":itemsOrder":["breadcrumb"]},":type":"snowflake-site/components/flexible-column-container","isBlogPage":false,"isActiveTOC":false,"appliedCssClassNames":"snowflake-flexible-column-container-gray-10-bg"},"flexible_column_cont_939100716":{"id":"flexible-column-container-f4399778d9","propertiesId":"hub-hero","type":"2-column-even","alignColumns":"center","containerMaxWidth":"extra-large","topPadding":"extra-small","bottomPadding":"extra-small","spaceBetween":"small","reverseOnMobile":false,"carouselOnMobile":false,"propertiesCSSClasses":"page-section","backgroundImageOption":"none","flexible_column_content_container_1":{"layout":"SIMPLE","id":"container-67dcee7421",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"title_v2":{"id":"title-v2-ab0f3e5416","additionalClasses":"hub-hero__headline","type":"heading1","lines":["How Embeddings Power Semantic Search, RAG and Recommendations"],":type":"snowflake-site/components/title-v2","appliedCssClassNames":"left-alignment"},"text":{"id":"text-e8316de233","additionalClasses":"hub-hero__subheadline","text":"\u003Cp\u003EEmbeddings help AI systems recognize relationships that keywords can miss. By turning content into comparable numerical representations, they support semantic search, recommendations and context-aware AI applications.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-size-regular text-color-text-05"},"container":{"additionalClasses":"hub-hero__authors","layout":"RESPONSIVE_GRID","columnCount":12,"columnClassNames":{"content_chip_copy":"aem-GridColumn aem-GridColumn--default--12","content_chip":"aem-GridColumn aem-GridColumn--default--12"},"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","id":"container-388fbc1bce",":type":"snowflake-site/components/container",":items":{"content_chip":{"id":"content-chip-32793de898","cta":{"id":"cta","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/blog/authors/Laurie-Macpherson/"},"linkTargetContentType":"GENERIC",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_INTERNAL","text":"Read bio"},"image":{"id":"image","height":"800","src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--a7e9fdff-6f08-4edc-9cd1-e213bb234aaa/laurie-macpherson.jpg?preferwebp=true&quality=85","lazyEnabled":true,"alt":"Laurie MacPherson","isLcpImage":true,"width":"800",":type":"snowflake-site/components/image"},"headline":{"id":"title","type":"heading5","lines":["Laurie MacPherson","Technical Writer, Snowflake"],":type":"snowflake-site/components/title-v2"},":type":"snowflake-site/components/content-chip"},"content_chip_copy":{"id":"content-chip-948a1ec2c0","cta":{"id":"cta","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/blog/authors/david-gaule/"},"linkTargetContentType":"GENERIC",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_INTERNAL","text":"Read bio"},"image":{"id":"image","height":"512","src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--fd9454ea-3b59-4d19-95be-534774cbd226/david.jpg?preferwebp=true&quality=85","lazyEnabled":true,"alt":"David Gaule","isLcpImage":false,"width":"512",":type":"snowflake-site/components/image"},"headline":{"id":"title","type":"heading5","lines":["David Gaule","Technical Editor, Snowflake"],":type":"snowflake-site/components/title-v2"},":type":"snowflake-site/components/content-chip"}},":itemsOrder":["content_chip","content_chip_copy"],"appliedCssClassNames":"snowflake-responsive-container-inner-padding-small"}},":itemsOrder":["title_v2","text","container"],"appliedCssClassNames":"snowflake-responsive-container-inner-padding-extra-small"},"flexible_column_content_container_2":{"additionalClasses":"hub-hero__video-column","layout":"SIMPLE","id":"container-95c15ee2ee",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"youtube":{"id":"embed-cc9faf6c53","youtubeVideoId":"8rzNbJO1TN4","layout":"responsive","youtubeAspectRatio":"56.25","youtubeAutoPlay":false,"youtubeLoop":false,"youtubeMute":false,"youtubePlaysInline":false,"youtubeRel":false,"embeddableResourceType":"core/wcm/components/embed/v1/embed/embeddable/youtube","type":"EMBEDDABLE",":type":"snowflake-site/components/youtube"}},":itemsOrder":["youtube"]},":type":"snowflake-site/components/flexible-column-container","isBlogPage":false,"isActiveTOC":false,"appliedCssClassNames":"snowflake-flexible-column-container-gray-10-bg"},"flexible_column_cont_1398138236":{"id":"flexible-column-container-e6df108e37","propertiesId":"hub-hero-related-topics","type":"1-column","alignColumns":"top","containerMaxWidth":"extra-large","topPadding":"extra-small","bottomPadding":"medium","spaceBetween":"small","reverseOnMobile":false,"carouselOnMobile":false,"propertiesCSSClasses":"page-section","backgroundImageOption":"none","flexible_column_content_container_1":{"additionalClasses":"related-topics-outer-container border-top","layout":"SIMPLE","id":"container-fdac965843",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"text_894059747":{"id":"text-eebb9d74b8","additionalClasses":"seo-hub-hero__related-topic-label","text":"\u003Cp\u003EFeature Engineering Topics:\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-size-regular text-color-text-05"},"text":{"id":"text-916b0071c7","additionalClasses":"related-topics ","text":"\u003Cul\u003E\r\n\u003Cli\u003E\u003Ca href=\"https://www.snowflake.com/en/artificial-intelligence/machine-learning/feature-engineering/data-preprocessing/\"\u003EData Preprocessing\u003C/a\u003E\u003C/li\u003E\r\n\u003Cli\u003E\u003Ca href=\"https://www.snowflake.com/en/artificial-intelligence/machine-learning/feature-engineering/feature-extraction/\"\u003EFeature Extraction\u003C/a\u003E\u003C/li\u003E\r\n\u003Cli\u003E\u003Ca href=\"https://www.snowflake.com/en/artificial-intelligence/machine-learning/feature-engineering/feature-store/\"\u003EFeature Store\u003C/a\u003E\u003C/li\u003E\r\n\u003C/ul\u003E\r\n","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-size-small"}},":itemsOrder":["text_894059747","text"]},":type":"snowflake-site/components/flexible-column-container","isBlogPage":false,"isActiveTOC":false,"appliedCssClassNames":"snowflake-flexible-column-container-gray-10-bg"},"flexible_column_cont_663228916":{"id":"flexible-column-container-bd30d5df7b","propertiesId":"hub-body","type":"2-column-60-40","alignColumns":"top","containerMaxWidth":"extra-large","topPadding":"medium","bottomPadding":"medium","spaceBetween":"small","reverseOnMobile":false,"carouselOnMobile":false,"propertiesCSSClasses":"page-section","backgroundImageOption":"none","flexible_column_content_container_1":{"additionalClasses":"longform-content","layout":"SIMPLE","id":"hub-body-content",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"callout__0":{"id":"text-4e77fdce30","additionalClasses":"callout callout--general","text":"\u003Cp\u003E\u003Cstrong\u003EEMBEDDINGS DEFINITION\u003C/strong\u003E\u003C/p\u003E\n\u003Cp\u003EAn embedding is a learned numerical representation that captures meaningful characteristics of an item so it can be compared with other items.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-size-regular text-color-text-05"},"text__0":{"id":"text-c52f76dd43","text":"\u003Cp\u003EConsider a support request that reads, “I can’t get into my account.” Although those words barely overlap with a support article titled “Recover access after a failed login,” a text embedding model can place the two items near each other in a vector space, reflecting the relationship between their meanings.\u003C/p\u003E\n\u003Cp\u003EOnce both items have numerical representations, an application can measure their proximity. For example, a search system might compare the request with thousands of article embeddings and return the closest matches. Embeddings support semantic search, \u003Ca href=\"https://www.snowflake.com/en/fundamentals/rag/\"\u003Eretrieval-augmented generation\u003C/a\u003E (RAG), recommendations, clustering and many other AI applications.\u003C/p\u003E\n\u003Cp\u003EEach stored vector may be compact, but creating and maintaining embeddings across a large, frequently changing collection introduces a substantial inference workload. Existing records require an initial processing pass, revised content needs fresh embeddings and every incoming search produces a query vector. As semantic retrieval supports more RAG, agent and multimodal applications, practitioners need to evaluate two concerns together: whether the embeddings retrieve the right material and whether the surrounding architecture can generate, refresh and serve them at the required scale.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"},"title_what-are-embeddings":{"id":"title-v2-e2040ae461","additionalClasses":"anchor-title anchor-title--what-are-embeddings","type":"heading2","lines":["What are embeddings?"],":type":"snowflake-site/components/title-v2"},"text_what-are-embeddings_0":{"id":"text-2db2955752","text":"\u003Cp\u003EAn embedding is a vector — a fixed-length sequence of numbers — that represents an object such as a word, passage, image or product. An embedding model generates the sequence so the object can be compared mathematically with other items processed by the same model.\u003C/p\u003E\n\u003Cp\u003EEach number corresponds to one dimension of the vector. Current text embedding models often produce hundreds or thousands of dimensions, although diagrams usually reduce the space to two or three so it can be shown on a page. Taken together, the dimensions locate the object at a particular position within the model’s vector space.\u003C/p\u003E\n\u003Cp\u003EThe value of any one dimension usually has little meaning on its own. A practitioner generally can’t inspect the 217th number and conclude that it represents account access, for example. The useful information is distributed across the complete vector and emerges through its relationship to other vectors.\u003C/p\u003E\n\u003Cp\u003EDuring training, the model learns to position related items near one another. A model built for document retrieval might place passages together when they address similar questions, while a recommendation model could group products that appeal to people with similar purchasing patterns. Proximity reflects the kind of relationship the model was trained to recognize.\u003C/p\u003E\n\u003Cp\u003EThis representation gives applications a common numerical basis for search, recommendation, clustering and other tasks. The original sentence, image or product record remains available, but the embedding supplies an additional form that the system can compare efficiently.\u003C/p\u003E\n\u003Cp\u003EEmbeddings can serve as learned features, replacing or complementing manually engineered variables in classification, clustering, recommendation and other \u003Ca href=\"https://www.snowflake.com/en/artificial-intelligence/machine-learning/\"\u003Emachine learning\u003C/a\u003E workflows.\u003C/p\u003E\n\u003Cp\u003E\u003Cem\u003ELearn how to scale embeddings with GPUs from Snowflake Notebooks on Container Runtime:\u003C/em\u003E\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"},"yt_what-are-embeddings_0":{"id":"embed-c30afb961f","youtubeVideoId":"uvLiJtfNd-M","layout":"responsive","youtubeAspectRatio":"56.25","youtubeAutoPlay":false,"youtubeLoop":false,"youtubeMute":false,"youtubePlaysInline":false,"youtubeRel":false,"embeddableResourceType":"core/wcm/components/embed/v1/embed/embeddable/youtube","type":"EMBEDDABLE",":type":"snowflake-site/components/youtube"},"title_how-embedding-models-create-vector-representations":{"id":"title-v2-33f89fcc6a","additionalClasses":"anchor-title anchor-title--how-embedding-models-create-vector-representations","type":"heading2","lines":["How embedding models create vector representations"],":type":"snowflake-site/components/title-v2"},"text_how-embedding-models-create-vector-representations_0":{"id":"text-b996301cd3","text":"\u003Cp\u003EThe values in an embedding reflect patterns established during model training. Later, during inference, the trained model applies those patterns to new inputs.\u003C/p\u003E\n\u003Ch3\u003ELearning the embedding space\u003C/h3\u003E\n\u003Cp\u003EDuring training, embedding models typically use pairs or groups of examples to learn what proximity should mean for the task. As training proceeds, the model adjusts its parameters to organize the representation space according to patterns and relationships in the training data.\u003C/p\u003E\n\u003Cp\u003EMany embedding models use contrastive learning to organize those examples. The training process reduces the distance between related inputs while increasing the separation between unrelated ones. SimCSE, for instance, applies a contrastive objective to sentence embeddings, using positive and negative examples to produce a space in which semantically related sentences occupy nearby positions.\u003C/p\u003E\n\u003Cp\u003EThe resulting geometry reflects the task and data used for training. A model trained on product interactions, for example, will typically emphasize behavioral relationships, including associations that aren’t apparent from the product descriptions alone.\u003C/p\u003E\n\u003Cp\u003ECoverage affects the quality of those representations as well. Specialized abbreviations, languages or document formats that appear infrequently in the training data may receive less useful embeddings, which helps explain why a model with strong general benchmark results can still perform poorly on an organization’s internal content.\u003C/p\u003E\n\u003Ch3\u003EGenerating an embedding\u003C/h3\u003E\n\u003Cp\u003EAfter training, the model can generate a vector for a new input. The input-processing stage depends on the modality: text models process tokens, image models process pixel-based patches or features, and audio or video models may divide content into frames or time segments.\u003C/p\u003E\n\u003Cp\u003EFor text, the process usually begins with tokenization: the model first divides the content into tokens and processes them in context through the neural network. It then combines information from those token-level representations into a fixed-length vector for the selected unit of content.\u003C/p\u003E\n\u003Cp\u003EThat unit might be a sentence, paragraph or complete document. Regardless of input length, the model produces the same number of dimensions, allowing a short query and a longer passage to be compared within the same vector space.\u003C/p\u003E\n\u003Cp\u003EChoosing the unit of representation is part of the retrieval design. Paragraph embeddings allow a search system to return a focused section, while one vector for an entire document blends information from all of its parts. Smaller units can preserve local detail, while larger units retain more surrounding context. Practitioners typically evaluate that trade-off against the material users need to retrieve.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"},"callout_how-embedding-models-create-vector-representations_0":{"id":"text-61dd9be390","additionalClasses":"callout callout--warning","text":"\u003Cp\u003E\u003Cstrong\u003ECOMMON PITFALL\u003C/strong\u003E\u003C/p\u003E\n\u003Cp\u003EA common mistake is to embed an entire long document as one unit. A single vector can blend unrelated sections and make it difficult to retrieve the passage that actually answers the query.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-size-regular text-color-text-05"},"title_how-applications-use-embeddings-to-find-similar-items":{"id":"title-v2-c6ae989aed","additionalClasses":"anchor-title anchor-title--how-applications-use-embeddings-to-find-similar-items","type":"heading2","lines":["How applications use embeddings to find similar items"],":type":"snowflake-site/components/title-v2"},"text_how-applications-use-embeddings-to-find-similar-items_0":{"id":"text-c86bf3fad2","text":"\u003Cp\u003EA single embedding records the position of one item in the learned space. Applications compare embeddings to answer a practical question: which stored items occupy positions closest to a new query, product, image or other input?\u003C/p\u003E\n\u003Cp\u003EThe source content remains available for display, filtering and downstream processing. Vector comparison provides the ranking signal.\u003C/p\u003E\n\u003Ch3\u003ESimilarity and distance metrics\u003C/h3\u003E\n\u003Cp\u003EA similarity or distance metric assigns a numerical score to two vectors. Common measures include:\u003C/p\u003E\n\u003Cul\u003E\n\u003Cli\u003E\u003Cstrong\u003ECosine similarity,\u003C/strong\u003E which compares the direction of the vectors and largely disregards their magnitude\u003C/li\u003E\n\u003Cli\u003E\u003Cstrong\u003EDot product,\u003C/strong\u003E which incorporates both direction and magnitude\u003C/li\u003E\n\u003Cli\u003E\u003Cstrong\u003EEuclidean distance,\u003C/strong\u003E which measures the straight-line distance between two points\u003C/li\u003E\n\u003C/ul\u003E\n\u003Cp\u003EThe appropriate metric depends on the model’s training objective and whether the vectors have been normalized. \u003Ca href=\"https://aclanthology.org/D19-1410/\" target=\"_blank\"\u003ESentence-BERT,\u003C/a\u003E for example, was designed to create semantically meaningful sentence vectors that can be compared efficiently with cosine similarity.\u003C/p\u003E\n\u003Cp\u003EWith a similarity metric, higher scores usually indicate a closer relationship. Distance metrics use the opposite convention: smaller values indicate greater proximity.\u003C/p\u003E\n\u003Cp\u003EThose scores are only meaningful within the model and collection that produced them. A cosine similarity of 0.8, for example, has no universal interpretation across models or data sets, and a high score only shows that the model represents two inputs as similar.\u003C/p\u003E\n\u003Ch3\u003ENearest-neighbor search\u003C/h3\u003E\n\u003Cp\u003EA similarity metric can score one query vector against one stored vector. In practice, however, a search or recommendation system needs to repeat that comparison across an entire collection, then rank the results. Nearest-neighbor search provides that larger-scale operation, identifying the stored vectors closest to the query.\u003C/p\u003E\n\u003Cp\u003EA basic semantic retrieval flow follows four steps:\u003C/p\u003E\n\u003Col\u003E\n\u003Cli\u003EDivide the source content into retrievable units.\u003C/li\u003E\n\u003Cli\u003EGenerate and store an embedding for each unit.\u003C/li\u003E\n\u003Cli\u003EGenerate an embedding when a query arrives.\u003C/li\u003E\n\u003Cli\u003ERank the stored vectors according to their proximity to the query.\u003C/li\u003E\n\u003C/ol\u003E\n\u003Cp\u003EFor a small collection, the system can calculate a score against every vector. As the index grows, a full scan adds latency and compute. Approximate nearest-neighbor methods organize the vector space so the system can search promising regions first, accepting some possibility that it will miss the mathematically closest candidate.\u003C/p\u003E\n\u003Cp\u003EThe index can store metadata with each vector, including the source record, publication date, category, region and access permissions. An application can use those fields to limit the search to eligible records — such as documents a user is authorized to view or products available in a particular region — before ranking the remaining results by vector similarity.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"},"card_v2_how-applications-use-embeddings-to-find-similar-items_0":{"id":"card-v2-79e88f0cb6","additionalClasses":"seo-customer","configurationStatus":{"configured":true,"message":""},":type":"snowflake-site/components/card-v2","text":{"id":"text","text":"\u003Cp\u003ETS Imagine uses Snowflake and Snowflake Cortex AI to unify data, teams and technologies across more than 500 financial-services clients while scaling generative AI use cases across the business. With RAG-based workflows and Streamlit in Snowflake, TS Imagine automated manual email monitoring, accelerated customer-support triage and knowledge discovery reduced AI costs by 30% compared with external LLM APIs, and saved 4,000 hours of manual effort per year [as of September 2024].\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/text"},"layoutStyle":"horizontal","title":{"id":"title","type":"heading4","lines":["TS Imagine"],":type":"snowflake-site/components/title-v2"},"button":{"id":"button","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/customers/all-customers/case-study/ts-imagine/"},"linkTargetContentType":"GENERIC",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_INTERNAL","text":"Read the full case study"},"image":{"id":"image","height":"351","src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--6a0bbc3f-d212-4f4f-a564-8706239332a9/ts-imagine%25403x.png?preferwebp=true&quality=85","lazyEnabled":true,"alt":"TSImagine Logo","isLcpImage":false,"width":"624",":type":"snowflake-site/components/image"},"type":"content-card"},"title_types-of-embeddings":{"id":"title-v2-8cacc49dc2","additionalClasses":"anchor-title anchor-title--types-of-embeddings","type":"heading2","lines":["Types of embeddings"],":type":"snowflake-site/components/title-v2"},"text_types-of-embeddings_0":{"id":"text-2263329ae7","text":"\u003Cp\u003EDifferent embedding types preserve different kinds of relationships because they’re trained on different inputs and objectives. The categories below describe both what each vector represents and what proximity means within that space.\u003C/p\u003E\n\u003Ch3\u003EText embeddings\u003C/h3\u003E\n\u003Cp\u003EText embeddings can represent words, sentences, passages or complete documents. Static word embeddings assign a stored vector to each vocabulary item, while contextual models calculate representations from the surrounding sequence.\u003C/p\u003E\n\u003Cp\u003EFor semantic search and clustering, purpose-built sentence or passage models combine contextual information into one fixed-length vector. Sentence-BERT introduced an architecture designed to produce sentence representations that could be compared directly, reducing the computational work required for large-scale semantic similarity search.\u003C/p\u003E\n\u003Ch3\u003EImage embeddings\u003C/h3\u003E\n\u003Cp\u003EAn image embedding model represents visual content through features learned from training images. Depending on its objective, the vector may preserve relationships involving objects, composition, texture, style or broader semantic content. For example, an ecommerce application could compare the embedding of a selected jacket with vectors across its catalog, surfacing visually related products.\u003C/p\u003E\n\u003Cp\u003EBecause the model uses learned visual features, similarity doesn’t require identical pixels. Cropping, background changes and alternate viewpoints can still produce nearby vectors as long as the underlying subject remains recognizable to the model.\u003C/p\u003E\n\u003Ch3\u003EMultimodal embeddings\u003C/h3\u003E\n\u003Cp\u003EMultimodal embedding models align representations across content types. An image encoder and a text encoder, for example, can learn to place an image near the descriptions associated with it.\u003C/p\u003E\n\u003Cp\u003E\u003Ca href=\"https://proceedings.mlr.press/v139/radford21a\" target=\"_blank\"\u003ECLIP\u003C/a\u003E demonstrated this approach by jointly training image and text encoders to identify the correct pairings among images and captions. Once the representations are aligned, a text query such as “red hiking backpack beside a tent” can retrieve an image even though the query consists of words and the indexed item consists of pixels.\u003C/p\u003E\n\u003Cp\u003EFor long audio and video files, the unit of representation determines how precisely the content can be retrieved. One vector for the complete asset summarizes it broadly, while separate embeddings for scenes, time ranges, audio tracks or transcript segments preserve more local detail. Segment-level indexing allows a search result to point to the relevant moment rather than returning only the full file.\u003C/p\u003E\n\u003Ch3\u003EEntity and item embeddings\u003C/h3\u003E\n\u003Cp\u003EEntity embeddings represent objects such as products, users, songs or graph nodes using information about their attributes, interactions or connections. The training data determines which of those relationships shape the vector space. In a recommendation system, for example, two products may receive nearby embeddings because similar customers view or purchase them, even when their descriptions differ.\u003C/p\u003E\n\u003Cp\u003EA graph model may place two entities near each other because they connect to the same kinds of nodes or occupy similar positions within the network. Proximity therefore has a task-specific interpretation. For text embeddings, it often reflects linguistic or semantic relationships. For product or user embeddings, it typically reflects behavior. For graph embeddings, it may reflect connectivity and structure.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"},"title_common-embedding-use-cases":{"id":"title-v2-73f34a47f9","additionalClasses":"anchor-title anchor-title--common-embedding-use-cases","type":"heading2","lines":["Common embedding use cases"],":type":"snowflake-site/components/title-v2"},"text_common-embedding-use-cases_0":{"id":"text-d571a9d438","text":"\u003Cp\u003EEmbeddings are used when an application needs to retrieve, group or compare items according to learned relationships rather than exact matches. Common uses include semantic search, RAG, recommendations, classification, clustering and near-duplicate detection.\u003C/p\u003E\n\u003Ch3\u003ESemantic search and RAG\u003C/h3\u003E\n\u003Cp\u003ESemantic search retrieves content according to meaning, even when the query and the source use different wording. Production search systems often combine embedding-based retrieval with keyword matching, metadata filters and reranking. Keywords preserve exact matches for names, codes and specialized terminology, while filters restrict results by criteria such as date, region or access permissions. Vector databases and vector-aware search services store the embeddings and support retrieval across large collections.\u003C/p\u003E\n\u003Cp\u003ERAG uses those retrieved passages as context for a language model. The search layer selects the supporting material, and the model uses that material when generating its response. Retrieval quality therefore shapes which evidence reaches the model and, in turn, the quality of the final answer.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"},"callout_common-embedding-use-cases_0":{"id":"text-86315a8e78","additionalClasses":"callout callout--tip","text":"\u003Cp\u003E\u003Cstrong\u003EQUICK TIP\u003C/strong\u003E\u003C/p\u003E\n\u003Cp\u003ETest hybrid retrieval when the collection contains product names, identifiers, error codes or specialized terminology. Keyword search can preserve exact matches that semantic similarity may otherwise underweight.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-size-regular text-color-text-05"},"text_common-embedding-use-cases_1":{"id":"text-b16be4a9bd","text":"\u003Ch3\u003ERecommendations\u003C/h3\u003E\n\u003Cp\u003ERecommendation systems use embeddings to identify products, media or other items associated with similar users, behaviors or attributes, even when their descriptions and categories differ.\u003C/p\u003E\n\u003Cp\u003EThe vectors usually supply candidate items rather than the final ranked list. Availability, recency, business priorities and the user’s current context may all influence what appears. A streaming service, for instance, might begin with items near a viewer’s recent activity, then adjust the ranking according to language, region and previously watched content.\u003C/p\u003E\n\u003Ch3\u003EClassification and clustering\u003C/h3\u003E\n\u003Cp\u003EEmbeddings often serve as learned input features for a classifier. A support team might generate embeddings for previously labeled tickets, then train a smaller model to assign new requests to categories such as billing, account access or technical support.\u003C/p\u003E\n\u003Cp\u003EClustering works without predefined labels. By grouping records with nearby embeddings, practitioners can explore recurring themes in customer feedback, organize document collections or identify common incident patterns. The algorithm supplies the groups, but teams still need to inspect representative records and determine what each cluster means in the business context.\u003C/p\u003E\n\u003Ch3\u003ESimilarity detection and deduplication\u003C/h3\u003E\n\u003Cp\u003EEmbedding-based similarity helps identify records that are related without being exact copies, such as a passage that’s been paraphrased or an image that’s been cropped and resized.\u003C/p\u003E\n\u003Cp\u003EThese comparisons usually produce candidates for review or further processing. The application may apply a similarity threshold, a secondary model or deterministic checks before merging records or marking them as duplicates. The appropriate threshold depends on the content and the consequences of an incorrect match, so teams typically calibrate it against representative examples.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"},"title_generating-and-using-embeddings-on-snowflake":{"id":"title-v2-bf082c9ea8","additionalClasses":"anchor-title anchor-title--generating-and-using-embeddings-on-snowflake","type":"heading2","lines":["Generating and using embeddings on Snowflake"],":type":"snowflake-site/components/title-v2"},"text_generating-and-using-embeddings-on-snowflake_0":{"id":"text-b04d575c9c","text":"\u003Cp\u003ESnowflake supports embedding workflows within the same environment used to store, process and govern the underlying data. Teams can generate vectors directly, use them in SQL or build a managed retrieval service, depending on how much control the application requires.\u003C/p\u003E\n\u003Cp\u003EThe AI_EMBED function creates embeddings from text or images and returns Snowflake’s native VECTOR data type. Vectors can remain alongside the source records and business metadata, then be compared through functions for cosine similarity, inner product and vector distance.\u003C/p\u003E\n\u003Cp\u003EFor semantic search and RAG, \u003Ca href=\"https://www.snowflake.com/en/blog/cortex-search-ai-hybrid-search/\"\u003ECortex Search\u003C/a\u003E manages the indexing and serving layer. It combines vector retrieval with keyword search, supporting conceptual matches while preserving precise retrieval for names, codes and specialized terminology. Teams can use managed embeddings or provide vectors created with an open source, commercial or custom-trained model.\u003C/p\u003E\n\u003Cp\u003EWorkloads that need specialized dependencies or custom inference logic can run through \u003Ca href=\"https://www.snowflake.com/en/developers/guides/notebook-container-runtime/\"\u003ESnowflake Container Runtime\u003C/a\u003E, which supports CPU and GPU environments for experimentation and batch inference. For multimodal retrieval, AI_MULTI_EMBED can generate segment-level representations from text, images, audio and video, subject to current model and regional availability.\u003C/p\u003E\n\u003Cp\u003E\u003Ca href=\"https://www.snowflake.com/en/blog/introducing-snowflake-arctic-embed-snowflakes-state-of-the-art-text-embedding-family-of-models/\"\u003EArctic Embed 2.0\u003C/a\u003E models provide an open source option designed for retrieval, including multilingual support and \u003Ca href=\"https://arxiv.org/abs/2205.13147\" target=\"_blank\"\u003EMatryoshka Representation Learning\u003C/a\u003E to enable practitioners to test shorter, compressed vectors when storage and search efficiency are priorities.\u003C/p\u003E\n\u003Cp\u003ETogether, these capabilities support managed retrieval through Cortex Search, direct vector operations in SQL and custom embedding pipelines for workloads with specialized model or compute requirements.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"},"title_managing-embeddings-as-a-production-workload":{"id":"title-v2-2429682bd0","additionalClasses":"anchor-title anchor-title--managing-embeddings-as-a-production-workload","type":"heading2","lines":["Managing embeddings as a production workload"],":type":"snowflake-site/components/title-v2"},"text_managing-embeddings-as-a-production-workload_0":{"id":"text-5ef4152373","text":"\u003Cp\u003EEmbeddings give applications a consistent numerical representation for content and entities that would otherwise be difficult to compare. A model generates the vectors, a similarity measure establishes proximity and a retrieval or analytical process uses those relationships to rank, group or classify the original items.\u003C/p\u003E\n\u003Cp\u003EAt production scale, the work extends beyond creating the vector. Teams need to decide what receives an embedding, keep changing content synchronized with the index and serve query vectors within the application’s latency target. Model quality, inference throughput and retrieval design all contribute to the final result.\u003C/p\u003E\n\u003Cp\u003EWhen those components are evaluated together, embeddings can support search and AI workflows across text, images, media and structured entities without limiting relevance to exact keyword overlap.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"},"callout_managing-embeddings-as-a-production-workload_0":{"id":"text-7e32279102","additionalClasses":"callout callout--tip","text":"\u003Cp\u003E\u003Cstrong\u003EKEY TAKEAWAY\u003C/strong\u003E\u003C/p\u003E\n\u003Cp\u003EEmbeddings provide the representation layer for semantic search, RAG and recommendations, while the surrounding retrieval and inference architecture determines whether those representations produce useful results at scale.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-size-regular text-color-text-05"}},":itemsOrder":["callout__0","text__0","title_what-are-embeddings","text_what-are-embeddings_0","yt_what-are-embeddings_0","title_how-embedding-models-create-vector-representations","text_how-embedding-models-create-vector-representations_0","callout_how-embedding-models-create-vector-representations_0","title_how-applications-use-embeddings-to-find-similar-items","text_how-applications-use-embeddings-to-find-similar-items_0","card_v2_how-applications-use-embeddings-to-find-similar-items_0","title_types-of-embeddings","text_types-of-embeddings_0","title_common-embedding-use-cases","text_common-embedding-use-cases_0","callout_common-embedding-use-cases_0","text_common-embedding-use-cases_1","title_generating-and-using-embeddings-on-snowflake","text_generating-and-using-embeddings-on-snowflake_0","title_managing-embeddings-as-a-production-workload","text_managing-embeddings-as-a-production-workload_0","callout_managing-embeddings-as-a-production-workload_0"],"appliedCssClassNames":"snowflake-responsive-container-inner-padding-medium"},"flexible_column_content_container_2":{"additionalClasses":"hub-sidebar","layout":"SIMPLE","id":"hub-body-aside",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"container":{"additionalClasses":"sticky-sidebar","layout":"RESPONSIVE_GRID","columnCount":12,"columnClassNames":{"text_943981956_copy_":"aem-GridColumn aem-GridColumn--default--12","text_copy":"aem-GridColumn aem-GridColumn--default--12"},"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","id":"container-703e8b9152",":type":"snowflake-site/components/container",":items":{"text_943981956_copy_":{"id":"text-a255b3e842","additionalClasses":"eyebrow-text","text":"\u003Cp\u003EIn This Guide\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-size-regular"},"text_copy":{"id":"text-782cb37f80","additionalClasses":"page-toc","text":"\u003Cul\u003E\u003Cli data-anchor=\"what-are-embeddings\"\u003EWhat are embeddings?\u003C/li\u003E\u003Cli data-anchor=\"how-embedding-models-create-vector-representations\"\u003EHow embedding models create vector representations\u003C/li\u003E\u003Cli data-anchor=\"how-applications-use-embeddings-to-find-similar-items\"\u003EHow applications use embeddings to find similar items\u003C/li\u003E\u003Cli data-anchor=\"types-of-embeddings\"\u003ETypes of embeddings\u003C/li\u003E\u003Cli data-anchor=\"common-embedding-use-cases\"\u003ECommon embedding use cases\u003C/li\u003E\u003Cli data-anchor=\"generating-and-using-embeddings-on-snowflake\"\u003EGenerating and using embeddings on Snowflake\u003C/li\u003E\u003Cli data-anchor=\"managing-embeddings-as-a-production-workload\"\u003EManaging embeddings as a production workload\u003C/li\u003E\u003C/ul\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-size-small text-color-text-05"}},":itemsOrder":["text_943981956_copy_","text_copy"],"appliedCssClassNames":"snowflake-responsive-container-inner-padding-medium"}},":itemsOrder":["container"],"appliedCssClassNames":"snowflake-responsive-container-inner-padding-small"},":type":"snowflake-site/components/flexible-column-container","isBlogPage":false,"isActiveTOC":false},"flexible_column_cont_1786318617":{"id":"flexible-column-container-b3957ea166","propertiesId":"hub-faq","type":"2-column-40-60","alignColumns":"top","containerMaxWidth":"extra-large","topPadding":"large","bottomPadding":"large","spaceBetween":"small","reverseOnMobile":false,"carouselOnMobile":false,"backgroundImageOption":"none","flexible_column_content_container_1":{"layout":"SIMPLE","id":"hub-faq-intro",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"title_v2_copy":{"id":"title-v2-d438449443","additionalClasses":"hub-faq__headline","type":"heading2","lines":["Frequently Asked Questions"],":type":"snowflake-site/components/title-v2","appliedCssClassNames":"left-alignment"},"text_copy":{"id":"text-a3c90999fa","additionalClasses":"hub-faq__subheadline","text":"\u003Cp\u003EYour common questions about embeddings, answered by Snowflake experts.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"}},":itemsOrder":["title_v2_copy","text_copy"],"appliedCssClassNames":"snowflake-responsive-container-inner-padding-extra-small"},"flexible_column_content_container_2":{"layout":"SIMPLE","id":"hub-faq-accordions",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"simple_snowflake_acc":{"id":"simple-snowflake-accordion-9e0d3e72c9","additionalClasses":"seo-hub__faqs","showDivider":false,"accordionItemsList":[{"title":"What are word embeddings?","richText":"\u003Cp\u003EWord embeddings represent individual words or vocabulary items. Many modern text models use contextual representations, allowing the vector for a word to change according to its surrounding language.\u003C/p\u003E"},{"title":"What is an embedding model?","richText":"\u003Cp\u003EAn embedding model converts an input into a vector whose position reflects relationships learned during training. The trained model can generate vectors for new inputs during inference, allowing an application to compare them with representations created for other items.\u003C/p\u003E"},{"title":"What is the difference between an embedding and a vector?","richText":"\u003Cp\u003EA vector is an ordered sequence of numbers. An embedding is a vector that serves as the learned representation of an object within a particular model’s representation space. Vectors can also contain coordinates, measurements or manually engineered features, so the terms are related without being interchangeable.\u003C/p\u003E"},{"title":"How many dimensions should an embedding have?","richText":"\u003Cp\u003EThe model usually determines the available output dimensions. Longer vectors increase storage and comparison costs and may provide additional representational capacity; shorter or compressed vectors can improve efficiency. Teams should compare the available sizes against representative retrieval queries, latency requirements and infrastructure constraints.\u003C/p\u003E"}],":type":"snowflake-site/components/simple-snowflake-accordion"}},":itemsOrder":["simple_snowflake_acc"]},":type":"snowflake-site/components/flexible-column-container","isBlogPage":false,"isActiveTOC":false,"appliedCssClassNames":"snowflake-flexible-column-container-gray-10-bg"},"flexible_column_cont_1467213961":{"id":"flexible-column-container-2cc0483012","propertiesId":"hub-explore-resources-header","type":"1-column","alignColumns":"top","containerMaxWidth":"extra-large","topPadding":"large","bottomPadding":"none","spaceBetween":"small","reverseOnMobile":false,"carouselOnMobile":false,"backgroundImageOption":"none","flexible_column_content_container_1":{"layout":"SIMPLE","id":"container-24d60e9d94",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"title_v2":{"id":"title-v2-b987b2f8b2","additionalClasses":"hub-explore-resources-header__headline","type":"heading2","lines":["Explore AI Resources"],":type":"snowflake-site/components/title-v2","appliedCssClassNames":"left-alignment"}},":itemsOrder":["title_v2"]},":type":"snowflake-site/components/flexible-column-container","isBlogPage":false,"isActiveTOC":false,"appliedCssClassNames":"snowflake-flexible-column-container-gray-10-bg"},"flexible_column_cont_912630531":{"id":"flexible-column-container-4477be5d9f","propertiesId":"hub-explore-resources-grid","type":"1-column","alignColumns":"top","containerMaxWidth":"extra-large","topPadding":"small","bottomPadding":"large","spaceBetween":"small","reverseOnMobile":false,"carouselOnMobile":false,"backgroundImageOption":"none","flexible_column_content_container_1":{"layout":"SIMPLE","id":"hub-explore-resources-grid-inner",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"resource_chip_0":{"id":"content-chip-3ee289b2b0","tagText":"GUIDE","tagColor":"#AAE5EA","cta":{"id":"cta","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"https://www.snowflake.com/en/developers/guides/scale-embeddings-with-snowflake-notebooks-on-container-runtime/"},"linkTargetContentType":"GENERIC",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_EXTERNAL","text":"Read more"},"headline":{"id":"title","type":"heading5","lines":["Scale Embeddings with Snowflake Notebooks on Container Runtime"],":type":"snowflake-site/components/title-v2"},":type":"snowflake-site/components/content-chip","appliedCssClassNames":"snowflake-content-chip-white-bg"},"resource_chip_1":{"id":"content-chip-97a802ff21","tagText":"BLOG","tagColor":"#29B5E8","cta":{"id":"cta","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"https://www.snowflake.com/en/blog/engineering/embedding-inference-arctic-16x-faster/"},"linkTargetContentType":"GENERIC",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_EXTERNAL","text":"Read more"},"headline":{"id":"title","type":"heading5","lines":["Scaling vLLM for Embeddings: 16x Throughput and Cost Reduction"],":type":"snowflake-site/components/title-v2"},":type":"snowflake-site/components/content-chip","appliedCssClassNames":"snowflake-content-chip-white-bg"},"resource_chip_2":{"id":"content-chip-76a457759b","tagText":"BLOG","tagColor":"#29B5E8","cta":{"id":"cta","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"https://www.snowflake.com/en/blog/engineering/arctic-embed-joins-arctictraining/"},"linkTargetContentType":"GENERIC",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_EXTERNAL","text":"Read more"},"headline":{"id":"title","type":"heading5","lines":["Snowflake Arctic Embed Joins ArcticTraining: Simple and Scalable Embedding Model Training"],":type":"snowflake-site/components/title-v2"},":type":"snowflake-site/components/content-chip","appliedCssClassNames":"snowflake-content-chip-white-bg"},"resource_chip_3":{"id":"content-chip-5fc183dca8","tagText":"GUIDE","tagColor":"#AAE5EA","cta":{"id":"cta","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"https://www.snowflake.com/en/developers/guides/getting-started-with-access-controls-for-cortex-search/"},"linkTargetContentType":"GENERIC",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_EXTERNAL","text":"Read more"},"headline":{"id":"title","type":"heading5","lines":["Getting Started with Access Controls for RAGs (Cortex Search)"],":type":"snowflake-site/components/title-v2"},":type":"snowflake-site/components/content-chip","appliedCssClassNames":"snowflake-content-chip-white-bg"}},":itemsOrder":["resource_chip_0","resource_chip_1","resource_chip_2","resource_chip_3"]},":type":"snowflake-site/components/flexible-column-container","isBlogPage":false,"isActiveTOC":false,"appliedCssClassNames":"snowflake-flexible-column-container-gray-10-bg"},"flexible_column_cont_1158003461":{"id":"flexible-column-container-ad05967201","propertiesId":"hub-explore-topics-header","type":"1-column","alignColumns":"top","containerMaxWidth":"extra-large","topPadding":"large","bottomPadding":"none","spaceBetween":"small","reverseOnMobile":false,"carouselOnMobile":false,"backgroundImageOption":"none","flexible_column_content_container_1":{"layout":"SIMPLE","id":"container-9b937a0cc7",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"title_v2_copy_copy":{"id":"title-v2-c2667b3535","additionalClasses":"hub-explore-topics-header__headline","type":"heading2","lines":["Explore AI Topics"],":type":"snowflake-site/components/title-v2","appliedCssClassNames":"left-alignment"},"text_copy_copy":{"id":"text-6101239d08","additionalClasses":"hub-explore-topics-header__subheadline","text":"\u003Cp\u003EDeep dives into every aspect of artificial intelligence\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"}},":itemsOrder":["title_v2_copy_copy","text_copy_copy"],"appliedCssClassNames":"snowflake-responsive-container-inner-padding-extra-small"},":type":"snowflake-site/components/flexible-column-container","isBlogPage":false,"isActiveTOC":false,"appliedCssClassNames":"snowflake-flexible-column-container-white-bg"},"flexible_column_cont_1377146023":{"id":"flexible-column-container-b458987dc6","propertiesId":"hub-explore-topics-grid","type":"3-column-even","alignColumns":"top","containerMaxWidth":"extra-large","topPadding":"small","bottomPadding":"large","spaceBetween":"small","reverseOnMobile":false,"carouselOnMobile":false,"backgroundImageOption":"none","flexible_column_content_container_1":{"layout":"SIMPLE","id":"container-04688592fc",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"topic_card_0":{"id":"card-v2-8e9a8030d4","configurationStatus":{"configured":true,"message":""},":type":"snowflake-site/components/card-v2","text":{"id":"text","text":"\u003Cp\u003EFeature engineering turns raw data into meaningful signals that improve model accuracy, generalization and real-world reliability.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text"},"layoutStyle":"vertical","title":{"id":"title","type":"heading4","lines":["Feature Engineering"],":type":"snowflake-site/components/title-v2"},"button":{"id":"button","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/artificial-intelligence/machine-learning/feature-engineering/"},"linkTargetContentType":"GENERIC",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_INTERNAL","text":"Learn more"},"type":"content-card"}},":itemsOrder":["topic_card_0"]},"flexible_column_content_container_2":{"layout":"SIMPLE","id":"container-d71d0ef940",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"topic_card_1":{"id":"card-v2-dbb8b2a959","configurationStatus":{"configured":true,"message":""},":type":"snowflake-site/components/card-v2","text":{"id":"text","text":"\u003Cp\u003ERAG improves LLM outputs by grounding them in external knowledge, boosting accuracy and relevance without requiring retraining.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text"},"layoutStyle":"vertical","title":{"id":"title","type":"heading4","lines":["Retrieval-Augmented Generation"],":type":"snowflake-site/components/title-v2"},"button":{"id":"button","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/fundamentals/rag/"},"linkTargetContentType":"GENERIC",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_INTERNAL","text":"Learn more"},"type":"content-card"}},":itemsOrder":["topic_card_1"]},"flexible_column_content_container_3":{"layout":"SIMPLE","id":"container-cf127bb9fd",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"topic_card_2":{"id":"card-v2-d3843ad1f0","configurationStatus":{"configured":true,"message":""},":type":"snowflake-site/components/card-v2","text":{"id":"text","text":"\u003Cp\u003EData augmentation adds meaningful variation and coverage to help models perform more reliably in real-world conditions\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/text"},"layoutStyle":"vertical","title":{"id":"title","type":"heading4","lines":["Data Augmentation"],":type":"snowflake-site/components/title-v2"},"button":{"id":"button","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/artificial-intelligence/machine-learning/model-training/data-augmentation/"},"linkTargetContentType":"GENERIC",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_INTERNAL","text":"Learn more"},"type":"content-card"}},":itemsOrder":["topic_card_2"]},":type":"snowflake-site/components/flexible-column-container","isBlogPage":false,"isActiveTOC":false,"appliedCssClassNames":"snowflake-flexible-column-container-white-bg"}},":itemsOrder":["flexible_column_cont","flexible_column_cont_939100716","flexible_column_cont_1398138236","flexible_column_cont_663228916","flexible_column_cont_1786318617","flexible_column_cont_1467213961","flexible_column_cont_912630531","flexible_column_cont_1158003461","flexible_column_cont_1377146023"],":type":"wcm/foundation/components/responsivegrid"},"modal_container":{"layout":"SIMPLE","id":"container-e94a556d99",":type":"snowflake-site/components/modal/modal-container",":items":{},":itemsOrder":[]},"markup_editor_928258845":{"id":"markup-editor-d9865efe1c","title":" ","cssContent":".snowflake-flexible-column-container-gray-10-bg\u003E.snowflake-flexible-column-container{background-color:var(--ui-background-05) !important}.text-size-regular:has(.seo-hub-hero__related-topic-label){display:flex;align-items:center}.hub-hero__headline span{text-transform:none !important}.hub-hero__subheadline p{max-width:70ch;margin-top:8px}.hub-hero__authors \u003E .container \u003E .cmp-container \u003E .aem-container{display:flex;flex-direction:row}.hub-hero__authors \u003E .container \u003E .cmp-container \u003E .aem-container \u003E div{width:auto !important;margin:24px 48px 0 0 !important}.hub-hero__authors .heading-5-v2{gap:var(--spacing-00)}.hub-hero__authors .snowflake-content-chip-button{display:none !important}.hub-hero__authors .snowflake-person-chip-content .body-2,.hub-hero__authors .snowflake-content-chip-content .snowflake-title-v2-line{font-size:16px !important;line-height:20px !important;font-family:\"Lato\",sans-serif !important;color:#000 !important;font-weight:600 !important}.hub-hero__authors .snowflake-person-chip-content .body-3,.hub-hero__authors .snowflake-content-chip-content .snowflake-title-v2-line:not(:first-child){font-weight:400 !important;color:var(--text-05) !important;font-size:16px !important}.hub-hero__authors .snowflake-image-container img{aspect-ratio:1 !important;border-radius:100%;overflow:hidden}.hub-hero__authors .snowflake-person-chip-avatar{width:56px;height:56px}.hub-hero__authors .snowflake-content-chip{align-items:center;display:inline-flex}.hub-hero__authors .snowflake-content-chip-image{max-width:56px;line-height:0;margin-right:var(--spacing-03)}.hub-hero__authors .snowflake-person-chip-inner-horizontal{gap:var(--spacing-03)}@media screen and (min-width:1367px){.hub-hero__headline .heading-1-v2{font-size:48px;line-height:44px}}#hub-explore-resources-grid-inner\u003E.container\u003E.cmp-container\u003E.aem-container{display:flex;flex-direction:row;flex-wrap:wrap;gap:24px}#hub-explore-resources-grid-inner\u003E.container\u003E.cmp-container\u003E.aem-container::before,#hub-explore-resources-grid-inner\u003E.container\u003E.cmp-container\u003E.aem-container::after{display:none}#hub-explore-resources-grid-inner\u003E.container\u003E.cmp-container\u003E.aem-container\u003Ediv{width:calc(25% - 18px)}.text-color-text-05 .snowflake-text h2,.text-color-text-05.cq-Editable-dom h2,.text-color-text-05 .snowflake-text h3,.text-color-text-05.cq-Editable-dom h3,.text-color-text-05 .snowflake-text h4,.text-color-text-05.cq-Editable-dom h4,.text-color-text-05 .snowflake-text h5,.text-color-text-05.cq-Editable-dom h5,.text-color-text-05 .snowflake-text h6,.text-color-text-05.cq-Editable-dom h6{color:#000 !important}",":type":"snowflake-site/components/markup-editor","isGSAPEnabled":false},"markup_editor_597730182":{"id":"markup-editor-0366d32522","title":" ","cssContent":".sf-copy-markdown [data-copy-md]{display:inline-flex;align-items:center;gap:6px;padding:8px 16px;font-family:'Texta',sans-serif;text-transform:uppercase;font-weight:800 !important;font-size:14px;font-weight:500;color:#11567f;background:#f0faff;border:1px solid #b8e6f9;border-radius:24px;cursor:pointer;transition:background .2s ease,border-color .2s ease,color .2s ease}.sf-copy-markdown [data-copy-md]:hover{background:#ddf3fc;border-color:#29b5e8}.sf-copy-markdown [data-copy-md][data-copied=\"1\"]{color:#0f7b3e;background:#ecfdf5;border-color:#6ee7a0;pointer-events:none}.sf-copy-markdown [data-copy-md] svg{flex-shrink:0}.longform-conten .snowflake-content-chip-white-bg .snowflake-content-chip{box-shadow:0 0 24px 4px rgba(0,0,0,.02),0 4px 8px 0 rgba(0,0,0,.04);flex-direction:row-reverse;align-items:center}.longform-conten .snowflake-content-chip-button{display:none}.longform-conten .snowflake-content-chip-image__inner{aspect-ratio:5 / 3;display:flex;justify-content:center;align-items:center;background-color:var(--ui-01);border-radius:4px}.longform-conten .snowflake-content-chip-image{margin-right:0;margin-left:48px}.longform-conten .snowflake-content-chip-image img{width:50%;border-radius:0 !important;object-fit:contain}.longform-content .black-blue-text-color .snowflake-title-v2-line:not(:first-child){font-size:14px !important;font-weight:400 !important;color:rgba(0,0,0,.6) !important;margin-top:8px !important}.page-toc ul li:first-child{padding-top:0 !important}.page-toc ul li:last-child{padding-bottom:0 !important}.seo-hub__top-bar \u003E .container \u003E .cmp-container \u003E .aem-container \u003E div:first-child{flex-grow:1}.sf-copy-markdown{margin-top:40px !important}.page-toc ul{margin-top:16px !important}.seo-hub__top-bar \u003E .container \u003E .cmp-container \u003E .aem-container{display:flex;justify-content:space-between}.sf-copy-markdown{}.callout.snowflake-text p:not(:first-child){margin-top:var(--spacing-01)}.callout \u003E span \u003E p:first-child \u003E strong,.callout \u003E span \u003E p:first-child \u003E b{text-transform:uppercase;font-family:'Texta',sans-serif;font-size:16px !important;color:var(--ui-01) !important}.seo-hub-hero__subheadline p{max-width:50ch}.tag-group ul{list-style-type:none;padding:0;margin:0;display:flex;flex-direction:row;row-gap:12px;column-gap:8px;align-items:center;flex-wrap:wrap}.tag-group ul li:first-child{flex-shrink:0}.tag-group ul li a{display:inline-block;padding:2px 12px;border-radius:48px;background-color:#ededed;color:#666;font-size:14px !important}@media screen and (min-width:1367px){.seo-hub-hero__headline span.snowflake-title-v2-line{font-size:56px !important}}.callout.snowflake-text p:not(:first-child){margin-top:var(--spacing-01)}.callout \u003E span \u003E p:first-child \u003E b{text-transform:uppercase;font-family:'Texta',sans-serif;font-size:16px !important;color:var(--ui-01) !important}.seo-hub-hero__subheadline p{max-width:80ch}#hero:has(.snowflake-youtube-lite) .seo-hub-hero__subheadline p{max-width:50ch}.tag-group ul{list-style-type:none;padding:0;margin:0;display:flex;flex-direction:row;row-gap:12px;column-gap:8px;align-items:center;flex-wrap:wrap}.tag-group ul li:first-child{width:100%;flex-shrink:0}.tag-group ul li a{display:inline-block;padding:2px 12px;border-radius:48px;background-color:#ededed;color:#666;font-size:14px !important}@media screen and (min-width:1367px){.seo-hub-hero__headline span.snowflake-title-v2-line{font-size:56px !important}}",":type":"snowflake-site/components/markup-editor","isGSAPEnabled":false},"markup_editor":{"id":"markup-editor-73e1a5aa2b","title":" ","cssContent":"div.snowflake-breadcrumb a.snowflake-breadcrumb-item,.snowflake-breadcrumb div.snowflake-breadcrumb-item{text-transform:none;font-weight:500}.snowflake-breadcrumb svg{display:none !important}.snowflake-breadcrumb a:has(svg)::after{content:'/';margin:0 12px;color:#666}.hub-sidebar{padding:0 40px}.sticky-sidebar{max-width:340px;margin-left:auto}.page-toc ul{list-style-type:none;padding:0}.page-toc li{padding:8px 16px;border-left:4px solid var(--ui-01);cursor:pointer;transition:300ms ease all}.page-toc li:hover{color:var(--ui-01);border-color:#7fd3f1;transition:300ms ease all}.callout,.customer-card{background-color:#eef9fd;border-left:4px solid var(--ui-01);padding:24px 24px 24px 32px;border-radius:4px}.logo-container{max-width:180px}.longform-content li{margin-top:1rem !important}div.longform-content p{max-width:80ch}.bolder .snowflake-title-v2-line{font-weight:900 !important}.border-top\u003Ediv{border-top:1px solid #ccc;padding-top:48px}.related-topics ul{list-style-type:none;padding:0;margin:0;display:flex;gap:8px;flex-wrap:wrap}.related-topics li{display:inline-block;border:1px solid #ccc;padding:4px 12px;border-radius:24px}div.longform-content .snowflake-text h2,div.longform-content .snowflake-text .heading-2-v2,div.longform-content .snowflake-text h3,div.longform-content .snowflake-text .heading-3-v2,div.longform-content .snowflake-title-v2 .heading-3-v2,div.longform-content .snowflake-text h4,div.longform-content .snowflake-text .heading-4-v2,div.longform-content .snowflake-title-v2 .heading-4-v2,div.longform-content .snowflake-text h5,div.longform-content .snowflake-text .heading-5-v2,div.longform-content .snowflake-title-v2 .heading-5-v2,div.longform-content .snowflake-text h6,div.longform-content .snowflake-title-v2 .heading-6-v2,div.longform-content .snowflake-text .heading-6-v2{text-transform:none !important}div.longform-content .snowflake-text h2,div.longform-content .snowflake-text .heading-2-v2,div.longform-content .snowflake-text h3,div.longform-content .snowflake-text .heading-3-v2,div.longform-content .snowflake-text h4,div.longform-content .snowflake-text .heading-4-v2,div.longform-content .snowflake-text h5,div.longform-content .snowflake-text .heading-5-v2,div.longform-content .snowflake-text h6,div.longform-content .snowflake-text .heading-6-v2{margin-top:1.5rem !important;line-height:1.1 !important}div.longform-content .snowflake-text h3,div.longform-content .snowflake-text .heading-3-v2,div.longform-content .snowflake-title-v2 .heading-3-v2,div.longform-content .snowflake-text h4,div.longform-content .snowflake-text .heading-4-v2,div.longform-content .snowflake-title-v2 .heading-4-v2,div.longform-content .snowflake-text h5,div.longform-content .snowflake-text .heading-5-v2,div.longform-content .snowflake-title-v2 .heading-5-v2,div.longform-content .snowflake-text h6,div.longform-content .snowflake-text .heading-6-v2,div.longform-content .snowflake-title-v2 .heading-6-v2{font-family:Lato,sans-serif !important;font-weight:800 !important}div.longform-content .snowflake-text h2,div.longform-content .snowflake-text .heading-2-v2,div.longform-content .snowflake-title-v2 .heading-2-v2{text-transform:none !important;font-size:28px !important}div.longform-content .snowflake-text h3,div.longform-content .snowflake-text .heading-3-v2,div.longform-content .snowflake-title-v2 .heading-3-v2{font-size:22px !important}div.longform-content .snowflake-text h4,div.longform-content .snowflake-text .heading-4-v2,div.longform-content .snowflake-title-v2 .heading-4-v2{font-size:18px !important}div.longform-content .snowflake-text h5,div.longform-content .snowflake-text .heading-5-v2,div.longform-content .snowflake-title-v2 .heading-5-v2{font-size:16px !important}div.longform-content .snowflake-text h6,div.longform-content .snowflake-text .heading-6-v2,div.longform-content .snowflake-title-v2 .heading-6-v2{font-size:14px !important}@media screen and (min-width:992px){div.longform-content .snowflake-text h2,div.longform-content .snowflake-text .heading-2-v2,div.longform-content .snowflake-title-v2 .heading-2-v2{font-size:38px !important}div.longform-content .snowflake-text h3,div.longform-content .snowflake-text .heading-3-v2,div.longform-content .snowflake-title-v2 .heading-3-v2{font-size:26px !important}div.longform-content .snowflake-text h4,div.longform-content .snowflake-text .heading-4-v2,div.longform-content .snowflake-title-v2 .heading-4-v2{font-size:22px !important}div.longform-content .snowflake-text h5,div.longform-content .snowflake-text .heading-5-v2,div.longform-content .snowflake-title-v2 .heading-5-v2{font-size:18px !important}div.longform-content .snowflake-text h6,div.longform-content .snowflake-text .heading-6-v2,div.longform-content .snowflake-title-v2 .heading-6-v2{font-size:16px !important}}.sticky-sidebar .page-toc li.is-active{font-weight:600;color:var(--snow-blue,#29b5e8)}.sticky-sidebar .page-toc li[data-anchor]{cursor:pointer}.longform-content table{margin-top:24px;margin-bottom:24px;width:100%;background-color:var(--ui-background-01);border-collapse:collapse;border:2px solid var(--ui-background-09);font-family:'Lato',sans-serif;color:var(--ui-background-09)}.longform-content table thead{background-color:var(--ui-01)}.longform-content th,.longform-content td{min-width:120px;border:2px solid var(--ui-background-09);padding:var(--spacing-01)}.longform-content ol{margin-top:0 !important}.longform-content ol li{margin-bottom:1rem !important}.longform-content ul li{margin:0;padding:0 0 0 32px;position:relative}.longform-content ul{list-style-type:none}.longform-content ul li::before{content:\"\";display:block;border-radius:100%;background:#29b5e8;width:18px;height:18px;position:absolute;top:4px;left:0;border:5px solid #e5f2f7;box-sizing:border-box}.seo-customer.snowflake-card-v2-advanced-horizontal .snowflake-card-v2-advanced-image-container{max-width:200px}.seo-customer.snowflake-card-v2-advanced-horizontal .snowflake-card-v2-advanced-image-container img{object-fit:contain}.related-topics-outer-container \u003E .container \u003E .cmp-container \u003E .aem-container{display:flex;flex-direction:row}.related-topics-outer-container \u003E .container \u003E .cmp-container \u003E .aem-container \u003E div{width:auto !important;margin:0 !important}.related-topics-outer-container \u003E .container \u003E .cmp-container \u003E .aem-container \u003E div:first-child{margin-right:16px !important;flex-shrink:0}","jsContent":"(function(){var OFFSET=100;if(window.gsap&&window.ScrollTrigger){gsap.registerPlugin(ScrollTrigger);var sidebar=document.querySelector('.sticky-sidebar');var body=document.querySelector('.longform-content');if(sidebar&&body){ScrollTrigger.create({trigger:sidebar,start:'top 100px',endTrigger:body,end:'bottom bottom',pin:sidebar,pinSpacing:false});}}document.addEventListener('click',function(e){var li=e.target.closest('li[data-anchor]');if(!li)return;var slug=li.getAttribute('data-anchor');var heading=document.querySelector('.anchor-title--'+CSS.escape(slug));if(!heading)return;e.preventDefault();var top=heading.getBoundingClientRect().top+window.pageYOffset-OFFSET;window.scrollTo({top:top,behavior:'smooth'});history.replaceState(null,'','#'+slug);},false);var headings=document.querySelectorAll('[class*=\"anchor-title--\"]');if(headings.length&&'IntersectionObserver'in window){var io=new IntersectionObserver(function(entries){entries.forEach(function(entry){if(!entry.isIntersecting)return;var cls=Array.from(entry.target.classList).find(function(c){return c.indexOf('anchor-title--')===0;});if(!cls)return;var slug=cls.replace('anchor-title--','');document.querySelectorAll('li[data-anchor]').forEach(function(li){li.classList.toggle('is-active',li.getAttribute('data-anchor')===slug);});});},{rootMargin:'-20% 0px -70% 0px',threshold:0});headings.forEach(function(h){io.observe(h);});}})();",":type":"snowflake-site/components/markup-editor","isGSAPEnabled":true},"experiencefragment-footer":{"id":"experiencefragment-4eca63f2da","localizedFragmentVariationPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/footer/master/jcr:content","configured":true,":type":"snowflake-site/components/experiencefragment","xfModelPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/footer/master.xfmodel.json"},"experiencefragment":{"id":"experiencefragment-a78a4fd044","localizedFragmentVariationPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/footer-legal-disclaimers/master/jcr:content","configured":true,":type":"snowflake-site/components/experiencefragment","xfModelPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/footer-legal-disclaimers/master.xfmodel.json"}},":itemsOrder":["experiencefragment-banner","experiencefragment-header","responsivegrid","modal_container","markup_editor_928258845","markup_editor_597730182","markup_editor","experiencefragment-footer","experiencefragment"],":type":"wcm/foundation/components/responsivegrid"}},":itemsOrder":["root"],":hierarchyType":"page","isPasswordProtected":false,"analyticsContentTags":[],"analyticsEnabled":true,"coveoConfig":{"apiKey":"xx335921a6-2a0a-40f2-a167-e390b4766c3d","pipeline":"snowflake.com","searchHub":"snowflake.com","organizationId":"snowflakecomputingproduction8neljofn"},"analyticsDebugMode":false,"analyticsData":{"excludeFromAnalytics":false,"subCategory":"","pageType":"homepage","templateName":"base-page-template54","siteName":"snowflake","pageUrl":"/content/snowflake-site/global/en/artificial-intelligence/machine-learning/feature-engineering/embeddings","language":"en","category":"general","pageName":"How Embeddings Power Semantic Search, RAG and Recommendations","contentTags":[]},"locale":"en"}
  