{"templateName":"base-page-template54","canonicalLink":"https://www.snowflake.com/en/artificial-intelligence/ai-engineering/ai-cost-optimization/","robotsTags":[],"cssClassNames":"page basicpage summit-page","allowedRenditionsWidth":["320","480","640","768","960","1200","1440","1920"],"description":"AI cost optimization cuts LLM and inference costs without losing quality. Learn to trim context, route models, cache, and right-size infra on Snowflake.","language":"en","title":"AI Cost Optimization: How to Reduce LLM Costs | Snowflake","analyticsPageType":"homepage","analyticsCategory":"general","analyticsSubCategory":"","excludeFromAnalytics":false,":mappedPath":"/en/artificial-intelligence/ai-engineering/ai-cost-optimization/",":type":"snowflake-site/components/structure/page",":items":{"root":{"columnCount":12,"columnClassNames":{"markup_editor_928258845":"aem-GridColumn aem-GridColumn--default--12","experiencefragment-banner":"aem-GridColumn aem-GridColumn--default--12","experiencefragment-header":"aem-GridColumn aem-GridColumn--default--12","responsivegrid":"aem-GridColumn aem-GridColumn--default--12","markup_editor_597730182":"aem-GridColumn aem-GridColumn--default--12","experiencefragment-footer":"aem-GridColumn aem-GridColumn--default--12","experiencefragment":"aem-GridColumn aem-GridColumn--default--12","modal_container":"aem-GridColumn aem-GridColumn--default--12","markup_editor":"aem-GridColumn aem-GridColumn--default--12"},"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12",":items":{"experiencefragment-banner":{"id":"experiencefragment-dc667679c8","localizedFragmentVariationPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/pushdown-banner/master/jcr:content","configured":true,":type":"snowflake-site/components/experiencefragment","xfModelPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/pushdown-banner/master.xfmodel.json"},"experiencefragment-header":{"id":"experiencefragment-eddd91c8e5","localizedFragmentVariationPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/mega-nav-header/master/jcr:content","configured":true,":type":"snowflake-site/components/experiencefragment","xfModelPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/mega-nav-header/master.xfmodel.json","languageNavPath":"/content/snowflake-site/global/en/artificial-intelligence/ai-engineering/ai-cost-optimization.languagenav.json"},"responsivegrid":{"columnCount":12,"columnClassNames":{"flexible_column_cont_939100716":"aem-GridColumn aem-GridColumn--default--12","flexible_column_cont":"aem-GridColumn aem-GridColumn--default--12","flexible_column_cont_1158003461":"aem-GridColumn aem-GridColumn--default--12","flexible_column_cont_663228916":"aem-GridColumn aem-GridColumn--default--12","flexible_column_cont_912630531":"aem-GridColumn aem-GridColumn--default--12","flexible_column_cont_1398138236":"aem-GridColumn aem-GridColumn--default--12","flexible_column_cont_1786318617":"aem-GridColumn aem-GridColumn--default--12","flexible_column_cont_1467213961":"aem-GridColumn aem-GridColumn--default--12","flexible_column_cont_1377146023":"aem-GridColumn aem-GridColumn--default--12"},"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12",":items":{"flexible_column_cont":{"id":"flexible-column-container-8089920582","propertiesId":"hub-hero-breadcrumbs","type":"1-column","alignColumns":"top","containerMaxWidth":"extra-large","topPadding":"small","bottomPadding":"none","spaceBetween":"small","reverseOnMobile":false,"carouselOnMobile":false,"propertiesCSSClasses":"page-section","backgroundImageOption":"none","flexible_column_content_container_1":{"layout":"SIMPLE","id":"container-09a4e2e8cb",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"breadcrumb":{"id":"breadcrumb-117a1bd455","items":[{"id":"breadcrumb-117a1bd455-item-4b75ed29e6","link":{"valid":true,"url":"/en/artificial-intelligence/"},"active":false,"current":false,"title":"Artificial Intelligence",":type":"snowflake-site/components/structure/page","appliedCssClassNames":"summit-page"},{"id":"breadcrumb-117a1bd455-item-d5e33a6013","link":{"valid":true,"url":"/en/artificial-intelligence/ai-engineering/"},"active":false,"current":false,"title":"AI Engineering",":type":"snowflake-site/components/structure/page","appliedCssClassNames":"summit-page"},{"id":"breadcrumb-117a1bd455-item-4f4ade45da","link":{"valid":true,"url":"/en/artificial-intelligence/ai-engineering/ai-cost-optimization/"},"active":true,"current":true,"title":"AI Cost Optimization",":type":"snowflake-site/components/structure/page","appliedCssClassNames":"summit-page"}],":type":"snowflake-site/components/breadcrumb"}},":itemsOrder":["breadcrumb"]},":type":"snowflake-site/components/flexible-column-container","isActiveTOC":false,"isBlogPage":false,"appliedCssClassNames":"snowflake-flexible-column-container-gray-10-bg"},"flexible_column_cont_939100716":{"id":"flexible-column-container-307d845c35","propertiesId":"hub-hero","type":"2-column-even","alignColumns":"center","containerMaxWidth":"extra-large","topPadding":"extra-small","bottomPadding":"extra-small","spaceBetween":"small","reverseOnMobile":false,"carouselOnMobile":false,"propertiesCSSClasses":"page-section","backgroundImageOption":"none","flexible_column_content_container_1":{"layout":"SIMPLE","id":"container-eccb94e828",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container","appliedCssClassNames":"snowflake-responsive-container-inner-padding-extra-small",":items":{"title_v2":{"id":"title-v2-f8e3d5f8a3","additionalClasses":"hub-hero__headline","type":"heading1","lines":["AI Cost Optimization: How to Reduce LLM and Inference Costs in Production"],":type":"snowflake-site/components/title-v2","appliedCssClassNames":"left-alignment"},"text":{"id":"text-0fb078812a","additionalClasses":"hub-hero__subheadline","text":"\u003Cp\u003ELLM tokens are getting cheaper, but production AI bills can still climb fast when applications use long context, repeated model calls and agentic workflows. Here’s how AI engineering teams can reduce inference costs without sacrificing quality.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-size-regular text-color-text-05"},"container":{"additionalClasses":"hub-hero__authors","layout":"RESPONSIVE_GRID","columnCount":12,"columnClassNames":{"content_chip_copy":"aem-GridColumn aem-GridColumn--default--12","content_chip":"aem-GridColumn aem-GridColumn--default--12"},"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","id":"container-a7fcefb576",":type":"snowflake-site/components/container","appliedCssClassNames":"snowflake-responsive-container-inner-padding-small",":items":{"content_chip":{"id":"content-chip-c70d99d2b9","cta":{"id":"cta","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/blog/authors/Laurie-Macpherson/"},"linkTargetContentType":"GENERIC",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_INTERNAL","text":"Read bio"},"image":{"id":"image","height":"800","src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--a7e9fdff-6f08-4edc-9cd1-e213bb234aaa/laurie-macpherson.jpg?quality=85&preferwebp=true","alt":"Laurie MacPherson","lazyEnabled":true,"isLcpImage":true,"width":"800",":type":"snowflake-site/components/image"},"headline":{"id":"title","type":"heading5","lines":["Laurie MacPherson","Technical Writer, Snowflake"],":type":"snowflake-site/components/title-v2"},":type":"snowflake-site/components/content-chip"},"content_chip_copy":{"id":"content-chip-f7ce5bbc01","cta":{"id":"cta","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/blog/authors/david-gaule/"},"linkTargetContentType":"GENERIC",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_INTERNAL","text":"Read bio"},"image":{"id":"image","height":"512","src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--fd9454ea-3b59-4d19-95be-534774cbd226/david.jpg?quality=85&preferwebp=true","alt":"David Gaule","lazyEnabled":true,"isLcpImage":false,"width":"512",":type":"snowflake-site/components/image"},"headline":{"id":"title","type":"heading5","lines":["David Gaule","Technical Editor, Snowflake"],":type":"snowflake-site/components/title-v2"},":type":"snowflake-site/components/content-chip"}},":itemsOrder":["content_chip","content_chip_copy"]}},":itemsOrder":["title_v2","text","container"]},"flexible_column_content_container_2":{"additionalClasses":"hub-hero__video-column","layout":"SIMPLE","id":"container-4c5e1471ab",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"youtube":{"id":"embed-b427e59335","youtubeVideoId":"gA-R_MgZQAk","layout":"responsive","youtubeAspectRatio":"56.25","youtubeAutoPlay":false,"youtubeLoop":false,"youtubeMute":false,"youtubePlaysInline":false,"youtubeRel":false,"embeddableResourceType":"core/wcm/components/embed/v1/embed/embeddable/youtube","type":"EMBEDDABLE",":type":"snowflake-site/components/youtube"}},":itemsOrder":["youtube"]},":type":"snowflake-site/components/flexible-column-container","isActiveTOC":false,"isBlogPage":false,"appliedCssClassNames":"snowflake-flexible-column-container-gray-10-bg"},"flexible_column_cont_1398138236":{"id":"flexible-column-container-f58f94898b","propertiesId":"hub-hero-related-topics","type":"1-column","alignColumns":"top","containerMaxWidth":"extra-large","topPadding":"extra-small","bottomPadding":"medium","spaceBetween":"small","reverseOnMobile":false,"carouselOnMobile":false,"propertiesCSSClasses":"page-section","backgroundImageOption":"none","flexible_column_content_container_1":{"additionalClasses":"related-topics-outer-container border-top","layout":"SIMPLE","id":"container-54b33a8736",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"text_894059747":{"id":"text-d0d38226cc","additionalClasses":"seo-hub-hero__related-topic-label","text":"\u003Cp\u003EAI Engineering Topics:\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-size-regular text-color-text-05"},"text":{"id":"text-a0f808b4c4","additionalClasses":"related-topics ","text":"\u003Cul\u003E\r\n\u003Cli\u003E\u003Ca href=\"https://www.snowflake.com/en/artificial-intelligence/ai-engineering/context-engineering/\"\u003EContext Engineering\u003C/a\u003E\u003C/li\u003E\r\n\u003Cli\u003E\u003Ca href=\"https://www.snowflake.com/en/artificial-intelligence/ai-engineering/harness-engineering/\"\u003EHarness Engineering\u003C/a\u003E\u003C/li\u003E\r\n\u003C/ul\u003E\r\n","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-size-small"}},":itemsOrder":["text_894059747","text"]},":type":"snowflake-site/components/flexible-column-container","isActiveTOC":false,"isBlogPage":false,"appliedCssClassNames":"snowflake-flexible-column-container-gray-10-bg"},"flexible_column_cont_663228916":{"id":"flexible-column-container-0b5c7f9674","propertiesId":"hub-body","type":"2-column-60-40","alignColumns":"top","containerMaxWidth":"extra-large","topPadding":"medium","bottomPadding":"medium","spaceBetween":"small","reverseOnMobile":false,"carouselOnMobile":false,"propertiesCSSClasses":"page-section","backgroundImageOption":"none","flexible_column_content_container_1":{"additionalClasses":"longform-content","layout":"SIMPLE","id":"hub-body-content",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container","appliedCssClassNames":"snowflake-responsive-container-inner-padding-medium",":items":{"callout__0":{"id":"text-2726977536","additionalClasses":"callout callout--general","text":"\u003Cp\u003E\u003Cstrong\u003EAI COST OPTIMIZATION DEFINED\u003C/strong\u003E\u003C/p\u003E\n\u003Cp\u003EAI cost optimization is the practice of measuring and reducing the cost of building and running AI applications while maintaining the required quality, latency and reliability.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-size-regular text-color-text-05"},"text__0":{"id":"text-2a7ad61831","text":"\u003Cp\u003EThe cost of running a capable language model has been falling at extraordinary speed. Between November 2022 and October 2024, the price of querying a model with GPT-3.5-level performance dropped from $20 to $0.07 per million tokens, according to \u003Ca href=\"https://hai.stanford.edu/news/ai-index-2025-state-of-ai-in-10-charts\" target=\"_blank\"\u003EStanford’s 2025 AI Index\u003C/a\u003E — a decline of more than 280-fold.\u003C/p\u003E\n\u003Cp\u003EProduction AI costs have not followed the same curve, however. As applications move beyond single-turn interactions, one user request may set off a long sequence of model calls, retrieved context, tool use and intermediate reasoning. In \u003Ca href=\"https://arxiv.org/abs/2604.22750\" target=\"_blank\"\u003Ea recent study\u003C/a\u003E of coding agents, agentic tasks consumed 1,000 times more tokens than code chat and code reasoning in the workloads evaluated; even repeated runs of the same task varied by as much as 30x.\u003C/p\u003E\n\u003Cp\u003ECheaper tokens, in other words, can still produce a significantly larger bill when applications use many more of them. Bringing that bill under control starts with understanding where the work is happening, then reducing unnecessary context, avoiding redundant model calls, matching tasks to appropriately sized models and improving inference efficiency and infrastructure utilization where relevant. Each change needs to be measured against the latency and output quality the application requires, which makes AI cost optimization an ongoing AI engineering practice rather than a single cost reduction exercise.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"},"title_what-drives-ai-and-llm-costs":{"id":"title-v2-2d02be53db","additionalClasses":"anchor-title anchor-title--what-drives-ai-and-llm-costs","type":"heading2","lines":["What drives AI and LLM costs?"],":type":"snowflake-site/components/title-v2"},"text_what-drives-ai-and-llm-costs_0":{"id":"text-907b0e46a1","text":"\u003Cp\u003EIn many enterprise LLM workloads, the model processes far more text than it generates. A request may include a system prompt, retrieved documents, conversation history, tool results and user instructions, then return only a short answer. \u003Ca href=\"https://www.snowflake.com/en/blog/engineering/swiftkv-llm-compute-reduction\"\u003ESnowflake AI Research\u003C/a\u003E observed an approximate 10:1 ratio between prompt and generated tokens across many enterprise workloads, which places much of the inference burden on processing the input.\u003C/p\u003E\n\u003Cp\u003EThat workload shape directs attention to the prefill phase. During prefill, the model processes the full prompt and creates the key-value cache (KV cache) used during generation. Decode follows, producing the response one token at a time. For applications with long prompts and relatively concise outputs, prefill can consume a substantial share of the compute required for each request.\u003C/p\u003E\n\u003Cp\u003EToken pricing translates that processing into the API bill. Providers generally charge separately for input and output tokens, often at different rates, so both contribute to cost. Yet counting only the visible response can give teams the wrong impression: Retrieved context, repeated instructions and accumulated conversation history may account for most of the tokens sent through the model.\u003C/p\u003E\n\u003Cp\u003ELeo Rodriguez, Principal Product Marketing Manager, AI/ML at Snowflake, says this is one of the easiest places for teams to underestimate cost: “If a task takes 20 prompts to complete, you’re not just paying for the 20th prompt. In many cases, the model is processing the prior prompts, long context windows and all the information it needs to decide where to look. That’s where teams can get surprised by the compounding cost of a single query.”\u003C/p\u003E\n\u003Ch3\u003EInference creates the recurring bill\u003C/h3\u003E\n\u003Cp\u003EFor teams building on foundation models, inference usually makes up the major continuing production expense. Every request initiates another round of prompt processing and generation, and multistep workflows can invoke a model several times before returning one result.\u003C/p\u003E\n\u003Cp\u003EAgentic applications make call-level metrics especially important. A user may submit one request while the agent plans a task, retrieves information, calls tools, evaluates intermediate results and generates a final response. Measuring only the final call leaves most of the workflow’s consumption unaccounted for.\u003C/p\u003E\n\u003Cp\u003ETwo operational metrics provide a useful starting point:\u003C/p\u003E\n\u003Cul\u003E\n\u003Cli\u003E\u003Cstrong\u003ECost per 1,000 API calls:\u003C/strong\u003E Useful for services with a relatively consistent request pattern.\u003C/li\u003E\n\u003Cli\u003E\u003Cstrong\u003ECost per workflow or agent run:\u003C/strong\u003E More informative when one user action triggers multiple models, tools or retrieval steps.\u003C/li\u003E\n\u003C/ul\u003E\n\u003Cp\u003ECost per successful workflow can add another layer of context. An inexpensive run that fails and must be repeated may cost more overall than a slightly more expensive run that completes reliably.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"},"callout_what-drives-ai-and-llm-costs_0":{"id":"text-79adb7fa0b","additionalClasses":"callout callout--tip","text":"\u003Cp\u003E\u003Cstrong\u003EQUICK TIP\u003C/strong\u003E\u003C/p\u003E\n\u003Cp\u003ETrack cost at the workflow level, not just the model-call level. A single user request may trigger retrieval, tool calls, retries and multiple model invocations, so cost per completed workflow is often the clearest signal for optimization.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-size-regular text-color-text-05"},"text_what-drives-ai-and-llm-costs_1":{"id":"text-14d998d3b8","text":"\u003Ch3\u003ESelf-hosting shifts the cost model\u003C/h3\u003E\n\u003Cp\u003EWith self-hosted models, token counts still indicate workload volume, although GPU economics usually dominate the infrastructure bill. The selected GPU type, model size, batch size, memory footprint and utilization rate determine how much inference capacity the organization receives from each instance.\u003C/p\u003E\n\u003Cp\u003EIdle time is particularly expensive. A GPU provisioned for peak demand may sit underused for much of the day, while an undersized serving pool can create queues, latency spikes and failed requests. Long context windows also consume KV-cache memory, limiting the number of requests a GPU can process concurrently.\u003C/p\u003E\n\u003Cp\u003EBefore changing the model or serving stack, teams need a baseline that connects infrastructure consumption to application activity: GPU-hours, utilization, throughput, latency and cost per completed request. Without that connection, lower infrastructure spending can conceal declining service quality.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"},"card_v2_what-drives-ai-and-llm-costs_0":{"id":"card-v2-f446d1b885","additionalClasses":"seo-customer","configurationStatus":{"configured":true,"message":""},":type":"snowflake-site/components/card-v2","text":{"id":"text","text":"\u003Cp\u003EYieldmo uses Snowflake and Snowpark Container Services to accelerate AI-powered advertising predictions while reducing the cost and complexity of its ML inference layer. By running H2O eScorer as a service in Snowflake, Yieldmo avoids large data transfers, brings predictive workloads closer to its data and enables data science teams to work with SQL instead of managing extra infrastructure. Initial testing shows predictions running 20x faster at 75% lower cost, helping Yieldmo improve A/B testing, speed time to market and optimize ad performance in near real time.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/text"},"layoutStyle":"horizontal","title":{"id":"title","type":"heading4","lines":["Yieldmo"],":type":"snowflake-site/components/title-v2"},"button":{"id":"button","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/customers/all-customers/case-study/yieldmo/"},"linkTargetContentType":"GENERIC",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_INTERNAL","text":"Read the full case study"},"image":{"id":"image","height":"351","src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--bdb54e97-c4bb-47b2-bea3-5a3340b7d57b/yieldmo.png?quality=85&preferwebp=true","alt":"yieldmo logo","lazyEnabled":true,"isLcpImage":false,"width":"624",":type":"snowflake-site/components/image"},"type":"content-card"},"title_techniques-to-reduce-llm-cost":{"id":"title-v2-a9423693d2","additionalClasses":"anchor-title anchor-title--techniques-to-reduce-llm-cost","type":"heading2","lines":["Techniques to reduce LLM cost"],":type":"snowflake-site/components/title-v2"},"text_techniques-to-reduce-llm-cost_0":{"id":"text-4d5f69eed5","text":"\u003Cp\u003ENo single technique addresses every source of AI cost. The right combination depends on request volume, context length, task complexity, latency requirements and how often prompts repeat.\u003C/p\u003E\r\n\u003Ch3\u003EReduce prompt and context length\u003C/h3\u003E\r\n\u003Cp\u003EFor workloads dominated by input tokens, prompt compression often offers the most direct savings. Teams can shorten system instructions, remove duplicated guidance, summarize older conversation turns and retrieve only the passages needed for the current request.\u003C/p\u003E\r\n\u003Cp\u003EResearch systems have demonstrated substantial compression, although results vary by task and by the amount of information that can safely be removed. Microsoft Research’s \u003Ca rel=\"noopener noreferrer\" target=\"_blank\" href=\"https://arxiv.org/abs/2310.05736\"\u003ELLMLingua\u003C/a\u003E, for example, achieved up to 20x prompt compression with little performance loss in its evaluations. In \u003Ca rel=\"noopener noreferrer\" target=\"_blank\" href=\"https://arxiv.org/abs/2605.11774\"\u003Ea separate 2026 study\u003C/a\u003E focused on clinical prediction, Medical Token-Pair Encoding reduced input length by as much as 31% and inference latency by 34% to 63% while maintaining or improving performance across the tasks evaluated.\u003C/p\u003E\r\n\u003Cp\u003EProduction implementations need a more cautious target. Aggressive compression may remove a policy instruction, relevant record or example that the model needs. Start with deterministic reductions: eliminate duplicated context, cap retrieved passages and replace full documents with structured extracts where possible. After those changes, evaluate summarization or learned compression against representative requests.\u003C/p\u003E\r\n\u003Cp\u003EContext engineering provides the broader discipline around deciding which instructions, data and prior state the model receives. Cost is one consideration within that design, while grounding, security and output quality remain part of the same decision.\u003C/p\u003E\r\n\u003Ch3\u003ERoute requests according to task difficulty\u003C/h3\u003E\r\n\u003Cp\u003EA frontier model may be appropriate for complex reasoning, ambiguous requests or high-consequence decisions, but classification, extraction, formatting and straightforward summarization can often run on a smaller model at a lower token rate.\u003C/p\u003E\r\n\u003Cp\u003ERodriguez frames model selection as a practical cost-control discipline: “Cost optimization starts with the harness knowing which models or tools are right for the task. For complex work, you may need a more capable model. For simpler or less time-sensitive work, a less expensive model may deliver the same outcome at a much lower cost. In either case, a well-defined model harness can help route the task to the most appropriate model.”\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"},"quote_techniques-to-reduce-llm-cost_0":{"id":"quote-item-a7d6e6aa9b","alignment":"left","layout":"simple","quoteSourceName":"Leo Rodriguez","quoteSourceTitle":"Principal Product Marketing Manager, AI/ML, Snowflake","showQuoteIcon":true,"quote":"Cost optimization starts with the harness knowing which models or tools are right for the task.","showHangingPunctuation":false,":type":"snowflake-site/components/quote-item"},"text_techniques-to-reduce-llm-cost_1":{"id":"text-8363dccda6","text":"\u003Cp\u003EModel routing, sometimes called model cascading, assigns each request to the least expensive model expected to meet its quality threshold. The router may use task type, prompt characteristics, confidence scores or the output of an initial model. When the smaller model cannot answer reliably, the workflow escalates the request.\u003C/p\u003E\n\u003Cp\u003EPublished results illustrate the potential range. The \u003Ca href=\"https://www.microsoft.com/en-us/research/publication/hybrid-llm-cost-efficient-and-quality-aware-query-routing/\" target=\"_blank\"\u003EHybrid LLM approach\u003C/a\u003E made up to 40% fewer calls to the larger model without reducing response quality in its experiments. \u003Ca href=\"https://arxiv.org/abs/2506.22716\" target=\"_blank\"\u003EBEST-Route\u003C/a\u003E reported cost reductions of up to 60% with less than a 1% performance decline on its evaluated data sets. These figures reflect controlled experiments, so teams should validate their own routing policy against production traffic and evaluation criteria.\u003C/p\u003E\n\u003Cp\u003EA router also introduces an additional failure mode: A request may be sent to a model that lacks the required capability. Quality thresholds, fallback rules and periodic evaluation help keep those decisions aligned with application requirements.\u003C/p\u003E\n\u003Ch3\u003EReuse responses through semantic caching\u003C/h3\u003E\n\u003Cp\u003EExact-match caches work well when identical prompts recur. Semantic caching broadens the match by identifying prompts with sufficiently similar meaning, then serving a stored response instead of issuing another model call.\u003C/p\u003E\n\u003Cp\u003EThe savings depend on repetition within the workload. A customer-support assistant answering variations of the same policy questions may achieve a high cache-hit rate, while an analytical application receiving unique queries may find fewer reuse opportunities. \u003Ca href=\"https://arxiv.org/html/2411.05276v2\" target=\"_blank\"\u003EIn one study\u003C/a\u003E, a semantic cache reduced model API calls by as much as 68.8%, with positive-hit rates above 97% in the evaluated query categories. \u003Ca href=\"https://arxiv.org/html/2403.02694v3\" target=\"_blank\"\u003EAnother system\u003C/a\u003E reported reductions of up to one-third in inference cost for semantically similar requests.\u003C/p\u003E\n\u003Cp\u003ECaching requires clear boundaries. Responses based on rapidly changing data need short expiry periods or cache invalidation rules. User-specific, permission-sensitive or nondeterministic outputs may be unsuitable for reuse. Similarity thresholds should also be tested carefully — a permissive threshold raises the hit rate, but it can return an answer generated for a meaningfully different question.\u003C/p\u003E\n\u003Ch3\u003ERight-size or specialize the model\u003C/h3\u003E\n\u003Cp\u003EA smaller model can lower cost in two ways: It may carry a lower API price and, when self-hosted, it generally requires less memory and compute. Distillation transfers behavior from a larger teacher model into a smaller student model, while fine-tuning can improve a compact model on a defined task.\u003C/p\u003E\n\u003Cp\u003EThis approach works best when the workload has stable boundaries. A specialized extraction model, for example, can learn a known schema and document type. A general-purpose assistant handling open-ended questions has a broader capability requirement and may need a larger model or an escalation path.\u003C/p\u003E\n\u003Cp\u003EBefore replacing a model, compare both average performance and important edge cases. Aggregate benchmark scores can hide failures on uncommon inputs, languages, long documents or requests requiring multiple reasoning steps.\u003C/p\u003E\n\u003Ch3\u003EConstrain output with structured schemas\u003C/h3\u003E\n\u003Cp\u003EUnbounded generation can produce explanatory text the application discards. When the downstream system needs a category, field set or tool argument, structured outputs can restrict the response to a defined schema.\u003C/p\u003E\n\u003Cp\u003EA schema can reduce wasted output tokens, simplify parsing and lower retry rates caused by invalid formatting. The savings per call may be modest, particularly for already concise responses, but high-volume classification and extraction workloads can accumulate meaningful reductions.\u003C/p\u003E\n\u003Cp\u003ESchema constraints don’t guarantee factual accuracy. The model may return a valid structure containing an incorrect value, so validation and evaluation remain necessary.\u003C/p\u003E\n\u003Ch3\u003EBatch compatible requests\u003C/h3\u003E\n\u003Cp\u003EGPUs process parallel work more efficiently than a stream of isolated requests. Batching groups requests so the serving system can run them together, increasing throughput and distributing infrastructure cost across more inferences.\u003C/p\u003E\n\u003Cp\u003EContinuous batching improves this approach by adding and removing requests as generation proceeds, instead of waiting for every request in a fixed batch to finish. The benefit depends on concurrency and latency targets. Offline document processing can tolerate larger batches, while an interactive assistant may need smaller batches to protect time to first token.\u003C/p\u003E\n\u003Cp\u003EThe \u003Ca href=\"https://arxiv.org/abs/2309.06180\" target=\"_blank\"\u003Eoriginal vLLM research\u003C/a\u003E reported 2x to 4x higher throughput at comparable latency than the systems used as baselines, largely through PagedAttention and more efficient KV-cache memory management. Those gains represent serving-engine benchmarks, not a guaranteed reduction in a production bill, but they show how infrastructure efficiency can increase the useful output obtained from the same GPU capacity.\u003C/p\u003E\n\u003Ch3\u003EUse quantization and KV-cache optimization\u003C/h3\u003E\n\u003Cp\u003EQuantization stores model weights, and sometimes activations, at lower numerical precision. Moving model weights from 16-bit to 4-bit precision reduces their raw memory footprint \u003Ca href=\"https://arxiv.org/abs/2306.00978\" target=\"_blank\"\u003Eby approximately 4x\u003C/a\u003E. Techniques such as activation-aware weight quantization (AWQ) are designed to preserve model quality under that lower-precision representation by protecting the weight channels that contribute most to the output. With less memory devoted to model weights, teams may be able to use lower-capacity hardware or leave more GPU memory available for KV caches and concurrent requests.\u003C/p\u003E\n\u003Cp\u003EThe quality trade-off depends on the model, quantization method and task. Reasoning-heavy workloads can be sensitive to aggressive quantization: One evaluation found sizable degradation on mathematical reasoning for some low-bit configurations. Test the quantized model against the actual application data, especially where small numerical or logical errors carry consequences.\u003C/p\u003E\n\u003Cp\u003EKV-cache optimization addresses a different memory consumer. Each active request stores attention state for the tokens already processed, and long prompts can use considerable cache capacity. More efficient allocation, prefix reuse and cache compression can increase batch size or reduce the number of GPUs needed to support a given traffic level.\u003C/p\u003E\n\u003Ch3\u003EStack compatible levers\u003C/h3\u003E\n\u003Cp\u003EIndividual savings percentages cannot simply be added together. Compression changes the request sent to the router, routing changes which calls reach each model and caching removes some calls before inference begins. Infrastructure improvements then apply only to the requests that remain.\u003C/p\u003E\n\u003Cp\u003EA practical sequence might compress prompt context, serve recurring queries from a semantic cache, route routine work to a smaller model and run the remaining self-hosted traffic through an efficient batching system. Measure the combined result at the workflow level, where interactions among the techniques are visible.\u003C/p\u003E\n\u003Cp\u003E\u003Cem\u003EWatch Snowflake leaders discuss key considerations for enterprises moving AI into production:\u003C/em\u003E\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"},"yt_techniques-to-reduce-llm-cost_0":{"id":"embed-645615b51a","youtubeVideoId":"cJGebAnHbB4","layout":"responsive","youtubeAspectRatio":"56.25","youtubeAutoPlay":false,"youtubeLoop":false,"youtubeMute":false,"youtubePlaysInline":false,"youtubeRel":false,"embeddableResourceType":"core/wcm/components/embed/v1/embed/embeddable/youtube","type":"EMBEDDABLE",":type":"snowflake-site/components/youtube"},"title_measuring-and-governing-ai-cost-with-finops":{"id":"title-v2-c02a5cc9f7","additionalClasses":"anchor-title anchor-title--measuring-and-governing-ai-cost-with-finops","type":"heading2","lines":["Measuring and governing AI cost with FinOps"],":type":"snowflake-site/components/title-v2"},"text_measuring-and-governing-ai-cost-with-finops_0":{"id":"text-0142616a01","text":"\u003Cp\u003EAI cost optimization starts with attribution. Provider invoices show total consumption, while token counts and application telemetry explain which teams, features and workflows generated it.\u003C/p\u003E\n\u003Cp\u003EAt minimum, production monitoring should capture:\u003C/p\u003E\n\u003Cul\u003E\n\u003Cli\u003EInput and output tokens by model\u003C/li\u003E\n\u003Cli\u003EModel calls per user request\u003C/li\u003E\n\u003Cli\u003ECache-hit and escalation rates\u003C/li\u003E\n\u003Cli\u003ECost per 1,000 calls\u003C/li\u003E\n\u003Cli\u003ECost per workflow or agent run\u003C/li\u003E\n\u003Cli\u003ELatency and completion rate\u003C/li\u003E\n\u003Cli\u003EEvaluation results for each model or routing path\u003C/li\u003E\n\u003C/ul\u003E\n\u003Cp\u003ETagging usage by team, environment, feature and workflow supports showback or chargeback. A platform team can then identify whether a billing increase came from user growth, a new feature, longer prompts, repeated agent loops or a shift toward a more expensive model.\u003C/p\u003E\n\u003Cp\u003EBudgets and alerts should use operational units as well as account-level spend. A monthly threshold may reveal a problem late in the billing cycle, whereas an alert on cost per workflow can surface a regression soon after deployment. Sudden increases in input tokens, tool calls or retry rates often indicate a code or configuration change that deserves investigation.\u003C/p\u003E\n\u003Ch3\u003EEvaluate every cost change\u003C/h3\u003E\n\u003Cp\u003EA lower bill doesn’t mean a successful optimization. Prompt compression may omit relevant context, a smaller model may fail on difficult requests and an overly broad cache may serve stale or mismatched answers.\u003C/p\u003E\n\u003Cp\u003EEvaluation gates connect cost changes to application behavior. Before rollout, compare the proposed configuration with the current one on a representative evaluation set. Measure task-specific quality, policy compliance, latency and cost, then define the degradation the application can accept. In production, monitor the same signals for distribution shifts and uncommon failures.\u003C/p\u003E\n\u003Cp\u003EThis discipline helps teams avoid several recurring mistakes: assigning every request to a frontier model, overlooking repeated prompts, leaving token consumption unattributed, maintaining excess GPU capacity and assuming the cost of a workflow is fixed.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"},"title_optimizing-ai-cost-on-snowflake":{"id":"title-v2-7f5a64acfe","additionalClasses":"anchor-title anchor-title--optimizing-ai-cost-on-snowflake","type":"heading2","lines":["Optimizing AI cost on Snowflake"],":type":"snowflake-site/components/title-v2"},"text_optimizing-ai-cost-on-snowflake_0":{"id":"text-a684727bf9","text":"\u003Cp\u003E\u003Ca href=\"https://www.snowflake.com/en/product/features/cortex/\"\u003ESnowflake Cortex AI\u003C/a\u003E provides managed inference, so teams can use supported models without provisioning and maintaining their own GPU serving infrastructure. Consumption is metered according to the Cortex AI service and model used, while Snowflake usage views expose token and credit consumption for monitoring. Cortex Agents, for example, are billed according to the tokens processed by the orchestration model, with additional consumption from the underlying services called during the workflow.\u003C/p\u003E\r\n\u003Cp\u003EFor batch-oriented work over tables, Cortex AI Functions run directly through SQL and are optimized for throughput. Teams can apply AI operations across many rows without building a separate request pipeline, while interactive applications can use the relevant REST APIs when latency carries greater weight.\u003C/p\u003E\r\n\u003Cp\u003EModel selection provides another cost-control point. Simpler classification, extraction or summarization tasks can use an appropriately sized model, while more capable models remain available for requests that need them. Because Cortex AI model rates vary, the selected model directly affects the token cost of the workload.\u003C/p\u003E\r\n\u003Cp\u003EFor Snowflake workloads, Rodriguez says cost and quality improve when your AI data harness is purpose-built for the data environment where the work happens. “If you’re working with Snowflake data, using Snowflake CoCo gives you a better outcome in terms of accuracy and cost efficiency because the harness has been purpose-built for Snowflake. You’re not pointing a generic LLM at enterprise data and hoping it finds the right answer.”\u003C/p\u003E\r\n\u003Cp\u003ESnowflake has also developed inference optimizations for prompt-heavy enterprise traffic. SwiftKV reuses hidden states from earlier transformer layers while generating the KV cache, reducing repeated prefill computation. Snowflake has reported \u003Ca href=\"https://www.snowflake.com/en/blog/engineering/swiftkv-llm-compute-reduction/\"\u003Eup to 50% lower prefill compute and up to 2x higher throughput for enterprise workloads with long prompts\u003C/a\u003E. Earlier published testing showed approximately 40% to 50% throughput improvement in the configurations (compared to compression-only methods).\u003C/p\u003E\r\n\u003Cp\u003EKeeping AI processing close to governed data can reduce additional architecture and data-transfer overhead. Applications can apply Snowflake access controls to the data used for inference, run Cortex AI Functions through SQL and inspect usage through account-level views without maintaining a separate copy solely for model processing. For workloads that otherwise export large context sets to another environment, reducing that movement can also lower transfer, storage and pipeline-management costs.\u003C/p\u003E\r\n\u003Cp\u003EAcross these controls, the underlying method remains consistent: Measure consumption at the workflow level, assign each task to suitable compute and validate every optimization against the quality the application is expected to deliver.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"},"title_scale-ai-with-cost-discipline":{"id":"title-v2-10164a2161","additionalClasses":"anchor-title anchor-title--scale-ai-with-cost-discipline","type":"heading2","lines":["Scale AI with cost discipline"],":type":"snowflake-site/components/title-v2"},"text_scale-ai-with-cost-discipline_0":{"id":"text-0f78ba0277","text":"\u003Cp\u003EAs applications add longer context, more tool calls and more autonomous workflows, the teams that control cost will be the ones that understand the full path of each request and tune it deliberately. AI cost optimization isn’t about doing less with AI, but rather matching every task to the right context, model and infrastructure so production systems can scale with both quality and financial confidence.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"},"callout_scale-ai-with-cost-discipline_0":{"id":"text-126af88fab","additionalClasses":"callout callout--general","text":"\u003Cp\u003E\u003Cb\u003EKEY TAKEAWAY\u003C/b\u003E \u003C/p\u003E\r\n\u003Cp\u003EAI cost optimization requires measuring the full cost of each workflow, then reducing unnecessary context, redundant calls and oversized model use while improving serving efficiency. The most effective approach combines multiple techniques — such as prompt compression, caching, model routing and infrastructure optimization — and validates every change against quality, latency and reliability requirements.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-size-regular text-color-text-05"}},":itemsOrder":["callout__0","text__0","title_what-drives-ai-and-llm-costs","text_what-drives-ai-and-llm-costs_0","callout_what-drives-ai-and-llm-costs_0","text_what-drives-ai-and-llm-costs_1","card_v2_what-drives-ai-and-llm-costs_0","title_techniques-to-reduce-llm-cost","text_techniques-to-reduce-llm-cost_0","quote_techniques-to-reduce-llm-cost_0","text_techniques-to-reduce-llm-cost_1","yt_techniques-to-reduce-llm-cost_0","title_measuring-and-governing-ai-cost-with-finops","text_measuring-and-governing-ai-cost-with-finops_0","title_optimizing-ai-cost-on-snowflake","text_optimizing-ai-cost-on-snowflake_0","title_scale-ai-with-cost-discipline","text_scale-ai-with-cost-discipline_0","callout_scale-ai-with-cost-discipline_0"]},"flexible_column_content_container_2":{"additionalClasses":"hub-sidebar","layout":"SIMPLE","id":"hub-body-aside",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container","appliedCssClassNames":"snowflake-responsive-container-inner-padding-small",":items":{"container":{"additionalClasses":"sticky-sidebar","layout":"RESPONSIVE_GRID","columnCount":12,"columnClassNames":{"text_943981956_copy_":"aem-GridColumn aem-GridColumn--default--12","text_copy":"aem-GridColumn aem-GridColumn--default--12"},"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","id":"container-09abef702b",":type":"snowflake-site/components/container","appliedCssClassNames":"snowflake-responsive-container-inner-padding-medium",":items":{"text_943981956_copy_":{"id":"text-1d634e874b","additionalClasses":"eyebrow-text","text":"\u003Cp\u003EIn This Guide\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-size-regular"},"text_copy":{"id":"text-6694125847","additionalClasses":"page-toc","text":"\u003Cul\u003E\u003Cli data-anchor=\"what-drives-ai-and-llm-costs\"\u003EWhat drives AI and LLM costs?\u003C/li\u003E\u003Cli data-anchor=\"techniques-to-reduce-llm-cost\"\u003ETechniques to reduce LLM cost\u003C/li\u003E\u003Cli data-anchor=\"measuring-and-governing-ai-cost-with-finops\"\u003EMeasuring and governing AI cost with FinOps\u003C/li\u003E\u003Cli data-anchor=\"optimizing-ai-cost-on-snowflake\"\u003EOptimizing AI cost on Snowflake\u003C/li\u003E\u003Cli data-anchor=\"scale-ai-with-cost-discipline\"\u003EScale AI with cost discipline\u003C/li\u003E\u003C/ul\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-size-small text-color-text-05"}},":itemsOrder":["text_943981956_copy_","text_copy"]}},":itemsOrder":["container"]},":type":"snowflake-site/components/flexible-column-container","isActiveTOC":false,"isBlogPage":false},"flexible_column_cont_1786318617":{"id":"flexible-column-container-ef5e865db2","propertiesId":"hub-faq","type":"2-column-40-60","alignColumns":"top","containerMaxWidth":"extra-large","topPadding":"large","bottomPadding":"large","spaceBetween":"small","reverseOnMobile":false,"carouselOnMobile":false,"backgroundImageOption":"none","flexible_column_content_container_1":{"layout":"SIMPLE","id":"hub-faq-intro",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container","appliedCssClassNames":"snowflake-responsive-container-inner-padding-extra-small",":items":{"title_v2_copy":{"id":"title-v2-a5bc9e9ab2","additionalClasses":"hub-faq__headline","type":"heading2","lines":["Frequently Asked Questions"],":type":"snowflake-site/components/title-v2","appliedCssClassNames":"left-alignment"},"text_copy":{"id":"text-a9991ee490","additionalClasses":"hub-faq__subheadline","text":"\u003Cp\u003EYour common questions about AI cost optimization, answered by Snowflake experts.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"}},":itemsOrder":["title_v2_copy","text_copy"]},"flexible_column_content_container_2":{"layout":"SIMPLE","id":"hub-faq-accordions",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"simple_snowflake_acc":{"id":"simple-snowflake-accordion-9a4759529d","additionalClasses":"seo-hub__faqs","showDivider":false,"accordionItemsList":[{"title":"How do I reduce LLM inference costs?","richText":"\u003Cp\u003ETo reduce LLM inference costs, start by measuring input tokens, output tokens, model calls and cost per completed workflow. Remove duplicated or irrelevant prompt context, then evaluate semantic caching and model routing for workloads with repeated or mixed-complexity requests. For self-hosted models, improve GPU utilization through batching, quantization and efficient KV-cache management. Test each change against latency and quality requirements before broad deployment.\u003C/p\u003E"},{"title":"What is the biggest driver of LLM costs?","richText":"\u003Cp\u003EThe biggest driver of LLM costs depends on the deployment model. For API-based enterprise applications, input-token processing can drive a large share of cost because prompts often contain much more text than the model generates. For self-hosted models, GPU selection, utilization and idle capacity frequently dominate the bill. Workflow design also contributes: Agents and multistep applications may invoke several models for one user request, making cost per workflow more useful than cost per individual call.\u003C/p\u003E"}],":type":"snowflake-site/components/simple-snowflake-accordion"}},":itemsOrder":["simple_snowflake_acc"]},":type":"snowflake-site/components/flexible-column-container","isActiveTOC":false,"isBlogPage":false,"appliedCssClassNames":"snowflake-flexible-column-container-gray-10-bg"},"flexible_column_cont_1467213961":{"id":"flexible-column-container-3217dbc907","propertiesId":"hub-explore-resources-header","type":"1-column","alignColumns":"top","containerMaxWidth":"extra-large","topPadding":"large","bottomPadding":"none","spaceBetween":"small","reverseOnMobile":false,"carouselOnMobile":false,"backgroundImageOption":"none","flexible_column_content_container_1":{"layout":"SIMPLE","id":"container-fe8997d754",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"title_v2":{"id":"title-v2-2d75113ff7","additionalClasses":"hub-explore-resources-header__headline","type":"heading2","lines":["Explore AI Resources"],":type":"snowflake-site/components/title-v2","appliedCssClassNames":"left-alignment"}},":itemsOrder":["title_v2"]},":type":"snowflake-site/components/flexible-column-container","isActiveTOC":false,"isBlogPage":false,"appliedCssClassNames":"snowflake-flexible-column-container-gray-10-bg"},"flexible_column_cont_912630531":{"id":"flexible-column-container-2392180c22","propertiesId":"hub-explore-resources-grid","type":"1-column","alignColumns":"top","containerMaxWidth":"extra-large","topPadding":"small","bottomPadding":"large","spaceBetween":"small","reverseOnMobile":false,"carouselOnMobile":false,"backgroundImageOption":"none","flexible_column_content_container_1":{"layout":"SIMPLE","id":"hub-explore-resources-grid-inner",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"resource_chip_0":{"id":"content-chip-381fb704f3","tagText":"WEBINAR","tagColor":"#EEBDD3","cta":{"id":"cta","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"https://www.snowflake.com/en/webinars/virtual-hands-on-lab/build-an-llmpowered-app-in-10-min-with-snowflake-cortex-ai-2026-05-27/"},"linkTargetContentType":"GENERIC",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_EXTERNAL","text":"Read more"},"headline":{"id":"title","type":"heading5","lines":["Build an LLM-Powered App in 10 min with Snowflake Cortex AI"],":type":"snowflake-site/components/title-v2"},":type":"snowflake-site/components/content-chip","appliedCssClassNames":"snowflake-content-chip-white-bg"},"resource_chip_1":{"id":"content-chip-07a8071817","tagText":"GUIDE","tagColor":"#AAE5EA","cta":{"id":"cta","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"https://www.snowflake.com/en/developers/guides/cost-optimization/"},"linkTargetContentType":"GENERIC",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_EXTERNAL","text":"Read more"},"headline":{"id":"title","type":"heading5","lines":["Cost Optimization on Snowflake"],":type":"snowflake-site/components/title-v2"},":type":"snowflake-site/components/content-chip","appliedCssClassNames":"snowflake-content-chip-white-bg"},"resource_chip_2":{"id":"content-chip-54cc8d7aaf","tagText":"BLOG","tagColor":"#29B5E8","cta":{"id":"cta","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"https://www.snowflake.com/en/blog/engineering/llm-model-serving-vllm-inference/"},"linkTargetContentType":"GENERIC",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_EXTERNAL","text":"Read more"},"headline":{"id":"title","type":"heading5","lines":["LLM Serving Made Easy: High-Throughput Inference with vLLM"],":type":"snowflake-site/components/title-v2"},":type":"snowflake-site/components/content-chip","appliedCssClassNames":"snowflake-content-chip-white-bg"},"resource_chip_3":{"id":"content-chip-9878955f15","tagText":"EBOOK","tagColor":"#71D3DC","cta":{"id":"cta","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"https://www.snowflake.com/en/resources/ebook/building-trusted-ai--data-platform-essentials/"},"linkTargetContentType":"GENERIC",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_EXTERNAL","text":"Read more"},"headline":{"id":"title","type":"heading5","lines":["Building Trusted AI: Data Platform Essentials"],":type":"snowflake-site/components/title-v2"},":type":"snowflake-site/components/content-chip","appliedCssClassNames":"snowflake-content-chip-white-bg"}},":itemsOrder":["resource_chip_0","resource_chip_1","resource_chip_2","resource_chip_3"]},":type":"snowflake-site/components/flexible-column-container","isActiveTOC":false,"isBlogPage":false,"appliedCssClassNames":"snowflake-flexible-column-container-gray-10-bg"},"flexible_column_cont_1158003461":{"id":"flexible-column-container-84d6b993f2","propertiesId":"hub-explore-topics-header","type":"1-column","alignColumns":"top","containerMaxWidth":"extra-large","topPadding":"large","bottomPadding":"none","spaceBetween":"small","reverseOnMobile":false,"carouselOnMobile":false,"backgroundImageOption":"none","flexible_column_content_container_1":{"layout":"SIMPLE","id":"container-417da6d528",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container","appliedCssClassNames":"snowflake-responsive-container-inner-padding-extra-small",":items":{"title_v2_copy_copy":{"id":"title-v2-74f12f0b7d","additionalClasses":"hub-explore-topics-header__headline","type":"heading2","lines":["Explore AI Topics"],":type":"snowflake-site/components/title-v2","appliedCssClassNames":"left-alignment"},"text_copy_copy":{"id":"text-a21e39de51","additionalClasses":"hub-explore-topics-header__subheadline","text":"\u003Cp\u003EDeep dives into every aspect of artificial intelligence\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"}},":itemsOrder":["title_v2_copy_copy","text_copy_copy"]},":type":"snowflake-site/components/flexible-column-container","isActiveTOC":false,"isBlogPage":false,"appliedCssClassNames":"snowflake-flexible-column-container-white-bg"},"flexible_column_cont_1377146023":{"id":"flexible-column-container-e6361cbdf8","propertiesId":"hub-explore-topics-grid","type":"3-column-even","alignColumns":"top","containerMaxWidth":"extra-large","topPadding":"small","bottomPadding":"large","spaceBetween":"small","reverseOnMobile":false,"carouselOnMobile":false,"backgroundImageOption":"none","flexible_column_content_container_1":{"layout":"SIMPLE","id":"container-286c23e2cb",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"topic_card_0":{"id":"card-v2-a923bfc78d","configurationStatus":{"configured":true,"message":""},":type":"snowflake-site/components/card-v2","text":{"id":"text","text":"\u003Cp\u003EAI engineering turns capable models into reliable production systems by combining governed data, application logic, orchestration, evaluation and operational controls.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/text"},"layoutStyle":"vertical","title":{"id":"title","type":"heading4","lines":["AI Engineering"],":type":"snowflake-site/components/title-v2"},"button":{"id":"button","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/artificial-intelligence/ai-engineering/"},"linkTargetContentType":"GENERIC",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_INTERNAL","text":"Learn more"},"type":"content-card"}},":itemsOrder":["topic_card_0"]},"flexible_column_content_container_2":{"layout":"SIMPLE","id":"container-1451a2baee",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"topic_card_1":{"id":"card-v2-a8f47c6ac9","configurationStatus":{"configured":true,"message":""},":type":"snowflake-site/components/card-v2","text":{"id":"text","text":"\u003Cp\u003EContext engineering gives AI models the right instructions, data, tools and memory to produce reliable results.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/text"},"layoutStyle":"vertical","title":{"id":"title","type":"heading4","lines":["Context Engineering"],":type":"snowflake-site/components/title-v2"},"button":{"id":"button","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/artificial-intelligence/ai-engineering/context-engineering/"},"linkTargetContentType":"GENERIC",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_INTERNAL","text":"Learn more"},"type":"content-card"}},":itemsOrder":["topic_card_1"]},"flexible_column_content_container_3":{"layout":"SIMPLE","id":"container-575c6a1003",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"topic_card_2":{"id":"card-v2-990911d48e","configurationStatus":{"configured":true,"message":""},":type":"snowflake-site/components/card-v2","text":{"id":"text","text":"\u003Cp\u003EHarness engineering brings together tools, context, memory and guardrails so AI agents can handle multistep work safely and consistently.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/text"},"layoutStyle":"vertical","title":{"id":"title","type":"heading4","lines":["Harness Engineering"],":type":"snowflake-site/components/title-v2"},"button":{"id":"button","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/artificial-intelligence/ai-engineering/harness-engineering/"},"linkTargetContentType":"GENERIC",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_INTERNAL","text":"Learn more"},"type":"content-card"}},":itemsOrder":["topic_card_2"]},":type":"snowflake-site/components/flexible-column-container","isActiveTOC":false,"isBlogPage":false,"appliedCssClassNames":"snowflake-flexible-column-container-white-bg"}},":itemsOrder":["flexible_column_cont","flexible_column_cont_939100716","flexible_column_cont_1398138236","flexible_column_cont_663228916","flexible_column_cont_1786318617","flexible_column_cont_1467213961","flexible_column_cont_912630531","flexible_column_cont_1158003461","flexible_column_cont_1377146023"],":type":"wcm/foundation/components/responsivegrid"},"modal_container":{"layout":"SIMPLE","id":"container-aae56141aa",":type":"snowflake-site/components/modal/modal-container",":items":{},":itemsOrder":[]},"markup_editor_928258845":{"id":"markup-editor-e092c18c67","title":" ","cssContent":".snowflake-flexible-column-container-gray-10-bg\u003E.snowflake-flexible-column-container{background-color:var(--ui-background-05) !important}.text-size-regular:has(.seo-hub-hero__related-topic-label){display:flex;align-items:center}.hub-hero__headline span{text-transform:none !important}.hub-hero__subheadline p{max-width:70ch;margin-top:8px}.hub-hero__authors \u003E .container \u003E .cmp-container \u003E .aem-container{display:flex;flex-direction:row}.hub-hero__authors \u003E .container \u003E .cmp-container \u003E .aem-container \u003E div{width:auto !important;margin:24px 48px 0 0 !important}.hub-hero__authors .heading-5-v2{gap:var(--spacing-00)}.hub-hero__authors .snowflake-content-chip-button{display:none !important}.hub-hero__authors .snowflake-person-chip-content .body-2,.hub-hero__authors .snowflake-content-chip-content .snowflake-title-v2-line{font-size:16px !important;line-height:20px !important;font-family:\"Lato\",sans-serif !important;color:#000 !important;font-weight:600 !important}.hub-hero__authors .snowflake-person-chip-content .body-3,.hub-hero__authors .snowflake-content-chip-content .snowflake-title-v2-line:not(:first-child){font-weight:400 !important;color:var(--text-05) !important;font-size:16px !important}.hub-hero__authors .snowflake-image-container img{aspect-ratio:1 !important;border-radius:100%;overflow:hidden}.hub-hero__authors .snowflake-person-chip-avatar{width:56px;height:56px}.hub-hero__authors .snowflake-content-chip{align-items:center;display:inline-flex}.hub-hero__authors .snowflake-content-chip-image{max-width:56px;line-height:0;margin-right:var(--spacing-03)}.hub-hero__authors .snowflake-person-chip-inner-horizontal{gap:var(--spacing-03)}@media screen and (min-width:1367px){.hub-hero__headline .heading-1-v2{font-size:48px;line-height:44px}}#hub-explore-resources-grid-inner\u003E.container\u003E.cmp-container\u003E.aem-container{display:flex;flex-direction:row;flex-wrap:wrap;gap:24px}#hub-explore-resources-grid-inner\u003E.container\u003E.cmp-container\u003E.aem-container::before,#hub-explore-resources-grid-inner\u003E.container\u003E.cmp-container\u003E.aem-container::after{display:none}#hub-explore-resources-grid-inner\u003E.container\u003E.cmp-container\u003E.aem-container\u003Ediv{width:calc(25% - 18px)}.text-color-text-05 .snowflake-text h2,.text-color-text-05.cq-Editable-dom h2,.text-color-text-05 .snowflake-text h3,.text-color-text-05.cq-Editable-dom h3,.text-color-text-05 .snowflake-text h4,.text-color-text-05.cq-Editable-dom h4,.text-color-text-05 .snowflake-text h5,.text-color-text-05.cq-Editable-dom h5,.text-color-text-05 .snowflake-text h6,.text-color-text-05.cq-Editable-dom h6{color:#000 !important}",":type":"snowflake-site/components/markup-editor","isGSAPEnabled":false},"markup_editor_597730182":{"id":"markup-editor-d68292df76","title":" ","cssContent":".sf-copy-markdown [data-copy-md]{display:inline-flex;align-items:center;gap:6px;padding:8px 16px;font-family:'Texta',sans-serif;text-transform:uppercase;font-weight:800 !important;font-size:14px;font-weight:500;color:#11567f;background:#f0faff;border:1px solid #b8e6f9;border-radius:24px;cursor:pointer;transition:background .2s ease,border-color .2s ease,color .2s ease}.sf-copy-markdown [data-copy-md]:hover{background:#ddf3fc;border-color:#29b5e8}.sf-copy-markdown [data-copy-md][data-copied=\"1\"]{color:#0f7b3e;background:#ecfdf5;border-color:#6ee7a0;pointer-events:none}.sf-copy-markdown [data-copy-md] svg{flex-shrink:0}.longform-conten .snowflake-content-chip-white-bg .snowflake-content-chip{box-shadow:0 0 24px 4px rgba(0,0,0,.02),0 4px 8px 0 rgba(0,0,0,.04);flex-direction:row-reverse;align-items:center}.longform-conten .snowflake-content-chip-button{display:none}.longform-conten .snowflake-content-chip-image__inner{aspect-ratio:5 / 3;display:flex;justify-content:center;align-items:center;background-color:var(--ui-01);border-radius:4px}.longform-conten .snowflake-content-chip-image{margin-right:0;margin-left:48px}.longform-conten .snowflake-content-chip-image img{width:50%;border-radius:0 !important;object-fit:contain}.longform-content .black-blue-text-color .snowflake-title-v2-line:not(:first-child){font-size:14px !important;font-weight:400 !important;color:rgba(0,0,0,.6) !important;margin-top:8px !important}.page-toc ul li:first-child{padding-top:0 !important}.page-toc ul li:last-child{padding-bottom:0 !important}.seo-hub__top-bar \u003E .container \u003E .cmp-container \u003E .aem-container \u003E div:first-child{flex-grow:1}.sf-copy-markdown{margin-top:40px !important}.page-toc ul{margin-top:16px !important}.seo-hub__top-bar \u003E .container \u003E .cmp-container \u003E .aem-container{display:flex;justify-content:space-between}.sf-copy-markdown{}.callout.snowflake-text p:not(:first-child){margin-top:var(--spacing-01)}.callout \u003E span \u003E p:first-child \u003E strong,.callout \u003E span \u003E p:first-child \u003E b{text-transform:uppercase;font-family:'Texta',sans-serif;font-size:16px !important;color:var(--ui-01) !important}.seo-hub-hero__subheadline p{max-width:50ch}.tag-group ul{list-style-type:none;padding:0;margin:0;display:flex;flex-direction:row;row-gap:12px;column-gap:8px;align-items:center;flex-wrap:wrap}.tag-group ul li:first-child{flex-shrink:0}.tag-group ul li a{display:inline-block;padding:2px 12px;border-radius:48px;background-color:#ededed;color:#666;font-size:14px !important}@media screen and (min-width:1367px){.seo-hub-hero__headline span.snowflake-title-v2-line{font-size:56px !important}}.callout.snowflake-text p:not(:first-child){margin-top:var(--spacing-01)}.callout \u003E span \u003E p:first-child \u003E b{text-transform:uppercase;font-family:'Texta',sans-serif;font-size:16px !important;color:var(--ui-01) !important}.seo-hub-hero__subheadline p{max-width:80ch}#hero:has(.snowflake-youtube-lite) .seo-hub-hero__subheadline p{max-width:50ch}.tag-group ul{list-style-type:none;padding:0;margin:0;display:flex;flex-direction:row;row-gap:12px;column-gap:8px;align-items:center;flex-wrap:wrap}.tag-group ul li:first-child{width:100%;flex-shrink:0}.tag-group ul li a{display:inline-block;padding:2px 12px;border-radius:48px;background-color:#ededed;color:#666;font-size:14px !important}@media screen and (min-width:1367px){.seo-hub-hero__headline span.snowflake-title-v2-line{font-size:56px !important}}",":type":"snowflake-site/components/markup-editor","isGSAPEnabled":false},"markup_editor":{"id":"markup-editor-12e194e3de","title":" ","cssContent":"div.snowflake-breadcrumb a.snowflake-breadcrumb-item,.snowflake-breadcrumb div.snowflake-breadcrumb-item{text-transform:none;font-weight:500}.snowflake-breadcrumb svg{display:none !important}.snowflake-breadcrumb a:has(svg)::after{content:'/';margin:0 12px;color:#666}.hub-sidebar{padding:0 40px}.sticky-sidebar{max-width:340px;margin-left:auto}.page-toc ul{list-style-type:none;padding:0}.page-toc li{padding:8px 16px;border-left:4px solid var(--ui-01);cursor:pointer;transition:300ms ease all}.page-toc li:hover{color:var(--ui-01);border-color:#7fd3f1;transition:300ms ease all}.callout,.customer-card{background-color:#eef9fd;border-left:4px solid var(--ui-01);padding:24px 24px 24px 32px;border-radius:4px}.logo-container{max-width:180px}.longform-content li{margin-top:1rem !important}div.longform-content p{max-width:80ch}.bolder .snowflake-title-v2-line{font-weight:900 !important}.border-top\u003Ediv{border-top:1px solid #ccc;padding-top:48px}.related-topics ul{list-style-type:none;padding:0;margin:0;display:flex;gap:8px;flex-wrap:wrap}.related-topics li{display:inline-block;border:1px solid #ccc;padding:4px 12px;border-radius:24px}div.longform-content .snowflake-text h2,div.longform-content .snowflake-text .heading-2-v2,div.longform-content .snowflake-text h3,div.longform-content .snowflake-text .heading-3-v2,div.longform-content .snowflake-title-v2 .heading-3-v2,div.longform-content .snowflake-text h4,div.longform-content .snowflake-text .heading-4-v2,div.longform-content .snowflake-title-v2 .heading-4-v2,div.longform-content .snowflake-text h5,div.longform-content .snowflake-text .heading-5-v2,div.longform-content .snowflake-title-v2 .heading-5-v2,div.longform-content .snowflake-text h6,div.longform-content .snowflake-title-v2 .heading-6-v2,div.longform-content .snowflake-text .heading-6-v2{text-transform:none !important}div.longform-content .snowflake-text h2,div.longform-content .snowflake-text .heading-2-v2,div.longform-content .snowflake-text h3,div.longform-content .snowflake-text .heading-3-v2,div.longform-content .snowflake-text h4,div.longform-content .snowflake-text .heading-4-v2,div.longform-content .snowflake-text h5,div.longform-content .snowflake-text .heading-5-v2,div.longform-content .snowflake-text h6,div.longform-content .snowflake-text .heading-6-v2{margin-top:1.5rem !important;line-height:1.1 !important}div.longform-content .snowflake-text h3,div.longform-content .snowflake-text .heading-3-v2,div.longform-content .snowflake-title-v2 .heading-3-v2,div.longform-content .snowflake-text h4,div.longform-content .snowflake-text .heading-4-v2,div.longform-content .snowflake-title-v2 .heading-4-v2,div.longform-content .snowflake-text h5,div.longform-content .snowflake-text .heading-5-v2,div.longform-content .snowflake-title-v2 .heading-5-v2,div.longform-content .snowflake-text h6,div.longform-content .snowflake-text .heading-6-v2,div.longform-content .snowflake-title-v2 .heading-6-v2{font-family:Lato,sans-serif !important;font-weight:800 !important}div.longform-content .snowflake-text h2,div.longform-content .snowflake-text .heading-2-v2,div.longform-content .snowflake-title-v2 .heading-2-v2{text-transform:none !important;font-size:28px !important}div.longform-content .snowflake-text h3,div.longform-content .snowflake-text .heading-3-v2,div.longform-content .snowflake-title-v2 .heading-3-v2{font-size:22px !important}div.longform-content .snowflake-text h4,div.longform-content .snowflake-text .heading-4-v2,div.longform-content .snowflake-title-v2 .heading-4-v2{font-size:18px !important}div.longform-content .snowflake-text h5,div.longform-content .snowflake-text .heading-5-v2,div.longform-content .snowflake-title-v2 .heading-5-v2{font-size:16px !important}div.longform-content .snowflake-text h6,div.longform-content .snowflake-text .heading-6-v2,div.longform-content .snowflake-title-v2 .heading-6-v2{font-size:14px !important}@media screen and (min-width:992px){div.longform-content .snowflake-text h2,div.longform-content .snowflake-text .heading-2-v2,div.longform-content .snowflake-title-v2 .heading-2-v2{font-size:38px !important}div.longform-content .snowflake-text h3,div.longform-content .snowflake-text .heading-3-v2,div.longform-content .snowflake-title-v2 .heading-3-v2{font-size:26px !important}div.longform-content .snowflake-text h4,div.longform-content .snowflake-text .heading-4-v2,div.longform-content .snowflake-title-v2 .heading-4-v2{font-size:22px !important}div.longform-content .snowflake-text h5,div.longform-content .snowflake-text .heading-5-v2,div.longform-content .snowflake-title-v2 .heading-5-v2{font-size:18px !important}div.longform-content .snowflake-text h6,div.longform-content .snowflake-text .heading-6-v2,div.longform-content .snowflake-title-v2 .heading-6-v2{font-size:16px !important}}.sticky-sidebar .page-toc li.is-active{font-weight:600;color:var(--snow-blue,#29b5e8)}.sticky-sidebar .page-toc li[data-anchor]{cursor:pointer}.longform-content table{margin-top:24px;margin-bottom:24px;width:100%;background-color:var(--ui-background-01);border-collapse:collapse;border:2px solid var(--ui-background-09);font-family:'Lato',sans-serif;color:var(--ui-background-09)}.longform-content table thead{background-color:var(--ui-01)}.longform-content th,.longform-content td{min-width:120px;border:2px solid var(--ui-background-09);padding:var(--spacing-01)}.longform-content ol{margin-top:0 !important}.longform-content ol li{margin-bottom:1rem !important}.longform-content ul li{margin:0;padding:0 0 0 32px;position:relative}.longform-content ul{list-style-type:none}.longform-content ul li::before{content:\"\";display:block;border-radius:100%;background:#29b5e8;width:18px;height:18px;position:absolute;top:4px;left:0;border:5px solid #e5f2f7;box-sizing:border-box}.seo-customer.snowflake-card-v2-advanced-horizontal .snowflake-card-v2-advanced-image-container{max-width:200px}.seo-customer.snowflake-card-v2-advanced-horizontal .snowflake-card-v2-advanced-image-container img{object-fit:contain}.related-topics-outer-container \u003E .container \u003E .cmp-container \u003E .aem-container{display:flex;flex-direction:row}.related-topics-outer-container \u003E .container \u003E .cmp-container \u003E .aem-container \u003E div{width:auto !important;margin:0 !important}.related-topics-outer-container \u003E .container \u003E .cmp-container \u003E .aem-container \u003E div:first-child{margin-right:16px !important;flex-shrink:0}","jsContent":"(function(){var OFFSET=100;if(window.gsap&&window.ScrollTrigger){gsap.registerPlugin(ScrollTrigger);var sidebar=document.querySelector('.sticky-sidebar');var body=document.querySelector('.longform-content');if(sidebar&&body){ScrollTrigger.create({trigger:sidebar,start:'top 100px',endTrigger:body,end:'bottom bottom',pin:sidebar,pinSpacing:false});}}document.addEventListener('click',function(e){var li=e.target.closest('li[data-anchor]');if(!li)return;var slug=li.getAttribute('data-anchor');var heading=document.querySelector('.anchor-title--'+CSS.escape(slug));if(!heading)return;e.preventDefault();var top=heading.getBoundingClientRect().top+window.pageYOffset-OFFSET;window.scrollTo({top:top,behavior:'smooth'});history.replaceState(null,'','#'+slug);},false);var headings=document.querySelectorAll('[class*=\"anchor-title--\"]');if(headings.length&&'IntersectionObserver'in window){var io=new IntersectionObserver(function(entries){entries.forEach(function(entry){if(!entry.isIntersecting)return;var cls=Array.from(entry.target.classList).find(function(c){return c.indexOf('anchor-title--')===0;});if(!cls)return;var slug=cls.replace('anchor-title--','');document.querySelectorAll('li[data-anchor]').forEach(function(li){li.classList.toggle('is-active',li.getAttribute('data-anchor')===slug);});});},{rootMargin:'-20% 0px -70% 0px',threshold:0});headings.forEach(function(h){io.observe(h);});}})();",":type":"snowflake-site/components/markup-editor","isGSAPEnabled":true},"experiencefragment-footer":{"id":"experiencefragment-b67dd4dc7b","localizedFragmentVariationPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/footer/master/jcr:content","configured":true,":type":"snowflake-site/components/experiencefragment","xfModelPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/footer/master.xfmodel.json"},"experiencefragment":{"id":"experiencefragment-3b4c1cf7a5","localizedFragmentVariationPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/footer-legal-disclaimers/master/jcr:content","configured":true,":type":"snowflake-site/components/experiencefragment","xfModelPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/footer-legal-disclaimers/master.xfmodel.json"}},":itemsOrder":["experiencefragment-banner","experiencefragment-header","responsivegrid","modal_container","markup_editor_928258845","markup_editor_597730182","markup_editor","experiencefragment-footer","experiencefragment"],":type":"wcm/foundation/components/responsivegrid"}},":itemsOrder":["root"],":hierarchyType":"page",":path":"/content/snowflake-site/global/en/artificial-intelligence/ai-engineering/ai-cost-optimization","analyticsEnabled":true,"coveoConfig":{"searchHub":"snowflake.com","organizationId":"snowflakecomputingproduction8neljofn","apiKey":"xx335921a6-2a0a-40f2-a167-e390b4766c3d","pipeline":"snowflake.com"},"analyticsDebugMode":false,"analyticsData":{"excludeFromAnalytics":false,"subCategory":"","pageType":"homepage","templateName":"base-page-template54","siteName":"snowflake","pageUrl":"/content/snowflake-site/global/en/artificial-intelligence/ai-engineering/ai-cost-optimization","language":"en","category":"general","pageName":"AI Cost Optimization: How to Reduce LLM and Inference Costs in Production","contentTags":[]},"isPasswordProtected":false,"analyticsContentTags":[],"locale":"en"}
  