{"cssClassNames":"page basicpage summit-page","templateName":"base-page-template54","canonicalLink":"https://www.snowflake.com/en/artificial-intelligence/machine-learning/model-training/gradient-descent/","robotsTags":[],"allowedRenditionsWidth":["320","480","640","768","960","1200","1440","1920"],"description":"Learn how gradient descent trains machine learning models, including backpropagation, learning rates, batch sizes, optimizers and loss curves.","language":"en","title":"Gradient Descent: How ML Models Learn | Snowflake","analyticsPageType":"homepage","analyticsCategory":"general","analyticsSubCategory":"","excludeFromAnalytics":false,":mappedPath":"/en/artificial-intelligence/machine-learning/model-training/gradient-descent/",":type":"snowflake-site/components/structure/page",":items":{"root":{"columnCount":12,"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","columnClassNames":{"markup_editor_928258845":"aem-GridColumn aem-GridColumn--default--12","experiencefragment-banner":"aem-GridColumn aem-GridColumn--default--12","experiencefragment-header":"aem-GridColumn aem-GridColumn--default--12","responsivegrid":"aem-GridColumn aem-GridColumn--default--12","markup_editor_597730182":"aem-GridColumn aem-GridColumn--default--12","experiencefragment-footer":"aem-GridColumn aem-GridColumn--default--12","experiencefragment":"aem-GridColumn aem-GridColumn--default--12","modal_container":"aem-GridColumn aem-GridColumn--default--12","markup_editor":"aem-GridColumn aem-GridColumn--default--12"},":items":{"experiencefragment-banner":{"id":"experiencefragment-13193137b8","localizedFragmentVariationPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/pushdown-banner/master/jcr:content","configured":true,":type":"snowflake-site/components/experiencefragment","xfModelPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/pushdown-banner/master.xfmodel.json"},"experiencefragment-header":{"id":"experiencefragment-708e63efd1","localizedFragmentVariationPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/mega-nav-header/master/jcr:content","configured":true,":type":"snowflake-site/components/experiencefragment","xfModelPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/mega-nav-header/master.xfmodel.json","languageNavPath":"/content/snowflake-site/global/en/artificial-intelligence/machine-learning/model-training/gradient-descent.languagenav.json"},"responsivegrid":{"columnCount":12,"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","columnClassNames":{"flexible_column_cont_939100716":"aem-GridColumn aem-GridColumn--default--12","flexible_column_cont":"aem-GridColumn aem-GridColumn--default--12","flexible_column_cont_1158003461":"aem-GridColumn aem-GridColumn--default--12","flexible_column_cont_663228916":"aem-GridColumn aem-GridColumn--default--12","flexible_column_cont_912630531":"aem-GridColumn aem-GridColumn--default--12","flexible_column_cont_1398138236":"aem-GridColumn aem-GridColumn--default--12","flexible_column_cont_1786318617":"aem-GridColumn aem-GridColumn--default--12","flexible_column_cont_1467213961":"aem-GridColumn aem-GridColumn--default--12","flexible_column_cont_1377146023":"aem-GridColumn aem-GridColumn--default--12"},":items":{"flexible_column_cont":{"id":"flexible-column-container-82c78a8e6f","propertiesId":"hub-hero-breadcrumbs","type":"1-column","alignColumns":"top","containerMaxWidth":"extra-large","topPadding":"small","bottomPadding":"none","spaceBetween":"small","reverseOnMobile":false,"carouselOnMobile":false,"propertiesCSSClasses":"page-section","backgroundImageOption":"none","flexible_column_content_container_1":{"layout":"SIMPLE","id":"container-d0ac5add93",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"breadcrumb":{"id":"breadcrumb-11c0bf84c4","items":[{"id":"breadcrumb-11c0bf84c4-item-f710b50fc9","link":{"valid":true,"url":"/en/artificial-intelligence/machine-learning/"},"active":false,"current":false,"title":"Machine Learning",":type":"snowflake-site/components/structure/page","appliedCssClassNames":"summit-page"},{"id":"breadcrumb-11c0bf84c4-item-7b414e79ce","link":{"valid":true,"url":"/en/artificial-intelligence/machine-learning/model-training/"},"active":false,"current":false,"title":"ML Model Training",":type":"snowflake-site/components/structure/page","appliedCssClassNames":"summit-page"},{"id":"breadcrumb-11c0bf84c4-item-af1757153f","link":{"valid":true,"url":"/en/artificial-intelligence/machine-learning/model-training/gradient-descent/"},"active":true,"current":true,"title":"Gradient Descent",":type":"snowflake-site/components/structure/page","appliedCssClassNames":"summit-page"}],":type":"snowflake-site/components/breadcrumb"}},":itemsOrder":["breadcrumb"]},":type":"snowflake-site/components/flexible-column-container","isBlogPage":false,"isActiveTOC":false,"appliedCssClassNames":"snowflake-flexible-column-container-gray-10-bg"},"flexible_column_cont_939100716":{"id":"flexible-column-container-d8d23f1292","propertiesId":"hub-hero","type":"2-column-even","alignColumns":"center","containerMaxWidth":"extra-large","topPadding":"extra-small","bottomPadding":"extra-small","spaceBetween":"small","reverseOnMobile":false,"carouselOnMobile":false,"propertiesCSSClasses":"page-section","backgroundImageOption":"none","flexible_column_content_container_1":{"layout":"SIMPLE","id":"container-4ac50ab647",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"title_v2":{"id":"title-v2-b76a9de4d7","additionalClasses":"hub-hero__headline","type":"heading1","lines":["Gradient Descent: How Machine Learning Models Learn From Error"],":type":"snowflake-site/components/title-v2","appliedCssClassNames":"left-alignment"},"text":{"id":"text-29be3c491e","additionalClasses":"hub-hero__subheadline","text":"\u003Cp\u003EEvery model update begins with the same question: Which weights contributed to the error, and how should they change? Gradient descent provides the answer.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-size-regular text-color-text-05"},"container":{"additionalClasses":"hub-hero__authors","layout":"RESPONSIVE_GRID","columnCount":12,"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","columnClassNames":{"content_chip_copy":"aem-GridColumn aem-GridColumn--default--12","content_chip":"aem-GridColumn aem-GridColumn--default--12"},"id":"container-2da4213f6d",":type":"snowflake-site/components/container",":items":{"content_chip":{"id":"content-chip-f10cf3eecf","cta":{"id":"cta","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/blog/authors/Laurie-Macpherson/"},"linkTargetContentType":"GENERIC",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_INTERNAL","text":"Read bio"},"image":{"id":"image","height":"800","alt":"Laurie MacPherson","src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--a7e9fdff-6f08-4edc-9cd1-e213bb234aaa/laurie-macpherson.jpg?preferwebp=true&quality=85","lazyEnabled":true,"isLcpImage":true,"width":"800",":type":"snowflake-site/components/image"},"headline":{"id":"title","type":"heading5","lines":["Laurie MacPherson","Technical Writer, Snowflake"],":type":"snowflake-site/components/title-v2"},":type":"snowflake-site/components/content-chip"},"content_chip_copy":{"id":"content-chip-9e8686b79d","cta":{"id":"cta","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/blog/authors/david-gaule/"},"linkTargetContentType":"GENERIC",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_INTERNAL","text":"Read bio"},"image":{"id":"image","height":"512","alt":"David Gaule","src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--fd9454ea-3b59-4d19-95be-534774cbd226/david.jpg?preferwebp=true&quality=85","lazyEnabled":true,"isLcpImage":false,"width":"512",":type":"snowflake-site/components/image"},"headline":{"id":"title","type":"heading5","lines":["David Gaule","Technical Editor, Snowflake"],":type":"snowflake-site/components/title-v2"},":type":"snowflake-site/components/content-chip"}},":itemsOrder":["content_chip","content_chip_copy"],"appliedCssClassNames":"snowflake-responsive-container-inner-padding-small"}},":itemsOrder":["title_v2","text","container"],"appliedCssClassNames":"snowflake-responsive-container-inner-padding-extra-small"},"flexible_column_content_container_2":{"additionalClasses":"hub-hero__video-column","layout":"SIMPLE","id":"container-6c6713c5ee",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"youtube":{"id":"embed-1d66f96e7e","youtubeVideoId":"F34xlRoQ3eQ","layout":"responsive","youtubeAspectRatio":"56.25","youtubeAutoPlay":false,"youtubeLoop":false,"youtubeMute":false,"youtubePlaysInline":false,"youtubeRel":false,"embeddableResourceType":"core/wcm/components/embed/v1/embed/embeddable/youtube","type":"EMBEDDABLE",":type":"snowflake-site/components/youtube"}},":itemsOrder":["youtube"]},":type":"snowflake-site/components/flexible-column-container","isBlogPage":false,"isActiveTOC":false,"appliedCssClassNames":"snowflake-flexible-column-container-gray-10-bg"},"flexible_column_cont_1398138236":{"id":"flexible-column-container-9e6efd82f5","propertiesId":"hub-hero-related-topics","type":"1-column","alignColumns":"top","containerMaxWidth":"extra-large","topPadding":"extra-small","bottomPadding":"medium","spaceBetween":"small","reverseOnMobile":false,"carouselOnMobile":false,"propertiesCSSClasses":"page-section","backgroundImageOption":"none","flexible_column_content_container_1":{"additionalClasses":"related-topics-outer-container border-top","layout":"SIMPLE","id":"container-78bf15607d",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"text_894059747":{"id":"text-f5e46d92cd","additionalClasses":"seo-hub-hero__related-topic-label","text":"\u003Cp\u003EML Model Training Topics:\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-size-regular text-color-text-05"},"text":{"id":"text-20701a4c2d","additionalClasses":"related-topics ","text":"\u003Cul\u003E\r\n\u003Cli\u003E\u003Ca href=\"https://www.snowflake.com/en/artificial-intelligence/machine-learning/model-training/data-augmentation/\"\u003EData Augmentation\u003C/a\u003E\u003C/li\u003E\r\n\u003C/ul\u003E\r\n","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-size-small"}},":itemsOrder":["text_894059747","text"]},":type":"snowflake-site/components/flexible-column-container","isBlogPage":false,"isActiveTOC":false,"appliedCssClassNames":"snowflake-flexible-column-container-gray-10-bg"},"flexible_column_cont_663228916":{"id":"flexible-column-container-0ba59a4950","propertiesId":"hub-body","type":"2-column-60-40","alignColumns":"top","containerMaxWidth":"extra-large","topPadding":"medium","bottomPadding":"medium","spaceBetween":"small","reverseOnMobile":false,"carouselOnMobile":false,"propertiesCSSClasses":"page-section","backgroundImageOption":"none","flexible_column_content_container_1":{"additionalClasses":"longform-content","layout":"SIMPLE","id":"hub-body-content",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"callout__0":{"id":"text-9a74a3ddbd","additionalClasses":"callout callout--general","text":"\u003Cp\u003E\u003Cstrong\u003EGRADIENT DESCENT DEFINED\u003C/strong\u003E\u003C/p\u003E\n\u003Cp\u003EGradient descent is a method for improving a machine learning model by repeatedly estimating which parameter changes would lower its current error.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-size-regular text-color-text-05"},"text__0":{"id":"text-c10e573e90","text":"\u003Cp\u003EWhen a \u003Ca href=\"https://www.snowflake.com/en/artificial-intelligence/machine-learning/models/\"\u003Emachine learning model\u003C/a\u003E produces a wrong prediction, how does it determine which of its millions of weights contributed to the error — and how each one should change to improve performance?\u003C/p\u003E\n\u003Cp\u003EGradient descent provides the basic rule. During training, the model makes a prediction, the loss function measures how far that prediction is from the expected result, and backpropagation calculates how the loss would change in response to a small change in each weight. The optimizer then uses those gradients to adjust the weights in a direction expected to reduce the loss.\u003C/p\u003E\n\u003Cp\u003EThe rule is simple, but applying it well is not. The learning rate controls the size of each update. The batch determines which training examples contribute to the gradient. The optimizer determines how the current gradients, and in many cases information from earlier updates, translate into changes to the weights. Together, those choices influence whether the loss falls steadily, fluctuates, stalls or becomes unstable.\u003C/p\u003E\n\u003Cp\u003EModern frameworks calculate gradients automatically, so practitioners rarely implement gradient descent themselves. They configure the process around it — choosing batch sizes, learning-rate schedules and optimizer variants, then interpreting the loss curve to determine whether those choices are working.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"},"title_what-is-gradient-descent":{"id":"title-v2-576e0711f0","additionalClasses":"anchor-title anchor-title--what-is-gradient-descent","type":"heading2","lines":["What is gradient descent?"],":type":"snowflake-site/components/title-v2"},"text_what-is-gradient-descent_0":{"id":"text-8f1107a47d","text":"\u003Cp\u003EGradient descent is an optimization algorithm that adjusts a model’s weights to reduce its loss, meaning the numerical measure of how far its predictions are from the expected result. It does this iteratively, using the gradient of the loss to guide each update.\u003C/p\u003E\n\u003Cp\u003EThose weights determine how the model transforms an input into an output. In a linear regression, for example, they define the slope and intercept of the fitted line. In a \u003Ca href=\"https://www.snowflake.com/en/artificial-intelligence/machine-learning/neural-network/\"\u003Eneural network\u003C/a\u003E, the same basic idea extends across many layers and potentially billions of parameters.\u003C/p\u003E\n\u003Cp\u003EGradient descent should be distinguished from gradient boosting. Gradient descent updates parameters within a model. Gradient boosting builds an ensemble by adding models sequentially, with each one fitted to the errors left by the models before it.\u003C/p\u003E\n\u003Cp\u003E\u003Cem\u003EWatch this Snowflake BUILD session to learn how teams are using Snowflake ML to build and operationalize large-scale models:\u003C/em\u003E\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"},"yt_what-is-gradient-descent_0":{"id":"embed-81c7937af9","youtubeVideoId":"kBYZVIXgh1o","layout":"responsive","youtubeAspectRatio":"56.25","youtubeAutoPlay":false,"youtubeLoop":false,"youtubeMute":false,"youtubePlaysInline":false,"youtubeRel":false,"embeddableResourceType":"core/wcm/components/embed/v1/embed/embeddable/youtube","type":"EMBEDDABLE",":type":"snowflake-site/components/youtube"},"title_how-gradient-descent-works-from-prediction-to-the-next-update":{"id":"title-v2-6abad7b66f","additionalClasses":"anchor-title anchor-title--how-gradient-descent-works-from-prediction-to-the-next-update","type":"heading2","lines":["How gradient descent works: from prediction to the next update"],":type":"snowflake-site/components/title-v2"},"text_how-gradient-descent-works-from-prediction-to-the-next-update_0":{"id":"text-7b0207e8a7","text":"\u003Cp\u003ETo understand the full process, consider a simple linear regression model that predicts a house’s sale price from its square footage. The model uses two trainable parameters: a weight that controls how sharply predicted price rises with square footage and a bias that sets the starting level of the fitted line. During training, gradient descent adjusts those values so the line fits the observed sales data more closely.\u003C/p\u003E\r\n\u003Cp\u003ESuppose the model begins with poorly chosen weights. For a 1,500-square-foot house that sold for $500,000, it predicts $400,000. Training now needs to measure the error, determine how the current weights contributed to it and revise them before the model predicts again.\u003C/p\u003E\r\n\u003Ch3\u003E1. The forward pass produces a prediction\u003C/h3\u003E\r\n\u003Cp\u003EDuring the forward pass, the model applies its current weights to the input and produces a predicted sale price.\u003C/p\u003E\r\n\u003Cp\u003EA real training step would typically process a mini-batch rather than one house at a time. The model might predict prices for 256 properties in parallel, using the same current weights for every example in the batch.\u003C/p\u003E\r\n\u003Ch3\u003E2. The loss function measures the error\u003C/h3\u003E\r\n\u003Cp\u003EThe loss function compares those predictions with the recorded sale prices. A regression model commonly uses mean squared error, which calculates the difference between each predicted and observed value, squares those differences and averages them across the batch.\u003C/p\u003E\r\n\u003Cp\u003ESquaring prevents positive and negative errors from canceling each other out, while giving larger misses more influence on the final score. The result is one loss value representing how well the current weights performed on that batch.\u003C/p\u003E\r\n\u003Cp\u003EAt this point, the training process knows how large the error is. The loss value alone, however, doesn’t show how each weight should change.\u003C/p\u003E\r\n\u003Ch3\u003E3. Backpropagation calculates the gradients\u003C/h3\u003E\r\n\u003Cp\u003EBackpropagation calculates the gradient of the loss with respect to each trainable weight. For a given weight, the gradient describes how the loss would respond to a small change in its value.\u003C/p\u003E\r\n\u003Cp\u003EThe gradient’s sign indicates the direction of the local change. A positive gradient means that a small increase in the weight would increase the loss, while a negative gradient means that a small increase would reduce it. Its magnitude shows how sensitive the loss is to that weight at the model’s current settings.\u003C/p\u003E\r\n\u003Cp\u003EIn the linear regression example, the calculation involves only a few operations, while a neural network may contain millions or billions of weights connected across many layers. Backpropagation handles that larger calculation by applying the chain rule from the loss backward through every operation that contributed to the prediction.\u003C/p\u003E\r\n\u003Ch3\u003E4. The optimizer updates the weights\u003C/h3\u003E\r\n\u003Cp\u003EFor plain gradient descent, the weight update can be written as:\u003C/p\u003E\r\n\u003Cp style=\"text-align: center;\"\u003Ewt+1​=wt​−η∇L(wt​)\u003C/p\u003E\r\n\u003Cp\u003EThe current weight is wt​, the gradient is ∇L(wt​), and the learning rate η controls the size of the change.\u003C/p\u003E\r\n\u003Cp\u003EBecause the optimizer subtracts the gradient, it moves the weight in the direction expected to reduce the loss. A positive gradient lowers the weight, while a negative gradient raises it.\u003C/p\u003E\r\n\u003Cp\u003EIn the house-price example, suppose the model is underestimating how strongly prices rise with square footage. Backpropagation may produce a negative gradient for the weight controlling that relationship, indicating that increasing the weight should reduce the loss. The optimizer raises it by an amount determined by the learning rate.\u003C/p\u003E\r\n\u003Cp\u003ESome optimizers use more elaborate update rules, but they begin with the same gradients. They differ in how they scale, combine or transform that information before changing the weights.\u003C/p\u003E\r\n\u003Ch3\u003E5. The model repeats the loop\u003C/h3\u003E\r\n\u003Cp\u003EWith the revised weights, the model processes the next mini-batch. It produces new predictions, calculates another loss, computes new gradients and updates the weights again.\u003C/p\u003E\r\n\u003Cp\u003EThe loss doesn’t have to decline after every batch. Each mini-batch contains a different sample of the training data, so an update that improves the model overall may still produce a slightly higher loss on the next group of examples. Practitioners look for a downward trend across many updates rather than expecting a perfectly smooth sequence.\u003C/p\u003E\r\n\u003Cp\u003EAfter the model has processed every training example once, it has completed what is called an epoch. Training usually continues for multiple epochs, allowing successive updates to refine the weights as the model encounters the data again.\u003C/p\u003E\r\n\u003Cp\u003EModern frameworks automate most of this process. PyTorch, JAX and TensorFlow record the operations used during the forward pass and apply automatic differentiation to calculate the gradients. Practitioners define the model and loss function, select an optimizer and configure settings such as the learning rate and batch size.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"},"card_v2_how-gradient-descent-works-from-prediction-to-the-next-update_0":{"id":"card-v2-0deb06ef0c","additionalClasses":"seo-customer","configurationStatus":{"configured":true,"message":""},":type":"snowflake-site/components/card-v2","title":{"id":"title","type":"heading4","lines":["Customer story: Yieldmo"],":type":"snowflake-site/components/title-v2"},"button":{"id":"button","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/customers/all-customers/case-study/yieldmo/"},"linkTargetContentType":"GENERIC",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_INTERNAL","text":"Read the full case study"},"image":{"id":"image","height":"351","alt":"yieldmo logo","src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--ba6d49bf-dac9-45c9-a888-3fd951e9ba58/yieldmo.png?preferwebp=true&quality=85","lazyEnabled":true,"isLcpImage":false,"width":"624",":type":"snowflake-site/components/image"},"type":"content-card","text":{"id":"text","text":"\u003Cp\u003EYieldmo uses Snowflake and Snowpark Container Services to accelerate AI-powered advertising predictions while reducing the cost and complexity of its ML inference layer. By running H2O eScorer as a service in Snowflake, Yieldmo avoids large data transfers, brings predictive workloads closer to its data and enables data science teams to work with SQL instead of managing extra infrastructure. Initial testing shows predictions running 20x faster at 75% lower cost, helping Yieldmo improve A/B testing, speed time to market and optimize ad performance in near real time.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/text"},"layoutStyle":"horizontal"},"title_batch-stochastic-and-mini-batch-gradient-descent":{"id":"title-v2-ab9628c985","additionalClasses":"anchor-title anchor-title--batch-stochastic-and-mini-batch-gradient-descent","type":"heading2","lines":["Batch, stochastic and mini-batch gradient descent"],":type":"snowflake-site/components/title-v2"},"text_batch-stochastic-and-mini-batch-gradient-descent_0":{"id":"text-5a51a8b50c","text":"\u003Cp\u003EA gradient can be calculated from the entire training data set, from one example or from a subset of examples. The choice changes both the cost of an update and the amount of variation in the direction it provides.\u003C/p\u003E\n\u003Ch3\u003EBatch gradient descent\u003C/h3\u003E\n\u003Cp\u003EBatch gradient descent uses every training example before updating the weights. In the house-price example, the model would predict a value for every property, calculate one loss across the full data set and derive a gradient from all of those observations.\u003C/p\u003E\n\u003Cp\u003EBecause every example contributes, the gradient represents the direction that reduces loss across the complete training set. The update is relatively stable, but each step requires a full pass through the data. At large data volumes, that approach is expensive and difficult to fit into memory.\u003C/p\u003E\n\u003Ch3\u003EStochastic gradient descent\u003C/h3\u003E\n\u003Cp\u003EStochastic gradient descent, in its strict mathematical definition, updates the weights after a single example. Each step is inexpensive, although one observation may point in a direction that differs considerably from the direction favored by the data set as a whole.\u003C/p\u003E\n\u003Cp\u003EThe resulting path is noisy. Loss may rise on one update and fall on the next even while the longer-term trend improves. Single-example updates also make poor use of the parallel processing available on modern accelerators.\u003C/p\u003E\n\u003Ch3\u003EMini-batch gradient descent\u003C/h3\u003E\n\u003Cp\u003EMini-batch gradient descent uses a subset of examples for each update. A batch might contain 32, 256 or several thousand records, depending on the model, data and available hardware. This gives the framework enough work to process efficiently in parallel while preserving frequent updates.\u003C/p\u003E\n\u003Cp\u003EIn everyday deep-learning usage, the term stochastic gradient descent (or SGD) often includes mini-batch training because each batch provides a sampled estimate of the full-data gradient.\u003C/p\u003E\n\u003Cp\u003EMini-batch noise isn’t purely a cost. Each batch provides a slightly different estimate of the full-data gradient. That variation can sometimes help the optimizer move out of shallow or low-gradient regions, although it can also make the training path less stable. Under some conditions, the noise acts as a form of implicit regularization and contributes to solutions that perform better on unseen data.\u003C/p\u003E\n\u003Cp\u003EBatch size therefore affects more than speed. Larger batches produce smoother, lower-variance gradient estimates. Smaller batches yield more frequent and variable updates. Changing the batch size also changes the number and character of the updates, so practitioners usually revisit the learning rate and schedule at the same time. Larger batches can sometimes support larger learning rates, but the relationship depends on the optimizer, architecture and training setup.\u003C/p\u003E\n\u003Cp\u003EAvailable memory places an upper bound on the batch. Inputs, activations, gradients, model weights and optimizer state must all coexist during training, with activations frequently consuming a large share of GPU memory.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"},"title_how-learning-rate-controls-training":{"id":"title-v2-d75e69df5e","additionalClasses":"anchor-title anchor-title--how-learning-rate-controls-training","type":"heading2","lines":["How learning rate controls training"],":type":"snowflake-site/components/title-v2"},"text_how-learning-rate-controls-training_0":{"id":"text-930977efdd","text":"\u003Cp\u003EThe learning rate determines how far the weights move during each update. Even when the model, data and gradient remain unchanged, a learning rate of 0.1 produces a much larger adjustment than a learning rate of 0.001.\u003C/p\u003E\n\u003Cp\u003EWhen the rate is too high, the weights may leap past lower-loss values. One update crosses a useful region, the next crosses back and the loss begins to oscillate. With still larger steps, the calculations may become unstable and produce an \u003Ca href=\"https://codemia.io/knowledge-hub/path/deep-learning_nan_loss_reasons_closed_1\" target=\"_blank\"\u003Einfinite or NaN loss\u003C/a\u003E.\u003C/p\u003E\n\u003Cp\u003EA very small rate creates slower progress. The loss may decline, but each adjustment changes the model so little that training requires far more steps than the available budget supports. In a flat part of the landscape, the curve may appear stalled.\u003C/p\u003E\n\u003Cp\u003EReturning to the regression model, suppose the fitted line lies below most of the observed house prices. The gradient indicates that the slope should increase. A moderate learning rate moves the line closer over several updates. An excessive rate sends the slope past the useful range, while a very small one barely changes it.\u003C/p\u003E\n\u003Cp\u003EThe appropriate step size can shift during training. Early in a run, weights may be far from a useful solution and gradients may vary sharply. Later, after the loss has declined, smaller updates can refine the parameters without repeatedly crossing low-loss regions.\u003C/p\u003E\n\u003Cp\u003ELarge neural-network training runs often use a learning-rate schedule. Warmup begins with small steps and increases the rate over an initial sequence of updates. Once the model and optimizer state have stabilized, the schedule reaches a peak and later reduces the rate.\u003C/p\u003E\n\u003Cp\u003ECosine decay is one common approach. It lowers the learning rate along a smooth cosine-shaped curve, allowing larger updates during the main phase of training and finer adjustments near the end. Other schedules use linear decay, stepwise reductions or performance-based changes.\u003C/p\u003E\n\u003Cp\u003EA schedule still requires sensible starting values. The peak rate, warmup duration and final rate interact with the batch size, optimizer and architecture. Changing the batch without revisiting the learning rate can turn a stable training configuration into a slow or unstable one.\u003C/p\u003E\n\u003Cp\u003EWhen loss rises sharply, oscillates or becomes NaN, an excessive learning rate is one of the first causes to investigate. The diagnostic section later in this article covers the other conditions that can produce similar patterns.\u003C/p\u003E\n\u003Cp\u003EPractitioners commonly test several learning rates and batch sizes rather than treating either as an isolated choice. Hyperparameter optimization tools such as \u003Ca href=\"https://optuna.org/\" target=\"_blank\"\u003EOptuna\u003C/a\u003E and \u003Ca href=\"https://docs.ray.io/en/latest/tune/index.html\" target=\"_blank\"\u003ERay Tune\u003C/a\u003E can run candidate configurations concurrently, although each candidate still incurs its own training cost.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"},"title_how-optimizers-modify-the-update":{"id":"title-v2-ef13e498d7","additionalClasses":"anchor-title anchor-title--how-optimizers-modify-the-update","type":"heading2","lines":["How optimizers modify the update"],":type":"snowflake-site/components/title-v2"},"text_how-optimizers-modify-the-update_0":{"id":"text-e592f458ed","text":"\u003Cp\u003EPlain gradient descent scales the current gradient by the learning rate and applies it directly. Modern optimizers change that calculation by incorporating earlier gradients, adapting the step for individual parameters or using structural information from the model’s weight matrices.\u003C/p\u003E\n\u003Ch3\u003EMomentum methods\u003C/h3\u003E\n\u003Cp\u003EMomentum maintains a running direction based on recent updates. When several gradients point the same way, their effect accumulates. A single mini-batch that points sharply elsewhere has less influence because the optimizer carries information from the preceding steps.\u003C/p\u003E\n\u003Cp\u003EThis is useful on uneven loss landscapes. In a long, narrow valley, plain gradient descent may repeatedly cross from one wall to the other. Momentum preserves more motion along the valley while damping some of that lateral oscillation.\u003C/p\u003E\n\u003Ch3\u003ECoordinate-wise adaptive methods: AdaGrad, RMSprop, Adam and AdamW\u003C/h3\u003E\n\u003Cp\u003EAdaptive methods adjust the effective learning rate for each parameter. \u003Ca href=\"https://optimization.cbe.cornell.edu/index.php?title=AdaGrad\" target=\"_blank\"\u003EAdaGrad\u003C/a\u003E introduced parameter-specific rates based on accumulated squared gradients, although its rates can shrink too aggressively over a long run. \u003Ca href=\"https://optimization.cbe.cornell.edu/index.php?title=RMSProp\" target=\"_blank\"\u003ERMSprop\u003C/a\u003E uses a moving average instead, allowing recent gradient behavior to carry more weight.\u003C/p\u003E\n\u003Cp\u003E\u003Ca href=\"https://www.geeksforgeeks.org/deep-learning/adam-optimizer/\" target=\"_blank\"\u003EAdam\u003C/a\u003E combines a momentum-like estimate of the gradient’s direction with a moving estimate of its squared magnitude. Each parameter receives an update informed by both. This often makes Adam easier to tune than plain stochastic gradient descent, especially when gradients vary widely across parameters, although the learning rate still has a major effect on stability and final performance.\u003C/p\u003E\n\u003Cp\u003E\u003Ca href=\"https://optimization.cbe.cornell.edu/index.php?title=AdamW\" target=\"_blank\"\u003EAdamW\u003C/a\u003E separates weight decay from Adam’s adaptive gradient update. In the original Adam formulation, adding an L2 penalty to the loss doesn't behave exactly like conventional weight decay because the optimizer’s parameter-specific scaling also affects that penalty. Applying weight decay separately makes its effect easier to interpret and gives practitioners more direct control over optimization and regularization.\u003C/p\u003E\n\u003Cp\u003EAdamW is a common default for transformer training, while stochastic gradient descent with momentum remains competitive in areas such as \u003Ca href=\"https://www.snowflake.com/en/artificial-intelligence/computer-vision/\"\u003Ecomputer vision\u003C/a\u003E. A carefully tuned SGD configuration can generalize well and uses less optimizer state, but it often demands more experimentation.\u003C/p\u003E\n\u003Cp\u003EMemory is part of that choice. Adam and AdamW typically maintain two additional values for every trainable parameter. On a large model, those moment estimates can occupy a substantial share of the total memory footprint alongside the weights and gradients.\u003C/p\u003E\n\u003Ch3\u003EStructured or matrix-aware methods: Shampoo, SOAP and Muon\u003C/h3\u003E\n\u003Cp\u003E\u003Ca href=\"https://arxiv.org/pdf/2607.20548\" target=\"_blank\"\u003EStructured optimizers\u003C/a\u003E take advantage of the fact that many neural-network parameters are organized as matrices or higher-dimensional tensors. Rather than adapting every scalar entry independently, these methods use relationships across rows, columns or tensor dimensions when constructing the update.\u003C/p\u003E\n\u003Cp\u003EShampoo maintains preconditioning information for each dimension of a tensor, while SOAP applies Adam-style adaptation within Shampoo’s preconditioned coordinate system. These methods use richer structural information than coordinate-wise optimizers and can capture some of the benefits associated with curvature-aware optimization.\u003C/p\u003E\n\u003Cp\u003EMuon takes a different approach. It transforms momentum updates for two-dimensional weight matrices through an \u003Ca href=\"https://bvanderlei.github.io/jupyter-guide-to-linear-algebra/Orthogonalization.html\" target=\"_blank\"\u003Eorthogonalization\u003C/a\u003E step before applying them, rather than directly estimating the curvature of the loss.\u003C/p\u003E\n\u003Cp\u003EInterest has grown as large-model experiments have shown that a more effective update may reduce the training required to reach a target quality. \u003Ca href=\"https://pytorch.org/blog/using-muon-optimizer-with-deepspeed/\" target=\"_blank\"\u003EIn one 1.5-billion-parameter experiment\u003C/a\u003E, Muon reached GPT-2 XL-level performance approximately 25% faster than AdamW. The trade-off is workload-specific: additional computation or optimizer state during each step must be balanced against any reduction in the total number of steps.\u003C/p\u003E\n\u003Cp\u003EFor most training projects, the practical starting point remains an optimizer already proven for the model family. Learning rate and schedule usually deserve attention before replacing a well-supported default. At the scale of frontier-model training, however, optimizer choice has reemerged as a meaningful compute and performance decision.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"},"title_when-training-stalls-reading-the-loss-curve":{"id":"title-v2-4743f98435","additionalClasses":"anchor-title anchor-title--when-training-stalls-reading-the-loss-curve","type":"heading2","lines":["When training stalls: reading the loss curve"],":type":"snowflake-site/components/title-v2"},"text_when-training-stalls-reading-the-loss-curve_0":{"id":"text-a9df79ff15","text":"\u003Cp\u003EA loss curve summarizes how the model’s error changes across updates. Its shape can’t identify a cause by itself, but it can show whether training is progressing, becoming unstable or separating from validation performance.\u003C/p\u003E\n\u003Cp\u003EA healthy curve often drops quickly at first, then declines more gradually. Mini-batch variation may make the raw line uneven, especially with smaller batches, so practitioners frequently examine both raw and smoothed values.\u003C/p\u003E\n\u003Ch3\u003ELoss is flat from the beginning\u003C/h3\u003E\n\u003Cp\u003EWhen loss barely changes from the first updates, the model may not be receiving a useful signal. The learning rate could be too small, the gradients could be close to zero or the inputs and labels could be misaligned.\u003C/p\u003E\n\u003Cp\u003EVanishing gradients are one possible cause. As backpropagation moves through a deep network, the sequence of derivatives can repeatedly weaken the signal. The gradients reaching earlier layers may then become too small to change their weights meaningfully.\u003C/p\u003E\n\u003Cp\u003EActivation functions, initialization and architecture all influence gradient flow. \u003Ca href=\"https://www.geeksforgeeks.org/deep-learning/relu-activation-function-in-deep-learning/\" target=\"_blank\"\u003EReLU-family activations\u003C/a\u003E preserve stronger gradients across much of their range than saturating functions. Initialization methods scale starting weights according to the number of incoming or outgoing connections, while residual connections provide shorter paths through deep networks.\u003C/p\u003E\n\u003Cp\u003EGradient norms help distinguish among causes. Near-zero values across many layers suggest weak gradient flow or a disconnected computational graph. Healthy gradients paired with a flat loss point toward the learning rate, loss calculation or data.\u003C/p\u003E\n\u003Ch3\u003ELoss rises, oscillates or becomes NaN\u003C/h3\u003E\n\u003Cp\u003EAggressive updates often produce a rising or sharply oscillating curve. Lowering the learning rate is a sensible first test, followed by checks for exploding gradients, invalid input values and numerical instability.\u003C/p\u003E\n\u003Cp\u003E\u003Ca href=\"https://www.geeksforgeeks.org/deep-learning/vanishing-and-exploding-gradients-problems-in-deep-learning/\" target=\"_blank\"\u003EExploding gradients\u003C/a\u003E occur when the chain of operations used during backpropagation repeatedly amplifies the gradient. In deep or recurrent networks, those effects can compound until the resulting updates become extremely large. Gradient clipping can limit the damage. One common form, global-norm clipping, rescales the full set of gradients when their combined norm exceeds a threshold.\u003C/p\u003E\n\u003Cp\u003EClipping is especially useful in architectures prone to occasional spikes, although frequent clipping indicates that the learning rate, initialization, numerical precision or model design may still need attention.\u003C/p\u003E\n\u003Ch3\u003ELoss declines and then plateaus\u003C/h3\u003E\n\u003Cp\u003EA plateau may indicate that the model is approaching the best solution available under the current configuration. It can also appear when the learning rate has decayed too far, the model lacks capacity or the optimizer has entered a flat or poorly conditioned region.\u003C/p\u003E\n\u003Cp\u003ESaddle points are one possible explanation, although the loss curve alone cannot establish that diagnosis. Along some directions, the loss curves upward; along others, it curves downward. Near the center, the gradient may be small even though lower-loss routes still exist. Momentum, adaptive updates and mini-batch variation can sometimes help the optimizer continue moving.\u003C/p\u003E\n\u003Cp\u003EA plateau can also reflect limits in the data. Noisy labels, missing predictors or irreducible variation place a floor under the loss that optimization alone cannot remove.\u003C/p\u003E\n\u003Cp\u003EThe current learning rate, gradient norms and validation behavior provide more context than the training curve alone. If both training and validation loss flatten at acceptable levels, further optimization may offer little value.\u003C/p\u003E\n\u003Ch3\u003ETraining and validation loss diverge\u003C/h3\u003E\n\u003Cp\u003EWhen training loss continues to fall while validation loss rises, the optimizer is still reducing error on the training examples, but the model’s performance on unseen data is deteriorating.\u003C/p\u003E\n\u003Cp\u003EThis pattern points toward overfitting rather than a failure of gradient descent. Regularization, more representative training data, augmentation or an earlier stopping point may improve generalization.\u003C/p\u003E\n\u003Ch3\u003ENormalization can stabilize training\u003C/h3\u003E\n\u003Cp\u003ENormalization can improve optimization stability by keeping intermediate activations within a more manageable range. Batch normalization uses statistics calculated across a mini-batch, often supporting higher learning rates and smoother training. Its original explanation focused on reducing \u003Ca href=\"https://www.geeksforgeeks.org/deep-learning/internal-covariant-shift-problem-in-deep-learning/\" target=\"_blank\"\u003Einternal covariate shift\u003C/a\u003E, while later research has placed greater emphasis on smoothing the loss landscape.\u003C/p\u003E\n\u003Cp\u003EAt very small batch sizes, batch-level statistics become less reliable. Transformers therefore typically use layer normalization, which normalizes across each token’s hidden features and doesn’t depend on the other examples in the mini-batch.\u003C/p\u003E\n\u003Cp\u003EThe curve supplies a starting point for investigation. Gradient measurements, validation metrics, data checks and system logs establish which part of the training process actually needs adjustment.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"},"callout_when-training-stalls-reading-the-loss-curve_0":{"id":"text-5e877987a3","additionalClasses":"callout callout--warning","text":"\u003Cp\u003E\u003Cstrong\u003EQUICK TIP\u003C/strong\u003E\u003C/p\u003E\n\u003Cp\u003EDon’t judge training from a single mini-batch. Look for a downward trend across many updates, since batch-to-batch loss naturally fluctuates.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-size-regular text-color-text-05"},"title_how-data-and-compute-shape-training-performance":{"id":"title-v2-4d7a2a019c","additionalClasses":"anchor-title anchor-title--how-data-and-compute-shape-training-performance","type":"heading2","lines":["How data and compute shape training performance"],":type":"snowflake-site/components/title-v2"},"text_how-data-and-compute-shape-training-performance_0":{"id":"text-6d8bbf2a6a","text":"\u003Cp\u003EEach gradient update begins only after the training system has prepared another mini-batch. Reading records, applying transformations, moving tensors into accelerator memory and running the forward and backward passes all contribute to the time required for one step.\u003C/p\u003E\n\u003Cp\u003EAt production scale, several constraints tend to determine how efficiently that loop runs:\u003C/p\u003E\n\u003Cul\u003E\n\u003Cli\u003E\u003Cstrong\u003EData loading and preprocessing:\u003C/strong\u003E When the input pipeline cannot supply examples as quickly as the accelerator can process them, the GPU waits between updates. Tokenization, image decoding, augmentation and feature construction often run on CPUs, so loader concurrency, storage throughput and preprocessing design can affect training time even when the loss curve looks normal.\u003C/li\u003E\n\u003Cli\u003E\u003Cstrong\u003EAccelerator memory:\u003C/strong\u003E Training must hold model weights, forward-pass activations, gradients and optimizer state at the same time. Larger batches increase activation memory, while optimizers such as AdamW maintain additional tensors for every parameter. When the workload no longer fits, practitioners may reduce batch size, accumulate gradients across smaller batches, recompute selected activations during backpropagation or use reduced-precision formats.\u003C/li\u003E\n\u003Cli\u003E\u003Cstrong\u003EDistributed execution:\u003C/strong\u003E Larger models and batches may require parameters, gradients or optimizer state to be spread across multiple devices. Under data parallelism, each GPU processes a different mini-batch, then participates in an all-reduce operation so the workers apply a consistent update. As device counts grow, synchronization and interconnect bandwidth can consume a larger share of each step.\u003C/li\u003E\n\u003Cli\u003E\u003Cstrong\u003EUtilization and throughput:\u003C/strong\u003E Adding GPUs doesn't produce a proportional reduction in run time. In a \u003Ca href=\"https://people.eecs.berkeley.edu/~matei/papers/2021/sc_megatron_lm.pdf\" target=\"_blank\"\u003Elarge-scale Megatron-LM study\u003C/a\u003E, a 1-trillion-parameter training run reached 52% of theoretical per-GPU peak performance across 3,072 A100 GPUs, illustrating how memory access, communication and other overhead remain significant even in a highly optimized system. Step time and examples processed per second often provide a clearer operational view than device count alone.\u003C/li\u003E\n\u003Cli\u003E\u003Cstrong\u003ECheckpointing and recovery:\u003C/strong\u003E Long training runs periodically save model weights, optimizer state and progress metadata so they can resume after an interruption. Frequent checkpoints reduce the amount of work at risk, but writing large states too often adds storage traffic and delays subsequent steps.\u003C/li\u003E\n\u003C/ul\u003E\n\u003Cp\u003EOptimization choices determine how many updates the model needs. Data throughput, memory use and distributed execution determine how long those updates take.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"},"title_running-model-training-on-snowflake":{"id":"title-v2-456c50d65d","additionalClasses":"anchor-title anchor-title--running-model-training-on-snowflake","type":"heading2","lines":["Running model training on Snowflake"],":type":"snowflake-site/components/title-v2"},"text_running-model-training-on-snowflake_0":{"id":"text-ef01a2e5bd","text":"\u003Cp\u003E\u003Ca href=\"https://www.snowflake.com/en/product/features/end-to-end-ml-workflows/\"\u003ESnowflake ML\u003C/a\u003E brings training code, governed data and CPU or GPU compute into the same environment. Within \u003Ca href=\"https://docs.snowflake.com/en/developer-guide/snowflake-ml/container-runtime-ml\"\u003EContainer Runtime\u003C/a\u003E, practitioners can use familiar open source frameworks while Snowflake manages the infrastructure that runs them. Teams choose the workload configuration, dependencies and resource requirements, without having to build and maintain the underlying cluster themselves.\u003C/p\u003E\n\u003Ch3\u003ESupply the training loop with data\u003C/h3\u003E\n\u003Cp\u003EContainer Runtime provides optimized access to Snowflake tables and converts query results into objects used by common ML libraries. Training can operate near data already held and governed in Snowflake, reducing the need to export and maintain a separate training copy. \u003Ca href=\"https://www.snowflake.com/en/blog/engineering/machine-learning-container-runtime/\"\u003EIn one Snowflake test\u003C/a\u003E, Container Runtime loaded an 81 GB training data set in about 50 seconds, compared with approximately 9.5 minutes in the external environment used for the comparison.\u003C/p\u003E\n\u003Cp\u003ESnowflake Notebooks provide an interactive environment for exploring the data, developing features and testing training code. Since the notebook runs against Snowflake data and compute, practitioners can move from query results to framework-native training objects within the same workflow.\u003C/p\u003E\n\u003Ch3\u003EMatch compute to the workload\u003C/h3\u003E\n\u003Cp\u003ECompute pools supply the CPU or GPU resources used by Container Runtime. Teams can select an environment suited to the model’s memory and processing requirements without provisioning and maintaining the underlying cluster themselves.\u003C/p\u003E\n\u003Cp\u003EFor models that require more than one worker, Snowflake distributed trainers coordinate execution across nodes and GPUs for supported model families and frameworks, including PyTorch-based neural networks and distributed XGBoost workloads. The training code continues to define the model, loss function and optimizer, while Snowflake handles worker setup and resource orchestration.\u003C/p\u003E\n\u003Cp\u003EThe framework still controls techniques such as gradient accumulation, mixed-precision training and model sharding. Container Runtime provides the managed environment in which those techniques run.\u003C/p\u003E\n\u003Ch3\u003ERun repeatable jobs and compare experiments\u003C/h3\u003E\n\u003Cp\u003EML Jobs lets teams submit training workloads to Snowflake compute from Snowflake Notebooks or an existing development environment. This separates interactive development from repeatable execution while preserving the same underlying code.\u003C/p\u003E\n\u003Cp\u003EWhen teams test several learning rates, batch sizes or optimizer configurations, parallel ML Jobs can run the candidates concurrently. Snowflake ML Experiments records parameters, metrics and artifacts across those runs, giving practitioners a consistent basis for comparison.\u003C/p\u003E\n\u003Cp\u003EOnce training is complete, the model can be logged in the \u003Ca href=\"https://docs.snowflake.com/en/developer-guide/snowflake-ml/model-registry/overview\"\u003ESnowflake Model Registry\u003C/a\u003E. The registry stores model versions and metadata as governed Snowflake objects and provides a path into inference and downstream lifecycle management.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"},"title_gradient-descent-turns-error-into-progress":{"id":"title-v2-a5ac2c5e4f","additionalClasses":"anchor-title anchor-title--gradient-descent-turns-error-into-progress","type":"heading2","lines":["Gradient descent turns error into progress"],":type":"snowflake-site/components/title-v2"},"text_gradient-descent-turns-error-into-progress_0":{"id":"text-623116e05f","text":"\u003Cp\u003EGradient descent gives model training its basic feedback loop: make a prediction, measure the error, calculate the gradients and update the parameters. Whether that loop works well depends on the choices around it. Batch size shapes the gradient, the learning rate controls the size of each step and the optimizer determines how those signals become an update. The loss curve shows the result, while gradient norms, validation metrics and data checks help explain what is happening.\u003C/p\u003E\n\u003Cp\u003EAt production scale, training is also a systems problem. An effective configuration reduces the number of updates needed, while efficient data loading, memory use and distributed execution determine how quickly those updates can run.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"},"callout_gradient-descent-turns-error-into-progress_0":{"id":"text-7c97f64c54","additionalClasses":"callout callout--general","text":"\u003Cp\u003E\u003Cstrong\u003EKEY TAKEAWAY\u003C/strong\u003E\u003C/p\u003E\n\u003Cp\u003EModel training is a repeated loop of prediction, loss calculation, backpropagation and parameter updates. Stable training depends on how that loop is configured.\u003C/p\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-size-regular text-color-text-05"}},":itemsOrder":["callout__0","text__0","title_what-is-gradient-descent","text_what-is-gradient-descent_0","yt_what-is-gradient-descent_0","title_how-gradient-descent-works-from-prediction-to-the-next-update","text_how-gradient-descent-works-from-prediction-to-the-next-update_0","card_v2_how-gradient-descent-works-from-prediction-to-the-next-update_0","title_batch-stochastic-and-mini-batch-gradient-descent","text_batch-stochastic-and-mini-batch-gradient-descent_0","title_how-learning-rate-controls-training","text_how-learning-rate-controls-training_0","title_how-optimizers-modify-the-update","text_how-optimizers-modify-the-update_0","title_when-training-stalls-reading-the-loss-curve","text_when-training-stalls-reading-the-loss-curve_0","callout_when-training-stalls-reading-the-loss-curve_0","title_how-data-and-compute-shape-training-performance","text_how-data-and-compute-shape-training-performance_0","title_running-model-training-on-snowflake","text_running-model-training-on-snowflake_0","title_gradient-descent-turns-error-into-progress","text_gradient-descent-turns-error-into-progress_0","callout_gradient-descent-turns-error-into-progress_0"],"appliedCssClassNames":"snowflake-responsive-container-inner-padding-medium"},"flexible_column_content_container_2":{"additionalClasses":"hub-sidebar","layout":"SIMPLE","id":"hub-body-aside",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"container":{"additionalClasses":"sticky-sidebar","layout":"RESPONSIVE_GRID","columnCount":12,"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","columnClassNames":{"text_943981956_copy_":"aem-GridColumn aem-GridColumn--default--12","text_copy":"aem-GridColumn aem-GridColumn--default--12"},"id":"container-35e32fae6f",":type":"snowflake-site/components/container",":items":{"text_943981956_copy_":{"id":"text-c2d270ac43","additionalClasses":"eyebrow-text","text":"\u003Cp\u003EIn This Guide\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-size-regular"},"text_copy":{"id":"text-e6d74f43f2","additionalClasses":"page-toc","text":"\u003Cul\u003E\u003Cli data-anchor=\"what-is-gradient-descent\"\u003EWhat is gradient descent?\u003C/li\u003E\u003Cli data-anchor=\"how-gradient-descent-works-from-prediction-to-the-next-update\"\u003EHow gradient descent works: from prediction to the next update\u003C/li\u003E\u003Cli data-anchor=\"batch-stochastic-and-mini-batch-gradient-descent\"\u003EBatch, stochastic and mini-batch gradient descent\u003C/li\u003E\u003Cli data-anchor=\"how-learning-rate-controls-training\"\u003EHow learning rate controls training\u003C/li\u003E\u003Cli data-anchor=\"how-optimizers-modify-the-update\"\u003EHow optimizers modify the update\u003C/li\u003E\u003Cli data-anchor=\"when-training-stalls-reading-the-loss-curve\"\u003EWhen training stalls: reading the loss curve\u003C/li\u003E\u003Cli data-anchor=\"how-data-and-compute-shape-training-performance\"\u003EHow data and compute shape training performance\u003C/li\u003E\u003Cli data-anchor=\"running-model-training-on-snowflake\"\u003ERunning model training on Snowflake\u003C/li\u003E\u003Cli data-anchor=\"gradient-descent-turns-error-into-progress\"\u003EGradient descent turns error into progress\u003C/li\u003E\u003C/ul\u003E","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-size-small text-color-text-05"}},":itemsOrder":["text_943981956_copy_","text_copy"],"appliedCssClassNames":"snowflake-responsive-container-inner-padding-medium"}},":itemsOrder":["container"],"appliedCssClassNames":"snowflake-responsive-container-inner-padding-small"},":type":"snowflake-site/components/flexible-column-container","isBlogPage":false,"isActiveTOC":false},"flexible_column_cont_1786318617":{"id":"flexible-column-container-60a2834d36","propertiesId":"hub-faq","type":"2-column-40-60","alignColumns":"top","containerMaxWidth":"extra-large","topPadding":"large","bottomPadding":"large","spaceBetween":"small","reverseOnMobile":false,"carouselOnMobile":false,"backgroundImageOption":"none","flexible_column_content_container_1":{"layout":"SIMPLE","id":"hub-faq-intro",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"title_v2_copy":{"id":"title-v2-878918f2ac","additionalClasses":"hub-faq__headline","type":"heading2","lines":["Frequently Asked Questions"],":type":"snowflake-site/components/title-v2","appliedCssClassNames":"left-alignment"},"text_copy":{"id":"text-7fa2b62b87","additionalClasses":"hub-faq__subheadline","text":"\u003Cp\u003EYour common questions about gradient descent, answered by Snowflake experts.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"}},":itemsOrder":["title_v2_copy","text_copy"],"appliedCssClassNames":"snowflake-responsive-container-inner-padding-extra-small"},"flexible_column_content_container_2":{"layout":"SIMPLE","id":"hub-faq-accordions",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"simple_snowflake_acc":{"id":"simple-snowflake-accordion-057f77456d","additionalClasses":"seo-hub__faqs","showDivider":false,"accordionItemsList":[{"title":"Do LLMs use gradient descent?","richText":"\u003Cp\u003EYes. \u003Ca href=\"https://www.snowflake.com/en/fundamentals/large-language-model/\"\u003ELarge language models\u003C/a\u003E are trained with gradient-based optimization: backpropagation calculates gradients, and an optimizer such as AdamW uses them to update the model’s parameters across many mini-batches. Distributed computation changes the scale of the process, not the basic mechanism.\u003C/p\u003E"},{"title":"Is gradient descent guaranteed to find the global minimum?","richText":"\u003Cp\u003EOnly for certain convex problems under suitable conditions. Neural networks have nonconvex loss landscapes, so gradient descent isn’t guaranteed to find the global minimum; in practice, teams evaluate whether training is stable and validation performance is good enough for the task.\u003C/p\u003E"}],":type":"snowflake-site/components/simple-snowflake-accordion"}},":itemsOrder":["simple_snowflake_acc"]},":type":"snowflake-site/components/flexible-column-container","isBlogPage":false,"isActiveTOC":false,"appliedCssClassNames":"snowflake-flexible-column-container-gray-10-bg"},"flexible_column_cont_1467213961":{"id":"flexible-column-container-667e4a3996","propertiesId":"hub-explore-resources-header","type":"1-column","alignColumns":"top","containerMaxWidth":"extra-large","topPadding":"large","bottomPadding":"none","spaceBetween":"small","reverseOnMobile":false,"carouselOnMobile":false,"backgroundImageOption":"none","flexible_column_content_container_1":{"layout":"SIMPLE","id":"container-d59646f2c4",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"title_v2":{"id":"title-v2-20badcb9f2","additionalClasses":"hub-explore-resources-header__headline","type":"heading2","lines":["Explore AI Resources"],":type":"snowflake-site/components/title-v2","appliedCssClassNames":"left-alignment"}},":itemsOrder":["title_v2"]},":type":"snowflake-site/components/flexible-column-container","isBlogPage":false,"isActiveTOC":false,"appliedCssClassNames":"snowflake-flexible-column-container-gray-10-bg"},"flexible_column_cont_912630531":{"id":"flexible-column-container-58bb363903","propertiesId":"hub-explore-resources-grid","type":"1-column","alignColumns":"top","containerMaxWidth":"extra-large","topPadding":"small","bottomPadding":"large","spaceBetween":"small","reverseOnMobile":false,"carouselOnMobile":false,"backgroundImageOption":"none","flexible_column_content_container_1":{"layout":"SIMPLE","id":"hub-explore-resources-grid-inner",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"resource_chip_0":{"id":"content-chip-58d3d8276c","tagText":"GUIDE","tagColor":"#AAE5EA","cta":{"id":"cta","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"https://www.snowflake.com/en/developers/guides/develop-and-manage-ml-models-with-feature-store-and-model-registry/"},"linkTargetContentType":"GENERIC",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_EXTERNAL","text":"Read more"},"headline":{"id":"title","type":"heading5","lines":["Develop and Manage ML Models with Feature Store and Model Registry"],":type":"snowflake-site/components/title-v2"},":type":"snowflake-site/components/content-chip","appliedCssClassNames":"snowflake-content-chip-white-bg"},"resource_chip_1":{"id":"content-chip-c9af8675eb","tagText":"BLOG","tagColor":"#29B5E8","cta":{"id":"cta","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"https://www.snowflake.com/en/blog/scalable-model-development-production-snowflake-ml/"},"linkTargetContentType":"GENERIC",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_EXTERNAL","text":"Read more"},"headline":{"id":"title","type":"heading5","lines":["Build, Deploy and Serve Models at Scale with Snowflake ML"],":type":"snowflake-site/components/title-v2"},":type":"snowflake-site/components/content-chip","appliedCssClassNames":"snowflake-content-chip-white-bg"},"resource_chip_2":{"id":"content-chip-078edba819","tagText":"GUIDE","tagColor":"#AAE5EA","cta":{"id":"cta","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"https://www.snowflake.com/en/developers/guides/agentic-machine-learning-best-practices-coco/"},"linkTargetContentType":"GENERIC",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_EXTERNAL","text":"Read more"},"headline":{"id":"title","type":"heading5","lines":["Agentic Machine Learning Best Practices with Snowflake CoCo"],":type":"snowflake-site/components/title-v2"},":type":"snowflake-site/components/content-chip","appliedCssClassNames":"snowflake-content-chip-white-bg"},"resource_chip_3":{"id":"content-chip-6b15ff69a2","tagText":"BLOG","tagColor":"#29B5E8","cta":{"id":"cta","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"https://www.snowflake.com/en/blog/engineering/scale-real-time-model-serving/"},"linkTargetContentType":"GENERIC",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_EXTERNAL","text":"Read more"},"headline":{"id":"title","type":"heading5","lines":["How to Scale Real-Time Model Serving for Low-Latency ML Inference"],":type":"snowflake-site/components/title-v2"},":type":"snowflake-site/components/content-chip","appliedCssClassNames":"snowflake-content-chip-white-bg"}},":itemsOrder":["resource_chip_0","resource_chip_1","resource_chip_2","resource_chip_3"]},":type":"snowflake-site/components/flexible-column-container","isBlogPage":false,"isActiveTOC":false,"appliedCssClassNames":"snowflake-flexible-column-container-gray-10-bg"},"flexible_column_cont_1158003461":{"id":"flexible-column-container-5f69184853","propertiesId":"hub-explore-topics-header","type":"1-column","alignColumns":"top","containerMaxWidth":"extra-large","topPadding":"large","bottomPadding":"none","spaceBetween":"small","reverseOnMobile":false,"carouselOnMobile":false,"backgroundImageOption":"none","flexible_column_content_container_1":{"layout":"SIMPLE","id":"container-d194973dcd",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"title_v2_copy_copy":{"id":"title-v2-118e3ffc99","additionalClasses":"hub-explore-topics-header__headline","type":"heading2","lines":["Explore AI Topics"],":type":"snowflake-site/components/title-v2","appliedCssClassNames":"left-alignment"},"text_copy_copy":{"id":"text-eaf23101a9","additionalClasses":"hub-explore-topics-header__subheadline","text":"\u003Cp\u003EDeep dives into every aspect of artificial intelligence\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/text","appliedCssClassNames":"text-color-text-05"}},":itemsOrder":["title_v2_copy_copy","text_copy_copy"],"appliedCssClassNames":"snowflake-responsive-container-inner-padding-extra-small"},":type":"snowflake-site/components/flexible-column-container","isBlogPage":false,"isActiveTOC":false,"appliedCssClassNames":"snowflake-flexible-column-container-white-bg"},"flexible_column_cont_1377146023":{"id":"flexible-column-container-46c8129453","propertiesId":"hub-explore-topics-grid","type":"3-column-even","alignColumns":"top","containerMaxWidth":"extra-large","topPadding":"small","bottomPadding":"large","spaceBetween":"small","reverseOnMobile":false,"carouselOnMobile":false,"backgroundImageOption":"none","flexible_column_content_container_1":{"layout":"SIMPLE","id":"container-1b730e73d2",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"topic_card_0":{"id":"card-v2-9aad5f5a60","configurationStatus":{"configured":true,"message":""},":type":"snowflake-site/components/card-v2","title":{"id":"title","type":"heading4","lines":["ML Model Training"],":type":"snowflake-site/components/title-v2"},"button":{"id":"button","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/artificial-intelligence/machine-learning/model-training/"},"linkTargetContentType":"GENERIC",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_INTERNAL","text":"Learn more"},"type":"content-card","text":{"id":"text","text":"\u003Cp\u003EMachine learning training teaches a model to improve its performance by repeatedly evaluating its outputs and adjusting its parameters based on data.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/text"},"layoutStyle":"vertical"}},":itemsOrder":["topic_card_0"]},"flexible_column_content_container_2":{"layout":"SIMPLE","id":"container-2b55b89757",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"topic_card_1":{"id":"card-v2-9b6a0bc68b","configurationStatus":{"configured":true,"message":""},":type":"snowflake-site/components/card-v2","title":{"id":"title","type":"heading4","lines":["Machine Learning"],":type":"snowflake-site/components/title-v2"},"button":{"id":"button","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/artificial-intelligence/machine-learning/"},"linkTargetContentType":"GENERIC",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_INTERNAL","text":"Learn more"},"type":"content-card","text":{"id":"text","text":"\u003Cp\u003EML enables systems to learn patterns from data to automate decisions, generate predictions, uncover insights, and improve outcomes over time.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/text"},"layoutStyle":"vertical"}},":itemsOrder":["topic_card_1"]},"flexible_column_content_container_3":{"layout":"SIMPLE","id":"container-b3af64b3a8",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"topic_card_2":{"id":"card-v2-610f9d5645","configurationStatus":{"configured":true,"message":""},":type":"snowflake-site/components/card-v2","title":{"id":"title","type":"heading4","lines":["Deep Learning"],":type":"snowflake-site/components/title-v2"},"button":{"id":"button","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/artificial-intelligence/machine-learning/deep-learning/"},"linkTargetContentType":"GENERIC",":type":"snowflake-site/components/button","linkType":"SNOWFLAKE_INTERNAL","text":"Learn more"},"type":"content-card","text":{"id":"text","text":"\u003Cp\u003EDeep learning uses multilayer neural networks to progressively learn more complex patterns and features from data.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/text"},"layoutStyle":"vertical"}},":itemsOrder":["topic_card_2"]},":type":"snowflake-site/components/flexible-column-container","isBlogPage":false,"isActiveTOC":false,"appliedCssClassNames":"snowflake-flexible-column-container-white-bg"}},":itemsOrder":["flexible_column_cont","flexible_column_cont_939100716","flexible_column_cont_1398138236","flexible_column_cont_663228916","flexible_column_cont_1786318617","flexible_column_cont_1467213961","flexible_column_cont_912630531","flexible_column_cont_1158003461","flexible_column_cont_1377146023"],":type":"wcm/foundation/components/responsivegrid"},"modal_container":{"layout":"SIMPLE","id":"container-90236636a6",":type":"snowflake-site/components/modal/modal-container",":items":{},":itemsOrder":[]},"markup_editor_928258845":{"id":"markup-editor-3494e7b17e","title":" ","cssContent":".snowflake-flexible-column-container-gray-10-bg\u003E.snowflake-flexible-column-container{background-color:var(--ui-background-05) !important}.text-size-regular:has(.seo-hub-hero__related-topic-label){display:flex;align-items:center}.hub-hero__headline span{text-transform:none !important}.hub-hero__subheadline p{max-width:70ch;margin-top:8px}.hub-hero__authors \u003E .container \u003E .cmp-container \u003E .aem-container{display:flex;flex-direction:row}.hub-hero__authors \u003E .container \u003E .cmp-container \u003E .aem-container \u003E div{width:auto !important;margin:24px 48px 0 0 !important}.hub-hero__authors .heading-5-v2{gap:var(--spacing-00)}.hub-hero__authors .snowflake-content-chip-button{display:none !important}.hub-hero__authors .snowflake-person-chip-content .body-2,.hub-hero__authors .snowflake-content-chip-content .snowflake-title-v2-line{font-size:16px !important;line-height:20px !important;font-family:\"Lato\",sans-serif !important;color:#000 !important;font-weight:600 !important}.hub-hero__authors .snowflake-person-chip-content .body-3,.hub-hero__authors .snowflake-content-chip-content .snowflake-title-v2-line:not(:first-child){font-weight:400 !important;color:var(--text-05) !important;font-size:16px !important}.hub-hero__authors .snowflake-image-container img{aspect-ratio:1 !important;border-radius:100%;overflow:hidden}.hub-hero__authors .snowflake-person-chip-avatar{width:56px;height:56px}.hub-hero__authors .snowflake-content-chip{align-items:center;display:inline-flex}.hub-hero__authors .snowflake-content-chip-image{max-width:56px;line-height:0;margin-right:var(--spacing-03)}.hub-hero__authors .snowflake-person-chip-inner-horizontal{gap:var(--spacing-03)}@media screen and (min-width:1367px){.hub-hero__headline .heading-1-v2{font-size:48px;line-height:44px}}#hub-explore-resources-grid-inner\u003E.container\u003E.cmp-container\u003E.aem-container{display:flex;flex-direction:row;flex-wrap:wrap;gap:24px}#hub-explore-resources-grid-inner\u003E.container\u003E.cmp-container\u003E.aem-container::before,#hub-explore-resources-grid-inner\u003E.container\u003E.cmp-container\u003E.aem-container::after{display:none}#hub-explore-resources-grid-inner\u003E.container\u003E.cmp-container\u003E.aem-container\u003Ediv{width:calc(25% - 18px)}.text-color-text-05 .snowflake-text h2,.text-color-text-05.cq-Editable-dom h2,.text-color-text-05 .snowflake-text h3,.text-color-text-05.cq-Editable-dom h3,.text-color-text-05 .snowflake-text h4,.text-color-text-05.cq-Editable-dom h4,.text-color-text-05 .snowflake-text h5,.text-color-text-05.cq-Editable-dom h5,.text-color-text-05 .snowflake-text h6,.text-color-text-05.cq-Editable-dom h6{color:#000 !important}",":type":"snowflake-site/components/markup-editor","isGSAPEnabled":false},"markup_editor_597730182":{"id":"markup-editor-0834146200","title":" ","cssContent":".sf-copy-markdown [data-copy-md]{display:inline-flex;align-items:center;gap:6px;padding:8px 16px;font-family:'Texta',sans-serif;text-transform:uppercase;font-weight:800 !important;font-size:14px;font-weight:500;color:#11567f;background:#f0faff;border:1px solid #b8e6f9;border-radius:24px;cursor:pointer;transition:background .2s ease,border-color .2s ease,color .2s ease}.sf-copy-markdown [data-copy-md]:hover{background:#ddf3fc;border-color:#29b5e8}.sf-copy-markdown [data-copy-md][data-copied=\"1\"]{color:#0f7b3e;background:#ecfdf5;border-color:#6ee7a0;pointer-events:none}.sf-copy-markdown [data-copy-md] svg{flex-shrink:0}.longform-conten .snowflake-content-chip-white-bg .snowflake-content-chip{box-shadow:0 0 24px 4px rgba(0,0,0,.02),0 4px 8px 0 rgba(0,0,0,.04);flex-direction:row-reverse;align-items:center}.longform-conten .snowflake-content-chip-button{display:none}.longform-conten .snowflake-content-chip-image__inner{aspect-ratio:5 / 3;display:flex;justify-content:center;align-items:center;background-color:var(--ui-01);border-radius:4px}.longform-conten .snowflake-content-chip-image{margin-right:0;margin-left:48px}.longform-conten .snowflake-content-chip-image img{width:50%;border-radius:0 !important;object-fit:contain}.longform-content .black-blue-text-color .snowflake-title-v2-line:not(:first-child){font-size:14px !important;font-weight:400 !important;color:rgba(0,0,0,.6) !important;margin-top:8px !important}.page-toc ul li:first-child{padding-top:0 !important}.page-toc ul li:last-child{padding-bottom:0 !important}.seo-hub__top-bar \u003E .container \u003E .cmp-container \u003E .aem-container \u003E div:first-child{flex-grow:1}.sf-copy-markdown{margin-top:40px !important}.page-toc ul{margin-top:16px !important}.seo-hub__top-bar \u003E .container \u003E .cmp-container \u003E .aem-container{display:flex;justify-content:space-between}.sf-copy-markdown{}.callout.snowflake-text p:not(:first-child){margin-top:var(--spacing-01)}.callout \u003E span \u003E p:first-child \u003E strong,.callout \u003E span \u003E p:first-child \u003E b{text-transform:uppercase;font-family:'Texta',sans-serif;font-size:16px !important;color:var(--ui-01) !important}.seo-hub-hero__subheadline p{max-width:50ch}.tag-group ul{list-style-type:none;padding:0;margin:0;display:flex;flex-direction:row;row-gap:12px;column-gap:8px;align-items:center;flex-wrap:wrap}.tag-group ul li:first-child{flex-shrink:0}.tag-group ul li a{display:inline-block;padding:2px 12px;border-radius:48px;background-color:#ededed;color:#666;font-size:14px !important}@media screen and (min-width:1367px){.seo-hub-hero__headline span.snowflake-title-v2-line{font-size:56px !important}}.callout.snowflake-text p:not(:first-child){margin-top:var(--spacing-01)}.callout \u003E span \u003E p:first-child \u003E b{text-transform:uppercase;font-family:'Texta',sans-serif;font-size:16px !important;color:var(--ui-01) !important}.seo-hub-hero__subheadline p{max-width:80ch}#hero:has(.snowflake-youtube-lite) .seo-hub-hero__subheadline p{max-width:50ch}.tag-group ul{list-style-type:none;padding:0;margin:0;display:flex;flex-direction:row;row-gap:12px;column-gap:8px;align-items:center;flex-wrap:wrap}.tag-group ul li:first-child{width:100%;flex-shrink:0}.tag-group ul li a{display:inline-block;padding:2px 12px;border-radius:48px;background-color:#ededed;color:#666;font-size:14px !important}@media screen and (min-width:1367px){.seo-hub-hero__headline span.snowflake-title-v2-line{font-size:56px !important}}",":type":"snowflake-site/components/markup-editor","isGSAPEnabled":false},"markup_editor":{"id":"markup-editor-b51fe5301c","title":" ","cssContent":"div.snowflake-breadcrumb a.snowflake-breadcrumb-item,.snowflake-breadcrumb div.snowflake-breadcrumb-item{text-transform:none;font-weight:500}.snowflake-breadcrumb svg{display:none !important}.snowflake-breadcrumb a:has(svg)::after{content:'/';margin:0 12px;color:#666}.hub-sidebar{padding:0 40px}.sticky-sidebar{max-width:340px;margin-left:auto}.page-toc ul{list-style-type:none;padding:0}.page-toc li{padding:8px 16px;border-left:4px solid var(--ui-01);cursor:pointer;transition:300ms ease all}.page-toc li:hover{color:var(--ui-01);border-color:#7fd3f1;transition:300ms ease all}.callout,.customer-card{background-color:#eef9fd;border-left:4px solid var(--ui-01);padding:24px 24px 24px 32px;border-radius:4px}.logo-container{max-width:180px}.longform-content li{margin-top:1rem !important}div.longform-content p{max-width:80ch}.bolder .snowflake-title-v2-line{font-weight:900 !important}.border-top\u003Ediv{border-top:1px solid #ccc;padding-top:48px}.related-topics ul{list-style-type:none;padding:0;margin:0;display:flex;gap:8px;flex-wrap:wrap}.related-topics li{display:inline-block;border:1px solid #ccc;padding:4px 12px;border-radius:24px}div.longform-content .snowflake-text h2,div.longform-content .snowflake-text .heading-2-v2,div.longform-content .snowflake-text h3,div.longform-content .snowflake-text .heading-3-v2,div.longform-content .snowflake-title-v2 .heading-3-v2,div.longform-content .snowflake-text h4,div.longform-content .snowflake-text .heading-4-v2,div.longform-content .snowflake-title-v2 .heading-4-v2,div.longform-content .snowflake-text h5,div.longform-content .snowflake-text .heading-5-v2,div.longform-content .snowflake-title-v2 .heading-5-v2,div.longform-content .snowflake-text h6,div.longform-content .snowflake-title-v2 .heading-6-v2,div.longform-content .snowflake-text .heading-6-v2{text-transform:none !important}div.longform-content .snowflake-text h2,div.longform-content .snowflake-text .heading-2-v2,div.longform-content .snowflake-text h3,div.longform-content .snowflake-text .heading-3-v2,div.longform-content .snowflake-text h4,div.longform-content .snowflake-text .heading-4-v2,div.longform-content .snowflake-text h5,div.longform-content .snowflake-text .heading-5-v2,div.longform-content .snowflake-text h6,div.longform-content .snowflake-text .heading-6-v2{margin-top:1.5rem !important;line-height:1.1 !important}div.longform-content .snowflake-text h3,div.longform-content .snowflake-text .heading-3-v2,div.longform-content .snowflake-title-v2 .heading-3-v2,div.longform-content .snowflake-text h4,div.longform-content .snowflake-text .heading-4-v2,div.longform-content .snowflake-title-v2 .heading-4-v2,div.longform-content .snowflake-text h5,div.longform-content .snowflake-text .heading-5-v2,div.longform-content .snowflake-title-v2 .heading-5-v2,div.longform-content .snowflake-text h6,div.longform-content .snowflake-text .heading-6-v2,div.longform-content .snowflake-title-v2 .heading-6-v2{font-family:Lato,sans-serif !important;font-weight:800 !important}div.longform-content .snowflake-text h2,div.longform-content .snowflake-text .heading-2-v2,div.longform-content .snowflake-title-v2 .heading-2-v2{text-transform:none !important;font-size:28px !important}div.longform-content .snowflake-text h3,div.longform-content .snowflake-text .heading-3-v2,div.longform-content .snowflake-title-v2 .heading-3-v2{font-size:22px !important}div.longform-content .snowflake-text h4,div.longform-content .snowflake-text .heading-4-v2,div.longform-content .snowflake-title-v2 .heading-4-v2{font-size:18px !important}div.longform-content .snowflake-text h5,div.longform-content .snowflake-text .heading-5-v2,div.longform-content .snowflake-title-v2 .heading-5-v2{font-size:16px !important}div.longform-content .snowflake-text h6,div.longform-content .snowflake-text .heading-6-v2,div.longform-content .snowflake-title-v2 .heading-6-v2{font-size:14px !important}@media screen and (min-width:992px){div.longform-content .snowflake-text h2,div.longform-content .snowflake-text .heading-2-v2,div.longform-content .snowflake-title-v2 .heading-2-v2{font-size:38px !important}div.longform-content .snowflake-text h3,div.longform-content .snowflake-text .heading-3-v2,div.longform-content .snowflake-title-v2 .heading-3-v2{font-size:26px !important}div.longform-content .snowflake-text h4,div.longform-content .snowflake-text .heading-4-v2,div.longform-content .snowflake-title-v2 .heading-4-v2{font-size:22px !important}div.longform-content .snowflake-text h5,div.longform-content .snowflake-text .heading-5-v2,div.longform-content .snowflake-title-v2 .heading-5-v2{font-size:18px !important}div.longform-content .snowflake-text h6,div.longform-content .snowflake-text .heading-6-v2,div.longform-content .snowflake-title-v2 .heading-6-v2{font-size:16px !important}}.sticky-sidebar .page-toc li.is-active{font-weight:600;color:var(--snow-blue,#29b5e8)}.sticky-sidebar .page-toc li[data-anchor]{cursor:pointer}.longform-content table{margin-top:24px;margin-bottom:24px;width:100%;background-color:var(--ui-background-01);border-collapse:collapse;border:2px solid var(--ui-background-09);font-family:'Lato',sans-serif;color:var(--ui-background-09)}.longform-content table thead{background-color:var(--ui-01)}.longform-content th,.longform-content td{min-width:120px;border:2px solid var(--ui-background-09);padding:var(--spacing-01)}.longform-content ol{margin-top:0 !important}.longform-content ol li{margin-bottom:1rem !important}.longform-content ul li{margin:0;padding:0 0 0 32px;position:relative}.longform-content ul{list-style-type:none}.longform-content ul li::before{content:\"\";display:block;border-radius:100%;background:#29b5e8;width:18px;height:18px;position:absolute;top:4px;left:0;border:5px solid #e5f2f7;box-sizing:border-box}.seo-customer.snowflake-card-v2-advanced-horizontal .snowflake-card-v2-advanced-image-container{max-width:200px}.seo-customer.snowflake-card-v2-advanced-horizontal .snowflake-card-v2-advanced-image-container img{object-fit:contain}.related-topics-outer-container \u003E .container \u003E .cmp-container \u003E .aem-container{display:flex;flex-direction:row}.related-topics-outer-container \u003E .container \u003E .cmp-container \u003E .aem-container \u003E div{width:auto !important;margin:0 !important}.related-topics-outer-container \u003E .container \u003E .cmp-container \u003E .aem-container \u003E div:first-child{margin-right:16px !important;flex-shrink:0}","jsContent":"(function(){var OFFSET=100;if(window.gsap&&window.ScrollTrigger){gsap.registerPlugin(ScrollTrigger);var sidebar=document.querySelector('.sticky-sidebar');var body=document.querySelector('.longform-content');if(sidebar&&body){ScrollTrigger.create({trigger:sidebar,start:'top 100px',endTrigger:body,end:'bottom bottom',pin:sidebar,pinSpacing:false});}}document.addEventListener('click',function(e){var li=e.target.closest('li[data-anchor]');if(!li)return;var slug=li.getAttribute('data-anchor');var heading=document.querySelector('.anchor-title--'+CSS.escape(slug));if(!heading)return;e.preventDefault();var top=heading.getBoundingClientRect().top+window.pageYOffset-OFFSET;window.scrollTo({top:top,behavior:'smooth'});history.replaceState(null,'','#'+slug);},false);var headings=document.querySelectorAll('[class*=\"anchor-title--\"]');if(headings.length&&'IntersectionObserver'in window){var io=new IntersectionObserver(function(entries){entries.forEach(function(entry){if(!entry.isIntersecting)return;var cls=Array.from(entry.target.classList).find(function(c){return c.indexOf('anchor-title--')===0;});if(!cls)return;var slug=cls.replace('anchor-title--','');document.querySelectorAll('li[data-anchor]').forEach(function(li){li.classList.toggle('is-active',li.getAttribute('data-anchor')===slug);});});},{rootMargin:'-20% 0px -70% 0px',threshold:0});headings.forEach(function(h){io.observe(h);});}})();",":type":"snowflake-site/components/markup-editor","isGSAPEnabled":true},"experiencefragment-footer":{"id":"experiencefragment-d066904cf4","localizedFragmentVariationPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/footer/master/jcr:content","configured":true,":type":"snowflake-site/components/experiencefragment","xfModelPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/footer/master.xfmodel.json"},"experiencefragment":{"id":"experiencefragment-e546bd0e63","localizedFragmentVariationPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/footer-legal-disclaimers/master/jcr:content","configured":true,":type":"snowflake-site/components/experiencefragment","xfModelPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/footer-legal-disclaimers/master.xfmodel.json"}},":itemsOrder":["experiencefragment-banner","experiencefragment-header","responsivegrid","modal_container","markup_editor_928258845","markup_editor_597730182","markup_editor","experiencefragment-footer","experiencefragment"],":type":"wcm/foundation/components/responsivegrid"}},":itemsOrder":["root"],":hierarchyType":"page",":path":"/content/snowflake-site/global/en/artificial-intelligence/machine-learning/model-training/gradient-descent","isPasswordProtected":false,"analyticsContentTags":[],"analyticsEnabled":true,"coveoConfig":{"searchHub":"snowflake.com","organizationId":"snowflakecomputingproduction8neljofn","apiKey":"xx335921a6-2a0a-40f2-a167-e390b4766c3d","pipeline":"snowflake.com"},"analyticsDebugMode":false,"analyticsData":{"excludeFromAnalytics":false,"subCategory":"","pageType":"homepage","templateName":"base-page-template54","siteName":"snowflake","pageUrl":"/content/snowflake-site/global/en/artificial-intelligence/machine-learning/model-training/gradient-descent","language":"en","category":"general","pageName":"Gradient Descent: How Machine Learning Models Learn From Error","contentTags":[]},"locale":"en"}
  