{"cssClassNames":"blog-page page basicpage summit-page","canonicalLink":"https://www.snowflake.com/en/blog/engineering/semi-persistence-gpu-model-swapping/","robotsTags":["index","follow"],"templateName":"blog-page","description":"Discover how Semi-Persistence speeds up GPU model swapping by 5.6x to 19.9x over vLLM’s Level 2 sleep mode in Snowflake benchmarks. Sub-second swaps for single-GPU serving.","language":"en","title":"Semi-Persistence: Fast Model Swapping for vLLM & GPUs","allowedRenditionsWidth":["320","480","640","768","960","1200","1440","1920"],"analyticsPageType":"homepage","analyticsCategory":"general","analyticsSubCategory":"","excludeFromAnalytics":false,"analyticsDebugMode":false,"analyticsContentTags":["snowflake-site:taxonomy/blog/engineering-blog/machine-learning"],"analyticsEnabled":true,":mappedPath":"/en/blog/engineering/semi-persistence-gpu-model-swapping/",":type":"snowflake-site/components/structure/page",":items":{"root":{"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","columnClassNames":{"experiencefragment-banner":"aem-GridColumn aem-GridColumn--default--12","experiencefragment-sub-header":"aem-GridColumn aem-GridColumn--default--12","experiencefragment-pre-footer":"aem-GridColumn aem-GridColumn--default--12","experiencefragment-header":"aem-GridColumn aem-GridColumn--default--12","markup_editor-table":"aem-GridColumn aem-GridColumn--default--12","responsivegrid":"aem-GridColumn aem-GridColumn--default--12","experiencefragment-footer":"aem-GridColumn aem-GridColumn--default--12","markup_editor":"aem-GridColumn aem-GridColumn--default--12","container_47873732":"aem-GridColumn aem-GridColumn--default--12"},"columnCount":12,":items":{"experiencefragment-banner":{"id":"experiencefragment-f1359efc15","localizedFragmentVariationPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/pushdown-banner/pushdown-banner-blank/jcr:content","configured":true,":type":"snowflake-site/components/experiencefragment","xfModelPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/pushdown-banner/pushdown-banner-blank.xfmodel.json"},"experiencefragment-header":{"id":"experiencefragment-6601f7a24d","localizedFragmentVariationPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/mega-nav-header/master/jcr:content","configured":true,":type":"snowflake-site/components/experiencefragment","languageNavPath":"/content/snowflake-site/global/en/blog/engineering/semi-persistence-gpu-model-swapping.languagenav.json","xfModelPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/mega-nav-header/master.xfmodel.json","appliedCssClassNames":"snowflake-sticky-nav-host"},"experiencefragment-sub-header":{"id":"experiencefragment-b0d9e18cb9","localizedFragmentVariationPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/sub-navigation/engineering-blog-sub-nav/jcr:content","configured":true,":type":"snowflake-site/components/experiencefragment","xfModelPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/sub-navigation/engineering-blog-sub-nav.xfmodel.json"},"responsivegrid":{"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","columnClassNames":{"container_breadcrumb":"aem-GridColumn aem-GridColumn--default--12","container_main_content":"aem-GridColumn aem-GridColumn--default--12"},"columnCount":12,":items":{"container_breadcrumb":{"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","columnClassNames":{"breadcrumb":"aem-GridColumn aem-GridColumn--default--12"},"layout":"RESPONSIVE_GRID","columnCount":12,"id":"blog-page-breadcrumb-indentation",":type":"snowflake-site/components/container",":items":{"breadcrumb":{"id":"breadcrumb-d7d1e08f49","breadcrumbItems":[{"title":"Blog","path":"/en/blog/engineering/","active":false},{"title":"Machine Learning","path":"/en/blog/engineering/machine-learning/","active":false},{"title":"Semi-Persistence: Blazing-Fast Model Swapping for Complex Scheduling","path":"/en/blog/engineering/semi-persistence-gpu-model-swapping/","active":false}],":type":"snowflake-site/components/blog/breadcrumb"}},":itemsOrder":["breadcrumb"],"appliedCssClassNames":"snowflake-container"},"container_main_content":{"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","columnClassNames":{"flexible_column_container":"aem-GridColumn aem-GridColumn--default--12","related_content":"aem-GridColumn aem-GridColumn--default--12"},"layout":"RESPONSIVE_GRID","columnCount":12,"id":"main-content",":type":"snowflake-site/components/container",":items":{"flexible_column_container":{"id":"flexible-column-container-09a4d60eaf","propertiesId":"snowflake-blog-template-main-container","type":"2-column-60-40","alignColumns":"top","containerMaxWidth":"extra-large","topPadding":"none","bottomPadding":"none","spaceBetween":"none","reverseOnMobile":true,"carouselOnMobile":false,"backgroundImageOption":"none","flexible_column_content_container_1":{"layout":"SIMPLE","id":"container-39b666c444",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"container_hero":{"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","columnClassNames":{"blog_hero":"aem-GridColumn aem-GridColumn--default--12"},"layout":"RESPONSIVE_GRID","columnCount":12,"id":"container-4b29955cec",":type":"snowflake-site/components/container",":items":{"blog_hero":{"id":"blog-hero-e4acba8ab7","showClaude":true,"showChatGpt":true,"timeToRead":"14","publicationDate":"AUG 27, 2026","image":{"id":"image-45108f7a15","lazyEnabled":true,"src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--cc092a50-6db2-40a7-b106-c4713a0b5246/sf-eng-blog-ml-1.png?quality=85&preferwebp=true","isLcpImage":false,"height":"720","width":"1680",":type":"snowflake-site/components/image"},"tag":{"tagText":"Machine Learning","tagColor":"#29B5E8"},"title":{"lines":["Semi-Persistence: Blazing-Fast Model Swapping for Complex Scheduling"],"type":"heading2",":type":"snowflake-site/components/title-v2"},"authors":[{"authorImage":{"id":"image-d16adbec17","lazyEnabled":true,"src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--e33877c7-18fd-4041-babd-c3b1846ca4bc/mert-hidayetoglu.jpg?quality=85&preferwebp=true","isLcpImage":false,":type":"snowflake-site/components/image"},"authorCta":{"id":"button-9141513b2f","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/blog/authors/mert-hidayetoglu0/"},"linkTargetContentType":"DOCUMENT_LEARN","linkType":"SNOWFLAKE_INTERNAL",":type":"snowflake-site/components/button","text":"Mert Hidayetoglu"}},{"authorImage":{"id":"image-2cb4132415","alt":"Yuxiong He","lazyEnabled":true,"src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--91e118d5-5e68-4422-b71b-2a33781308d9/yuxiong-headshot.png?quality=85&preferwebp=true","isLcpImage":false,"height":"328","width":"216",":type":"snowflake-site/components/image"},"authorCta":{"id":"button-ee41f1a3b2","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/blog/authors/yuxiong-he/"},"linkTargetContentType":"DOCUMENT_LEARN","linkType":"SNOWFLAKE_INTERNAL",":type":"snowflake-site/components/button","text":"Yuxiong He"}},{"authorImage":{"id":"image-66669b5940","lazyEnabled":true,"src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--43554496-ead2-43f3-a9a0-b5c8117eaa68/jeffrasleyhead.png?quality=85&preferwebp=true","isLcpImage":false,"height":"512","width":"512",":type":"snowflake-site/components/image"},"authorCta":{"id":"button-6293acdd0d","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/blog/authors/jeff-rasley/"},"linkTargetContentType":"DOCUMENT_LEARN","linkType":"SNOWFLAKE_INTERNAL",":type":"snowflake-site/components/button","text":"Jeff Rasley"}},{"authorImage":{"id":"image-7f9b405acc","alt":"Samyam Rajbhandari","lazyEnabled":true,"src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--746d3344-2e70-4178-b5cc-01726d6b5dc5/samyam-headshot.png?quality=85&preferwebp=true","isLcpImage":false,"height":"326","width":"245",":type":"snowflake-site/components/image"},"authorCta":{"id":"button-3095dd2a31","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/blog/authors/samyam-rajbhandari/"},"linkTargetContentType":"DOCUMENT_LEARN","linkType":"SNOWFLAKE_INTERNAL",":type":"snowflake-site/components/button","text":"Samyam Rajbhandari"}},{"authorImage":{"id":"image-b49fda74bf","lazyEnabled":true,"src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--99e42d00-f962-453d-81c1-0b51cd5e6815/yanlin-du.png?quality=85&preferwebp=true","isLcpImage":false,"height":"282","width":"286",":type":"snowflake-site/components/image"},"authorCta":{"id":"button-5ed1017ede","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/blog/authors/yanlin-du/"},"linkTargetContentType":"DOCUMENT_LEARN","linkType":"SNOWFLAKE_INTERNAL",":type":"snowflake-site/components/button","text":"Yanlin Du"}}],":type":"snowflake-site/components/blog/blog-hero"}},":itemsOrder":["blog_hero"]},"responsivegrid_content":{"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","columnClassNames":{"image_261254347":"aem-GridColumn aem-GridColumn--default--12","wistia_video":"aem-GridColumn aem-GridColumn--default--12","wistia_video_1699350611":"aem-GridColumn aem-GridColumn--default--12","blog_text_461129563":"aem-GridColumn aem-GridColumn--default--12","blog_text_1921441906":"aem-GridColumn aem-GridColumn--default--12","blog_text_1296122968":"aem-GridColumn aem-GridColumn--default--12","blog_text_1812033667":"aem-GridColumn aem-GridColumn--default--12","blog_text_511212293":"aem-GridColumn aem-GridColumn--default--12","blog_text_1598519303":"aem-GridColumn aem-GridColumn--default--12","image_1199967082":"aem-GridColumn aem-GridColumn--default--12","blog_text_1453361308":"aem-GridColumn aem-GridColumn--default--12","blog_spacer_1203754731":"aem-GridColumn aem-GridColumn--default--12","blog_text":"aem-GridColumn aem-GridColumn--default--12","blog_spacer_232704412":"aem-GridColumn aem-GridColumn--default--12","blog_text_832219046":"aem-GridColumn aem-GridColumn--default--12","blog_text_550937975":"aem-GridColumn aem-GridColumn--default--12","blog_text_9944376":"aem-GridColumn aem-GridColumn--default--12","blog_text_2096807707":"aem-GridColumn aem-GridColumn--default--12","blog_spacer_1072704398":"aem-GridColumn aem-GridColumn--default--12","image_1501883167":"aem-GridColumn aem-GridColumn--default--12","blog_text_895422757":"aem-GridColumn aem-GridColumn--default--12","image":"aem-GridColumn aem-GridColumn--default--12","blog_text_1710010257":"aem-GridColumn aem-GridColumn--default--12","image_2089512201":"aem-GridColumn aem-GridColumn--default--12","blog_text_577774234":"aem-GridColumn aem-GridColumn--default--12","blog_text_696046921":"aem-GridColumn aem-GridColumn--default--12","blog_text_908945214":"aem-GridColumn aem-GridColumn--default--12","blog_text_1090329809":"aem-GridColumn aem-GridColumn--default--12","code_snippet_1193783962":"aem-GridColumn aem-GridColumn--default--12","blog_spacer_389189345":"aem-GridColumn aem-GridColumn--default--12","blog_text_1476376714":"aem-GridColumn aem-GridColumn--default--12","image_708995999":"aem-GridColumn aem-GridColumn--default--12","image_1275475633":"aem-GridColumn aem-GridColumn--default--12","blog_text_947923851":"aem-GridColumn aem-GridColumn--default--12","blog_text_1779129571":"aem-GridColumn aem-GridColumn--default--12","blog_text_2067859004":"aem-GridColumn aem-GridColumn--default--12","image_2035384667":"aem-GridColumn aem-GridColumn--default--12","code_snippet":"aem-GridColumn aem-GridColumn--default--12","blog_text_1661123118":"aem-GridColumn aem-GridColumn--default--12","blog_spacer_239078418":"aem-GridColumn aem-GridColumn--default--12","blog_text_2138780205":"aem-GridColumn aem-GridColumn--default--12","blog_spacer":"aem-GridColumn aem-GridColumn--default--12","blog_text_161603577":"aem-GridColumn aem-GridColumn--default--12","image_934802533":"aem-GridColumn aem-GridColumn--default--12","blog_text_1066533398":"aem-GridColumn aem-GridColumn--default--12","code_snippet_2125324529":"aem-GridColumn aem-GridColumn--default--12","blog_text_1777416854":"aem-GridColumn aem-GridColumn--default--12"},"columnCount":12,"appliedCssClassNames":"snowflake-layout-container-inner-padding-small",":items":{"blog_text":{"id":"blog-text-14fb55d913","text":"\u003Cp\u003EOpen-source models are becoming increasingly compelling on both quality and cost. Specialization can push these economics even further: smaller models trained for focused tasks can match or exceed frontier-model quality at a fraction of the inference cost. Snowflake's \u003Ca href=\"https://www.snowflake.com/en/blog/engineering/enterprise-text-to-sql-arctic-r2/\" target=\"_blank\" rel=\"noopener noreferrer\"\u003EArctic Text-to-SQL models\u003C/a\u003E, for example, demonstrate frontier-level quality at about 25× lower inference cost per token than a closed-source frontier API.\u003Csup\u003E1\u003C/sup\u003E\u003C/p\u003E\r\n\u003Cp\u003ESnowflake AI Research is also working to reduce the cost and complexity of post-training open source models. This includes open source RL backends such as \u003Ca href=\"https://www.snowflake.com/en/blog/engineering/zorro-enterprise-rl-training/\" target=\"_blank\" rel=\"noopener noreferrer\"\u003EZoRRo\u003C/a\u003E and \u003Ca href=\"https://www.snowflake.com/en/blog/engineering/arctic-rl-open-source-backend/\" target=\"_blank\" rel=\"noopener noreferrer\"\u003EArctic RL\u003C/a\u003E, as well as Snowflake-managed services such as \u003Ca href=\"https://www.snowflake.com/en/blog/snowflake-for-ai-enterprise-ai-platform/\" target=\"_blank\" rel=\"noopener noreferrer\"\u003ESnowflake Cortex Training\u003C/a\u003E.\u003C/p\u003E\r\n\u003Cp\u003EServing presents a different system problem. A serving platform may need to host many models whose demand changes over time. Keeping every model permanently loaded wastes expensive GPU capacity; moving models in and out of GPU memory is useful only if the swap itself is fast enough to stay out of the critical path.\u003C/p\u003E\r\n\u003Cp\u003E\u003Cb\u003ESemi-Persistence makes model swapping practical.\u003C/b\u003E It keeps model weights in a persistent CPU-memory pool while allowing their GPU copies to come and go with demand. The lightweight model skeleton moves independently across disk, CPU, and GPU, while weights can be restored rapidly from CPU memory when the model is needed again.\u003C/p\u003E\r\n\u003Cp\u003EAcross models from 2B to 397B parameters, Snowflake internal benchmarks show that Semi-Persistence reduces the end-to-end sleep and wake-up cycle by \u003Cb\u003E5.6× to 19.9×\u003C/b\u003E, achieving sub-second model swapping for single-GPU models (Figure 1).\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"blog_spacer_389189345":{"id":"blog-spacer-5360a8f9b5",":type":"snowflake-site/components/blog/blog-spacer"},"image":{"id":"image-5c3d9bb8d3","alt":"Figure 1.","lazyEnabled":true,"src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--1a012e76-005f-4277-90eb-5da9d9e6d4fc/figure-1.-semi-persistence--blazing-fast-model-swapping-for-complex-scheduling.png?quality=85&preferwebp=true","isLcpImage":true,"height":"728","width":"1994",":type":"snowflake-site/components/image"},"blog_text_1476376714":{"id":"blog-text-7f8c2e46c3","text":"\u003Cp\u003E\u003Ci\u003EFigure 1: End-to-end sleep and wake-up latency for vLLM and Semi-Persistence. Semi-Persistence persists the weights in CPU memory and optimizes their movement back to the GPUs.\u003C/i\u003E\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"blog_text_2067859004":{"id":"blog-text-47b06d6178","text":"\u003Ch2\u003EWhy existing model swapping falls short\u003C/h2\u003E\r\n\u003Cp\u003EModel swapping lets multiple models time-share the same GPUs: when a model becomes idle, its GPU memory can be reclaimed and used by another model; when demand returns, the model is restored and resumes serving. This makes it possible to increase GPU utilization without dedicating capacity to every model at all times.\u003C/p\u003E\r\n\u003Cp\u003EBut model swapping only works as a serving primitive if the transition is fast. A slow eviction delays reuse of the GPU, while a slow restore adds directly to the time between request arrival and token generation. In practice, the swap path needs to avoid repeatedly moving or reloading the model's full weights whenever its GPU residency changes.\u003C/p\u003E\r\n\u003Cp\u003EThe standard way to swap models in vLLM is through its two \u003Ca rel=\"noopener noreferrer\" target=\"_blank\" href=\"https://docs.vllm.ai/en/latest/features/sleep_mode/\"\u003Esleep modes\u003C/a\u003E, but each pays one of these expensive paths.\u003C/p\u003E\r\n\u003Cp\u003E\u003Cb\u003ELevel 1\u003C/b\u003E copies model weights from GPU memory to CPU memory before releasing the GPU. This supports a faster wake-up, but every sleep requires a full GPU-to-CPU transfer.\u003C/p\u003E\r\n\u003Cp\u003E\u003Cb\u003ELevel 2\u003C/b\u003E discards the GPU weights immediately, making sleep faster, but the weights must be loaded again from storage before the model can serve requests. Either way, every sleep/wake cycle pays a slow path — exactly what the swap budget cannot afford.\u003C/p\u003E\r\n\u003Cp\u003EInference weights often do not change while a model is running. Instead of copying them back from the GPU (as in Level 1) or repeatedly reading them from storage (as in Level 2), we can keep one long-lived CPU copy and treat the GPU copy as temporary.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"blog_text_511212293":{"id":"blog-text-e922564e3a","text":"\u003Ch2\u003ESemi-Persistence\u003C/h2\u003E\r\n\u003Cp\u003ESemi-Persistence is a model-serving approach that makes model swapping fast — and manages many models across shared GPUs — by keeping model weights ready in CPU memory and moving models in and out of GPU memory as demand changes.\u003C/p\u003E\r\n\u003Cp\u003E\u003Cb\u003EKeep weights ready in CPU memory.\u003C/b\u003E Semi-Persistence keeps model weights in a persistent pinned CPU-memory pool. When a model becomes idle, its GPU copy can be discarded immediately because the weights remain available in CPU memory. The lightweight model skeleton, meanwhile, moves between disk, CPU, and GPU independently of the weights.\u003C/p\u003E\r\n\u003Cp\u003E\u003Cb\u003ERestore them quickly.\u003C/b\u003E When requests arrive, Semi-Persistence transfers the weights back to the GPUs. It shards the weights across the node, loads them through multiple devices in parallel and uses both PCIe and NVLink to maximize transfer bandwidth.\u003C/p\u003E\r\n\u003Cp\u003E\u003Cb\u003EAdapt as demand changes.\u003C/b\u003E Semi-Persistence also manages the models sharing the GPUs. It can load and evict models, migrate them between GPUs and nodes, consolidate smaller models to free capacity, and pause and resume requests during a move.\u003C/p\u003E\r\n\u003Cp\u003EThese operations happen asynchronously, allowing model movement and request processing to overlap. We describe how each of these works in the Deep dive below.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"blog_text_2096807707":{"id":"blog-text-ce9ab35e91","text":"\u003Ch2\u003EPerformance\u003C/h2\u003E\r\n\u003Cp\u003EFigure 1 shows benchmarks of the end-to-end sleep and wake-up cycle — the time to release a model's GPU memory plus the time to make it ready to serve again — across eight open-weight models, comparing vLLM's Level 2 sleep mode against Semi-Persistence under otherwise identical configuration. For single-GPU models, latency falls from \u003Cb\u003E1.2–13.5 seconds to 214–801 milliseconds\u003C/b\u003E. For multi-GPU models, it falls from \u003Cb\u003E10.5–40.6 seconds to 1.75–7 seconds\u003C/b\u003E. Overall, Semi-Persistence delivers a \u003Cb\u003E5.6× to 19.9× speedup\u003C/b\u003E across models ranging from 2B to 397B parameters.\u003Csup\u003E2\u003C/sup\u003E\u003C/p\u003E\r\n\u003Cp\u003EThe advantage grows for the largest frontier models. For trillion-parameter models such as DeepSeek-v4-pro and GLM-5.2-FP8, a cold start in vLLM takes 13.3–15.5 minutes, whereas Semi-Persistence restores them in 32.8 seconds (Figure 8, detailed under Frontier models) — roughly \u003Cb\u003E24× to 28× faster\u003C/b\u003E.\u003C/p\u003E\r\n\u003Ch2\u003EWhat are we open-sourcing today?\u003C/h2\u003E\r\n\u003Cp\u003EWe implemented Semi-Persistence around the vLLM backend that we use today, but the same instance-based design applies to other engines. We expose the semi-persistent capabilities to an instance as composable primitives. Anyone can use these primitives directly to add Semi-Persistence to their own serving stack or implement a new backend behind the same interface.\u003C/p\u003E\r\n\u003Cp\u003EOn top of the Semi-Persistence \u003Cb\u003Ewrapper\u003C/b\u003E, we are also open-sourcing our experimental complex-scheduling system: an \u003Cb\u003Eorchestrator\u003C/b\u003E that drives many decentralized instances dynamically, and a \u003Cb\u003Edashboard\u003C/b\u003E for observing the live state of the system. Everything is available at \u003Ca href=\"https://github.com/snowflakedb/ArcticInference/tree/main/arctic_inference/semi_persistence\" target=\"_blank\" rel=\"noopener noreferrer\"\u003EArcticInference/semi_persistence\u003C/a\u003E.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"blog_text_1598519303":{"id":"blog-text-f3a705b0ad","text":"\u003Ch2\u003EDeep dive\u003C/h2\u003E\r\n\u003Cp\u003ESemi-Persistence is built around four ideas. First, it separates model skeletons from weights, allowing many specialized models to share the same runtime. Second, it restores weights at very high speed by saturating multiple interconnects in the communication hierarchy at the same time. Third, it orchestrates models asynchronously, enabling many instances to load, sleep, evict and migrate concurrently without a centralized bottleneck. Finally, it turns these primitives into complex, self-healing scheduling patterns that keep GPUs highly utilized as workloads change. Together, these techniques make dynamic model swapping practical for production serving.\u003C/p\u003E\r\n\u003Cp\u003EThe Deep dive is organized around these four highlights:\u003C/p\u003E\r\n\u003Col\u003E\r\n\u003Cli\u003E\u003Cb\u003ESeparating skeletons from weights:\u003C/b\u003E the reusable model skeleton is cached once while only the weights change per specialization, so many models share a single runtime.\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EFast weight restoration:\u003C/b\u003E weights are streamed from a pinned CPU pool and assembled on the GPUs in parallel over PCIe and NVLink, restoring a model at full node bandwidth.\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EAsynchronous orchestration:\u003C/b\u003E each instance runs as an independent state machine coordinated by a lightweight reservation system, overlapping data movement without a central scheduler.\u003C/li\u003E\r\n\u003Cli\u003E\u003Cb\u003EComplex scheduling:\u003C/b\u003E a few simple per-instance rules give rise to self-healing behaviors — eviction, migration and consolidation — that keep GPUs busy as traffic shifts.\u003C/li\u003E\r\n\u003C/ol\u003E\r\n\u003Ch3\u003ESeparating skeletons from weights\u003C/h3\u003E\r\n\u003Cp\u003ESemi-Persistence manages the skeleton and weights of a model independently. The skeleton contains the model structure and runtime state — typically a few gigabytes — while the weights define a particular specialization.\u003C/p\u003E\r\n\u003Cp\u003EMany specialized models share the same skeleton but differ only in their weights. Semi-Persistence caches each skeleton once and stores all weights in a persistent pinned CPU-memory pool. Because the weights are stored in vLLM's native format, any compatible skeleton can restore immediately without rebuilding or converting the model. The first time a configuration is seen, Semi-Persistence saves the skeleton image to disk and later restores it with CRIU (Checkpoint/Restore In Userspace), so a single runtime can efficiently serve many specialized models.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"image_2035384667":{"id":"image-672b4ccfc0","alt":"Figure 2.","lazyEnabled":true,"src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--2ed23fd1-e841-476a-9b07-f3af59bc2e0f/figure-2.-semi-persistence--blazing-fast-model-swapping-for-complex-scheduling.png?quality=85&preferwebp=true","isLcpImage":false,"height":"748","width":"1446",":type":"snowflake-site/components/image"},"blog_text_161603577":{"id":"blog-text-01e102a288","text":"\u003Cp\u003E\u003Ci\u003EFigure 2: Semi-Persistence tracks each instance's state (Saved, Checkpoint, Sleep, Up) and manages weights and skeleton independently. Weights live in the pinned memory pool; the skeleton moves between disk image, CPU and GPU via CRIU and CUDA checkpoint/restore.\u003C/i\u003E\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"blog_text_908945214":{"id":"blog-text-078cb68a67","text":"\u003Cp\u003EFigure 2 defines the states an instance's skeleton moves through in the hardware hierarchy. \u003Cb\u003ESaved\u003C/b\u003E keeps the instance image ready on disk; \u003Cb\u003ECheckpoint\u003C/b\u003E holds the model entirely on CPU with no GPU context; \u003Cb\u003ESleep\u003C/b\u003E keeps the skeleton on the GPU but consumes no compute and negligible GPU memory; and \u003Cb\u003EUp\u003C/b\u003E is fully deployed, serving generation with no added latency. We use different sequences of optimized primitives, shown in the figure, to move instances across the hierarchy.\u003C/p\u003E\r\n\u003Ch3\u003EFast weight restoration\u003C/h3\u003E\r\n\u003Cp\u003EPersistent weights are only useful if they can be restored at GPU speed. The main bottleneck in weight restoration is the PCIe path between CPU and GPU. To saturate every switch, Semi-Persistence uses a pinned memory pool that (1) shards the weights across the pool, (2) transfers them in parallel with multiple GPUs over distinct PCIe switches, and (3) gathers them over NVLink. As a result, we use the aggregate bandwidth of the node (full PCIe and NVLink) rather than a single PCIe-only path. Figure 3 shows the weight restore for TP=1 and TP=2 models.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"image_261254347":{"id":"image-56d994a3f5","alt":"Figure 3.","lazyEnabled":true,"src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--2d3d32e8-dcad-4afc-ab02-c8d2ccb1e789/figure-3.-semi-persistence--blazing-fast-model-swapping-for-complex-scheduling.png?quality=85&preferwebp=true","isLcpImage":false,"height":"1236","width":"2048",":type":"snowflake-site/components/image"},"blog_text_2138780205":{"id":"blog-text-e8ef2aadf8","text":"\u003Cp\u003E\u003Ci\u003EFigure 3: Parallel weight restoration from pinned memory. Weights are sharded across the node, streamed from the pinned pool into 8 GB staging buffers over PCIe (H2D) and assembled on the target GPUs over NVLink (D2D).\u003C/i\u003E\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"blog_spacer":{"id":"blog-spacer-65f52f6caf",":type":"snowflake-site/components/blog/blog-spacer"},"blog_text_1090329809":{"id":"blog-text-6e18ef6945","text":"\u003Cp\u003EThe memory pool has a set of long-lived daemon processes that perform the host-to-device (H2D) transfers simultaneously, while the instance initiates the device-to-device (D2D) communication that gathers the shards. If instance and pool processes are not co-located, the shards are transferred using inter-process communication (IPC) over NVLink. We pipeline the hops in Figure 3 for overlapping H2D and D2D communications, keeping PCIe saturated throughout the restore.\u003C/p\u003E\r\n\u003Cp\u003EOn our test platform (an AWS p5en.48xlarge with 192 vCPUs, 2 TiB of memory and 8 H200 GPUs), PCIe bandwidth is highly sensitive to GPU placement: when multiple GPUs share a PCIe switch, bandwidth drops sharply (Figure 4). Even with the best PCIe-only placement, adding NVLink to gather the shards improves restore throughput by about \u003Cb\u003E3.7× (TP1), 2× (TP2) and 1.8× (TP4)\u003C/b\u003E.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"image_934802533":{"id":"image-78f7f5eb6b","alt":"Figure 4.","lazyEnabled":true,"src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--f1ca57f7-50d8-44b6-baa6-d2d8faa592ec/figure-4.-semi-persistence--blazing-fast-model-swapping-for-complex-scheduling.png?quality=85&preferwebp=true","isLcpImage":false,"height":"760","width":"1360",":type":"snowflake-site/components/image"},"blog_text_550937975":{"id":"blog-text-c02a1d3bd4","text":"\u003Cp\u003E\u003Cb\u003EFigure 4A. Restore over PCIe\u003C/b\u003E \u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"image_1501883167":{"id":"image-5040eb57ec","alt":"Figure 5.","lazyEnabled":true,"src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--d862386d-e01b-4d2c-9487-5991f66133a5/figure-5.-semi-persistence--blazing-fast-model-swapping-for-complex-scheduling.png?quality=85&preferwebp=true","isLcpImage":false,"height":"700","width":"700",":type":"snowflake-site/components/image"},"blog_text_1921441906":{"id":"blog-text-8bcf35a3f5","text":"\u003Cp\u003E\u003Cb\u003EFigure 4B. Restore over PCIe + NVLink\u003C/b\u003E\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"image_708995999":{"id":"image-666006de77","alt":"Figure 6.","lazyEnabled":true,"src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--18d1be84-ef93-4c38-8366-f2fb4b122c07/figure-6.-semi-persistence--blazing-fast-model-swapping-for-complex-scheduling.png?quality=85&preferwebp=true","isLcpImage":false,"height":"700","width":"700",":type":"snowflake-site/components/image"},"blog_text_832219046":{"id":"blog-text-9b015c579d","text":"\u003Cp\u003E\u003Ci\u003EFigure 4: GPU placement and the NUMA/PCIe topology (top). Bandwidth suffers when GPUs share a switch (left); routing weights through staging GPUs and NVLink recovers throughput (right).\u003C/i\u003E\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"blog_text_9944376":{"id":"blog-text-c28278928e","text":"\u003Ch3\u003EAsynchronous orchestration\u003C/h3\u003E\r\n\u003Cp\u003EFast restores alone are not enough. A production serving system must continuously load, evict, migrate and pause models as demand changes, and controlling every instance and interaction from a centralized scheduler causes significant software overhead. Therefore, Semi-Persistence treats each instance as an asynchronous state machine. Instances independently decide their own state transitions when a request arrives, which decentralizes scheduling and lets operations proceed concurrently, naturally overlapping disk I/O, CPU-GPU transfers and execution.\u003C/p\u003E\r\n\u003Cp\u003ETo coordinate these independent decisions safely, Semi-Persistence uses a global reservation system that reserves a GPU \u003Ci\u003Eslot\u003C/i\u003E the moment a request arrives — with low latency. Slots logically partition GPU memory: a quarter, half, or full GPU for small models, up to multiple GPUs for larger ones (Figure 5). A slot does not actually allocate memory on the hardware but blocks other instances from the same memory region; an instance that cannot obtain a slot of its size waits in the Checkpoint state, and a freed slot is handed to the next instance in the queue. This lets instances make independent moves without ever running out of memory.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"blog_spacer_1203754731":{"id":"blog-spacer-229ec88108",":type":"snowflake-site/components/blog/blog-spacer"},"image_1199967082":{"id":"image-18e473480e","alt":"Figure 7.","lazyEnabled":true,"src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--68ac3cb5-ac60-4201-9b1f-dea9d8e6c1a7/figure-7.-semi-persistence--blazing-fast-model-swapping-for-complex-scheduling.png?quality=85&preferwebp=true","isLcpImage":false,"height":"1141","width":"2048",":type":"snowflake-site/components/image"},"blog_text_1779129571":{"id":"blog-text-fe111e2812","text":"\u003Cp\u003E\u003Ci\u003EFigure 5: Orchestrator dashboard (schematic). Each GPU is divided into quarter-, half-, or full-GPU slots — shown as stars in brackets, for example, GPU 3 [**|*|*] — so several models can share one GPU (here, three models on GPU 3). The dashboard also surfaces GPU memory, instance state (up vs. waiting), the CPU pinned pool, the live request log and the image cache.\u003C/i\u003E\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"blog_text_1777416854":{"id":"blog-text-efcc774d8a","text":"\u003Ch3\u003EComplex scheduling\u003C/h3\u003E\r\n\u003Cp\u003EThe most powerful behaviors in Semi-Persistence are not programmed centrally — they emerge. From a few simple per-instance rules, complex scheduling behaviors such as eviction, migration, consolidation and waiting arise on their own, and the system effectively self-heals fragmentation with no central brain.\u003C/p\u003E\r\n\u003Cp\u003EConsider a concrete example of head-of-line (HOL) blocking (Figure 6). Small models run on GPU 1 and GPU 3, and a large model runs on GPU 2. Two requests then arrive back-to-back: one for a large model and one for a small model. In a naive FIFO queue, the large model head-of-line-blocks the smaller one — even though enough capacity for the small model already exists — so both wait here for 7.6 seconds.\u003C/p\u003E\r\n\u003Cp\u003ESemi-Persistence self-heals this situation by migrating and consolidating the small models onto GPU 1, freeing a contiguous slot for the large model and dropping the wait time to zero. Migration is driven by two primitives: \u003Cb\u003Epause\u003C/b\u003E saves the in-flight token IDs to CPU and releases the instance's GPU memory (making it evictable) and \u003Cb\u003Eresume\u003C/b\u003E automatically moves the instance to its new GPU, re-prefills the KV cache and continues generation exactly where it left off — no additional commands required.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"blog_spacer_1072704398":{"id":"blog-spacer-23f7c26a65",":type":"snowflake-site/components/blog/blog-spacer"},"blog_text_1661123118":{"id":"blog-text-89cc4b9f07","text":"\u003Cp\u003E\u003Cb\u003EFigure 6A. With HOL blocking: 7.6 s wait\u003C/b\u003E\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"wistia_video":{"id":"wistia-video-ba29a1383d","title":{"id":"title","type":"heading1","lines":["Semi-Persistence: Blazing-Fast Model Swapping for Complex Scheduling"],":type":"snowflake-site/components/title"},"wistiaVideoId":"iqmx7uzy29","showVideoLength":true,"showRewatchButton":true,":type":"snowflake-site/components/wistia-video"},"blog_text_895422757":{"id":"blog-text-32c50a55ec","text":"\u003Cp\u003E\u003Cb\u003E\u003Cb\u003EFigure 6B. Without HOL blocking: 0 s wait\u003C/b\u003E\u003C/b\u003E\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"wistia_video_1699350611":{"id":"wistia-video-315efc507e","title":{"id":"title","type":"heading1","lines":["Semi-Persistence: Blazing-Fast Model Swapping for Complex Scheduling"],":type":"snowflake-site/components/title"},"wistiaVideoId":"ffw3riotuy","showVideoLength":true,"showRewatchButton":true,":type":"snowflake-site/components/wistia-video"},"blog_text_461129563":{"id":"blog-text-bbbc6154d5","text":"\u003Cp\u003E\u003Ci\u003EFigure 6: Self-healing scheduling. Migrating and consolidating small models frees a contiguous slot for the waiting large model, eliminating head-of-line blocking (7.6 s → 0 s).\u003C/i\u003E\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"blog_spacer_232704412":{"id":"blog-spacer-9d8f828a6c",":type":"snowflake-site/components/blog/blog-spacer"},"blog_text_1710010257":{"id":"blog-text-1df5699a00","text":"\u003Cp\u003EMigration involves more than moving weights. The skeleton is &quot;sticky&quot; to GPUs: it owns CUDA runtime state and distributed communication infrastructure, and it must leave no residual CUDA context behind when checkpointed. Semi-Persistence uses the CUDA checkpoint driver to capture the context — only the lightweight skeleton, so it stays fast — and CUDA restore to remap GPU addresses on the destination. For multi-GPU instances, it tears down vLLM's static communication infrastructure (NCCL, IPC buffers) and reinitializes it on the new GPUs; because reinitializing NCCL invalidates the captured CUDA graph, the graph is recaptured. As a result, migration latency is ultimately bounded by CUDA checkpoint and restore (a driver operation), as seen in Figure 7.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"blog_spacer_239078418":{"id":"blog-spacer-9b1584d27c",":type":"snowflake-site/components/blog/blog-spacer"},"image_2089512201":{"id":"image-2a90d6087e","alt":"Figure 10.","lazyEnabled":true,"src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--5d364089-aff6-445f-9cd6-81c8ddec4510/figure-10.-semi-persistence--blazing-fast-model-swapping-for-complex-scheduling.png?quality=85&preferwebp=true","isLcpImage":false,"height":"947","width":"2048",":type":"snowflake-site/components/image"},"blog_text_947923851":{"id":"blog-text-0b61cc582d","text":"\u003Cp\u003E\u003Ci\u003EFigure 7: Migration latency breakdown. After optimization, cost is dominated by CUDA checkpoint/restore rather than restoring weights.\u003C/i\u003E\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"blog_text_1296122968":{"id":"blog-text-cc5c10fc7a","text":"\u003Ch2\u003EFrontier models\u003C/h2\u003E\r\n\u003Cp\u003ESemi-Persistence is not limited to small specialized models — it applies equally to the largest open source frontier models, such as DeepSeek, GLM, and Kimi. Teams can post-train and specialize these trillion-parameter models for specific tasks, then serve them at roughly 5×–10× lower inference cost per token than closed-source frontier APIs.\u003Csup\u003E3\u003C/sup\u003E Because such models occupy a full node, fast swapping matters even more: without it, simply bringing a model up can take minutes.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"image_1275475633":{"id":"image-9d3ed446e0","alt":"Figure 11.","lazyEnabled":true,"src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--afb6ff0f-2abe-47e8-b5c2-d3b16fb78d9a/figure-11.-semi-persistence--blazing-fast-model-swapping-for-complex-scheduling.png?quality=85&preferwebp=true","isLcpImage":false,"height":"671","width":"2048",":type":"snowflake-site/components/image"},"blog_text_1812033667":{"id":"blog-text-954e5c4c07","text":"\u003Cp\u003E\u003Ci\u003EFigure 8: (left) Large-model sleep/wake cycles and (right) vLLM cold-start vs. Semi-Persistence load-from-image times.\u003C/i\u003E\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"blog_text_1453361308":{"id":"blog-text-b0ce7aa740","text":"\u003Cp\u003EFigure 8 makes the difference concrete. A cold start in vLLM initializes each model from a local cache and still takes several minutes — about 15.5 minutes for DeepSeek-v4-pro and 6.26 minutes for Kimi-K2.6. Semi-Persistence instead restores the same models directly from the image cache in roughly 33 seconds, more than an order of magnitude faster, so even the largest models can be swapped in and out on demand.\u003Csup\u003E4\u003C/sup\u003E\u003C/p\u003E\r\n\u003Ch2\u003EGet started\u003C/h2\u003E\r\n\u003Cp\u003EGetting started takes only a few steps:\u003C/p\u003E\r\n\u003Col\u003E\r\n\u003Cli\u003EInstall \u003Ca href=\"https://github.com/vllm-project/vllm\" target=\"_blank\" rel=\"noopener noreferrer\"\u003EvLLM\u003C/a\u003E.\u003C/li\u003E\r\n\u003Cli\u003EInstall \u003Ca href=\"https://github.com/snowflakedb/ArcticInference\" target=\"_blank\" rel=\"noopener noreferrer\"\u003EArcticInference\u003C/a\u003E\u003C/li\u003E\r\n\u003Cli\u003EExplore the examples, then run the scripts below to reproduce our results.\u003C/li\u003E\r\n\u003C/ol\u003E\r\n\u003Cp\u003E\u003Cb\u003ERun the server\u003C/b\u003E\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"code_snippet":{"id":"code-snippet-7c9e7bf2e8","codeSnippet":"sudo python semi_persistence/server.py --gpu 1,2,3 --image-cache /code/image-cache","multiLine":true,":type":"snowflake-site/components/code-snippet"},"blog_text_1066533398":{"id":"blog-text-52903c09fd","text":"\u003Cp\u003E\u003Cb\u003ERun the client\u003C/b\u003E\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"code_snippet_2125324529":{"id":"code-snippet-e69c4c0777","language":"python","codeSnippet":"from client import OrchestratorClient as cl cl.init(\"demo_hol.jsonl\") cl.generate(\"32b bird\", \"who are you?\", 2000) cl.generate(\"model 8\", \"what is the capital of France?\", 3000) cl.generate(\"model 10\", \"What is the meaning of life?\", 6000) cl.pause(\"model 8\") cl.generate(\"spec 30b\", \"who are you?\", 2000) cl.generate(\"spec 8b\", \"who are you?\", 3000) cl.resume(\"model 8\") cl.wait()","multiLine":true,":type":"snowflake-site/components/code-snippet"},"blog_text_577774234":{"id":"blog-text-f6828edd33","text":"\u003Cp\u003E\u003Cb\u003EMonitor with the dashboard\u003C/b\u003E\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"},"code_snippet_1193783962":{"id":"code-snippet-2d61a297e8","language":"python","codeSnippet":"python semi_persistence/dashboard.py --interval 0.1","multiLine":true,":type":"snowflake-site/components/code-snippet"},"blog_text_696046921":{"id":"blog-text-4523179985","text":"\u003Cp\u003EThat's it. Semi-Persistence is available at \u003Ca href=\"https://github.com/snowflakedb/ArcticInference/tree/main/arctic_inference/semi_persistence\" target=\"_blank\" rel=\"noopener noreferrer\"\u003Ehttps://github.com/snowflakedb/ArcticInference/tree/main/arctic_inference/semi_persistence\u003C/a\u003E — reproduce our results, or serve your own specialized models on shared GPUs.\u003C/p\u003E\r\n\u003Col\u003E\r\n\u003Cli id=\"fn1\"\u003EBased on Opus 4.7 API ($6.82 per million tokens) vs. our cost of running Arctic-Text2SQL-R2 32B on H200 ($0.27 per million tokens), assuming 20K input + 2K output tokens per request.\u003C/li\u003E\r\n\u003Cli id=\"fn2\"\u003EThe Figure 1 results were measured on AWS p5en.48xlarge instances (192 vCPUs, 2 TiB of host memory and 8 NVIDIA H200 GPUs) using vLLM v0.18.0.\u003C/li\u003E\r\n\u003Cli id=\"fn3\"\u003EBased on publicly listed per-token prices on OpenRouter as of August 2026. GLM-5.2 is listed at $1.19 per million input tokens and $3.74 per million output tokens, compared with $10.00 / $50.00 for Claude Fable 5 and $5.00 / $25.00 for Claude Opus 4.7. At a 3:1 input-to-output token mix that works out to about $1.83 per million tokens for GLM-5.2 versus $20.00 for Claude Fable 5 — roughly 10× lower — and roughly 5× lower than Claude Opus 4.7. Listed prices for open-weight models vary by provider and change frequently. Sources: \u003Ca href=\"https://openrouter.ai/z-ai/glm-5.2#providers\" target=\"_blank\" rel=\"noopener noreferrer\"\u003EGLM-5.2\u003C/a\u003E, \u003Ca href=\"https://openrouter.ai/anthropic/claude-fable-5\" target=\"_blank\" rel=\"noopener noreferrer\"\u003EClaude Fable 5\u003C/a\u003E, \u003Ca href=\"https://openrouter.ai/anthropic/claude-opus-4.7\" target=\"_blank\" rel=\"noopener noreferrer\"\u003EClaude Opus 4.7\u003C/a\u003E, \u003Ca href=\"https://openrouter.ai/qwen/qwen3-32b\" target=\"_blank\" rel=\"noopener noreferrer\"\u003EQwen3-32B\u003C/a\u003E.\u003C/li\u003E\r\n\u003Cli id=\"fn4\"\u003EExperiments with Frontier models (Figure 8) were made on 8×B300 nodes using vLLM v0.24.0. All the other experiments were made on the 8×H200 nodes.\u003C/li\u003E\r\n\u003C/ol\u003E\r\n","richText":true,":type":"snowflake-site/components/blog/blog-text"}},":itemsOrder":["blog_text","blog_spacer_389189345","image","blog_text_1476376714","blog_text_2067859004","blog_text_511212293","blog_text_2096807707","blog_text_1598519303","image_2035384667","blog_text_161603577","blog_text_908945214","image_261254347","blog_text_2138780205","blog_spacer","blog_text_1090329809","image_934802533","blog_text_550937975","image_1501883167","blog_text_1921441906","image_708995999","blog_text_832219046","blog_text_9944376","blog_spacer_1203754731","image_1199967082","blog_text_1779129571","blog_text_1777416854","blog_spacer_1072704398","blog_text_1661123118","wistia_video","blog_text_895422757","wistia_video_1699350611","blog_text_461129563","blog_spacer_232704412","blog_text_1710010257","blog_spacer_239078418","image_2089512201","blog_text_947923851","blog_text_1296122968","image_1275475633","blog_text_1812033667","blog_text_1453361308","code_snippet","blog_text_1066533398","code_snippet_2125324529","blog_text_577774234","code_snippet_1193783962","blog_text_696046921"],":type":"wcm/foundation/components/responsivegrid"},"responsivegrid_premium_content_banner":{"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","columnClassNames":{},"columnCount":12,"appliedCssClassNames":"snowflake-responsive-component-top-padding-medium",":items":{},":itemsOrder":[],":type":"wcm/foundation/components/responsivegrid"},"container_author_chip":{"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","columnClassNames":{"author_chip":"aem-GridColumn aem-GridColumn--default--12"},"layout":"RESPONSIVE_GRID","columnCount":12,"id":"container-f38c991ace",":type":"snowflake-site/components/container",":items":{"author_chip":{"id":"author-chip-826492c174","title":{"id":"title","type":"heading2","lines":["Learn more about the authors"],":type":"snowflake-site/components/title-v2"},"authors":[{"authorImage":{"id":"image-d16adbec17","lazyEnabled":true,"src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--e33877c7-18fd-4041-babd-c3b1846ca4bc/mert-hidayetoglu.jpg?quality=85&preferwebp=true","isLcpImage":false,":type":"snowflake-site/components/image"},"authorCta":{"id":"button-9141513b2f","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/blog/authors/mert-hidayetoglu0/"},"linkTargetContentType":"DOCUMENT_LEARN","linkType":"SNOWFLAKE_INTERNAL",":type":"snowflake-site/components/button","text":"Mert Hidayetoglu"},"authorTitle":"Software Engineer"},{"authorImage":{"id":"image-2cb4132415","alt":"Yuxiong He","lazyEnabled":true,"src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--91e118d5-5e68-4422-b71b-2a33781308d9/yuxiong-headshot.png?quality=85&preferwebp=true","isLcpImage":false,"height":"328","width":"216",":type":"snowflake-site/components/image"},"authorCta":{"id":"button-ee41f1a3b2","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/blog/authors/yuxiong-he/"},"linkTargetContentType":"DOCUMENT_LEARN","linkType":"SNOWFLAKE_INTERNAL",":type":"snowflake-site/components/button","text":"Yuxiong He"},"authorTitle":"Sr. Director, Software Engineering"},{"authorImage":{"id":"image-66669b5940","lazyEnabled":true,"src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--43554496-ead2-43f3-a9a0-b5c8117eaa68/jeffrasleyhead.png?quality=85&preferwebp=true","isLcpImage":false,"height":"512","width":"512",":type":"snowflake-site/components/image"},"authorCta":{"id":"button-6293acdd0d","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/blog/authors/jeff-rasley/"},"linkTargetContentType":"DOCUMENT_LEARN","linkType":"SNOWFLAKE_INTERNAL",":type":"snowflake-site/components/button","text":"Jeff Rasley"},"authorTitle":"Senior Software Engineer"},{"authorImage":{"id":"image-7f9b405acc","alt":"Samyam Rajbhandari","lazyEnabled":true,"src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--746d3344-2e70-4178-b5cc-01726d6b5dc5/samyam-headshot.png?quality=85&preferwebp=true","isLcpImage":false,"height":"326","width":"245",":type":"snowflake-site/components/image"},"authorCta":{"id":"button-3095dd2a31","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/blog/authors/samyam-rajbhandari/"},"linkTargetContentType":"DOCUMENT_LEARN","linkType":"SNOWFLAKE_INTERNAL",":type":"snowflake-site/components/button","text":"Samyam Rajbhandari"},"authorTitle":"Principal AI Architect, Snowflake"},{"authorImage":{"id":"image-b49fda74bf","lazyEnabled":true,"src":"https://www.snowflake.com/adobe/dynamicmedia/deliver/dm-aid--99e42d00-f962-453d-81c1-0b51cd5e6815/yanlin-du.png?quality=85&preferwebp=true","isLcpImage":false,"height":"282","width":"286",":type":"snowflake-site/components/image"},"authorCta":{"id":"button-5ed1017ede","showOutboundIcon":false,"buttonLink":{"valid":true,"url":"/en/blog/authors/yanlin-du/"},"linkTargetContentType":"DOCUMENT_LEARN","linkType":"SNOWFLAKE_INTERNAL",":type":"snowflake-site/components/button","text":"Yanlin Du"},"authorTitle":"Research Intern"}],":type":"snowflake-site/components/blog/author-chip"}},":itemsOrder":["author_chip"],"appliedCssClassNames":"snowflake-responsive-component-top-padding-medium"},"container_share_article":{"gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","columnClassNames":{"share_article":"aem-GridColumn aem-GridColumn--default--12"},"layout":"RESPONSIVE_GRID","columnCount":12,"id":"container-f2b4abca3a",":type":"snowflake-site/components/container",":items":{"share_article":{"id":"share-article-a8a5c0977b","linkedInShareUrl":"https://www.linkedin.com/shareArticle?mini=true&url=https%3A%2F%2Fwww.snowflake.com%2Fcontent%2Fsnowflake-site%2Fglobal%2Fen%2Fblog%2Fengineering%2Fsemi-persistence-gpu-model-swapping&title=Semi-Persistence%3A+Blazing-Fast+Model+Swapping+for+Complex+Scheduling","twitterShareUrl":"https://x.com/intent/post?url=https%3A%2F%2Fwww.snowflake.com%2Fcontent%2Fsnowflake-site%2Fglobal%2Fen%2Fblog%2Fengineering%2Fsemi-persistence-gpu-model-swapping&text=Semi-Persistence%3A+Blazing-Fast+Model+Swapping+for+Complex+Scheduling","facebookShareUrl":"https://www.facebook.com/sharer/sharer.php?u=https%3A%2F%2Fwww.snowflake.com%2Fcontent%2Fsnowflake-site%2Fglobal%2Fen%2Fblog%2Fengineering%2Fsemi-persistence-gpu-model-swapping",":type":"snowflake-site/components/blog/share-article"}},":itemsOrder":["share_article"],"appliedCssClassNames":"snowflake-responsive-component-top-padding-small"}},":itemsOrder":["container_hero","responsivegrid_content","responsivegrid_premium_content_banner","container_author_chip","container_share_article"]},"flexible_column_content_container_2":{"layout":"SIMPLE","id":"container-19650485f1",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"blog_table_of_content":{"id":"blog-table-of-content-ee7a9f3e3e",":type":"snowflake-site/components/blog/blog-table-of-content","tableOfContents":[]}},":itemsOrder":["blog_table_of_content"]},":type":"snowflake-site/components/flexible-column-container","isBlogPage":true,"isActiveTOC":false},"related_content":{"id":"related-content-8dc19aaf28","relatedContent":[],":type":"snowflake-site/components/blog/related-content","isBlogPage":true}},":itemsOrder":["flexible_column_container","related_content"],"appliedCssClassNames":"snowflake-container"}},":itemsOrder":["container_breadcrumb","container_main_content"],":type":"wcm/foundation/components/responsivegrid"},"container_47873732":{"additionalClasses":"section--blog-newsletter","gridClassNames":"aem-Grid aem-Grid--12 aem-Grid--default--12","columnClassNames":{"flexible_column_cont":"aem-GridColumn aem-GridColumn--default--12"},"layout":"RESPONSIVE_GRID","columnCount":12,"id":"container-49f6bca3d3",":type":"snowflake-site/components/container",":items":{"flexible_column_cont":{"id":"flexible-column-container-5cb0936988","type":"1-column","alignColumns":"top","containerMaxWidth":"extra-large","topPadding":"small","bottomPadding":"none","spaceBetween":"small","reverseOnMobile":false,"carouselOnMobile":false,"propertiesCSSClasses":"section--blog-newsletter","backgroundImageOption":"none","flexible_column_content_container_1":{"layout":"SIMPLE","id":"container-c1b9dc2911",":type":"snowflake-site/components/flexible-column-container/flexible-column-content-container",":items":{"marketo_v2":{"id":"marketo-v2-07d773fe1d","marketoForm":{"formId":"3320","edit":false,"successUrl":null,"hidden":null,"script":null,"values":null},"title":{"id":"title","type":"heading3","lines":["Subscribe to our blog newsletter","Get the best, coolest and latest delivered to your inbox each week"],":type":"snowflake-site/components/title-v2"},"munchkinId":"252-RFO-227","serverInstance":"252-RFO-227.mktoweb.com","marketoConfigured":true,"formConfigured":true,"partnerCampaignPage":false,":type":"snowflake-site/components/form/marketo-v2"},"text":{"id":"text-509efbdc3f","additionalClasses":"newsletter-disclaimer","text":"\u003Cp\u003EBy submitting this form, I understand Snowflake will process my personal information in accordance with their Privacy Notice.\u003C/p\u003E\r\n","richText":true,":type":"snowflake-site/components/text"}},":itemsOrder":["marketo_v2","text"]},":type":"snowflake-site/components/flexible-column-container","isBlogPage":true,"isActiveTOC":false}},":itemsOrder":["flexible_column_cont"],"appliedCssClassNames":"snowflake-container"},"experiencefragment-pre-footer":{"id":"experiencefragment-3f103f4311","localizedFragmentVariationPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/get-started-pre-footer/get-started-pre-footer/jcr:content","configured":true,":type":"snowflake-site/components/experiencefragment","xfModelPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/get-started-pre-footer/get-started-pre-footer.xfmodel.json"},"markup_editor":{"id":"markup-editor-db8ed04c59","title":"Page CSS","cssContent":"@media screen and (min-width:768px){.snowflake-blog-author-chip-wrapper{justify-content:flex-start}.snowflake-blog-related-content-on-blog-page{max-width:1408px;margin-left:auto;margin-right:auto}.snowflake-text{font-family:Lato,sans-serif;font-weight:400;font-size:16px;line-height:24px}}.section--blog-newsletter{max-width:none;width:100%;padding-left:0;padding-right:0;margin-left:0;margin-right:0;margin-bottom:0}.section--blog-newsletter .mktoField{background-color:transparent !important}.section--blog-newsletter\u003E.container{padding-left:0;padding-right:0}@media screen and (min-width:768px){.section--blog-newsletter\u003E.container{padding-left:0;padding-right:0}}.newsletter-disclaimer p{font-size:14px !important}.section--blog-newsletter .snowflake-marketo-form-container{margin-bottom:24px;background-color:#f6f9fa;gap:48px;box-shadow:none}.section--blog-newsletter .snowflake-title p.snowflake-title-line:first-child{font-family:Texta;font-size:24px;line-height:26px;font-weight:700;margin-bottom:4px}.section--blog-newsletter .snowflake-title p.snowflake-title-line{text-transform:none;font-family:\"Lato\",sans-serif;font-size:16px;line-height:24px;font-weight:normal}@media screen and (min-width:1024px){.section--blog-newsletter .snowflake-marketo-form-container{display:flex;justify-content:center}.section--blog-newsletter .snowflake-title .snowflake-title-line{text-align:left}.section--blog-newsletter .snowflake-marketo-form .mktoFormRow:has(\u003E input[type=\"hidden\"]){flex-grow:0}.section--blog-newsletter .snowflake-marketo-form{display:flex;width:50% !important}.section--blog-newsletter .snowflake-marketo-form .mktoButtonRow{flex-grow:0;width:auto !important;margin-left:0;margin-right:0}.section--blog-newsletter .snowflake-marketo-form .mktoFormRow{flex-grow:1}.section--blog-newsletter\u003E.container{padding-left:0;padding-right:0}.section--blog-newsletter .snowflake-marketo-form-title{width:50%;margin-bottom:0 !important}.section--blog-newsletter .center .snowflake-title{align-items:flex-start}}.snowflake-sub-navigation a.snowflake-sub-navigation-primary-link{width:auto !important}.snowflake-blog-hero{align-items:stretch !important}",":type":"snowflake-site/components/markup-editor","isGSAPEnabled":false},"markup_editor-table":{"id":"markup-editor-bd80c058fd","title":"Table Styling CSS","cssContent":"#snowflake-blog-template-main-container table{width:100%;background-color:var(--ui-background-01);border-collapse:collapse;border:2px solid var(--ui-background-09);font-family:'Lato',sans-serif;color:var(--ui-background-09)}#snowflake-blog-template-main-container table thead{background-color:var(--ui-01)}#snowflake-blog-template-main-container table th,#snowflake-blog-template-main-container table td{border:2px solid var(--ui-background-09);padding:var(--spacing-01)}",":type":"snowflake-site/components/markup-editor","isGSAPEnabled":false},"experiencefragment-footer":{"id":"experiencefragment-97169dbf35","localizedFragmentVariationPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/footer/master/jcr:content","configured":true,":type":"snowflake-site/components/experiencefragment","xfModelPath":"/content/experience-fragments/snowflake-site/language-masters/en/site/footer/master.xfmodel.json"}},":itemsOrder":["experiencefragment-banner","experiencefragment-header","experiencefragment-sub-header","responsivegrid","container_47873732","experiencefragment-pre-footer","markup_editor","markup_editor-table","experiencefragment-footer"],":type":"wcm/foundation/components/responsivegrid"}},":itemsOrder":["root"],":hierarchyType":"page",":path":"/content/snowflake-site/global/en/blog/engineering/semi-persistence-gpu-model-swapping","analyticsData":{"excludeFromAnalytics":false,"subCategory":"","pageType":"homepage","templateName":"blog-page","siteName":"snowflake","pageUrl":"/content/snowflake-site/global/en/blog/engineering/semi-persistence-gpu-model-swapping","language":"en","category":"general","pageName":"Semi-Persistence: Blazing-Fast Model Swapping for Complex Scheduling","contentTags":["snowflake-site:taxonomy/blog/engineering-blog/machine-learning"]},"isPasswordProtected":false,"locale":"en","coveoConfig":{"pipeline":"snowflake.com","searchHub":"snowflake.com","organizationId":"snowflakecomputingproduction8neljofn","apiKey":"xx335921a6-2a0a-40f2-a167-e390b4766c3d"}}
  