Buckets:

download
raw
39.8 kB
<meta charset="utf-8" /><meta name="hf:doc:metadata" content="{&quot;title&quot;:&quot;Caching&quot;,&quot;local&quot;:&quot;caching&quot;,&quot;sections&quot;:[{&quot;title&quot;:&quot;Choose a cache method&quot;,&quot;local&quot;:&quot;choose-a-cache-method&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;Pyramid Attention Broadcast&quot;,&quot;local&quot;:&quot;pyramid-attention-broadcast&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;FasterCache&quot;,&quot;local&quot;:&quot;fastercache&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;SeaCache&quot;,&quot;local&quot;:&quot;seacache&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;FirstBlockCache&quot;,&quot;local&quot;:&quot;firstblockcache&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;TaylorSeer Cache&quot;,&quot;local&quot;:&quot;taylorseer-cache&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;MagCache&quot;,&quot;local&quot;:&quot;magcache&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;Text KV Cache&quot;,&quot;local&quot;:&quot;text-kv-cache&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2}],&quot;depth&quot;:1}"/>
<link href="/docs/diffusers/pr_14867/en/_app/immutable/entry/start.BY1X0FnC.js" rel="modulepreload">
<link href="/docs/diffusers/pr_14867/en/_app/immutable/chunks/BDChHepi.js" rel="modulepreload">
<link href="/docs/diffusers/pr_14867/en/_app/immutable/chunks/DK803DsY.js" rel="modulepreload">
<link href="/docs/diffusers/pr_14867/en/_app/immutable/entry/app.BcsduW5Q.js" rel="modulepreload">
<link href="/docs/diffusers/pr_14867/en/_app/immutable/chunks/DTwaC60R.js" rel="modulepreload">
<link href="/docs/diffusers/pr_14867/en/_app/immutable/chunks/BTASUwav.js" rel="modulepreload">
<link href="/docs/diffusers/pr_14867/en/_app/immutable/chunks/WKU8S240.js" rel="modulepreload">
<link href="/docs/diffusers/pr_14867/en/_app/immutable/chunks/DsnmJJEf.js" rel="modulepreload">
<link href="/docs/diffusers/pr_14867/en/_app/immutable/nodes/0.ByoTqlNi.js" rel="modulepreload">
<link href="/docs/diffusers/pr_14867/en/_app/immutable/chunks/BzOvRKAw.js" rel="modulepreload">
<link href="/docs/diffusers/pr_14867/en/_app/immutable/nodes/282.D8uks8VT.js" rel="modulepreload">
<link href="/docs/diffusers/pr_14867/en/_app/immutable/chunks/Bqo1LwON.js" rel="modulepreload">
<!--fs68f1--><meta name="hf:doc:metadata" content="{&quot;title&quot;:&quot;Caching&quot;,&quot;local&quot;:&quot;caching&quot;,&quot;sections&quot;:[{&quot;title&quot;:&quot;Choose a cache method&quot;,&quot;local&quot;:&quot;choose-a-cache-method&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;Pyramid Attention Broadcast&quot;,&quot;local&quot;:&quot;pyramid-attention-broadcast&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;FasterCache&quot;,&quot;local&quot;:&quot;fastercache&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;SeaCache&quot;,&quot;local&quot;:&quot;seacache&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;FirstBlockCache&quot;,&quot;local&quot;:&quot;firstblockcache&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;TaylorSeer Cache&quot;,&quot;local&quot;:&quot;taylorseer-cache&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;MagCache&quot;,&quot;local&quot;:&quot;magcache&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;Text KV Cache&quot;,&quot;local&quot;:&quot;text-kv-cache&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2}],&quot;depth&quot;:1}"/><!---->
<link href="/docs/diffusers/pr_14867/en/_app/immutable/assets/0.tn0RQdqM.css" rel="modulepreload"> <!--[--><!--[0--><!--[--><!--[0--><!--[--><p></p> <div class="items-center shrink-0 min-w-[100px] max-sm:min-w-[50px] justify-end ml-auto flex" style="float: right; margin-left: 10px; display: inline-flex; position: relative; z-index: 10;"><div class="inline-flex rounded-md max-sm:rounded-sm"><button class="inline-flex items-center gap-1 h-7 max-sm:h-7 px-2 max-sm:px-1.5 text-sm font-medium text-gray-800 border border-r-0 rounded-l-md max-sm:rounded-l-sm border-gray-200 bg-white hover:shadow-inner dark:border-gray-850 dark:bg-gray-950 dark:text-gray-200 dark:hover:bg-gray-800" aria-live="polite"><span class="inline-flex items-center justify-center rounded-md p-0.5 max-sm:p-0 hover:text-gray-800 dark:hover:text-gray-200"><svg class="sm:size-3.5 size-3" xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----></span> <span>Copy page</span></button> <button class="inline-flex items-center justify-center w-6 max-sm:w-5 h-7 max-sm:h-7 disabled:pointer-events-none text-sm text-gray-500 hover:text-gray-700 dark:hover:text-white rounded-r-md max-sm:rounded-r-sm border border-l transition border-gray-200 bg-white hover:shadow-inner dark:border-gray-850 dark:bg-gray-950 dark:text-gray-200 dark:hover:bg-gray-800" aria-haspopup="menu" aria-expanded="false" aria-label="Open copy menu"><svg class="transition-transform text-gray-400 overflow-visible sm:size-3.5 size-3 rotate-0" width="1em" height="1em" viewBox="0 0 12 7" fill="none" xmlns="http://www.w3.org/2000/svg"><path d="M1 1L6 6L11 1" stroke="currentColor"></path></svg><!----></button></div> <!--[-1--><!--]--></div><!----> <!--[0--><h1 class="relative group"><a id="caching" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#caching"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Caching</span></h1><!--]--><!----> <p>Caching reuses intermediate layer outputs across denoising steps to speed up inference. It uses more memory and doesn’t need training. Enable a method on the transformer with a config.</p> <!--[1--><h2 class="relative group"><a id="choose-a-cache-method" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#choose-a-cache-method"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Choose a cache method</span></h2><!--]--><!----> <p>Pick a method depending on how much config you will set, and the fit you need.</p> <table><thead><tr><th>Method</th><th>Use when</th><th>Tradeoff</th></tr></thead><tbody><tr><td>Text KV Cache</td><td>NucleusMoE image only, need exact text K/V reuse across steps</td><td>Lossless</td></tr><tr><td>SeaCache</td><td>Video transformers that already have a SeaCache path</td><td>Approximate, settings often do not transfer across models</td></tr><tr><td>FirstBlockCache</td><td>Want one main speed/quality knob on a registered transformer</td><td>Approximate</td></tr><tr><td>MagCache</td><td>Have magnitude ratios for your checkpoint and scheduler, or will calibrate first</td><td>Approximate, ratios are checkpoint and scheduler specific</td></tr><tr><td>TaylorSeer</td><td>Want to predict later activations from earlier steps</td><td>Approximate</td></tr><tr><td>PAB</td><td>Video, willing to tune attention reuse (block and timestep skip ranges per attention kind)</td><td>Approximate</td></tr><tr><td>FasterCache</td><td>Like PAB, plus optional CFG-branch skipping</td><td>Approximate, experimental</td></tr></tbody></table> <!--[1--><h2 class="relative group"><a id="pyramid-attention-broadcast" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#pyramid-attention-broadcast"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Pyramid Attention Broadcast</span></h2><!--]--><!----> <p><a href="https://huggingface.co/papers/2408.12588" rel="nofollow">Pyramid Attention Broadcast (PAB)</a> approximates attention across denoising steps by reusing attention outputs for some blocks and timesteps instead of recomputing every step. Config separates attention kinds (spatial, temporal, cross) when the model has them. Not every video model exposes all three, and set only the ranges that match the blocks you have.</p> <p>Each kind uses a <code>*_attention_block_skip_range</code> (how often to recompute vs reuse within the window) and a <code>*_attention_timestep_skip_range</code> (which denoising timesteps may skip). You must pass <code>current_timestep_callback</code> so the hook can read the pipeline’s current timestep. Wider or more aggressive skips usually mean more speed and more quality risk.</p> <p>Pass a <a href="/docs/diffusers/pr_14867/en/api/cache#diffusers.PyramidAttentionBroadcastConfig">PyramidAttentionBroadcastConfig</a> to enable it.</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-python "><!----><span class="hljs-keyword">import</span> torch
<span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> CogVideoXPipeline, PyramidAttentionBroadcastConfig
pipe = CogVideoXPipeline.from_pretrained(<span class="hljs-string">&quot;THUDM/CogVideoX-5b&quot;</span>, dtype=torch.bfloat16)
pipe.to(<span class="hljs-string">&quot;cuda&quot;</span>) <span class="hljs-comment"># or &quot;mps&quot;, &quot;xpu&quot;, &quot;cpu&quot;</span>
config = PyramidAttentionBroadcastConfig(
spatial_attention_block_skip_range=<span class="hljs-number">2</span>,
spatial_attention_timestep_skip_range=(<span class="hljs-number">100</span>, <span class="hljs-number">800</span>),
current_timestep_callback=<span class="hljs-keyword">lambda</span>: pipe.current_timestep,
)
pipe.transformer.enable_cache(config)<!----></pre></div><!----> <!--[1--><h2 class="relative group"><a id="fastercache" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#fastercache"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>FasterCache</span></h2><!--]--><!----> <p><a href="https://huggingface.co/papers/2410.19355" rel="nofollow">FasterCache</a> caches and reuses attention features similar to <a href="#pyramid-attention-broadcast">PAB</a>. It can also skip the unconditional branch under classifier-free guidance and estimate it from the conditional branch when successive latents are redundant enough.</p> <p>Pass a <a href="/docs/diffusers/pr_14867/en/api/cache#diffusers.FasterCacheConfig">FasterCacheConfig</a> to enable it. Like PAB, set <code>*_attention_block_skip_range</code> and <code>*_attention_timestep_skip_range</code> for the attention kinds you have, plus the CFG-branch skip options when you want them.</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-python "><!----><span class="hljs-keyword">import</span> torch
<span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> CogVideoXPipeline, FasterCacheConfig
pipe = CogVideoXPipeline.from_pretrained(<span class="hljs-string">&quot;THUDM/CogVideoX-5b&quot;</span>, dtype=torch.bfloat16)
pipe.to(<span class="hljs-string">&quot;cuda&quot;</span>) <span class="hljs-comment"># or &quot;mps&quot;, &quot;xpu&quot;, &quot;cpu&quot;</span>
config = FasterCacheConfig(
spatial_attention_block_skip_range=<span class="hljs-number">2</span>,
spatial_attention_timestep_skip_range=(-<span class="hljs-number">1</span>, <span class="hljs-number">681</span>),
current_timestep_callback=<span class="hljs-keyword">lambda</span>: pipe.current_timestep,
attention_weight_callback=<span class="hljs-keyword">lambda</span> _: <span class="hljs-number">0.3</span>,
unconditional_batch_skip_range=<span class="hljs-number">5</span>,
unconditional_batch_timestep_skip_range=(-<span class="hljs-number">1</span>, <span class="hljs-number">641</span>),
tensor_format=<span class="hljs-string">&quot;BFCHW&quot;</span>,
)
pipe.transformer.enable_cache(config)<!----></pre></div><!----> <!--[1--><h2 class="relative group"><a id="seacache" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#seacache"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>SeaCache</span></h2><!--]--><!----> <p><a href="https://huggingface.co/papers/2602.18993" rel="nofollow">SeaCache</a> compares Spectral Evolution Aware (SEA) indicators between successive denoising steps. When the accumulated change stays under a threshold, it skips the transformer block stack and predicts the output from cached residuals. The method is approximate and designed for video generation.</p> <p>Built-in adapters for SeaCache include:</p> <ul><li>Cosmos 3 is the primary optimized and benchmarked integration.</li> <li>Wan T2V uses the generic repeated-block path as a demo for how to provide the raw vision latents to SeaCache. The same cache parameters may not transfer to Wan or other Wan variants.</li></ul> <p>Enable SeaCache on the transformer. The Cosmos 3 denoising loop attaches scheduler step, sigma, and step count to each <code>cache_context</code>, so no extra parameters are needed.</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-python "><!----><span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> Cosmos3OmniPipeline, SeaCacheConfig
pipe = Cosmos3OmniPipeline.from_pretrained(<span class="hljs-string">&quot;nvidia/Cosmos3-Nano&quot;</span>)
pipe.transformer.enable_cache(SeaCacheConfig(threshold=<span class="hljs-number">0.2</span>, max_consecutive_cached=<span class="hljs-number">2</span>))<!----></pre></div><!----> <p>SeaCache may change outputs. Call <code>pipe.transformer.disable_cache()</code> when you need every step to run the full transformer. The same enable call works with <a href="/docs/diffusers/pr_14867/en/api/pipelines/cosmos3#diffusers.Cosmos3OmniPipeline">Cosmos3OmniPipeline</a>, <a href="/docs/diffusers/pr_14867/en/api/pipelines/cosmos3#diffusers.Cosmos3OmniModularPipeline">Cosmos3OmniModularPipeline</a>, and <a href="/docs/diffusers/pr_14867/en/api/pipelines/cosmos3#diffusers.Cosmos3DistilledModularPipeline">Cosmos3DistilledModularPipeline</a>.</p> <p>To integrate another video transformer, use <code>CacheMixin</code>, register the block layout in <code>TransformerBlockRegistry</code>, enter a <code>cache_context</code> on every call with <code>step_index</code>, <code>sigma</code>, and <code>num_inference_steps</code>, and pass a <code>raw_vision_callback</code> when no built-in adapter exists. Tune parameters per model and scheduler.</p> <!--[1--><h2 class="relative group"><a id="firstblockcache" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#firstblockcache"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>FirstBlockCache</span></h2><!--]--><!----> <p><a href="/docs/diffusers/pr_14867/en/api/cache#diffusers.FirstBlockCacheConfig">FirstBlockCacheConfig</a> checks how much the early layers of the denoiser change from one timestep to the next. If the change is small, the model skips the expensive later layers and reuses the previous output.</p> <p>Enable it through <code>enable_cache</code> so <code>disable_cache</code> and <code>is_cache_enabled</code> stay in sync. The default <code>threshold</code> is <code>0.05</code>. A higher value such as <code>0.2</code> skips more often for extra speed, but generation quality may drop.</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-python "><!----><span class="hljs-keyword">import</span> torch
<span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> DiffusionPipeline, FirstBlockCacheConfig
pipe = DiffusionPipeline.from_pretrained(
<span class="hljs-string">&quot;Qwen/Qwen-Image&quot;</span>, dtype=torch.bfloat16
)
pipe.transformer.enable_cache(FirstBlockCacheConfig(threshold=<span class="hljs-number">0.2</span>))<!----></pre></div><!----> <!--[1--><h2 class="relative group"><a id="taylorseer-cache" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#taylorseer-cache"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>TaylorSeer Cache</span></h2><!--]--><!----> <p><a href="https://huggingface.co/papers/2503.06923" rel="nofollow">TaylorSeer Cache</a> accelerates diffusion inference with Taylor series expansions across denoising steps. It predicts later-step activations from earlier ones and reuses those predictions for several steps so the transformer does less full work.</p> <ul><li><code>cache_interval</code>: Number of steps to reuse cached outputs before performing a full forward pass</li> <li><code>disable_cache_before_step</code>: Initial steps that use full computations to gather data for approximations</li> <li><code>max_order</code>: Higher Taylor orders can be more accurate but use more memory. Keep this at <code>1</code> unless you have a reason to change it.</li></ul> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-python "><!----><span class="hljs-keyword">import</span> torch
<span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> FluxPipeline, TaylorSeerCacheConfig
pipe = FluxPipeline.from_pretrained(
<span class="hljs-string">&quot;black-forest-labs/FLUX.1-dev&quot;</span>,
dtype=torch.bfloat16,
).to(<span class="hljs-string">&quot;cuda&quot;</span>) <span class="hljs-comment"># or &quot;mps&quot;, &quot;xpu&quot;, &quot;cpu&quot;</span>
config = TaylorSeerCacheConfig(
cache_interval=<span class="hljs-number">5</span>,
max_order=<span class="hljs-number">1</span>,
disable_cache_before_step=<span class="hljs-number">10</span>,
taylor_factors_dtype=torch.bfloat16,
)
pipe.transformer.enable_cache(config)<!----></pre></div><!----> <!--[1--><h2 class="relative group"><a id="magcache" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#magcache"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>MagCache</span></h2><!--]--><!----> <p><a href="https://github.com/Zehong-Ma/MagCache" rel="nofollow">MagCache</a> skips transformer blocks from the residual update magnitude. Update magnitudes decay predictably over denoising, and MagCache tracks an error budget from precomputed magnitude ratios (<code>mag_ratios</code>) to decide when reuse is safe. Those ratios are checkpoint and scheduler-specific. Ratios from a high step count can be interpolated down to fewer steps.</p> <p>MagCache follows two steps:</p> <ol><li>Calibration: Run inference once with <code>calibrate=True</code>. The hook measures residual magnitudes and prints the calculated ratios.</li> <li>Inference: Disable the calibration cache, then pass those ratios to <code>MagCacheConfig</code> for acceleration.</li></ol> <p>Classifier-free guidance may affect calibration. Pipelines that use true CFG with sequential contexts, such as Flux when <code>true_cfg_scale > 1</code>, enter <code>cache_context("cond")</code> and <code>cache_context("uncond")</code> separately. Calibration may print one array per context, but you should use the conditional array in most cases. Pipelines that batch CFG by concatenating conditional and unconditional inputs (for example, CogVideoX) produce a single joint array you can use directly.</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-python "><!----><span class="hljs-keyword">import</span> torch
<span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> FluxPipeline, MagCacheConfig
<span class="hljs-keyword">from</span> diffusers.hooks.mag_cache <span class="hljs-keyword">import</span> FLUX_MAG_RATIOS
pipe = FluxPipeline.from_pretrained(
<span class="hljs-string">&quot;black-forest-labs/FLUX.1-schnell&quot;</span>,
dtype=torch.bfloat16
).to(<span class="hljs-string">&quot;cuda&quot;</span>) <span class="hljs-comment"># or &quot;mps&quot;, &quot;xpu&quot;, &quot;cpu&quot;</span>
<span class="hljs-comment"># 1. Calibration Step</span>
<span class="hljs-comment"># Run full inference to measure model behavior.</span>
calib_config = MagCacheConfig(calibrate=<span class="hljs-literal">True</span>, num_inference_steps=<span class="hljs-number">4</span>)
pipe.transformer.enable_cache(calib_config)
<span class="hljs-comment"># Run a prompt to trigger calibration</span>
pipe(<span class="hljs-string">&quot;A cat playing chess&quot;</span>, num_inference_steps=<span class="hljs-number">4</span>)
<span class="hljs-comment"># Prints: [MagCache] Calibration Complete. Copy these values to MagCacheConfig(mag_ratios=...):</span>
<span class="hljs-comment"># 2. Inference Step</span>
<span class="hljs-comment"># Disable calibration hooks before enabling MagCache for inference.</span>
pipe.transformer.disable_cache()
<span class="hljs-comment"># Apply ratios from calibration, or use the Flux defaults:</span>
<span class="hljs-comment"># mag_ratios=FLUX_MAG_RATIOS</span>
mag_config = MagCacheConfig(
mag_ratios=[<span class="hljs-number">1.0</span>, <span class="hljs-number">1.37</span>, <span class="hljs-number">0.97</span>, <span class="hljs-number">0.87</span>],
num_inference_steps=<span class="hljs-number">4</span>
)
pipe.transformer.enable_cache(mag_config)
image = pipe(<span class="hljs-string">&quot;A cat playing chess&quot;</span>, num_inference_steps=<span class="hljs-number">4</span>).images[<span class="hljs-number">0</span>]<!----></pre></div><!----> <!--[1--><h2 class="relative group"><a id="text-kv-cache" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#text-kv-cache"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Text KV Cache</span></h2><!--]--><!----> <p><a href="/docs/diffusers/pr_14867/en/api/cache#diffusers.TextKVCacheConfig">TextKVCacheConfig</a> enables exact (lossless) reuse of text key and value projections across denoising steps. It is for NucleusMoE image only (<code>NucleusMoEImageTransformerBlock</code>, the architecture <a href="/docs/diffusers/pr_14867/en/api/cache#diffusers.apply_text_kv_cache">apply_text_kv_cache()</a> hooks). Enable it with <code>enable_cache</code>.</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-python "><!----><span class="hljs-keyword">import</span> torch
<span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> NucleusMoEImagePipeline, TextKVCacheConfig
pipe = NucleusMoEImagePipeline.from_pretrained(
<span class="hljs-string">&quot;NucleusAI/NucleusMoE-Image&quot;</span>, dtype=torch.bfloat16
)
pipe.to(<span class="hljs-string">&quot;cuda&quot;</span>) <span class="hljs-comment"># or &quot;mps&quot;, &quot;xpu&quot;, &quot;cpu&quot;</span>
pipe.transformer.enable_cache(TextKVCacheConfig())
image = pipe(<span class="hljs-string">&quot;A cat holding a sign that says hello world&quot;</span>, num_inference_steps=<span class="hljs-number">50</span>).images[<span class="hljs-number">0</span>]<!----></pre></div><!----> <a class="!text-gray-400 !no-underline text-sm flex items-center not-prose mt-4" href="https://github.com/huggingface/diffusers/blob/main/docs/source/en/optimization/cache.md" target="_blank"><svg class="mr-1" xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M31,16l-7,7l-1.41-1.41L28.17,16l-5.58-5.59L24,9l7,7z"></path><path d="M1,16l7-7l1.41,1.41L3.83,16l5.58,5.59L8,23l-7-7z"></path><path d="M12.419,25.484L17.639,6.552l1.932,0.518L14.351,26.002z"></path></svg><!----> <span><span class="underline">Update</span> on GitHub</span></a><!----> <p></p><!--]--><!----><!--]--><!--]--><!--]--> <!--[-1--><!--]--><!--]-->
<script>
{
__sveltekit_diewsf = {
base: "/docs/diffusers/pr_14867/en",
assets: "/docs/diffusers/pr_14867/en"
};
const element = document.currentScript.parentElement;
Promise.all([
import("/docs/diffusers/pr_14867/en/_app/immutable/entry/start.BY1X0FnC.js"),
import("/docs/diffusers/pr_14867/en/_app/immutable/entry/app.BcsduW5Q.js")
]).then(([kit, app]) => {
kit.start(app, element, {
node_ids: [0, 282],
data: [null,null],
form: null,
error: null
});
});
}
</script>

Xet Storage Details

Size:
39.8 kB
·
Xet hash:
cbcdb4275c7c665a36f414a63915a81140cdbe915b2e88f85ec462433566919c

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.