SUPIR

Restarting on Zero

App Files Files Community

Fabrice-TIERCELIN commited on Oct 18

Commit

c24bf2e

verified ·

1 Parent(s): 0e28f69

Upload 11 files

Browse files

Files changed (12) hide show

.gitattributes +6 -0
README.md +5 -14
app.py +0 -0
capyabara.webp +3 -0
capyabara_zoomed.png +3 -0
optimization.py +133 -0
optimization_utils.py +107 -0
poli_tower.png +3 -0
requirements.txt +10 -22
squatting_sonic.png +3 -0
tower_takes_off.png +3 -0
ugly_sonic.jpeg +3 -0

.gitattributes CHANGED Viewed

@@ -64,3 +64,9 @@ RealESRGAN_examples/Example1.mp4 filter=lfs diff=lfs merge=lfs -text
 cat.png filter=lfs diff=lfs merge=lfs -text
 flowers.png filter=lfs diff=lfs merge=lfs -text
 monster.png filter=lfs diff=lfs merge=lfs -text

 cat.png filter=lfs diff=lfs merge=lfs -text
 flowers.png filter=lfs diff=lfs merge=lfs -text
 monster.png filter=lfs diff=lfs merge=lfs -text
+capyabara_zoomed.png filter=lfs diff=lfs merge=lfs -text
+capyabara.webp filter=lfs diff=lfs merge=lfs -text
+poli_tower.png filter=lfs diff=lfs merge=lfs -text
+squatting_sonic.png filter=lfs diff=lfs merge=lfs -text
+tower_takes_off.png filter=lfs diff=lfs merge=lfs -text
+ugly_sonic.jpeg filter=lfs diff=lfs merge=lfs -text

README.md CHANGED Viewed

@@ -1,21 +1,12 @@
 ---
-title: FramePack/HunyuanVideo
-emoji: 🎥
-colorFrom: pink
 colorTo: gray
 sdk: gradio
-sdk_version: 5.29.1
 app_file: app.py
-license: apache-2.0
-short_description: Text-to-Video/Image-to-Video/Video extender (timed prompt)
-tags:
-  - Image-to-Video
-  - Image-2-Video
-  - Img-to-Vid
-  - Img-2-Vid
-  - language models
-  - LLMs
-suggested_hardware: zero-a10g
 ---
 Check out the configuration reference at https://huggingface.co/docs/hub/spaces-config-reference

 ---
+title: Wan 2 2 First Last Frame
+emoji: 💻
+colorFrom: purple
 colorTo: gray
 sdk: gradio
+sdk_version: 5.44.1
 app_file: app.py
+pinned: false
 ---
 Check out the configuration reference at https://huggingface.co/docs/hub/spaces-config-reference

app.py CHANGED Viewed

The diff for this file is too large to render. See raw diff

capyabara.webp ADDED Viewed

Git LFS Details

SHA256: 26f8ee938a1f453a81e85c2035e3787b1e5ddbb9a92acb01688b39abd987c1e8
Pointer size: 131 Bytes
Size of remote file: 467 kB

capyabara_zoomed.png ADDED Viewed

Git LFS Details

SHA256: 37c27e972f09ab9b1c7df8aaa4b7c2cdbb702466e5bb0fecf5cb502ee531a26c
Pointer size: 132 Bytes
Size of remote file: 1.58 MB

optimization.py ADDED Viewed

	@@ -0,0 +1,133 @@

+"""
+"""
+from typing import Any
+from typing import Callable
+from typing import ParamSpec
+import spaces
+import torch
+from torch.utils._pytree import tree_map_only
+from torchao.quantization import quantize_
+from torchao.quantization import Float8DynamicActivationFloat8WeightConfig
+from torchao.quantization import Int8WeightOnlyConfig
+from optimization_utils import capture_component_call
+from optimization_utils import aoti_compile
+from optimization_utils import drain_module_parameters
+P = ParamSpec('P')
+# --- CORRECTED DYNAMIC SHAPING ---
+# VAE temporal scale factor is 1, latent_frames = num_frames. Range is [8, 81].
+LATENT_FRAMES_DIM = torch.export.Dim('num_latent_frames', min=8, max=81)
+# The transformer has a patch_size of (1, 2, 2), which means the input latent height and width
+# are effectively divided by 2. This creates constraints that fail if the symbolic tracer
+# assumes odd numbers are possible.
+#
+# To solve this, we define the dynamic dimension for the *patched* (i.e., post-division) size,
+# and then express the input shape as 2 * this dimension. This mathematically guarantees
+# to the compiler that the input latent dimensions are always even, satisfying the constraints.
+# App range for pixel dimensions: [480, 832]. VAE scale factor is 8.
+# Latent dimension range: [480/8, 832/8] = [60, 104].
+# Patched latent dimension range: [60/2, 104/2] = [30, 52].
+LATENT_PATCHED_HEIGHT_DIM = torch.export.Dim('latent_patched_height', min=30, max=52)
+LATENT_PATCHED_WIDTH_DIM = torch.export.Dim('latent_patched_width', min=30, max=52)
+# Now, we define the dynamic shapes for the transformer's `hidden_states` input,
+# which has the shape (batch_size, channels, num_frames, height, width).
+TRANSFORMER_DYNAMIC_SHAPES = {
+    'hidden_states': {
+        2: LATENT_FRAMES_DIM,
+        3: 2 * LATENT_PATCHED_HEIGHT_DIM, # Guarantees even height
+        4: 2 * LATENT_PATCHED_WIDTH_DIM,  # Guarantees even width
+    },
+}
+# --- END OF CORRECTION ---
+INDUCTOR_CONFIGS = {
+    'conv_1x1_as_mm': True,
+    'epilogue_fusion': False,
+    'coordinate_descent_tuning': True,
+    'coordinate_descent_check_all_directions': True,
+    'max_autotune': True,
+    'triton.cudagraphs': True,
+}
+def optimize_pipeline_(pipeline: Callable[P, Any], *args: P.args, **kwargs: P.kwargs):
+    @spaces.GPU(duration=1500)
+    def compile_transformer():
+        # This LoRA fusion part remains the same
+        pipeline.load_lora_weights(
+            "Kijai/WanVideo_comfy",
+            weight_name="Lightx2v/lightx2v_I2V_14B_480p_cfg_step_distill_rank128_bf16.safetensors",
+            adapter_name="lightx2v"
+        )
+        kwargs_lora = {}
+        kwargs_lora["load_into_transformer_2"] = True
+        pipeline.load_lora_weights(
+            "Kijai/WanVideo_comfy",
+            weight_name="Lightx2v/lightx2v_I2V_14B_480p_cfg_step_distill_rank128_bf16.safetensors",
+            adapter_name="lightx2v_2", **kwargs_lora
+        )
+        pipeline.set_adapters(["lightx2v", "lightx2v_2"], adapter_weights=[1., 1.])
+        pipeline.fuse_lora(adapter_names=["lightx2v"], lora_scale=3., components=["transformer"])
+        pipeline.fuse_lora(adapter_names=["lightx2v_2"], lora_scale=1., components=["transformer_2"])
+        pipeline.unload_lora_weights()
+        # Capture a single call to get the args/kwargs structure
+        with capture_component_call(pipeline, 'transformer') as call:
+            pipeline(*args, **kwargs)
+        dynamic_shapes = tree_map_only((torch.Tensor, bool), lambda t: None, call.kwargs)
+        dynamic_shapes |= TRANSFORMER_DYNAMIC_SHAPES
+        # Quantization remains the same
+        quantize_(pipeline.transformer, Float8DynamicActivationFloat8WeightConfig())
+        quantize_(pipeline.transformer_2, Float8DynamicActivationFloat8WeightConfig())
+        # --- SIMPLIFIED COMPILATION ---
+        exported_1 = torch.export.export(
+            mod=pipeline.transformer,
+            args=call.args,
+            kwargs=call.kwargs,
+            dynamic_shapes=dynamic_shapes,
+        )
+        exported_2 = torch.export.export(
+            mod=pipeline.transformer_2,
+            args=call.args,
+            kwargs=call.kwargs,
+            dynamic_shapes=dynamic_shapes,
+        )
+        compiled_1 = aoti_compile(exported_1, INDUCTOR_CONFIGS)
+        compiled_2 = aoti_compile(exported_2, INDUCTOR_CONFIGS)
+        # Return the two compiled models
+        return compiled_1, compiled_2
+    # Quantize text encoder (same as before)
+    quantize_(pipeline.text_encoder, Int8WeightOnlyConfig())
+    # Get the two dynamically-shaped compiled models
+    compiled_transformer_1, compiled_transformer_2 = compile_transformer()
+    # --- SIMPLIFIED ASSIGNMENT ---
+    pipeline.transformer.forward = compiled_transformer_1
+    drain_module_parameters(pipeline.transformer)
+    pipeline.transformer_2.forward = compiled_transformer_2
+    drain_module_parameters(pipeline.transformer_2)

optimization_utils.py ADDED Viewed

	@@ -0,0 +1,107 @@

+"""
+"""
+import contextlib
+from contextvars import ContextVar
+from io import BytesIO
+from typing import Any
+from typing import cast
+from unittest.mock import patch
+import torch
+from torch._inductor.package.package import package_aoti
+from torch.export.pt2_archive._package import AOTICompiledModel
+from torch.export.pt2_archive._package_weights import Weights
+INDUCTOR_CONFIGS_OVERRIDES = {
+    'aot_inductor.package_constants_in_so': False,
+    'aot_inductor.package_constants_on_disk': True,
+    'aot_inductor.package': True,
+}
+class ZeroGPUWeights:
+    def __init__(self, constants_map: dict[str, torch.Tensor], to_cuda: bool = False):
+        if to_cuda:
+            self.constants_map = {name: tensor.to('cuda') for name, tensor in constants_map.items()}
+        else:
+            self.constants_map = constants_map
+    def __reduce__(self):
+        constants_map: dict[str, torch.Tensor] = {}
+        for name, tensor in self.constants_map.items():
+            tensor_ = torch.empty_like(tensor, device='cpu').pin_memory()
+            constants_map[name] = tensor_.copy_(tensor).detach().share_memory_()
+        return ZeroGPUWeights, (constants_map, True)
+class ZeroGPUCompiledModel:
+    def __init__(self, archive_file: torch.types.FileLike, weights: ZeroGPUWeights):
+        self.archive_file = archive_file
+        self.weights = weights
+        self.compiled_model: ContextVar[AOTICompiledModel | None] = ContextVar('compiled_model', default=None)
+    def __call__(self, *args, **kwargs):
+        if (compiled_model := self.compiled_model.get()) is None:
+            compiled_model = cast(AOTICompiledModel, torch._inductor.aoti_load_package(self.archive_file))
+            compiled_model.load_constants(self.weights.constants_map, check_full_update=True, user_managed=True)
+            self.compiled_model.set(compiled_model)
+        return compiled_model(*args, **kwargs)
+    def __reduce__(self):
+        return ZeroGPUCompiledModel, (self.archive_file, self.weights)
+def aoti_compile(
+    exported_program: torch.export.ExportedProgram,
+    inductor_configs: dict[str, Any] | None = None,
+):
+    inductor_configs = (inductor_configs or {}) | INDUCTOR_CONFIGS_OVERRIDES
+    gm = cast(torch.fx.GraphModule, exported_program.module())
+    assert exported_program.example_inputs is not None
+    args, kwargs = exported_program.example_inputs
+    artifacts = torch._inductor.aot_compile(gm, args, kwargs, options=inductor_configs)
+    archive_file = BytesIO()
+    files: list[str | Weights] = [file for file in artifacts if isinstance(file, str)]
+    package_aoti(archive_file, files)
+    weights, = (artifact for artifact in artifacts if isinstance(artifact, Weights))
+    zerogpu_weights = ZeroGPUWeights({name: weights.get_weight(name)[0] for name in weights})
+    return ZeroGPUCompiledModel(archive_file, zerogpu_weights)
+@contextlib.contextmanager
+def capture_component_call(
+    pipeline: Any,
+    component_name: str,
+    component_method='forward',
+):
+    class CapturedCallException(Exception):
+        def __init__(self, *args, **kwargs):
+            super().__init__()
+            self.args = args
+            self.kwargs = kwargs
+    class CapturedCall:
+        def __init__(self):
+            self.args: tuple[Any, ...] = ()
+            self.kwargs: dict[str, Any] = {}
+    component = getattr(pipeline, component_name)
+    captured_call = CapturedCall()
+    def capture_call(*args, **kwargs):
+        raise CapturedCallException(*args, **kwargs)
+    with patch.object(component, component_method, new=capture_call):
+        try:
+            yield captured_call
+        except CapturedCallException as e:
+            captured_call.args = e.args
+            captured_call.kwargs = e.kwargs
+def drain_module_parameters(module: torch.nn.Module):
+    state_dict_meta = {name: {'device': tensor.device, 'dtype': tensor.dtype} for name, tensor in module.state_dict().items()}
+    state_dict = {name: torch.nn.Parameter(torch.empty_like(tensor, device='cpu')) for name, tensor in module.state_dict().items()}
+    module.load_state_dict(state_dict, assign=True)
+    for name, param in state_dict.items():
+        meta = state_dict_meta[name]
+        param.data = torch.Tensor([]).to(**meta)

poli_tower.png ADDED Viewed

Git LFS Details

SHA256: 96bc0e056b5aee2d2f1ed7723bab4f9c928dfb519ec21380aff4bbb12d22b849
Pointer size: 132 Bytes
Size of remote file: 3.49 MB

requirements.txt CHANGED Viewed

@@ -1,23 +1,11 @@
-accelerate==1.7.0
-diffusers==0.33.1
-transformers==4.52.4
-sentencepiece==0.2.0
-pillow==11.2.1
-av==12.1.0
-numpy==1.26.2
-scipy==1.12.0
-requests==2.32.4
-torchsde==0.2.6
-torch>=2.0.0
-torchvision
-torchaudio
-einops==0.8.1
-opencv-contrib-python
 safetensors
-huggingface_hub==0.34.3
-decord==0.6.0
-imageio_ffmpeg==0.6.0
-sageattention==1.0.6
-xformers==0.0.29.post3
-bitsandbytes==0.46.0
-pillow-heif==0.22.0

+git+https://github.com/linoytsaban/diffusers.git@wan22-loras
+transformers
+accelerate
 safetensors
+sentencepiece
+peft
+ftfy
+imageio-ffmpeg
+opencv-python
+torchao==0.11.0

squatting_sonic.png ADDED Viewed

Git LFS Details

SHA256: d5675e8192c6274c22b07cb60af92b8577d9fcf26f79a10450a325e385e17e18
Pointer size: 132 Bytes
Size of remote file: 1.05 MB

tower_takes_off.png ADDED Viewed

Git LFS Details

SHA256: 3f824eae87d73d1b841354fcb96cfe5f7d08f8f2d6410bfaed864ecaf1500499
Pointer size: 132 Bytes
Size of remote file: 1.43 MB

ugly_sonic.jpeg ADDED Viewed

Git LFS Details

SHA256: 37f76cf1cbb3a3fa0a6eb26898c8f89f71fa280d13f30fcc9dfdd3709cb9824d
Pointer size: 131 Bytes
Size of remote file: 290 kB