forked from chenfengxu714/StreamDiffusionV2
-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathinference_common.py
More file actions
158 lines (124 loc) · 5.67 KB
/
Copy pathinference_common.py
File metadata and controls
158 lines (124 loc) · 5.67 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
"""Shared helpers for the StreamDiffusionV2 inference entrypoints."""
import os
from typing import Any
import av
import numpy as np
import torch
import torchvision
import torchvision.transforms.functional as TF
from einops import rearrange
from omegaconf import OmegaConf
def _read_video_with_av(video_path: str) -> torch.Tensor:
"""Read a video with PyAV when torchvision's legacy video API is absent."""
frames = []
with av.open(video_path) as container:
stream = container.streams.video[0]
for frame in container.decode(stream):
frames.append(frame.to_rgb().to_ndarray())
if not frames:
raise ValueError(f"No video frames decoded from {video_path}")
video = np.stack(frames, axis=0)
return torch.from_numpy(video).permute(0, 3, 1, 2).contiguous()
def load_mp4_as_tensor(
video_path: str,
max_frames: int = None,
resize_hw: tuple[int, int] = None,
normalize: bool = True,
) -> torch.Tensor:
"""Load an mp4 video as a tensor with shape [C, T, H, W]."""
assert os.path.exists(video_path), f"Video file not found: {video_path}"
if hasattr(torchvision.io, "read_video"):
video, _, _ = torchvision.io.read_video(video_path, output_format="TCHW")
else:
video = _read_video_with_av(video_path)
if max_frames is not None:
video = video[:max_frames]
video = rearrange(video, "t c h w -> c t h w")
if resize_hw is not None:
_, t, _, _ = video.shape
video = torch.stack(
[TF.resize(video[:, i], resize_hw, antialias=True) for i in range(t)],
dim=1,
)
if video.dtype != torch.float32:
video = video.float()
if normalize:
video = video / 127.5 - 1.0
return video
def resolve_config_path(config_path: str, args) -> str:
"""Select an alternate config file when runtime flags imply one."""
fast = bool(args.get("fast", False)) if isinstance(args, dict) else bool(getattr(args, "fast", False))
if not fast:
return config_path
base_name = os.path.basename(config_path)
if base_name != "wan_causal_dmd_v2v.yaml":
return config_path
fast_config_path = os.path.join(os.path.dirname(config_path), "wan_causal_dmd_v2v_fast.yaml")
return fast_config_path if os.path.exists(fast_config_path) else config_path
def merge_cli_config(config_path: str, args) -> OmegaConf:
"""Load a YAML config and overlay CLI arguments onto it."""
config_path = resolve_config_path(config_path, args)
config = OmegaConf.load(config_path)
cli_config = OmegaConf.create(vars(args) if not isinstance(args, dict) else args)
config = OmegaConf.merge(config, cli_config)
config = normalize_acceleration_flags(config)
# CLI --step should always select the first N non-zero denoising steps from
# the canonical YAML schedule, then append the terminal zero step back.
full_denoising_list = list(config.denoising_step_list)
non_terminal_steps = [step for step in full_denoising_list if int(step) != 0]
step_value = int(config.step)
config.denoising_step_list = non_terminal_steps[:step_value]
config.denoising_step_list.append(0)
return config
def load_generator_state_dict(checkpoint_folder: str):
"""Load the generator weights from a checkpoint folder.
Uses mmap so the file is paged in on demand rather than copied into RAM
all at once. Key remapping is done in-place to avoid a second full copy
of the tensors.
"""
ckpt_path = os.path.join(checkpoint_folder, "model.pt")
# mmap=True keeps tensors as memory-mapped views of the file — no RAM copy
# until the tensor is actually read. weights_only=True disables arbitrary
# pickle execution (also required for mmap on PyTorch >= 2.1).
try:
checkpoint = torch.load(ckpt_path, map_location="cpu", mmap=True, weights_only=True)
except TypeError:
# Older PyTorch without mmap support — fall back gracefully
checkpoint = torch.load(ckpt_path, map_location="cpu")
def add_model_prefix_inplace(state_dict: dict) -> dict:
"""Remap keys in-place so we don't allocate a second copy of all tensors."""
keys_to_rename = [k for k in state_dict if not k.startswith("model.")]
for key in keys_to_rename:
state_dict[f"model.{key}"] = state_dict.pop(key)
return state_dict
if isinstance(checkpoint, dict):
for key in ("generator", "generator_ema", "state_dict"):
if key in checkpoint:
raw = checkpoint.pop(key) # drop the top-level reference
del checkpoint # release any other keys (e.g. optimizer state)
return ckpt_path, add_model_prefix_inplace(raw)
return ckpt_path, add_model_prefix_inplace(checkpoint)
def _get_flag(config: Any, key: str, default=False):
if isinstance(config, dict):
return config.get(key, default)
return getattr(config, key, default)
def _set_flag(config: Any, key: str, value) -> None:
if isinstance(config, dict):
config[key] = value
else:
setattr(config, key, value)
def normalize_acceleration_flags(config):
"""Apply shared CLI/runtime flag semantics for fast and TensorRT modes."""
use_taehv = bool(_get_flag(config, "use_taehv", False))
use_tensorrt = bool(_get_flag(config, "use_tensorrt", False))
fast = bool(_get_flag(config, "fast", False))
if fast:
use_taehv = True
use_tensorrt = True
# The current TensorRT path is implemented on top of the TAEHV decoder.
if use_tensorrt:
use_taehv = True
_set_flag(config, "use_taehv", use_taehv)
_set_flag(config, "use_tensorrt", use_tensorrt)
_set_flag(config, "fast", fast)
return config