Adopt full MMH3 spatial split controls
This commit is contained in:
@@ -47,9 +47,9 @@
|
|||||||
- Per-shot directives now support `continuity:`, `ref_mode:`, `ref_noise_aug:`, `anchor_add:`, `soundscape:`, and `music:` in addition to the existing timing and wardrobe directives.
|
- Per-shot directives now support `continuity:`, `ref_mode:`, `ref_noise_aug:`, `anchor_add:`, `soundscape:`, and `music:` in addition to the existing timing and wardrobe directives.
|
||||||
|
|
||||||
- `Dumas H3 Latent Upscale Params`
|
- `Dumas H3 Latent Upscale Params`
|
||||||
- Inputs: `mode`, `model_name`, `method`, `width`, `height`, `device`, `precision`, `sampler_name`, `scheduler`, `steps`, `denoise`, `megapixels`, `tile_width`, `tile_height`, `overlap`, `fade_width`, `overlap_mode`, `overlap_blend`, `chunk_length`, `resize_conditioning`, `anchor_strength`
|
- Inputs: `mode`, `model_name`, `method`, `width`, `height`, `device`, `precision`, `sampler_name`, `scheduler`, `steps`, `denoise`, `megapixels`, `tile_width`, `tile_height`, `overlap`, `fade_width`, `fade_height`, `overlap_mode`, `overlap_blend`, `tile_size_mode`, `grid_rows`, `grid_cols`, `spatial_w_overlap`, `spatial_h_overlap`, `min_tile_size`, `masked_area_noise`, `brightness_match`, `dynamic_fade`, `dynamic_fade_min`, `chunk_length`, `resize_conditioning`, `anchor_strength`
|
||||||
- Output: `latent_upscale_param`
|
- Output: `latent_upscale_param`
|
||||||
- Bundles the optional latent-space upscaler settings used by `Dumas H3 Long Videos` before decode, so the main node can rebuild conditioning at the target size and run a short refinement pass with your chosen sampler, scheduler, step count, denoise, optional spatial batching, and the upstream split compatibility settings.
|
- Bundles the optional latent-space upscaler settings used by `Dumas H3 Long Videos` before decode, so the main node can rebuild conditioning at the target size and run a short refinement pass with your chosen sampler, scheduler, step count, denoise, and the full upstream spatial split controls.
|
||||||
|
|
||||||
- `Dumas H3 Beat Prompt`
|
- `Dumas H3 Beat Prompt`
|
||||||
- Inputs: authored through the custom front-end beat editor
|
- Inputs: authored through the custom front-end beat editor
|
||||||
|
|||||||
@@ -469,12 +469,34 @@ class H3LatentUpscaleParams:
|
|||||||
"tooltip": "Spatial tile height for the refinement stage in pixels. 512 matches the upstream latent-split default."}),
|
"tooltip": "Spatial tile height for the refinement stage in pixels. 512 matches the upstream latent-split default."}),
|
||||||
"overlap": ("INT", {"default": 64, "min": 0, "max": 4096, "step": 32,
|
"overlap": ("INT", {"default": 64, "min": 0, "max": 4096, "step": 32,
|
||||||
"tooltip": "Pixel overlap between neighbouring spatial tiles. 64 matches the upstream latent-split default."}),
|
"tooltip": "Pixel overlap between neighbouring spatial tiles. 64 matches the upstream latent-split default."}),
|
||||||
"fade_width": ("INT", {"default": 0, "min": 0, "max": 4096, "step": 32,
|
"fade_width": ("INT", {"default": 32, "min": 0, "max": 4096, "step": 32,
|
||||||
"tooltip": "Width in pixels of the freeze-to-free transition inside each overlap strip. 0 freezes the whole strip, matching the upstream default."}),
|
"tooltip": "Width in pixels of the freeze-to-free transition inside each overlap strip. 0 freezes the whole strip; 32 matches the upstream default."}),
|
||||||
|
"fade_height": ("INT", {"default": 32, "min": 0, "max": 4096, "step": 32,
|
||||||
|
"tooltip": "Height in pixels of the freeze-to-free transition inside each overlap strip. 0 freezes the whole strip; 32 matches the upstream default."}),
|
||||||
"overlap_mode": (["earlier", "later"], {"default": "earlier",
|
"overlap_mode": (["earlier", "later"], {"default": "earlier",
|
||||||
"tooltip": "Which tile wins the overlap band when stitching the spatial batches back together."}),
|
"tooltip": "Which tile wins the overlap band when stitching the spatial batches back together."}),
|
||||||
"overlap_blend": (["linear", "smoothstep", "overwrite", "midpoint"], {"default": "linear",
|
"overlap_blend": (["linear", "smoothstep", "overwrite", "midpoint"], {"default": "linear",
|
||||||
"tooltip": "How overlap bands are blended when the spatial batches are stitched back together."}),
|
"tooltip": "How overlap bands are blended when the spatial batches are stitched back together."}),
|
||||||
|
"tile_size_mode": (["specific_size", "rows_cols"], {"default": "specific_size",
|
||||||
|
"tooltip": "How the spatial tile size is determined. 'specific_size' uses tile_width/tile_height. 'rows_cols' solves equal-size tiles from the target width/height and grid counts."}),
|
||||||
|
"grid_rows": ("INT", {"default": 2, "min": 1, "max": 9, "step": 1,
|
||||||
|
"tooltip": "Number of tile rows used when tile_size_mode = rows_cols."}),
|
||||||
|
"grid_cols": ("INT", {"default": 2, "min": 1, "max": 9, "step": 1,
|
||||||
|
"tooltip": "Number of tile columns used when tile_size_mode = rows_cols."}),
|
||||||
|
"spatial_w_overlap": ("INT", {"default": 128, "min": 0, "max": 4096, "step": 32,
|
||||||
|
"tooltip": "Desired horizontal overlap when tile_size_mode = rows_cols. The node solves the actual equal-tile overlap from the requested target size."}),
|
||||||
|
"spatial_h_overlap": ("INT", {"default": 128, "min": 0, "max": 4096, "step": 32,
|
||||||
|
"tooltip": "Desired vertical overlap when tile_size_mode = rows_cols. The node solves the actual equal-tile overlap from the requested target size."}),
|
||||||
|
"min_tile_size": ("INT", {"default": 256, "min": 0, "max": 4096, "step": 32,
|
||||||
|
"tooltip": "Minimum leftover edge tile size allowed when the spatial grid is solved."}),
|
||||||
|
"masked_area_noise": ("FLOAT", {"default": 0.0, "min": 0.0, "max": 1.0, "step": 0.01,
|
||||||
|
"tooltip": "How much noise is allowed into the masked overlap band during spatial refinement."}),
|
||||||
|
"brightness_match": ("BOOLEAN", {"default": False,
|
||||||
|
"tooltip": "Match each sampled tile's brightness baseline to the source tile region after sampling."}),
|
||||||
|
"dynamic_fade": (["off", "narrowing", "widening"], {"default": "off",
|
||||||
|
"tooltip": "Temporal fade schedule over each tile's sampling. Off keeps the fade fixed; narrowing shrinks it over steps; widening grows it over steps."}),
|
||||||
|
"dynamic_fade_min": ("INT", {"default": 32, "min": 0, "max": 4096, "step": 32,
|
||||||
|
"tooltip": "Minimum fade width used by dynamic_fade when it is enabled."}),
|
||||||
"chunk_length": ("INT", {"default": 136, "min": 17, "max": 100000, "step": 17,
|
"chunk_length": ("INT", {"default": 136, "min": 17, "max": 100000, "step": 17,
|
||||||
"tooltip": "Reserved for upstream split compatibility. 136 matches the original chunk-length default."}),
|
"tooltip": "Reserved for upstream split compatibility. 136 matches the original chunk-length default."}),
|
||||||
"resize_conditioning": ("BOOLEAN", {"default": False,
|
"resize_conditioning": ("BOOLEAN", {"default": False,
|
||||||
@@ -484,7 +506,7 @@ class H3LatentUpscaleParams:
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
def build(self, mode, model_name, method, width, height, device, precision, sampler_name, scheduler, steps, denoise, megapixels, tile_width, tile_height, overlap, fade_width, overlap_mode, overlap_blend, chunk_length, resize_conditioning, anchor_strength):
|
def build(self, mode, model_name, method, width, height, device, precision, sampler_name, scheduler, steps, denoise, megapixels, tile_width, tile_height, overlap, fade_width, fade_height, overlap_mode, overlap_blend, tile_size_mode, grid_rows, grid_cols, spatial_w_overlap, spatial_h_overlap, min_tile_size, masked_area_noise, brightness_match, dynamic_fade, dynamic_fade_min, chunk_length, resize_conditioning, anchor_strength):
|
||||||
width = int(width)
|
width = int(width)
|
||||||
height = int(height)
|
height = int(height)
|
||||||
steps = int(steps)
|
steps = int(steps)
|
||||||
@@ -492,6 +514,14 @@ class H3LatentUpscaleParams:
|
|||||||
tile_height = int(tile_height)
|
tile_height = int(tile_height)
|
||||||
overlap = int(overlap)
|
overlap = int(overlap)
|
||||||
fade_width = int(fade_width)
|
fade_width = int(fade_width)
|
||||||
|
fade_height = int(fade_height)
|
||||||
|
grid_rows = int(grid_rows)
|
||||||
|
grid_cols = int(grid_cols)
|
||||||
|
spatial_w_overlap = int(spatial_w_overlap)
|
||||||
|
spatial_h_overlap = int(spatial_h_overlap)
|
||||||
|
min_tile_size = int(min_tile_size)
|
||||||
|
masked_area_noise = float(masked_area_noise)
|
||||||
|
dynamic_fade_min = int(dynamic_fade_min)
|
||||||
chunk_length = int(chunk_length)
|
chunk_length = int(chunk_length)
|
||||||
anchor_strength = float(anchor_strength)
|
anchor_strength = float(anchor_strength)
|
||||||
if mode == "off":
|
if mode == "off":
|
||||||
@@ -509,8 +539,19 @@ class H3LatentUpscaleParams:
|
|||||||
"tile_height": tile_height,
|
"tile_height": tile_height,
|
||||||
"overlap": overlap,
|
"overlap": overlap,
|
||||||
"fade_width": fade_width,
|
"fade_width": fade_width,
|
||||||
|
"fade_height": fade_height,
|
||||||
"overlap_mode": overlap_mode,
|
"overlap_mode": overlap_mode,
|
||||||
"overlap_blend": overlap_blend,
|
"overlap_blend": overlap_blend,
|
||||||
|
"tile_size_mode": tile_size_mode,
|
||||||
|
"grid_rows": grid_rows,
|
||||||
|
"grid_cols": grid_cols,
|
||||||
|
"spatial_w_overlap": spatial_w_overlap,
|
||||||
|
"spatial_h_overlap": spatial_h_overlap,
|
||||||
|
"min_tile_size": min_tile_size,
|
||||||
|
"masked_area_noise": masked_area_noise,
|
||||||
|
"brightness_match": bool(brightness_match),
|
||||||
|
"dynamic_fade": dynamic_fade,
|
||||||
|
"dynamic_fade_min": dynamic_fade_min,
|
||||||
"chunk_length": chunk_length,
|
"chunk_length": chunk_length,
|
||||||
"resize_conditioning": bool(resize_conditioning),
|
"resize_conditioning": bool(resize_conditioning),
|
||||||
"anchor_strength": anchor_strength,
|
"anchor_strength": anchor_strength,
|
||||||
@@ -537,8 +578,19 @@ class H3LatentUpscaleParams:
|
|||||||
"tile_height": tile_height,
|
"tile_height": tile_height,
|
||||||
"overlap": overlap,
|
"overlap": overlap,
|
||||||
"fade_width": fade_width,
|
"fade_width": fade_width,
|
||||||
|
"fade_height": fade_height,
|
||||||
"overlap_mode": overlap_mode,
|
"overlap_mode": overlap_mode,
|
||||||
"overlap_blend": overlap_blend,
|
"overlap_blend": overlap_blend,
|
||||||
|
"tile_size_mode": tile_size_mode,
|
||||||
|
"grid_rows": grid_rows,
|
||||||
|
"grid_cols": grid_cols,
|
||||||
|
"spatial_w_overlap": spatial_w_overlap,
|
||||||
|
"spatial_h_overlap": spatial_h_overlap,
|
||||||
|
"min_tile_size": min_tile_size,
|
||||||
|
"masked_area_noise": masked_area_noise,
|
||||||
|
"brightness_match": bool(brightness_match),
|
||||||
|
"dynamic_fade": dynamic_fade,
|
||||||
|
"dynamic_fade_min": dynamic_fade_min,
|
||||||
"chunk_length": chunk_length,
|
"chunk_length": chunk_length,
|
||||||
"resize_conditioning": bool(resize_conditioning),
|
"resize_conditioning": bool(resize_conditioning),
|
||||||
"anchor_strength": anchor_strength,
|
"anchor_strength": anchor_strength,
|
||||||
|
|||||||
+257
-13
@@ -4118,6 +4118,159 @@ def _latent_spatial_blend_weights(t, overlap_mode, overlap_blend="linear"):
|
|||||||
return 1.0 - base
|
return 1.0 - base
|
||||||
|
|
||||||
|
|
||||||
|
def _param_value(mapping, key, default):
|
||||||
|
value = mapping.get(key, default)
|
||||||
|
return default if value is None else value
|
||||||
|
|
||||||
|
|
||||||
|
def _grid_1d(size, tile, ol, min_tile):
|
||||||
|
if size <= tile:
|
||||||
|
return [0], [size], [0]
|
||||||
|
sh = tile - ol
|
||||||
|
n = math.ceil((size - ol) / sh)
|
||||||
|
if (n - 1) * sh + tile < size:
|
||||||
|
n += 1
|
||||||
|
rows = [i * sh for i in range(n)]
|
||||||
|
trows = [min(tile, size - r) for r in rows]
|
||||||
|
if min_tile > 0 and n >= 2:
|
||||||
|
edge = size - rows[-1]
|
||||||
|
if edge < min_tile:
|
||||||
|
new_last = size - min_tile
|
||||||
|
if rows[-2] < new_last < rows[-2] + trows[-2]:
|
||||||
|
rows[-1] = new_last
|
||||||
|
trows[-1] = size - new_last
|
||||||
|
ovl = [0] * n
|
||||||
|
for i in range(1, n):
|
||||||
|
ovl[i] = max(0, rows[i - 1] + trows[i - 1] - rows[i])
|
||||||
|
return rows, trows, ovl
|
||||||
|
|
||||||
|
|
||||||
|
def compute_spatial_grid(h, w, th, tw, ol_h, ol_w, min_th=0, min_tw=0):
|
||||||
|
if th <= 0 or tw <= 0:
|
||||||
|
raise ValueError("tile dimensions must be positive")
|
||||||
|
if ol_h >= th or ol_w >= tw:
|
||||||
|
raise ValueError("overlap must be smaller than the tile size")
|
||||||
|
if min_th < 0 or min_tw < 0:
|
||||||
|
raise ValueError("minimum tile size must be non-negative")
|
||||||
|
if min_th > th or min_tw > tw:
|
||||||
|
raise ValueError("minimum tile size must not exceed the tile size")
|
||||||
|
rows, trows, row_ovl = _grid_1d(h, th, ol_h, min_th)
|
||||||
|
cols, tcols, col_ovl = _grid_1d(w, tw, ol_w, min_tw)
|
||||||
|
return rows, cols, trows, tcols, row_ovl, col_ovl
|
||||||
|
|
||||||
|
|
||||||
|
def spatial_fade_mask(tile_h, tile_w, ol_h, ol_w, done_top, done_left, fade_h=0, fade_w=0):
|
||||||
|
mask = torch.ones(tile_h, tile_w, dtype=torch.float32)
|
||||||
|
if done_left and ol_w > 0:
|
||||||
|
if fade_w == 0:
|
||||||
|
mask[:, :ol_w] = 0.0
|
||||||
|
else:
|
||||||
|
f = min(fade_w, ol_w)
|
||||||
|
frozen_w = ol_w - f
|
||||||
|
w = torch.linspace(0.0, 1.0, f)
|
||||||
|
mask[:, :frozen_w] = 0.0
|
||||||
|
mask[:, frozen_w:ol_w] = torch.minimum(mask[:, frozen_w:ol_w], w[None, :])
|
||||||
|
if done_top and ol_h > 0:
|
||||||
|
if fade_h == 0:
|
||||||
|
mask[:ol_h, :] = 0.0
|
||||||
|
else:
|
||||||
|
f = min(fade_h, ol_h)
|
||||||
|
frozen_h = ol_h - f
|
||||||
|
w = torch.linspace(0.0, 1.0, f)
|
||||||
|
mask[:frozen_h, :] = 0.0
|
||||||
|
mask[frozen_h:ol_h, :] = torch.minimum(mask[frozen_h:ol_h, :], w[:, None])
|
||||||
|
return mask
|
||||||
|
|
||||||
|
|
||||||
|
def _fade_band(band, fade, axis):
|
||||||
|
n = band.shape[axis]
|
||||||
|
f = min(int(fade), n)
|
||||||
|
if f == 0:
|
||||||
|
band[:] = 0.0
|
||||||
|
return
|
||||||
|
w = torch.linspace(0.0, 1.0, f, dtype=band.dtype, device=band.device)
|
||||||
|
frozen = n - f
|
||||||
|
if axis == 1:
|
||||||
|
w = w[None, :]
|
||||||
|
band[:, :frozen] = torch.minimum(
|
||||||
|
band[:, :frozen], torch.zeros(frozen, dtype=band.dtype, device=band.device)
|
||||||
|
)
|
||||||
|
band[:, frozen:] = torch.minimum(band[:, frozen:], w)
|
||||||
|
else:
|
||||||
|
w = w[:, None]
|
||||||
|
band[:frozen, :] = torch.minimum(
|
||||||
|
band[:frozen, :], torch.zeros((frozen, 1), dtype=band.dtype, device=band.device)
|
||||||
|
)
|
||||||
|
band[frozen:, :] = torch.minimum(band[frozen:, :], w)
|
||||||
|
|
||||||
|
|
||||||
|
def make_fade_mask(tile_h, tile_w, ol_h, ol_w, done_top, done_left, fade_h=0, fade_w=0):
|
||||||
|
mask = torch.ones(tile_h, tile_w, dtype=torch.float32)
|
||||||
|
if done_left and ol_w > 0:
|
||||||
|
_fade_band(mask[:, :ol_w], fade_w, 1)
|
||||||
|
if done_top and ol_h > 0:
|
||||||
|
_fade_band(mask[:ol_h, :], fade_h, 0)
|
||||||
|
return mask
|
||||||
|
|
||||||
|
|
||||||
|
def bright_match_tile(tile, ref, clamp=0.05):
|
||||||
|
d = (tile - ref).float().reshape(tile.shape[0], tile.shape[1], tile.shape[2], -1)
|
||||||
|
dc = d.median(dim=-1).values.clamp(-clamp, clamp)
|
||||||
|
return tile - dc.to(tile.dtype).view(tile.shape[0], tile.shape[1], tile.shape[2], 1, 1)
|
||||||
|
|
||||||
|
|
||||||
|
def _dynamic_fade_closure(sp, fw, fh, tr, tc, tr_s, tc_s, ovh, ovw, done_top, done_left, video_flat, mn=0.0):
|
||||||
|
schedule = sp.get("dynamic_fade", "off")
|
||||||
|
if schedule == "off":
|
||||||
|
return None
|
||||||
|
fmin_w = int(sp.get("dynamic_fade_min", 0)) // 16
|
||||||
|
fmin_h = int(sp.get("dynamic_fade_min", 0)) // 16
|
||||||
|
if fw <= fmin_w and fh <= fmin_h:
|
||||||
|
return None
|
||||||
|
fw_start, fh_start = max(fw, 0), max(fh, 0)
|
||||||
|
fmin_w, fmin_h = min(fmin_w, fw_start), min(fmin_h, fh_start)
|
||||||
|
done_top = done_top and ovh > 0
|
||||||
|
done_left = done_left and ovw > 0
|
||||||
|
s_tok = tr_s * tc_s
|
||||||
|
n_frames = video_flat // s_tok
|
||||||
|
|
||||||
|
def fade_at(p):
|
||||||
|
if schedule == "widening":
|
||||||
|
return fmin_w + (fw_start - fmin_w) * p, fmin_h + (fh_start - fmin_h) * p
|
||||||
|
return fw_start - (fw_start - fmin_w) * p, fh_start - (fh_start - fmin_h) * p
|
||||||
|
|
||||||
|
cache = {}
|
||||||
|
|
||||||
|
def step_fn(sigma, denoise_mask, **kwargs):
|
||||||
|
sigmas = kwargs.get("extra_options", {}).get("sigmas")
|
||||||
|
n_sigmas = int(sigmas.numel()) if sigmas is not None else 0
|
||||||
|
masks = cache.get(n_sigmas)
|
||||||
|
if masks is None:
|
||||||
|
step_count = max(n_sigmas - 1, 1)
|
||||||
|
masks = []
|
||||||
|
for i in range(n_sigmas - 1):
|
||||||
|
p = i / (step_count - 1) if step_count > 1 else 0.0
|
||||||
|
cw, ch = fade_at(p)
|
||||||
|
m = make_fade_mask(tr_s, tc_s, ovh, ovw, done_top, done_left,
|
||||||
|
fade_h=round(ch), fade_w=round(cw))
|
||||||
|
m[tr:tr_s, :] = 0.0
|
||||||
|
m[:, tc:tc_s] = 0.0
|
||||||
|
if mn > 0:
|
||||||
|
m = m + mn * (1.0 - m)
|
||||||
|
masks.append(m)
|
||||||
|
cache[n_sigmas] = masks
|
||||||
|
idx = 0
|
||||||
|
if sigmas is not None:
|
||||||
|
idx = int((sigmas > sigma + 1e-6).sum())
|
||||||
|
m = masks[min(idx, len(masks) - 1)]
|
||||||
|
flat = denoise_mask.clone()
|
||||||
|
flat.reshape(denoise_mask.shape[0], -1)[:, :video_flat] = \
|
||||||
|
m.reshape(1, -1).repeat(denoise_mask.shape[0], n_frames)
|
||||||
|
return flat
|
||||||
|
|
||||||
|
return step_fn
|
||||||
|
|
||||||
|
|
||||||
def _nested_tensor_parts(samples):
|
def _nested_tensor_parts(samples):
|
||||||
if samples is None:
|
if samples is None:
|
||||||
return ()
|
return ()
|
||||||
@@ -6513,19 +6666,67 @@ class H3LongVideos:
|
|||||||
refine_scheduler = latent_upscale_param.get("scheduler", sch)
|
refine_scheduler = latent_upscale_param.get("scheduler", sch)
|
||||||
refine_denoise_value = latent_upscale_param.get("denoise", latent_upscale_param.get("refine_denoise", 0.2))
|
refine_denoise_value = latent_upscale_param.get("denoise", latent_upscale_param.get("refine_denoise", 0.2))
|
||||||
refine_denoise = 0.2 if refine_denoise_value is None else float(refine_denoise_value)
|
refine_denoise = 0.2 if refine_denoise_value is None else float(refine_denoise_value)
|
||||||
|
tile_size_mode = str(latent_upscale_param.get("tile_size_mode", "specific_size"))
|
||||||
tile_w_px = int(latent_upscale_param.get("tile_width", 512) or 512)
|
tile_w_px = int(latent_upscale_param.get("tile_width", 512) or 512)
|
||||||
tile_h_px = int(latent_upscale_param.get("tile_height", 512) or 512)
|
tile_h_px = int(latent_upscale_param.get("tile_height", 512) or 512)
|
||||||
overlap_px = max(0, int(latent_upscale_param.get("overlap", 64) or 64))
|
overlap_px = max(0, int(_param_value(latent_upscale_param, "overlap", 64)))
|
||||||
fade_px = max(0, int(latent_upscale_param.get("fade_width", 0) or 0))
|
fade_w_px = max(0, int(_param_value(latent_upscale_param, "fade_width", 32)))
|
||||||
|
fade_h_px = max(0, int(_param_value(latent_upscale_param, "fade_height", 32)))
|
||||||
overlap_mode = str(latent_upscale_param.get("overlap_mode", "earlier"))
|
overlap_mode = str(latent_upscale_param.get("overlap_mode", "earlier"))
|
||||||
overlap_blend = str(latent_upscale_param.get("overlap_blend", "linear"))
|
overlap_blend = str(latent_upscale_param.get("overlap_blend", "linear"))
|
||||||
tile_tw = max(1, min(int(up_w), max(1, tile_w_px // 16)))
|
grid_rows = max(1, int(_param_value(latent_upscale_param, "grid_rows", 2)))
|
||||||
tile_th = max(1, min(int(up_h), max(1, tile_h_px // 16)))
|
grid_cols = max(1, int(_param_value(latent_upscale_param, "grid_cols", 2)))
|
||||||
ol_tw = max(0, min(tile_tw - 1, overlap_px // 16))
|
spatial_w_overlap_px = max(0, int(_param_value(latent_upscale_param, "spatial_w_overlap", overlap_px)))
|
||||||
ol_th = max(0, min(tile_th - 1, overlap_px // 16))
|
spatial_h_overlap_px = max(0, int(_param_value(latent_upscale_param, "spatial_h_overlap", overlap_px)))
|
||||||
fw_tw = max(0, min(ol_tw, fade_px // 16))
|
min_tile_size_px = max(0, int(_param_value(latent_upscale_param, "min_tile_size", 256)))
|
||||||
fw_th = max(0, min(ol_th, fade_px // 16))
|
masked_area_noise = float(_param_value(latent_upscale_param, "masked_area_noise", 0.0))
|
||||||
rows, cols, trows, tcols = _latent_spatial_grid(int(up_h), int(up_w), tile_th, tile_tw, ol_th, ol_tw)
|
brightness_match = bool(latent_upscale_param.get("brightness_match", False))
|
||||||
|
dynamic_fade = str(latent_upscale_param.get("dynamic_fade", "off"))
|
||||||
|
dynamic_fade_min_px = max(0, int(_param_value(latent_upscale_param, "dynamic_fade_min", 32)))
|
||||||
|
if tile_size_mode == "rows_cols":
|
||||||
|
tile_w_px, spatial_w_overlap_px = _solve_equal_tiles(target_w, grid_cols, spatial_w_overlap_px, 16)
|
||||||
|
tile_h_px, spatial_h_overlap_px = _solve_equal_tiles(target_h, grid_rows, spatial_h_overlap_px, 16)
|
||||||
|
if tile_w_px < min_tile_size_px or tile_h_px < min_tile_size_px:
|
||||||
|
raise ValueError(
|
||||||
|
f"rows_cols mode: solved tile size is {tile_h_px}x{tile_w_px}px "
|
||||||
|
f"(grid {grid_rows}x{grid_cols} over {target_h}x{target_w}px), "
|
||||||
|
f"which is smaller than min_tile_size ({min_tile_size_px}px). "
|
||||||
|
f"Reduce grid_rows/grid_cols, or lower min_tile_size to at most "
|
||||||
|
f"{min(tile_w_px, tile_h_px)}px."
|
||||||
|
)
|
||||||
|
fade_w_px = min(fade_w_px, spatial_w_overlap_px)
|
||||||
|
fade_h_px = min(fade_h_px, spatial_h_overlap_px)
|
||||||
|
else:
|
||||||
|
for name, value in (
|
||||||
|
("tile_width", tile_w_px),
|
||||||
|
("tile_height", tile_h_px),
|
||||||
|
("overlap", overlap_px),
|
||||||
|
("fade_width", fade_w_px),
|
||||||
|
("fade_height", fade_h_px),
|
||||||
|
("min_tile_size", min_tile_size_px),
|
||||||
|
):
|
||||||
|
if value % 32 != 0:
|
||||||
|
raise ValueError(f"'{name}' must be a multiple of 32 pixels; got {value}.")
|
||||||
|
if overlap_px >= tile_w_px:
|
||||||
|
raise ValueError("overlap must be smaller than tile_width")
|
||||||
|
if overlap_px >= tile_h_px:
|
||||||
|
raise ValueError("overlap must be smaller than tile_height")
|
||||||
|
if fade_w_px > spatial_w_overlap_px:
|
||||||
|
raise ValueError("fade_width must not exceed spatial_w_overlap")
|
||||||
|
if fade_h_px > spatial_h_overlap_px:
|
||||||
|
raise ValueError("fade_height must not exceed spatial_h_overlap")
|
||||||
|
if min_tile_size_px > tile_w_px or min_tile_size_px > tile_h_px:
|
||||||
|
raise ValueError("min_tile_size must not exceed the tile size")
|
||||||
|
tile_tw = max(1, tile_w_px // 16)
|
||||||
|
tile_th = max(1, tile_h_px // 16)
|
||||||
|
ol_tw = max(0, min(tile_tw - 1, spatial_w_overlap_px // 16))
|
||||||
|
ol_th = max(0, min(tile_th - 1, spatial_h_overlap_px // 16))
|
||||||
|
fw_tw = max(0, min(ol_tw, fade_w_px // 16))
|
||||||
|
fw_th = max(0, min(ol_th, fade_h_px // 16))
|
||||||
|
min_tile_tw = max(0, min_tile_size_px // 16)
|
||||||
|
rows, cols, trows, tcols, row_ovl, col_ovl = compute_spatial_grid(
|
||||||
|
int(up_h), int(up_w), tile_th, tile_tw, ol_th, ol_tw, min_tile_tw, min_tile_tw
|
||||||
|
)
|
||||||
if len(rows) == 1 and len(cols) == 1:
|
if len(rows) == 1 and len(cols) == 1:
|
||||||
(refined_out,) = nodes.common_ksampler(
|
(refined_out,) = nodes.common_ksampler(
|
||||||
model, seed, refine_steps, cfg, refine_sampler, refine_scheduler, upscale_cond, negative, upscale_latent,
|
model, seed, refine_steps, cfg, refine_sampler, refine_scheduler, upscale_cond, negative, upscale_latent,
|
||||||
@@ -6535,8 +6736,10 @@ class H3LongVideos:
|
|||||||
full_audio = parts[1]
|
full_audio = parts[1]
|
||||||
for row_index, r0 in enumerate(rows):
|
for row_index, r0 in enumerate(rows):
|
||||||
tr = trows[row_index]
|
tr = trows[row_index]
|
||||||
|
ovh = row_ovl[row_index]
|
||||||
for col_index, c0 in enumerate(cols):
|
for col_index, c0 in enumerate(cols):
|
||||||
tc = tcols[col_index]
|
tc = tcols[col_index]
|
||||||
|
ovw = col_ovl[col_index]
|
||||||
tile_target_w = int(tc) * 16
|
tile_target_w = int(tc) * 16
|
||||||
tile_target_h = int(tr) * 16
|
tile_target_h = int(tr) * 16
|
||||||
tile_cond, tile_latent = _build_shot_conditioning(
|
tile_cond, tile_latent = _build_shot_conditioning(
|
||||||
@@ -6544,14 +6747,45 @@ class H3LongVideos:
|
|||||||
ref_images=refs, ref_image_size=ref_image_size,
|
ref_images=refs, ref_image_size=ref_image_size,
|
||||||
ref_noise_aug=ref_noise_aug, audio_vae=audio_vae, silent=silent)
|
ref_noise_aug=ref_noise_aug, audio_vae=audio_vae, silent=silent)
|
||||||
tile_video = upscaled_video[:, :, :, r0:r0 + tr, c0:c0 + tc].contiguous()
|
tile_video = upscaled_video[:, :, :, r0:r0 + tr, c0:c0 + tc].contiguous()
|
||||||
tile_latent["samples"] = comfy.nested_tensor.NestedTensor((tile_video, full_audio))
|
tr_s = tr + (tr % 2)
|
||||||
|
tc_s = tc + (tc % 2)
|
||||||
|
tile = torch.zeros((1, tile_video.shape[1], tile_video.shape[2], tr_s, tc_s),
|
||||||
|
device=tile_video.device, dtype=tile_video.dtype)
|
||||||
|
tile[:, :, :, :tr, :tc] = tile_video
|
||||||
|
if col_index > 0 and ovw > 0:
|
||||||
|
tile[:, :, :, :tr, :ovw] = refined_video[:, :, :, r0:r0 + tr, c0:c0 + ovw]
|
||||||
|
if row_index > 0 and ovh > 0:
|
||||||
|
tile[:, :, :, :ovh, :tc] = refined_video[:, :, :, r0:r0 + ovh, c0:c0 + tc]
|
||||||
|
mask = make_fade_mask(tr_s, tc_s, ovh, ovw, row_index > 0, col_index > 0,
|
||||||
|
fade_h=fw_th, fade_w=fw_tw)
|
||||||
|
mask[tr:tr_s, :] = 0.0
|
||||||
|
mask[:, tc:tc_s] = 0.0
|
||||||
|
mv = (mask + masked_area_noise * (1.0 - mask))[None, None, None].to(tile.dtype)
|
||||||
|
ma = torch.zeros((1, 32, 2, full_audio.shape[-1]), device=full_audio.device, dtype=full_audio.dtype)
|
||||||
|
tile_latent["samples"] = comfy.nested_tensor.NestedTensor((tile, full_audio))
|
||||||
|
tile_latent["noise_mask"] = comfy.nested_tensor.NestedTensor((mv, ma))
|
||||||
|
dynamic = _dynamic_fade_closure(
|
||||||
|
latent_upscale_param, fw_tw, fw_th, tr, tc, tr_s, tc_s, ovh, ovw,
|
||||||
|
row_index > 0, col_index > 0, math.prod(tile.shape[1:]), mn=masked_area_noise
|
||||||
|
)
|
||||||
|
if dynamic is not None:
|
||||||
|
model.set_model_denoise_mask_function(dynamic)
|
||||||
|
try:
|
||||||
tile_out, = nodes.common_ksampler(
|
tile_out, = nodes.common_ksampler(
|
||||||
model, seed, refine_steps, cfg, refine_sampler, refine_scheduler, tile_cond, negative, tile_latent,
|
model, seed, refine_steps, cfg, refine_sampler, refine_scheduler, tile_cond, negative, tile_latent,
|
||||||
denoise=refine_denoise)
|
denoise=refine_denoise)
|
||||||
|
finally:
|
||||||
|
if dynamic is not None:
|
||||||
|
model.model_options.pop("denoise_mask_function", None)
|
||||||
tile_out = _video_only_refined_latent(
|
tile_out = _video_only_refined_latent(
|
||||||
{"samples": comfy.nested_tensor.NestedTensor((tile_video, full_audio))},
|
{"samples": comfy.nested_tensor.NestedTensor((tile_video, full_audio))},
|
||||||
tile_out)
|
tile_out)
|
||||||
tile_video_out = tile_out["samples"].tensors[0]
|
tile_video_out = tile_out["samples"].tensors[0]
|
||||||
|
if brightness_match:
|
||||||
|
tile_video_out = bright_match_tile(
|
||||||
|
tile_video_out,
|
||||||
|
upscaled_video[:, :, :, r0:r0 + tr, c0:c0 + tc]
|
||||||
|
)
|
||||||
region = refined_video[:, :, :, r0:r0 + tr, c0:c0 + tc]
|
region = refined_video[:, :, :, r0:r0 + tr, c0:c0 + tc]
|
||||||
base_region = region.clone()
|
base_region = region.clone()
|
||||||
region.copy_(tile_video_out)
|
region.copy_(tile_video_out)
|
||||||
@@ -6713,12 +6947,22 @@ class H3LongVideos:
|
|||||||
refine_sampler = latent_upscale_param.get("sampler_name", "euler_ancestral")
|
refine_sampler = latent_upscale_param.get("sampler_name", "euler_ancestral")
|
||||||
refine_scheduler = latent_upscale_param.get("scheduler", "simple")
|
refine_scheduler = latent_upscale_param.get("scheduler", "simple")
|
||||||
batch_note = ""
|
batch_note = ""
|
||||||
|
tile_size_mode = str(latent_upscale_param.get("tile_size_mode", "specific_size"))
|
||||||
tile_w_px = int(latent_upscale_param.get("tile_width", 512) or 512)
|
tile_w_px = int(latent_upscale_param.get("tile_width", 512) or 512)
|
||||||
tile_h_px = int(latent_upscale_param.get("tile_height", 512) or 512)
|
tile_h_px = int(latent_upscale_param.get("tile_height", 512) or 512)
|
||||||
overlap_px = max(0, int(latent_upscale_param.get("overlap", 64) or 64))
|
overlap_px = max(0, int(_param_value(latent_upscale_param, "overlap", 64)))
|
||||||
overlap_blend = str(latent_upscale_param.get("overlap_blend", "linear"))
|
overlap_blend = str(latent_upscale_param.get("overlap_blend", "linear"))
|
||||||
if tile_w_px > 0 and tile_h_px > 0 and (tile_w_px < target_w or tile_h_px < target_h):
|
grid_rows = max(1, int(_param_value(latent_upscale_param, "grid_rows", 2)))
|
||||||
batch_note = f"; spatial batches {tile_w_px}x{tile_h_px}px overlap {overlap_px}px {overlap_blend}"
|
grid_cols = max(1, int(_param_value(latent_upscale_param, "grid_cols", 2)))
|
||||||
|
if tile_size_mode == "rows_cols":
|
||||||
|
batch_note = f"; spatial batches {grid_rows}x{grid_cols} rows_cols over {target_w}x{target_h}px"
|
||||||
|
elif tile_w_px > 0 and tile_h_px > 0 and (tile_w_px < target_w or tile_h_px < target_h):
|
||||||
|
spatial_w_overlap_px = max(0, int(_param_value(latent_upscale_param, "spatial_w_overlap", overlap_px)))
|
||||||
|
spatial_h_overlap_px = max(0, int(_param_value(latent_upscale_param, "spatial_h_overlap", overlap_px)))
|
||||||
|
batch_note = (
|
||||||
|
f"; spatial batches {tile_w_px}x{tile_h_px}px "
|
||||||
|
f"overlap {spatial_w_overlap_px}x{spatial_h_overlap_px}px {overlap_blend}"
|
||||||
|
)
|
||||||
latent_upscale_note = (
|
latent_upscale_note = (
|
||||||
f" latent upscale: target {target_w}x{target_h}px{detail}; "
|
f" latent upscale: target {target_w}x{target_h}px{detail}; "
|
||||||
f"{int(latent_upscale_param.get('steps', 2) or 2)}-step refinement "
|
f"{int(latent_upscale_param.get('steps', 2) or 2)}-step refinement "
|
||||||
|
|||||||
@@ -1130,9 +1130,20 @@ class DumasH3LongVideosHelperTests(unittest.TestCase):
|
|||||||
self.assertEqual(required["tile_width"][1]["default"], 512)
|
self.assertEqual(required["tile_width"][1]["default"], 512)
|
||||||
self.assertEqual(required["tile_height"][1]["default"], 512)
|
self.assertEqual(required["tile_height"][1]["default"], 512)
|
||||||
self.assertEqual(required["overlap"][1]["default"], 64)
|
self.assertEqual(required["overlap"][1]["default"], 64)
|
||||||
self.assertEqual(required["fade_width"][1]["default"], 0)
|
self.assertEqual(required["fade_width"][1]["default"], 32)
|
||||||
|
self.assertEqual(required["fade_height"][1]["default"], 32)
|
||||||
self.assertEqual(required["overlap_mode"][1]["default"], "earlier")
|
self.assertEqual(required["overlap_mode"][1]["default"], "earlier")
|
||||||
self.assertEqual(required["overlap_blend"][1]["default"], "linear")
|
self.assertEqual(required["overlap_blend"][1]["default"], "linear")
|
||||||
|
self.assertEqual(required["tile_size_mode"][1]["default"], "specific_size")
|
||||||
|
self.assertEqual(required["grid_rows"][1]["default"], 2)
|
||||||
|
self.assertEqual(required["grid_cols"][1]["default"], 2)
|
||||||
|
self.assertEqual(required["spatial_w_overlap"][1]["default"], 128)
|
||||||
|
self.assertEqual(required["spatial_h_overlap"][1]["default"], 128)
|
||||||
|
self.assertEqual(required["min_tile_size"][1]["default"], 256)
|
||||||
|
self.assertEqual(required["masked_area_noise"][1]["default"], 0.0)
|
||||||
|
self.assertFalse(required["brightness_match"][1]["default"])
|
||||||
|
self.assertEqual(required["dynamic_fade"][1]["default"], "off")
|
||||||
|
self.assertEqual(required["dynamic_fade_min"][1]["default"], 32)
|
||||||
self.assertEqual(required["chunk_length"][1]["default"], 136)
|
self.assertEqual(required["chunk_length"][1]["default"], 136)
|
||||||
self.assertFalse(required["resize_conditioning"][1]["default"])
|
self.assertFalse(required["resize_conditioning"][1]["default"])
|
||||||
self.assertEqual(required["anchor_strength"][1]["default"], 0.999)
|
self.assertEqual(required["anchor_strength"][1]["default"], 0.999)
|
||||||
|
|||||||
Reference in New Issue
Block a user