256 lines
11 KiB
Python
256 lines
11 KiB
Python
"""
|
|
PIL text overlays for H3 Long Videos -- watermark and intro title.
|
|
|
|
Text is COMPOSITED onto the decoded frames, never asked of the model. H3 (like
|
|
every video diffusion model) renders text as plausible-looking letterforms that
|
|
drift, warp and re-spell themselves frame to frame; a watermark that changes
|
|
shape every frame is worse than none. Compositing gives pixel-identical text on
|
|
every frame at zero sampling cost, and keeps the words out of the prompt where
|
|
they would otherwise steal conditioning from the actual shot.
|
|
|
|
Both overlays are WHITE text drawn on a fully transparent RGBA layer, then
|
|
alpha-blended over the video -- so only the glyphs themselves land on the frame
|
|
and the picture shows through everywhere else.
|
|
|
|
Everything here is best-effort: any failure returns the frames untouched with a
|
|
note, because a cosmetic overlay must never lose a finished render.
|
|
"""
|
|
|
|
import torch
|
|
|
|
BLEND_CHUNK = 64 # frames blended per slice -- bounds peak RAM on long chains
|
|
|
|
# Auto-fit: text is wrapped, then shrunk in FIT_SHRINK steps until the block fits
|
|
# inside the margins. MIN_FONT_PX is the point below which the text would be
|
|
# unreadable anyway, so the loop stops there and lets PIL clip rather than spin.
|
|
MIN_FONT_PX = 8
|
|
FIT_SHRINK = 0.92
|
|
FIT_STEPS = 48
|
|
|
|
# Fonts to try when the requested one cannot be loaded. PIL resolves bare names
|
|
# against the system font directory, so "arial.ttf" works on Windows as-is.
|
|
FONT_FALLBACKS = ("arial.ttf", "segoeui.ttf", "DejaVuSans.ttf", "LiberationSans-Regular.ttf")
|
|
|
|
# Anchor -> (x, y) as a fraction of the free space: 0 = hard against the left/top
|
|
# margin, 1 = hard against the right/bottom, 0.5 = centered.
|
|
POSITIONS = {
|
|
"bottom-right": (1.0, 1.0),
|
|
"bottom-left": (0.0, 1.0),
|
|
"bottom-center": (0.5, 1.0),
|
|
"top-right": (1.0, 0.0),
|
|
"top-left": (0.0, 0.0),
|
|
"top-center": (0.5, 0.0),
|
|
"center": (0.5, 0.5),
|
|
"lower-third": (0.5, 0.72),
|
|
}
|
|
|
|
|
|
def _load_font(name, px):
|
|
"""A truetype font at px, falling back through the known-present faces and
|
|
finally to PIL's bitmap default (which ignores size -- ugly, but never fatal)."""
|
|
from PIL import ImageFont
|
|
px = max(8, int(px))
|
|
for cand in ([name] if name else []) + list(FONT_FALLBACKS):
|
|
try:
|
|
return ImageFont.truetype(cand, px)
|
|
except Exception:
|
|
continue
|
|
return ImageFont.load_default()
|
|
|
|
|
|
def _measure(draw, text, font, stroke_px, spacing):
|
|
"""(x0, y0, x1, y1) of a multi-line block, tolerant of older Pillow builds."""
|
|
try:
|
|
return draw.multiline_textbbox((0, 0), text, font=font, align="center",
|
|
stroke_width=stroke_px, spacing=spacing)
|
|
except TypeError: # older Pillow: no stroke/spacing kwargs
|
|
return draw.multiline_textbbox((0, 0), text, font=font, align="center")
|
|
|
|
|
|
def _wrap(draw, text, font, max_w, stroke_px, spacing):
|
|
"""Greedy word-wrap every hard line to max_w. A single word wider than the
|
|
frame cannot be broken -- the shrink loop in render_text_layer handles that."""
|
|
out = []
|
|
for hard in text.split("\n"):
|
|
words = hard.split()
|
|
if not words:
|
|
out.append("")
|
|
continue
|
|
cur = words[0]
|
|
for wd in words[1:]:
|
|
trial = cur + " " + wd
|
|
b = _measure(draw, trial, font, stroke_px, spacing)
|
|
if b[2] - b[0] <= max_w:
|
|
cur = trial
|
|
else:
|
|
out.append(cur)
|
|
cur = wd
|
|
out.append(cur)
|
|
return "\n".join(out)
|
|
|
|
|
|
def _fit(draw, text, font_name, px, max_w, max_h, stroke_px, line_spacing, wrap=True):
|
|
"""Largest size at or below px whose wrapped block fits (max_w, max_h).
|
|
|
|
Without this, a title is drawn at the requested size and whatever runs past the
|
|
frame is simply CLIPPED by PIL -- silently, with no error and no note. That is
|
|
the whole "overlays don't work at other resolutions" failure: the size is a
|
|
percentage, so the same text that fits 1344x768 overflows a 512-wide portrait
|
|
canvas and loses its outer characters."""
|
|
px = max(MIN_FONT_PX, int(px))
|
|
for _ in range(FIT_STEPS):
|
|
font = _load_font(font_name, px)
|
|
spacing = int(max(0.0, px * (line_spacing - 1.0)))
|
|
fitted = _wrap(draw, text, font, max_w, stroke_px, spacing) if wrap else text
|
|
box = _measure(draw, fitted, font, stroke_px, spacing)
|
|
if (box[2] - box[0] <= max_w and box[3] - box[1] <= max_h) or px <= MIN_FONT_PX:
|
|
return font, fitted, box, spacing, px
|
|
px = max(MIN_FONT_PX, int(px * FIT_SHRINK))
|
|
return font, fitted, box, spacing, px
|
|
|
|
|
|
def render_text_layer(width, height, text, font_px, position="bottom-right",
|
|
margin_pct=3.0, font_name="", stroke_px=0, line_spacing=1.15,
|
|
wrap=True):
|
|
"""White text on a transparent RGBA canvas the size of one frame.
|
|
|
|
The block is WRAPPED and SHRUNK until it fits inside the margins, so the same
|
|
settings render legibly on every supported preset -- portrait canvases and the
|
|
512 tier included -- instead of being clipped at the frame edge.
|
|
|
|
Returns (rgb, alpha, bbox): rgb [H,W,3] float 0..1, alpha [H,W,1] float 0..1
|
|
(zero everywhere except the glyphs and their optional stroke), and the tight
|
|
(x0, y0, x1, y1) box of non-transparent pixels so the blend only has to touch
|
|
the region the text actually occupies. None when there is nothing to draw."""
|
|
from PIL import Image, ImageDraw
|
|
import numpy as np
|
|
|
|
text = (text or "").strip()
|
|
if not text:
|
|
return None
|
|
|
|
img = Image.new("RGBA", (int(width), int(height)), (0, 0, 0, 0))
|
|
draw = ImageDraw.Draw(img)
|
|
stroke_px = max(0, int(stroke_px))
|
|
|
|
margin = int(min(width, height) * max(0.0, margin_pct) / 100.0)
|
|
ax, ay = POSITIONS.get(position, POSITIONS["bottom-right"])
|
|
|
|
# Measure first, so the block is placed by its real size rather than a guess --
|
|
# and fit it to the space the margins actually leave.
|
|
max_w = max(1, int(width) - 2 * margin)
|
|
max_h = max(1, int(height) - 2 * margin)
|
|
font, text, box, spacing, font_px = _fit(draw, text, font_name, font_px, max_w, max_h,
|
|
stroke_px, line_spacing, wrap)
|
|
tw, th = box[2] - box[0], box[3] - box[1]
|
|
|
|
free_w = max(0, int(width) - 2 * margin - tw)
|
|
free_h = max(0, int(height) - 2 * margin - th)
|
|
x = margin + free_w * ax - box[0]
|
|
y = margin + free_h * ay - box[1]
|
|
|
|
kwargs = dict(font=font, fill=(255, 255, 255, 255), align="center")
|
|
if stroke_px:
|
|
kwargs.update(stroke_width=stroke_px, stroke_fill=(0, 0, 0, 255))
|
|
try:
|
|
draw.multiline_text((x, y), text, spacing=spacing, **kwargs)
|
|
except TypeError:
|
|
draw.multiline_text((x, y), text, **kwargs)
|
|
|
|
arr = np.asarray(img, dtype=np.float32) / 255.0 # [H, W, 4]
|
|
alpha = arr[..., 3:4]
|
|
if not alpha.any():
|
|
return None
|
|
# Tight bbox of drawn pixels: blending a whole 1344x768 frame for a corner
|
|
# watermark would cost ~50x more work on a 3000-frame chain.
|
|
ys, xs = np.nonzero(alpha[..., 0] > 0.0)
|
|
bbox = (int(xs.min()), int(ys.min()), int(xs.max()) + 1, int(ys.max()) + 1)
|
|
return (torch.from_numpy(arr[..., :3].copy()),
|
|
torch.from_numpy(alpha.copy()),
|
|
bbox)
|
|
|
|
|
|
def blend_layer(frames, layer, frame_alpha=None, opacity=1.0):
|
|
"""Alpha-composite a rendered layer over frames [N,H,W,3] in 0..1, in place.
|
|
|
|
frame_alpha is an optional per-frame multiplier (length N) -- that is what
|
|
makes an intro title hold and then fade instead of sitting on the whole
|
|
video. Frames whose multiplier is 0 are skipped entirely."""
|
|
if layer is None:
|
|
return frames
|
|
rgb, alpha, (x0, y0, x1, y1) = layer
|
|
n = frames.shape[0]
|
|
if frame_alpha is None:
|
|
frame_alpha = torch.ones(n, dtype=torch.float32)
|
|
frame_alpha = frame_alpha.to(torch.float32).clamp(0.0, 1.0) * float(opacity)
|
|
|
|
a_crop = alpha[y0:y1, x0:x1, :].to(frames.dtype)
|
|
c_crop = rgb[y0:y1, x0:x1, :].to(frames.dtype)
|
|
live = (frame_alpha > 0).nonzero().flatten().tolist()
|
|
for s in range(0, len(live), BLEND_CHUNK):
|
|
idx = live[s:s + BLEND_CHUNK]
|
|
fa = frame_alpha[idx].to(frames.dtype).view(-1, 1, 1, 1)
|
|
sub = frames[idx, y0:y1, x0:x1, :]
|
|
a = a_crop * fa
|
|
frames[idx, y0:y1, x0:x1, :] = sub * (1.0 - a) + c_crop * a
|
|
return frames
|
|
|
|
|
|
def hold_fade_alpha(total_frames, hold_frames, fade_frames):
|
|
"""Per-frame opacity for an intro: full through hold_frames, then a linear
|
|
ramp to zero over fade_frames, then nothing. Returns a length-N tensor."""
|
|
a = torch.zeros(int(total_frames), dtype=torch.float32)
|
|
hold = max(0, min(int(hold_frames), int(total_frames)))
|
|
a[:hold] = 1.0
|
|
fade = max(0, min(int(fade_frames), int(total_frames) - hold))
|
|
if fade:
|
|
a[hold:hold + fade] = torch.linspace(1.0, 0.0, fade + 2)[1:-1]
|
|
return a
|
|
|
|
|
|
def apply_overlays(frames, fps, watermark="", wm_position="bottom-right", wm_size_pct=4.0,
|
|
wm_opacity=0.75, wm_margin_pct=3.0, intro="", intro_seconds=3.0,
|
|
intro_fade=0.6, intro_size_pct=9.0, intro_position="center",
|
|
font_name="", stroke_px=0):
|
|
"""Composite the watermark (every frame) and the intro title (first seconds
|
|
only, then faded out). Returns (frames, note). Never raises -- a cosmetic
|
|
overlay must not be able to destroy a finished render."""
|
|
notes = []
|
|
if frames is None or frames.ndim != 4 or frames.shape[0] == 0:
|
|
return frames, ""
|
|
n, h, w = frames.shape[0], frames.shape[1], frames.shape[2]
|
|
frames = frames.contiguous()
|
|
# Size from the SHORT edge, not the height. Height is the long edge on every
|
|
# portrait preset, so a height-based percentage drew 9:16 text ~1.75x larger
|
|
# than the same setting at 16:9 -- on the canvas with the LEAST room for it.
|
|
# The short edge makes one setting mean the same apparent size at every ratio.
|
|
short = min(int(w), int(h))
|
|
|
|
if (watermark or "").strip():
|
|
try:
|
|
layer = render_text_layer(w, h, watermark, short * max(0.5, wm_size_pct) / 100.0,
|
|
wm_position, wm_margin_pct, font_name, stroke_px)
|
|
if layer is not None:
|
|
blend_layer(frames, layer, None, wm_opacity)
|
|
notes.append(f"watermark composited ({wm_position}, {wm_opacity:.0%})")
|
|
except Exception as e:
|
|
notes.append(f"watermark skipped ({type(e).__name__}: {e})")
|
|
|
|
if (intro or "").strip():
|
|
try:
|
|
layer = render_text_layer(w, h, intro, short * max(0.5, intro_size_pct) / 100.0,
|
|
intro_position, 6.0, font_name, stroke_px)
|
|
if layer is not None:
|
|
hold = round(max(0.0, float(intro_seconds)) * fps)
|
|
fade = round(max(0.0, float(intro_fade)) * fps)
|
|
fa = hold_fade_alpha(n, hold, fade)
|
|
if fa.max() > 0:
|
|
blend_layer(frames, layer, fa, 1.0)
|
|
notes.append(f"intro title composited ({hold}f hold + {fade}f fade)")
|
|
else:
|
|
notes.append("intro title skipped (no hold or fade frames)")
|
|
except Exception as e:
|
|
notes.append(f"intro title skipped ({type(e).__name__}: {e})")
|
|
|
|
return frames, "; ".join(notes)
|