From f0d51cf6dabd7f77b8e27f81f1f7cd29e7635527 Mon Sep 17 00:00:00 2001 From: Chris Dumas Date: Wed, 26 Aug 2026 22:46:02 +0000 Subject: [PATCH] Add per-shot H3 beat directives --- README.md | 6 + dumas_h3_beat_prompt.py | 4 +- dumas_h3_longvideos.py | 469 ++++++++++++++++++++---------- js/h3_beat_prompt/index.js | 194 +++++++++++- tests/test_dumas_h3_longvideos.py | 45 +++ 5 files changed, 560 insertions(+), 158 deletions(-) diff --git a/README.md b/README.md index 8fddc74..1f68614 100644 --- a/README.md +++ b/README.md @@ -42,6 +42,12 @@ - Prompt `` tags now map to the actual ref socket numbers you wire, even with gaps such as only `ref_2` and `ref_7` connected. - Character refs now contribute appearance and wardrobe context from the same structured object, while location refs contribute environment context from theirs. - The default ref2v bias is now stronger: `ref_mode` defaults to `auto ref2v` so untagged prompts condition every shot instead of only shot 1, and `ref_noise_aug` defaults to `0.95` rather than the upstream-literal `0.999`. + - Per-shot directives now support `continuity:`, `ref_mode:`, `ref_noise_aug:`, `anchor_add:`, `soundscape:`, and `music:` in addition to the existing timing and wardrobe directives. + +- `Dumas H3 Beat Prompt` + - Inputs: authored through the custom front-end beat editor + - Output: `prompt` + - Builds one H3 prompt block per beat, with quick controls for per-shot timing, continuity, ref behavior, anchor additions, soundscape, and music while staying compatible with direct text editing. - `Dumas H3 Shot Length` - Inputs: `shot_seconds`, `fps`, optional `cap_to_h3_max` diff --git a/dumas_h3_beat_prompt.py b/dumas_h3_beat_prompt.py index c0a3253..9f0cb96 100644 --- a/dumas_h3_beat_prompt.py +++ b/dumas_h3_beat_prompt.py @@ -47,8 +47,8 @@ def _assemble_beat_prompt(state): class DumasH3BeatPromptNode: DESCRIPTION = ( "Build a MiniMax H3 prompt from one textbox per beat, with a front-end beat " - "editor that can append directive examples such as wardrobe set/add/remove, " - "seconds, exit, and music." + "editor that can append directive examples and expose per-shot controls for " + "timing, continuity, ref behavior, anchor additions, soundscape, and music." ) RETURN_TYPES = ("STRING",) RETURN_NAMES = ("prompt",) diff --git a/dumas_h3_longvideos.py b/dumas_h3_longvideos.py index 306a209..e900518 100644 --- a/dumas_h3_longvideos.py +++ b/dumas_h3_longvideos.py @@ -281,8 +281,26 @@ ADDED_WIDGETS = ( NL = "\n" # Lines that CONFIGURE a beat rather than being one. They attach to the beat that # follows them, so a line-split never turns "wardrobe: ..." into its own shot. -DIRECTIVE_KEYS = ("wardrobe", "seconds", "duration", "exit", "enter", - "overall_soundscape", "non_diegetic_music", "soundscape", "music") +DIRECTIVE_KEYS = ("wardrobe", "seconds", "duration", "exit", "enter", + "overall_soundscape", "non_diegetic_music", "soundscape", "music", + "continuity", "ref_mode", "ref_noise_aug", "anchor_add") + +_REF_MODE_DIRECTIVE_MAP = { + "auto ref2v": "auto ref2v", + "where tagged": "where tagged", + "first shot": "first shot", + "every shot": "every shot", + "every shot + handoff ref": "every shot + handoff ref", +} + +_CONTINUITY_DIRECTIVE_MAP = { + "auto": "auto", + "default": "auto", + "soft carry": "soft carry", + "hard cut": "hard cut", + "keyframe carry": "keyframe carry", + "handoff ref": "handoff ref", +} def is_directive_line(line): @@ -2774,6 +2792,82 @@ def extract_directive(body, key): extract_directive = lru_cache(maxsize=4096)(extract_directive) + + +def extract_directive_aliases(body, keys): + """Pull all alias lines for one logical directive, returning the last value found.""" + cleaned = str(body or "") + value = None + for key in keys: + cleaned, found = extract_directive(cleaned, key) + if found: + value = found + return cleaned, value + + +def _normalize_choice_directive(value, allowed_map): + lowered = str(value or "").strip().lower() + if not lowered: + return None + return allowed_map.get(lowered) + + +def beat_ref_mode_directive(beat): + _, value = extract_directive((beat or ""), "ref_mode") + return _normalize_choice_directive(value, _REF_MODE_DIRECTIVE_MAP) + + +beat_ref_mode_directive = lru_cache(maxsize=2048)(beat_ref_mode_directive) + + +def beat_ref_noise_aug_directive(beat): + _, value = extract_directive((beat or ""), "ref_noise_aug") + if not value: + return None + match = re.search(r"([0-9]*\.?[0-9]+)", value) + if not match: + return None + try: + parsed = float(match.group(1)) + except ValueError: + return None + return parsed if parsed >= 0 else None + + +beat_ref_noise_aug_directive = lru_cache(maxsize=2048)(beat_ref_noise_aug_directive) + + +def beat_continuity_directive(beat): + _, value = extract_directive((beat or ""), "continuity") + return _normalize_choice_directive(value, _CONTINUITY_DIRECTIVE_MAP) + + +beat_continuity_directive = lru_cache(maxsize=2048)(beat_continuity_directive) + + +def beat_override_summary(beat, shot_number): + items = [] + continuity = beat_continuity_directive(beat) + if continuity and continuity != "auto": + items.append(f"continuity {continuity}") + ref_mode = beat_ref_mode_directive(beat) + if ref_mode: + items.append(f"ref_mode {ref_mode}") + ref_noise_aug = beat_ref_noise_aug_directive(beat) + if ref_noise_aug is not None: + items.append(f"ref_noise_aug {ref_noise_aug:g}") + _, anchor_add = extract_directive((beat or ""), "anchor_add") + if anchor_add: + items.append("anchor_add") + _, soundscape = extract_directive_aliases((beat or ""), ("overall_soundscape", "soundscape")) + if soundscape: + items.append("soundscape") + _, music = extract_directive_aliases((beat or ""), ("non_diegetic_music", "music")) + if music: + items.append("music") + if not items: + return "" + return f"shot {shot_number}: " + ", ".join(items) # "walks out OF THE BARN" is emerging INTO the scene, not leaving it -- and a false @@ -3223,15 +3317,21 @@ def distribute_generations(anchor, beats, gs, music="", char_memory="", auto_war # unmatched. Checked after the loop, once the full cast is known. seen_names = {k for k in active if k} blocks = [] - for gi, b in enumerate(beats, 1): - body, wardrobe_change = extract_wardrobe((b or "").strip()) - body, _ = extract_directive(body, "seconds") # shot length, not prose - body, _ = extract_directive(body, "duration") # ditto (alias) - body, exit_directive = extract_directive(body, "exit") # explicit 'exit: Jon' - body, enter_directive = extract_directive(body, "enter") # explicit 'enter: Jon' (undo) - if enter_directive: - for nm in _entries(enter_directive): - departed.discard(_norm_name(nm)) + for gi, b in enumerate(beats, 1): + body, wardrobe_change = extract_wardrobe((b or "").strip()) + body, _ = extract_directive(body, "seconds") # shot length, not prose + body, _ = extract_directive(body, "duration") # ditto (alias) + body, exit_directive = extract_directive(body, "exit") # explicit 'exit: Jon' + body, enter_directive = extract_directive(body, "enter") # explicit 'enter: Jon' (undo) + body, shot_soundscape = extract_directive_aliases(body, ("overall_soundscape", "soundscape")) + body, shot_music = extract_directive_aliases(body, ("non_diegetic_music", "music")) + body, shot_anchor_add = extract_directive(body, "anchor_add") + body, _ = extract_directive(body, "continuity") + body, _ = extract_directive(body, "ref_mode") + body, _ = extract_directive(body, "ref_noise_aug") + if enter_directive: + for nm in _entries(enter_directive): + departed.discard(_norm_name(nm)) # Naming a departed character again is intent to have them BACK. Without # this they stayed departed, so the beat carried their bare NAME with no # description while everyone else kept theirs -- and the described character @@ -3384,13 +3484,16 @@ def distribute_generations(anchor, beats, gs, music="", char_memory="", auto_war zones_bare = [z for z, mk in marks.items() if mk in active[nm]] if zones_bare: bare_now[nm] = zones_bare - persistent = compose_persistent(body, active, anchor_id, removed, departed, count_subjects, - speaking=has_speech(body), front_load=front_load, - count_auto=count_auto, - silence_nonspeech=bool(auto_silence_nonspeech)) - # State the DIRECTION of the change, in the shot that performs it. Only for - # people actually in this shot; an anchor-prose garment is stated - # impersonally, so it summons nobody. + persistent = compose_persistent(body, active, anchor_id, removed, departed, count_subjects, + speaking=has_speech(body), front_load=front_load, + count_auto=count_auto, + silence_nonspeech=bool(auto_silence_nonspeech)) + if shot_anchor_add: + persistent = persistent.rstrip(". ") + persistent = f"{persistent}. {shot_anchor_add}".strip(". ") if persistent else shot_anchor_add + # State the DIRECTION of the change, in the shot that performs it. Only for + # people actually in this shot; an anchor-prose garment is stated + # impersonally, so it summons nobody. speak_off = [(n, it) for n, it in off_now if not n or person_referenced(body, n, active)] off_clause = takes_off_clause(speak_off, active) @@ -3492,15 +3595,22 @@ def distribute_generations(anchor, beats, gs, music="", char_memory="", auto_war persistent = persistent.rstrip(". ") + "." + clause silent_shot = no_speech and not allow_vocals block = f"[Generation {gi}] {persistent}".strip() - # A silenced shot ALWAYS gets a soundscape line. Leaving the field out is - # what let H3 improvise a voice track under a shot whose picture was already - # told to keep its mouth shut -- the babble the lips-closed clause cannot - # reach, because it only constrains the frames. - if "soundscape:" not in block.lower(): - if gs: - if silent_shot: - block += f"\noverall_soundscape: {gs}{NO_VOICE_CLAUSE}" - elif no_speech: + # A silenced shot ALWAYS gets a soundscape line. Leaving the field out is + # what let H3 improvise a voice track under a shot whose picture was already + # told to keep its mouth shut -- the babble the lips-closed clause cannot + # reach, because it only constrains the frames. + if shot_soundscape: + if silent_shot: + block += f"\noverall_soundscape: {shot_soundscape}{NO_VOICE_CLAUSE}" + elif no_speech: + block += f"\noverall_soundscape: {shot_soundscape}{NO_VOICE_SPEECH_CLAUSE}" + else: + block += f"\noverall_soundscape: {shot_soundscape}" + elif "soundscape:" not in block.lower(): + if gs: + if silent_shot: + block += f"\noverall_soundscape: {gs}{NO_VOICE_CLAUSE}" + elif no_speech: block += f"\noverall_soundscape: {gs}{NO_VOICE_SPEECH_CLAUSE}" else: block += f"\noverall_soundscape: {gs}" @@ -3508,12 +3618,14 @@ def distribute_generations(anchor, beats, gs, music="", char_memory="", auto_war block += f"\noverall_soundscape: {NO_VOICE_SOUNDSCAPE}" elif no_speech: block += f"\noverall_soundscape: {NO_VOICE_SPEECH_SOUNDSCAPE}" - # Music is OPT-IN: a blank field emits the spec's silence token N/A on every - # shot, so H3 doesn't improvise a score. (Soundscape is NOT forced to N/A -- - # per the spec it takes N/A only when total silence is explicitly wanted, so a - # blank soundscape still lets H3 provide ambient sound.) - if "non_diegetic_music:" not in block.lower(): - block += f"\nnon_diegetic_music: {music if music else 'N/A'}" + # Music is OPT-IN: a blank field emits the spec's silence token N/A on every + # shot, so H3 doesn't improvise a score. (Soundscape is NOT forced to N/A -- + # per the spec it takes N/A only when total silence is explicitly wanted, so a + # blank soundscape still lets H3 provide ambient sound.) + if shot_music: + block += f"\nnon_diegetic_music: {shot_music}" + elif "non_diegetic_music:" not in block.lower(): + block += f"\nnon_diegetic_music: {music if music else 'N/A'}" blocks.append(block.strip()) # Exits stay DEFERRED, unlike removals: a character has to be visible in the # shot that shows them leaving, and the frame they leave in is the shot's own @@ -6204,18 +6316,21 @@ class H3LongVideos: f"{seed}..{seed + len(gens) - 1} rather than one field. Stochastic detail " f"resets at every boundary, which looks like a cut in a continuous take. " f"Turn it off unless the beats are meant to look separately shot") - preflight = [("SLA", sla_note), - ("LORA HINTS", "; ".join(hint_notes)), - ("", mp_note), - ("SCHEDULE", sched_note), - ("KERNELS", kernel_note), - ("AUDIO", audio_ratio_note), - ("CONTINUITY", "; ".join(cohesion_notes)), - ("SOUND", sound_note)] - preflight_txt = "".join(f"{(lbl + ' -- ') if lbl else ''}{txt}. " - for lbl, txt in preflight if txt) - - if plan_only: + preflight = [("SLA", sla_note), + ("LORA HINTS", "; ".join(hint_notes)), + ("", mp_note), + ("SCHEDULE", sched_note), + ("KERNELS", kernel_note), + ("AUDIO", audio_ratio_note), + ("CONTINUITY", "; ".join(cohesion_notes)), + ("SOUND", sound_note)] + preflight_txt = "".join(f"{(lbl + ' -- ') if lbl else ''}{txt}. " + for lbl, txt in preflight if txt) + override_notes = [beat_override_summary(beat, index) for index, beat in enumerate(beats, 1)] + override_notes = [note for note in override_notes if note] + any_tags_anywhere = any(picture_tags(g) for g in gens) + + if plan_only: # Preview the split using THIS node's own settings -- no render, near-instant. shots = len(gens) plan_lens = (lens + [ln] * shots)[:shots] @@ -6246,17 +6361,29 @@ class H3LongVideos: # prompts and falls back to first shot when nothing is tagged -- # reporting by ref_mode alone described shots the render never gave # references to. - tagged_mode = ref_mode in ("where tagged", "auto ref2v") - if tagged_mode and any(picture_tags(g) for g in gens): - on = [n + 1 for n, g in enumerate(gens) if resolve_tagged_refs(g, ref_slots)[1]] + tagged_used = False + on = [] + effective_modes = [] + for shot_index, gen in enumerate(gens): + shot_mode = beat_ref_mode_directive(beats[shot_index] if shot_index < len(beats) else "") or ref_mode + if shot_mode in ("where tagged", "auto ref2v") and any_tags_anywhere: + if resolve_tagged_refs(gen, ref_slots)[1]: + on.append(shot_index + 1) + tagged_used = True + else: + mode_eff = ("every shot" if shot_mode == "auto ref2v" + else "first shot" if shot_mode == "where tagged" else shot_mode) + effective_modes.append(mode_eff) + if shot_references(ref_slots, mode_eff, shot_index, 1 if shot_index else None): + on.append(shot_index + 1) + if tagged_used: how = "placed by tags" else: - mode_eff = ("every shot" if ref_mode == "auto ref2v" - else "first shot" if ref_mode == "where tagged" else ref_mode) - on = [n + 1 for n in range(shots) - if shot_references(ref_slots, mode_eff, n, 1 if n else None)] + distinct_modes = list(dict.fromkeys(effective_modes)) + mode_eff = distinct_modes[0] if len(distinct_modes) == 1 else "per-shot overrides" + global_tag_mode = ref_mode in ("where tagged", "auto ref2v") how = (f"ref_mode '{mode_eff}'" - + (" -- no tags found anywhere" if tagged_mode else "")) + + (" -- no tags found anywhere" if global_tag_mode and not any_tags_anywhere else "")) src = [] if direct_ref_count: src.append(f"{direct_ref_count} direct") @@ -6272,12 +6399,14 @@ class H3LongVideos: + (f" {beats_note}." if beats_note else "") + (" ANCHOR: " + "; ".join(anchor_hazards) + "." if anchor_hazards else "") - + (f"{plan_audio}." if plan_audio else "") - + (" EXPOSURE -- " + "; ".join(wardrobe_notes) + "." - if wardrobe_notes else "") - + (f"{plan_ref}." if plan_ref else "") - + (f" {fps_note}." if fps_note else "") - + (f" {ln_note}." if ln_note else "")) + + (f"{plan_audio}." if plan_audio else "") + + (" EXPOSURE -- " + "; ".join(wardrobe_notes) + "." + if wardrobe_notes else "") + + (" OVERRIDES -- " + "; ".join(override_notes) + "." + if override_notes else "") + + (f"{plan_ref}." if plan_ref else "") + + (f" {fps_note}." if fps_note else "") + + (f" {ln_note}." if ln_note else "")) ph_img = torch.zeros((1, 64, 64, 3)) ph_audio = {"waveform": torch.zeros((1, 2, 1)), "sample_rate": 44100} # plan_only samples nothing, so there is no latent to hand out. Emit a @@ -6301,88 +6430,111 @@ class H3LongVideos: ref_missing = [] # tags naming an unconnected slot ref_carried = [] # tagged shots that kept continuity as an extra ref ref_keyframed = [] # tagged shots that kept it as a real keyframe - # 'where tagged' reads the prompt instead of counting shots. If references are - # connected but nothing is tagged anywhere, fall back to first-shot placement - # rather than silently conditioning nothing at all. - tag_mode = ref_mode in ("where tagged", "auto ref2v") - tag_driven = bool(connected_ref_count) and tag_mode and any( - picture_tags(g) for g in gens) - if connected_ref_count and tag_mode and not tag_driven: - ref_mode = "every shot" if ref_mode == "auto ref2v" else "first shot" - if cleanup_between_shots: - _deep_cleanup() # start the first (heaviest) shot with max free VRAM + ref_mode_used = [] + continuity_used = [] + ref_aug_used = [] + if cleanup_between_shots: + _deep_cleanup() # start the first (heaviest) shot with max free VRAM shot_lens = (lens + [ln] * len(gens))[:len(gens)] for i, gen_prompt in enumerate(gens): # denoise is fixed at 1.0 (partial denoise desyncs the joint AV schedule). - sa = (seed + i if vary_seed_per_shot else seed, steps, cfg, sampler_name, scheduler, 1.0) - ln_i = shot_lens[i] # this beat's own length (<= the VRAM ceiling) - # Which conditioning channels this shot carries is decided here; see - # _build_shot_conditioning for how they are packed. On ComfyUI 0.31+ a - # shot may carry BOTH references and a keyframe. - carry_keyframe = False # tagged shot keeps its handoff as a keyframe - if tag_driven: - # The prompt itself says where each reference belongs: the shot whose - # text names gets image N, renumbered to match what that - # shot actually carries. Every untagged shot keeps its handoff. + sa = (seed + i if vary_seed_per_shot else seed, steps, cfg, sampler_name, scheduler, 1.0) + ln_i = shot_lens[i] # this beat's own length (<= the VRAM ceiling) + # Which conditioning channels this shot carries is decided here; see + # _build_shot_conditioning for how they are packed. On ComfyUI 0.31+ a + # shot may carry BOTH references and a keyframe. + beat_text = beats[i] if i < len(beats) else "" + shot_mode = beat_ref_mode_directive(beat_text) or ref_mode + shot_continuity = beat_continuity_directive(beat_text) or "auto" + shot_ref_noise_aug = beat_ref_noise_aug_directive(beat_text) + shot_aug = ref_noise_aug if shot_ref_noise_aug is None else shot_ref_noise_aug + shot_tag_driven = bool(connected_ref_count) and shot_mode in ("where tagged", "auto ref2v") and any_tags_anywhere + if shot_tag_driven: + shot_mode_eff = shot_mode + else: + shot_mode_eff = ("every shot" if shot_mode == "auto ref2v" + else "first shot" if shot_mode == "where tagged" else shot_mode) + carry_keyframe = False # tagged shot keeps its handoff as a keyframe + if shot_tag_driven: + # The prompt itself says where each reference belongs: the shot whose + # text names gets image N, renumbered to match what that + # shot actually carries. Every untagged shot keeps its handoff. gen_prompt, shot_refs, dropped = resolve_tagged_refs(gen_prompt, ref_slots) - for n in dropped: - if n not in ref_missing: - ref_missing.append(n) - # A tagged shot keeps its continuity, but by which channel depends on - # ref_noise_aug. On ComfyUI 0.31+ refs and keyframes coexist, so the - # handoff can be a REAL keyframe -- it anchors the first frame, which - # is what continuity means. Once references are softened that same - # aug would soften the keyframe too, so there it falls back to riding - # as an extra reference (the pre-0.31 workaround): weaker, but it - # leaves no anchor to compromise. Appended AFTER the tagged images, so - # their numbers are untouched. - if shot_refs and handoff is not None: - if keyframe_rides_with_refs(ref_noise_aug): - carry_keyframe = True - ref_keyframed.append(i + 1) - else: - shot_refs = shot_refs + [handoff] - ref_carried.append(i + 1) - else: - shot_refs = shot_references(ref_slots, ref_mode, i, handoff) - # ComfyUI 0.31+ lets references and a keyframe ride TOGETHER, and only - # the tagged branch above was ever updated for it. Everywhere else a - # ref-conditioned shot still dropped its handoff, as 0.30 required: - # 'every shot' -> no keyframe at all, so consecutive - # shots meet as CUTS - # 'every shot + handoff ref' -> the handoff demoted to a soft - # reference ("look like this") rather - # than an anchor ("start from this") - # 'first shot', shot 0 -> the start_image was ignored outright - # In each case the last frame of a shot does not become the first frame - # of the next, which is exactly the reported symptom. - if shot_refs and handoff is not None and keyframe_rides_with_refs(ref_noise_aug): - carry_keyframe = True - ref_keyframed.append(i + 1) - # It is anchoring as a keyframe now, so the SAME frame repeated in - # the ref channel would only spend rows saying it twice -- and say - # it more weakly. - shot_refs = [r for r in shot_refs if r is not handoff] - elif (shot_refs and handoff is not None - and ref_mode == "every shot + handoff ref"): - ref_carried.append(i + 1) # softened refs: the 0.30 fallback - # A shot that follows a strip starts FRESH. Continuing from a frame that - # still shows the garment is how it reappears -- the picture outvotes the - # text every time. Costs a cut exactly where the state changes, which is - # where a cut belongs anyway. - # No scripted line -> anchor this shot's audio branch to silence. - shot_silent = bool(auto_silence_nonspeech and not allow_nonspeech_vocals and i < len(spk) and not spk[i]) - after_strip = i in strip_shots # strip_shots is 1-based, i is 0-based - shot_handoff = (None if after_strip - else handoff if (carry_keyframe or not shot_refs) else None) - if shot_refs: - ref_shots.append(i + 1) - if i == 0: - while True: - try: - frames, audio, shot_latent = self._render(model, clip, vae, audio_vae, negative, gen_prompt, w, h, ln_i, fps, tiled, sa, shot_handoff, decode_tile_frames, decode_tile_size, - shot_refs, ref_image_size, ref_noise_aug, shot_silent) + for n in dropped: + if n not in ref_missing: + ref_missing.append(n) + else: + shot_refs = shot_references(ref_slots, shot_mode_eff, i, handoff) + # A shot that follows a strip starts FRESH. Continuing from a frame that + # still shows the garment is how it reappears -- the picture outvotes the + # text every time. Costs a cut exactly where the state changes, which is + # where a cut belongs anyway. + # No scripted line -> anchor this shot's audio branch to silence. + shot_silent = bool(auto_silence_nonspeech and not allow_nonspeech_vocals and i < len(spk) and not spk[i]) + after_strip = i in strip_shots # strip_shots is 1-based, i is 0-based + if handoff is not None and shot_refs and shot_tag_driven and keyframe_rides_with_refs(shot_aug): + carry_keyframe = True + elif handoff is not None and shot_refs and shot_tag_driven: + if handoff not in shot_refs: + shot_refs = shot_refs + [handoff] + if (i + 1) not in ref_carried: + ref_carried.append(i + 1) + elif handoff is not None and shot_refs and keyframe_rides_with_refs(shot_aug): + carry_keyframe = True + shot_refs = [r for r in shot_refs if r is not handoff] + elif handoff is not None and shot_refs and shot_mode_eff == "every shot + handoff ref": + if (i + 1) not in ref_carried: + ref_carried.append(i + 1) + + if after_strip: + shot_refs = [r for r in shot_refs if r is not handoff] + shot_handoff = None + carry_keyframe = False + continuity_label = "hard cut (post-strip)" + elif shot_continuity == "hard cut": + shot_refs = [r for r in shot_refs if r is not handoff] + shot_handoff = None + carry_keyframe = False + continuity_label = "hard cut" + elif shot_continuity == "keyframe carry": + shot_refs = [r for r in shot_refs if r is not handoff] + shot_handoff = handoff + carry_keyframe = handoff is not None + continuity_label = "keyframe carry" + elif shot_continuity == "handoff ref": + if handoff is not None and shot_refs: + if handoff not in shot_refs: + shot_refs = shot_refs + [handoff] + if (i + 1) not in ref_carried: + ref_carried.append(i + 1) + shot_handoff = None + else: + shot_handoff = handoff if handoff is not None else None + carry_keyframe = False + continuity_label = "handoff ref" + elif shot_continuity == "soft carry": + shot_refs = [r for r in shot_refs if r is not handoff] + shot_handoff = handoff if not shot_refs else None + carry_keyframe = False + continuity_label = "soft carry" + else: + shot_handoff = handoff if (carry_keyframe or not shot_refs) else None + continuity_label = ("keyframe carry" if carry_keyframe else + "handoff ref" if (handoff is not None and handoff in shot_refs) else + "soft carry" if shot_handoff is not None else "hard cut") + if carry_keyframe and (i + 1) not in ref_keyframed and handoff is not None: + ref_keyframed.append(i + 1) + ref_mode_used.append(shot_mode if shot_tag_driven else shot_mode_eff) + continuity_used.append(continuity_label) + ref_aug_used.append(shot_aug) + if shot_refs: + ref_shots.append(i + 1) + if i == 0: + while True: + try: + frames, audio, shot_latent = self._render(model, clip, vae, audio_vae, negative, gen_prompt, w, h, ln_i, fps, tiled, sa, shot_handoff, decode_tile_frames, decode_tile_size, + shot_refs, ref_image_size, shot_aug, shot_silent) break except (torch.cuda.OutOfMemoryError, RuntimeError) as e: if not _is_oom(e): @@ -6396,9 +6548,9 @@ class H3LongVideos: raise RuntimeError("H3 Long Videos: not enough VRAM even at the smallest size. " "Pick a smaller resolution, close other GPU apps, or use a smaller quant.") else: - try: - frames, audio, shot_latent = self._render(model, clip, vae, audio_vae, negative, gen_prompt, w, h, ln_i, fps, tiled, sa, shot_handoff, decode_tile_frames, decode_tile_size, - shot_refs, ref_image_size, ref_noise_aug, shot_silent) + try: + frames, audio, shot_latent = self._render(model, clip, vae, audio_vae, negative, gen_prompt, w, h, ln_i, fps, tiled, sa, shot_handoff, decode_tile_frames, decode_tile_size, + shot_refs, ref_image_size, shot_aug, shot_silent) except (torch.cuda.OutOfMemoryError, RuntimeError) as e: if _is_oom(e) and getattr(e, "_h3_stage", "") == "sampling": # Retrying with tiles would re-run the whole sampling pass and @@ -6409,9 +6561,9 @@ class H3LongVideos: ) from e if not _is_oom(e) or tiled: raise - mm.soft_empty_cache(True); tiled = True; backoff.append(f"shot {i+1}: tiled") - frames, audio, shot_latent = self._render(model, clip, vae, audio_vae, negative, gen_prompt, w, h, ln_i, fps, tiled, sa, shot_handoff, decode_tile_frames, decode_tile_size, - shot_refs, ref_image_size, ref_noise_aug, shot_silent) + mm.soft_empty_cache(True); tiled = True; backoff.append(f"shot {i+1}: tiled") + frames, audio, shot_latent = self._render(model, clip, vae, audio_vae, negative, gen_prompt, w, h, ln_i, fps, tiled, sa, shot_handoff, decode_tile_frames, decode_tile_size, + shot_refs, ref_image_size, shot_aug, shot_silent) if shot_latent is not None: latent_chunks.append(shot_latent) @@ -6603,19 +6755,30 @@ class H3LongVideos: ref_note_missing = (f" named in the prompt " f"but no image is connected to that ref_image input -- the tag(s) were " f"dropped from the text") - else: - ref_note_missing = "" + else: + ref_note_missing = "" if connected_ref_count and ref_shots: kept = [n for n in range(1, len(gens) + 1) if n not in ref_shots] - ref_placement = "placed by tags" if tag_driven else f"ref_mode '{ref_mode}'" + distinct_ref_modes = list(dict.fromkeys(ref_mode_used)) + tagged_used = any(mode in ("where tagged", "auto ref2v") for mode in ref_mode_used) and any_tags_anywhere + ref_placement = ("placed by tags" if tagged_used else + f"ref_mode '{distinct_ref_modes[0]}'" if len(distinct_ref_modes) == 1 else + "mixed per-shot ref_mode") ref_source = [] if direct_ref_count: ref_source.append(f"{direct_ref_count} direct") + aug_override_notes = [ + f"{index + 1}={value:.3f}" + for index, value in enumerate(ref_aug_used) + if ref_noise_aug is not None and float(value) != float(ref_noise_aug) + ] ref_note = (f" ref2va: {connected_ref_count} reference image(s) at '{ref_image_size}' on shot(s) " f"{','.join(str(n) for n in ref_shots)} " f"({ref_placement})" + (f", ref_noise_aug {ref_noise_aug:.3f}" if ref_noise_aug is not None and float(ref_noise_aug) < 0.999 else "") + + (f"; shot-specific ref_noise_aug shot(s) {', '.join(aug_override_notes)}" + if aug_override_notes else "") + (f"; source {' + '.join(ref_source)}" if ref_source else "") + (f"; shot(s) {','.join(str(n) for n in kept)} keep the handoff" if kept else "") @@ -6630,8 +6793,10 @@ class H3LongVideos: else ", so every cut between beats is a CUT, not a continuous take") + ref_note_missing) elif connected_ref_count: + distinct_ref_modes = list(dict.fromkeys(ref_mode_used)) + mode_label = distinct_ref_modes[0] if len(distinct_ref_modes) == 1 else "mixed per-shot ref_mode" ref_note = (f" ref2va: {connected_ref_count} reference image(s) connected but ref_mode " - f"'{ref_mode}' applied them to no shot" + f"'{mode_label}' applied them to no shot" + (f" (source {direct_ref_count} direct)" if direct_ref_count else "")) else: ref_note = "" @@ -6655,11 +6820,13 @@ class H3LongVideos: + (f" SLA LoRA '{os.path.basename(str(sla_name))}' paired with sparse attention." if sla_name and sparse_on else "") + (f" {beats_note}." if beats_note else "") - + (f"{audio_note}." if audio_note else "") - + (" EXPOSURE -- " + "; ".join(wardrobe_notes) + "." - if wardrobe_notes else "") - + (f"{ref_note}." if ref_note else "") - + (f" {fps_note}." if fps_note else "") + + (f"{audio_note}." if audio_note else "") + + (" EXPOSURE -- " + "; ".join(wardrobe_notes) + "." + if wardrobe_notes else "") + + (" OVERRIDES -- " + "; ".join(override_notes) + "." + if override_notes else "") + + (f"{ref_note}." if ref_note else "") + + (f" {fps_note}." if fps_note else "") + (f" {swap_note}." if swap_note else "") + (f" free VRAM/shot: {vram_trace}." if len(vram_trace) > 1 else "") + (f" {accel_note}." if accel_note else "") diff --git a/js/h3_beat_prompt/index.js b/js/h3_beat_prompt/index.js index 22ac23d..3202969 100644 --- a/js/h3_beat_prompt/index.js +++ b/js/h3_beat_prompt/index.js @@ -8,6 +8,17 @@ const DEFAULT_W = 520; const DEFAULT_H = 340; const DEFAULT_BEAT = "Describe this beat."; const STATE_PROPERTY = "dumas_h3_beat_prompt_state"; +const CONTINUITY_OPTIONS = ["", "soft carry", "hard cut", "keyframe carry", "handoff ref"]; +const REF_MODE_OPTIONS = ["", "auto ref2v", "where tagged", "first shot", "every shot", "every shot + handoff ref"]; +const MANAGED_DIRECTIVES = { + seconds: ["seconds", "duration"], + continuity: ["continuity"], + ref_mode: ["ref_mode"], + ref_noise_aug: ["ref_noise_aug"], + anchor_add: ["anchor_add"], + overall_soundscape: ["overall_soundscape", "soundscape"], + non_diegetic_music: ["non_diegetic_music", "music"], +}; const DIRECTIVE_EXAMPLES = [ ["wardrobe set", "wardrobe: Maya = grey shorts, red jacket"], ["wardrobe add", "wardrobe: Maya += red jacket"], @@ -15,6 +26,10 @@ const DIRECTIVE_EXAMPLES = [ ["seconds", "seconds: 8"], ["exit", "exit: Maya"], ["enter", "enter: Jon"], + ["continuity", "continuity: hard cut"], + ["ref_mode", "ref_mode: every shot"], + ["ref_noise_aug", "ref_noise_aug: 0.92"], + ["anchor_add", "anchor_add: harsh sodium-vapor spill, wet pavement, long-lens compression"], ["overall_soundscape", "overall_soundscape: soft rain, distant traffic"], ["non_diegetic_music", "non_diegetic_music: tense analog synth pulse"], ["soundscape", "soundscape: fluorescent room tone, faint HVAC hum"], @@ -113,6 +128,40 @@ function injectCSS() { flex-wrap: wrap; gap: 6px; } + .dh3bp-controls { + display: grid; + grid-template-columns: repeat(2, minmax(0, 1fr)); + gap: 8px; + margin-bottom: 10px; + } + .dh3bp-control { + display: flex; + flex-direction: column; + gap: 4px; + } + .dh3bp-control-wide { + grid-column: 1 / -1; + } + .dh3bp-control-label { + color: #b8bec8; + font-size: 10px; + text-transform: uppercase; + letter-spacing: 0.05em; + } + .dh3bp-input, + .dh3bp-select { + width: 100%; + box-sizing: border-box; + background: #121418; + color: #e8e8e8; + border: 1px solid #3c414a; + border-radius: 8px; + padding: 7px 9px; + font: 12px/1.35 "Segoe UI", sans-serif; + } + .dh3bp-input::placeholder { + color: #7f8793; + } .dh3bp-directive { background: #252a33; color: #d4d9e1; @@ -202,6 +251,41 @@ function appendDirectiveText(currentText, example) { return `${trimmed}\n${example}`; } +function splitBeatLines(text) { + return String(text || "").split(/\n/); +} + +function readDirectiveValue(text, directiveNames) { + let value = ""; + for (const line of splitBeatLines(text)) { + const trimmed = line.trim(); + for (const name of directiveNames) { + const lower = name.toLowerCase(); + if (trimmed.toLowerCase().startsWith(`${lower}:`)) { + value = trimmed.slice(trimmed.indexOf(":") + 1).trim(); + } + } + } + return value; +} + +function stripDirectiveValues(text, directiveNames) { + const lowered = directiveNames.map((name) => name.toLowerCase()); + const kept = splitBeatLines(text).filter((line) => { + const trimmed = line.trim().toLowerCase(); + return !lowered.some((name) => trimmed.startsWith(`${name}:`)); + }); + return kept.join("\n").replace(/\n{3,}/g, "\n\n").trim(); +} + +function setDirectiveValue(text, canonicalName, directiveNames, value) { + const cleaned = stripDirectiveValues(text, directiveNames); + const trimmedValue = String(value || "").trim(); + if (!trimmedValue) return cleaned; + const directiveLine = `${canonicalName}: ${trimmedValue}`; + return cleaned ? `${directiveLine}\n${cleaned}` : directiveLine; +} + function renderUI(node) { const ui = node._dh3bpUI; if (!ui) return; @@ -246,6 +330,108 @@ function renderUI(node) { }); textarea.addEventListener("keydown", (event) => event.stopImmediatePropagation()); + const applyTextUpdate = (nextText) => { + const next = readState(node); + next.beats[index].text = nextText; + writeState(node, next); + textarea.value = nextText; + updateTextareaHeight(textarea); + }; + + const controls = document.createElement("div"); + controls.className = "dh3bp-controls"; + + const buildField = ({ labelText, className = "", input }) => { + const wrap = document.createElement("label"); + wrap.className = `dh3bp-control ${className}`.trim(); + const labelEl = document.createElement("div"); + labelEl.className = "dh3bp-control-label"; + labelEl.textContent = labelText; + wrap.append(labelEl, input); + return wrap; + }; + + const secondsInput = document.createElement("input"); + secondsInput.className = "dh3bp-input"; + secondsInput.type = "text"; + secondsInput.placeholder = "8"; + secondsInput.value = readDirectiveValue(beat.text, MANAGED_DIRECTIVES.seconds); + secondsInput.addEventListener("input", () => { + applyTextUpdate(setDirectiveValue(textarea.value, "seconds", MANAGED_DIRECTIVES.seconds, secondsInput.value)); + }); + + const continuitySelect = document.createElement("select"); + continuitySelect.className = "dh3bp-select"; + CONTINUITY_OPTIONS.forEach((value) => { + const option = document.createElement("option"); + option.value = value; + option.textContent = value || "Default"; + continuitySelect.appendChild(option); + }); + continuitySelect.value = readDirectiveValue(beat.text, MANAGED_DIRECTIVES.continuity); + continuitySelect.addEventListener("change", () => { + applyTextUpdate(setDirectiveValue(textarea.value, "continuity", MANAGED_DIRECTIVES.continuity, continuitySelect.value)); + }); + + const refModeSelect = document.createElement("select"); + refModeSelect.className = "dh3bp-select"; + REF_MODE_OPTIONS.forEach((value) => { + const option = document.createElement("option"); + option.value = value; + option.textContent = value || "Global"; + refModeSelect.appendChild(option); + }); + refModeSelect.value = readDirectiveValue(beat.text, MANAGED_DIRECTIVES.ref_mode); + refModeSelect.addEventListener("change", () => { + applyTextUpdate(setDirectiveValue(textarea.value, "ref_mode", MANAGED_DIRECTIVES.ref_mode, refModeSelect.value)); + }); + + const refNoiseInput = document.createElement("input"); + refNoiseInput.className = "dh3bp-input"; + refNoiseInput.type = "text"; + refNoiseInput.placeholder = "0.95"; + refNoiseInput.value = readDirectiveValue(beat.text, MANAGED_DIRECTIVES.ref_noise_aug); + refNoiseInput.addEventListener("input", () => { + applyTextUpdate(setDirectiveValue(textarea.value, "ref_noise_aug", MANAGED_DIRECTIVES.ref_noise_aug, refNoiseInput.value)); + }); + + const anchorInput = document.createElement("input"); + anchorInput.className = "dh3bp-input"; + anchorInput.type = "text"; + anchorInput.placeholder = "extra per-shot style treatment"; + anchorInput.value = readDirectiveValue(beat.text, MANAGED_DIRECTIVES.anchor_add); + anchorInput.addEventListener("input", () => { + applyTextUpdate(setDirectiveValue(textarea.value, "anchor_add", MANAGED_DIRECTIVES.anchor_add, anchorInput.value)); + }); + + const soundscapeInput = document.createElement("input"); + soundscapeInput.className = "dh3bp-input"; + soundscapeInput.type = "text"; + soundscapeInput.placeholder = "faint traffic, loose sign rattle"; + soundscapeInput.value = readDirectiveValue(beat.text, MANAGED_DIRECTIVES.overall_soundscape); + soundscapeInput.addEventListener("input", () => { + applyTextUpdate(setDirectiveValue(textarea.value, "overall_soundscape", MANAGED_DIRECTIVES.overall_soundscape, soundscapeInput.value)); + }); + + const musicInput = document.createElement("input"); + musicInput.className = "dh3bp-input"; + musicInput.type = "text"; + musicInput.placeholder = "low pulsing synth tension"; + musicInput.value = readDirectiveValue(beat.text, MANAGED_DIRECTIVES.non_diegetic_music); + musicInput.addEventListener("input", () => { + applyTextUpdate(setDirectiveValue(textarea.value, "non_diegetic_music", MANAGED_DIRECTIVES.non_diegetic_music, musicInput.value)); + }); + + controls.append( + buildField({ labelText: "Seconds", input: secondsInput }), + buildField({ labelText: "Continuity", input: continuitySelect }), + buildField({ labelText: "Ref Mode", input: refModeSelect }), + buildField({ labelText: "Ref Noise Aug", input: refNoiseInput }), + buildField({ labelText: "Anchor Add", className: "dh3bp-control-wide", input: anchorInput }), + buildField({ labelText: "Shot Soundscape", className: "dh3bp-control-wide", input: soundscapeInput }), + buildField({ labelText: "Shot Music", className: "dh3bp-control-wide", input: musicInput }), + ); + const directives = document.createElement("div"); directives.className = "dh3bp-directives"; @@ -256,15 +442,13 @@ function renderUI(node) { button.textContent = labelText; button.title = example; button.addEventListener("click", () => { - const next = readState(node); - next.beats[index].text = appendDirectiveText(next.beats[index].text, example); - writeState(node, next); + applyTextUpdate(appendDirectiveText(textarea.value, example)); renderUI(node); }); directives.appendChild(button); }); - card.append(head, textarea, directives); + card.append(head, textarea, controls, directives); ui.list.appendChild(card); updateTextareaHeight(textarea); }); @@ -288,7 +472,7 @@ function setupNode(node) { title.textContent = "Beat Prompt Builder"; const subtitle = document.createElement("div"); subtitle.className = "dh3bp-subtitle"; - subtitle.textContent = "One textbox per H3 beat. Use the buttons for seconds and quick wardrobe set/add/remove syntax."; + subtitle.textContent = "One textbox per H3 beat, plus per-shot controls for timing, ref behavior, continuity, anchor adds, and audio directives."; titleWrap.append(title, subtitle); const addButton = document.createElement("button"); diff --git a/tests/test_dumas_h3_longvideos.py b/tests/test_dumas_h3_longvideos.py index 04e4fee..8411742 100644 --- a/tests/test_dumas_h3_longvideos.py +++ b/tests/test_dumas_h3_longvideos.py @@ -141,6 +141,51 @@ class DumasH3LongVideosHelperTests(unittest.TestCase): self.assertGreater(action_fn.cache_info().hits, 0) self.assertGreater(estimate_fn.cache_info().hits, 0) + def test_per_shot_directive_helpers_parse_new_controls(self): + beat = ( + "ref_mode: every shot + handoff ref\n" + "ref_noise_aug: 0.87\n" + "continuity: keyframe carry\n" + "The courier waits under the sign." + ) + + self.assertEqual( + self.module.beat_ref_mode_directive(beat), + "every shot + handoff ref", + ) + self.assertEqual(self.module.beat_ref_noise_aug_directive(beat), 0.87) + self.assertEqual( + self.module.beat_continuity_directive(beat), + "keyframe carry", + ) + + def test_distribute_generations_canonicalizes_per_shot_audio_and_anchor_directives(self): + generations = self.module.distribute_generations( + "", + [ + "anchor_add: harsh sodium spill, wet asphalt reflections\n" + "soundscape: distant traffic hiss, loose sign rattle\n" + "music: low pulsing synth tension\n" + "continuity: hard cut\n" + "ref_mode: every shot\n" + "ref_noise_aug: 0.88\n" + "A courier waits under the streetlight." + ], + "global rain", + "global score", + ) + + block = generations[0] + self.assertIn("harsh sodium spill, wet asphalt reflections", block) + self.assertIn("overall_soundscape: distant traffic hiss, loose sign rattle", block) + self.assertIn("non_diegetic_music: low pulsing synth tension", block) + self.assertNotIn("\nsoundscape:", block) + self.assertNotIn("\nmusic:", block) + self.assertNotIn("\ncontinuity:", block) + self.assertNotIn("\nref_mode:", block) + self.assertNotIn("\nref_noise_aug:", block) + self.assertNotIn("\nanchor_add:", block) + def test_has_speech_cache_respects_written_text_filter(self): fn = self.module.has_speech fn.cache_clear()