Refine H3 long videos defaults and aliases

This commit is contained in:
2026-08-26 15:20:45 +00:00
parent 18129327d9
commit 0b20381c5c
3 changed files with 168 additions and 63 deletions
+2
View File
@@ -38,8 +38,10 @@
- Outputs: `images`, `audio`, `info`, `script`, `frames_per_shot`, `total_frames`, `shots`, `video_seconds`, `fps`, `fps_int`, `latent`, `soundscape`
- First-pass Dumas port of the `MiniMax-H3-Longvideos` sampler, brought in as a local starting point for long-form H3 chaining work.
- Keeps the upstream split-beats / handoff / ref-routing behavior close to source so future Dumas-specific improvements can be compared against a known baseline.
- Only the canonical `DumasH3LongVideos` node key is exposed now; the older FL2VA/REF2VA alias entries are no longer duplicated in the Add Node menu.
- Prompt `<Picture N>` tags now map to the actual ref socket numbers you wire, even with gaps such as only `ref_image_2` and `ref_image_7` connected.
- A connected H3 plan can now supply the current scenes 9-image bundle directly; any directly-wired `ref_image_*` socket overrides the same numbered plan slot.
- The default ref2v bias is now stronger: `ref_mode` defaults to `auto ref2v` so untagged prompts condition every shot instead of only shot 1, and `ref_noise_aug` defaults to `0.95` rather than the upstream-literal `0.999`.
- `Dumas H3 Shot Length`
- Inputs: `shot_seconds`, `fps`, optional `cap_to_h3_max`
+72 -40
View File
@@ -1550,7 +1550,7 @@ def compose_persistent(body, active, anchor_id, removed=None, departed=None,
speakers = _speakers_in(body, names)
if speakers:
bound_names = set(refs) or (
{n for n in names} if _PLURAL_CAST.search(body) else set())
{n for n in names} if _plural_cast_matches_present(body, len(names)) else set())
if speakers & bound_names:
listeners = bound_names - speakers
@@ -1561,7 +1561,7 @@ def compose_persistent(body, active, anchor_id, removed=None, departed=None,
return speaking and n not in listeners
roll_call = ""
if not refs and len(names) > 1 and _PLURAL_CAST.search(body):
if not refs and _plural_cast_matches_present(body, len(names)):
bits = []
for n in names:
desc = ", ".join(_clean_items(active[n], n, drop_mouth_state=_drop_mouth(n)))
@@ -2315,7 +2315,41 @@ NO_VOICE_SPEECH_CLAUSE = ", no speech, no dialogue, no talking, no singing, no w
# singular pronoun, which is exactly the scenery-beat case that must stay empty.
_PLURAL_CAST = re.compile(
r"\b(?:they|themselves|both|each other|one another|"
r"the two of them|all of them)\b", re.I)
r"the two of them|the two characters|the three characters|"
r"the four characters|all of them|all three|all four)\b", re.I)
_PLURAL_COUNT_PATTERNS = (
(re.compile(r"\b(?:both|each other|the two of them|the two characters)\b", re.I), 2),
(re.compile(r"\b(?:the three characters|all three)\b", re.I), 3),
(re.compile(r"\b(?:the four characters|all four)\b", re.I), 4),
(re.compile(r"\ball of them\b", re.I), "all"),
# Bare "they"/"themselves"/"one another" is only safe when exactly two tracked
# people are active; with three or more it is ambiguous and should not summon
# the whole cast into the shot.
(re.compile(r"\b(?:they|themselves|one another)\b", re.I), "ambiguous"),
)
def _plural_cast_matches_present(body, present_count):
"""Does this beat unambiguously refer to the whole currently-active cast?
Plural wording used to pull EVERY tracked character from character_memory into a
shot whenever the beat said "they" or "both of them". That is only safe when
the count implied by the words matches the active cast exactly. Otherwise the
plural is ambiguous and must not be expanded into a full roll-call."""
text = body or ""
present_count = max(0, int(present_count or 0))
if present_count < 2:
return False
for rx, target in _PLURAL_COUNT_PATTERNS:
if not rx.search(text):
continue
if target == "all":
return True
if target == "ambiguous":
return present_count == 2
return present_count == target
return False
def person_referenced(body, name, active):
@@ -2351,7 +2385,7 @@ def person_in_shot(body, name, active, departed=()):
if person_referenced(body, name, active):
return True
present = [n for n in (active or {}) if n and n not in (departed or ())]
return len(present) > 1 and bool(_PLURAL_CAST.search(body or ""))
return _plural_cast_matches_present(body, len(present))
def _subject_term(name, active):
@@ -4074,6 +4108,10 @@ def shot_references(ref_list, ref_mode, shot_index, handoff):
mode contributes to the REFERENCE channel; they no longer describe a shot's whole
conditioning, and 'no handoff at all' is no longer a consequence of picking one:
'auto ref2v' -- use explicit prompt tags when they exist; otherwise carry
the references on every shot. This is the ref2v-biased
default: identity first, no need to tag a single-subject
chain by hand just to stop refs collapsing to shot 1.
'first shot' -- references establish the cast in shot 1; every later shot
uses the last-frame handoff. Continuity is unbroken and the
look propagates down the chain, but only through the frames.
@@ -4089,6 +4127,8 @@ def shot_references(ref_list, ref_mode, shot_index, handoff):
refs = _connected_refs(ref_list)
if not refs:
return []
if ref_mode == "auto ref2v":
return list(refs)
if ref_mode == "first shot":
return list(refs) if shot_index == 0 else []
if ref_mode == "every shot":
@@ -5481,28 +5521,27 @@ class H3LongVideos:
"overlay_stroke": ("INT", {"default": 0, "min": 0, "max": 20,
"tooltip": "Black outline thickness in pixels around the white text. 0 keeps it pure "
"white as asked; 2-3 makes it survive a bright sky or a white wall."}),
"ref_mode": (["where tagged", "first shot", "every shot", "every shot + handoff ref"],
{"default": "where tagged",
"tooltip": "Which shots the ref_image inputs condition. A shot carries EITHER "
"references or the last-frame handoff, never both. 'where tagged' "
"(default): write <Picture 1> in the beat where that character "
"appears and ONLY that shot gets the reference -- every other shot "
"keeps its handoff. This is the precise option: the other modes go by "
"shot NUMBER and are blind to who is actually in the shot, so a "
"character who first appears in shot 2 gets nothing while an empty "
"establishing shot 1 gets a portrait pushed into it. Tags are "
"renumbered per shot, so <Picture 2> alone still resolves. With refs "
"connected but no tags anywhere, falls back to first shot rather than "
"silently doing nothing. 'first shot' / 'every shot' / 'every shot + "
"handoff ref' go purely by position. Ignored when no ref_image is "
"connected."}),
"ref_noise_aug": ("FLOAT", {"default": 0.999, "min": 0.50, "max": 1.0, "step": 0.005,
"ref_mode": (["auto ref2v", "where tagged", "first shot", "every shot", "every shot + handoff ref"],
{"default": "auto ref2v",
"tooltip": "Which shots the ref_image inputs condition. 'auto ref2v' (default) is "
"the reference-to-video bias: if the prompt uses <Picture N> tags, those "
"tags decide which shot gets which ref; if there are NO tags anywhere, the "
"node conditions EVERY shot with the connected refs rather than collapsing "
"them to shot 1. That is the better default for single-subject ref2v and "
"for long chains where identity drift matters more than strict per-shot "
"routing. 'where tagged' keeps the old strict behavior, including the "
"first-shot fallback when no tags are found. Tags are renumbered per shot, "
"so <Picture 2> alone still resolves. 'first shot' / 'every shot' / "
"'every shot + handoff ref' go purely by position. Ignored when no "
"ref_image is connected."}),
"ref_noise_aug": ("FLOAT", {"default": 0.95, "min": 0.50, "max": 1.0, "step": 0.005,
"tooltip": "How CLEAN each reference is presented to the model. 0.999 (H3's own "
"default) hands it a finished, noise-free image -- which invites the "
"model to REPRODUCE the reference in the opening frames instead of just "
"taking an identity from it. Lower values blend the condition with "
"noise and label it as approximate, so it informs the face without "
"being copied: try 0.95, then 0.90. Too low (below ~0.8) and the "
"being copied. 0.95 is the ref2v-biased default here; 0.999 keeps the "
"upstream literal-reference behavior. Too low (below ~0.8) and the "
"reference stops holding identity at all. Applies ONLY to "
"ref-conditioned shots -- the last-frame handoff is never weakened, or "
"continuity would break."}),
@@ -6111,15 +6150,17 @@ class H3LongVideos:
# prompts and falls back to first shot when nothing is tagged --
# reporting by ref_mode alone described shots the render never gave
# references to.
if ref_mode == "where tagged" and any(picture_tags(g) for g in gens):
tagged_mode = ref_mode in ("where tagged", "auto ref2v")
if tagged_mode and any(picture_tags(g) for g in gens):
on = [n + 1 for n, g in enumerate(gens) if resolve_tagged_refs(g, ref_slots)[1]]
how = "placed by <Picture N> tags"
else:
mode_eff = "first shot" if ref_mode == "where tagged" else ref_mode
mode_eff = ("every shot" if ref_mode == "auto ref2v"
else "first shot" if ref_mode == "where tagged" else ref_mode)
on = [n + 1 for n in range(shots)
if shot_references(ref_slots, mode_eff, n, 1 if n else None)]
how = (f"ref_mode '{mode_eff}'"
+ (" -- no tags found anywhere" if ref_mode == "where tagged" else ""))
+ (" -- no tags found anywhere" if tagged_mode else ""))
src = []
if direct_ref_count:
src.append(f"{direct_ref_count} direct")
@@ -6169,10 +6210,11 @@ class H3LongVideos:
# 'where tagged' reads the prompt instead of counting shots. If references are
# connected but nothing is tagged anywhere, fall back to first-shot placement
# rather than silently conditioning nothing at all.
tag_driven = bool(connected_ref_count) and ref_mode == "where tagged" and any(
tag_mode = ref_mode in ("where tagged", "auto ref2v")
tag_driven = bool(connected_ref_count) and tag_mode and any(
picture_tags(g) for g in gens)
if connected_ref_count and ref_mode == "where tagged" and not tag_driven:
ref_mode = "first shot"
if connected_ref_count and tag_mode and not tag_driven:
ref_mode = "every shot" if ref_mode == "auto ref2v" else "first shot"
if cleanup_between_shots:
_deep_cleanup() # start the first (heaviest) shot with max free VRAM
@@ -6546,23 +6588,13 @@ class H3LongVideos:
float(fps), int(fps), latent_out, global_soundscape)
# REF2VA registers under its OWN key. The FL2VA pack one directory up keeps
# "H3LongVideosFL2VA" and the legacy "H3LongVideosV1" alias; ComfyUI builds one
# flat registry, so repeating either here would silently overwrite that node --
# same name in the search, no way to tell which copy a workflow is running.
# ONE node, three registration keys. ComfyUI stores the key verbatim in every saved
# workflow, so all three must keep resolving or existing graphs load as red "missing
# node" boxes: "H3LongVideosV1" is the original name, "H3LongVideosFL2VA" the rename,
# and "H3LongVideosREF2VA" the separate reference node that has now been folded in.
# They are aliases onto the same class -- there is no second implementation.
# The old FL2VA / REF2VA aliases were only alternate menu entries for the same class.
# Dumas workflows now use the canonical "DumasH3LongVideos" key, so expose a single
# node entry instead of triplicating the search results with duplicate aliases.
NODE_CLASS_MAPPINGS = {
"DumasH3LongVideos": H3LongVideos,
"DumasH3LongVideosFL2VA": H3LongVideos,
"DumasH3LongVideosREF2VA": H3LongVideos,
}
NODE_DISPLAY_NAME_MAPPINGS = {
"DumasH3LongVideos": "Dumas H3 Long Videos (FL2VA + REF2VA)",
"DumasH3LongVideosFL2VA": "Dumas H3 Long Videos (FL2VA + REF2VA)",
"DumasH3LongVideosREF2VA": "Dumas H3 Long Videos (FL2VA + REF2VA)",
}
__all__ = ["NODE_CLASS_MAPPINGS", "NODE_DISPLAY_NAME_MAPPINGS"]
+71
View File
@@ -183,6 +183,10 @@ class DumasH3LongVideosHelperTests(unittest.TestCase):
def test_shot_references_uses_all_connected_sparse_slots(self):
refs = [None, "img2", None, "img4", None, None, "img7", None, None]
self.assertEqual(
self.module.shot_references(refs, "auto ref2v", 0, None),
["img2", "img4", "img7"],
)
self.assertEqual(
self.module.shot_references(refs, "first shot", 0, None),
["img2", "img4", "img7"],
@@ -225,6 +229,73 @@ class DumasH3LongVideosHelperTests(unittest.TestCase):
("plan1", "direct2", "plan3", None, "direct5", None, "plan7", None, None),
)
def test_ref_mode_defaults_are_ref2v_biased(self):
optional = self.module.H3LongVideos.INPUT_TYPES()["optional"]
self.assertEqual(optional["ref_mode"][1]["default"], "auto ref2v")
self.assertEqual(optional["ref_noise_aug"][1]["default"], 0.95)
def test_only_canonical_h3_long_videos_node_is_exposed(self):
self.assertEqual(
self.module.NODE_CLASS_MAPPINGS,
{"DumasH3LongVideos": self.module.H3LongVideos},
)
self.assertEqual(
self.module.NODE_DISPLAY_NAME_MAPPINGS,
{"DumasH3LongVideos": "Dumas H3 Long Videos (FL2VA + REF2VA)"},
)
def test_compose_persistent_does_not_expand_ambiguous_plural_to_full_cast(self):
active = self.module.parse_wardrobe(
"Maya = she, red jacket\n"
"Jon = he, navy overalls\n"
"Becca = she, green coat"
)
shot = self.module.compose_persistent(
"Both of them walk to the door.",
active,
"",
speaking=False,
)
self.assertEqual(shot, "Both of them walk to the door.")
def test_compose_persistent_keeps_two_person_plural_binding(self):
active = self.module.parse_wardrobe(
"Maya = she, red jacket\n"
"Jon = he, navy overalls"
)
shot = self.module.compose_persistent(
"They walk to the door.",
active,
"",
speaking=False,
)
self.assertIn("Maya (red jacket)", shot)
self.assertIn("Jon (navy overalls)", shot)
self.assertIn("They walk to the door.", shot)
def test_compose_persistent_all_three_characters_binds_full_cast(self):
active = self.module.parse_wardrobe(
"Maya = she, red jacket\n"
"Jon = he, navy overalls\n"
"Becca = she, green coat"
)
shot = self.module.compose_persistent(
"The three characters walk to the door.",
active,
"",
speaking=False,
)
self.assertIn("Maya (red jacket)", shot)
self.assertIn("Jon (navy overalls)", shot)
self.assertIn("Becca (green coat)", shot)
if __name__ == "__main__":
unittest.main()