From a6afdeab38998c8a0a397225c7e9768f9e579264 Mon Sep 17 00:00:00 2001 From: Chris Dumas Date: Thu, 27 Aug 2026 06:21:26 +0000 Subject: [PATCH] Expose per-beat H3 clip outputs --- dumas_h3_longvideos.py | 27 +++++++++++++++------------ tests/test_dumas_h3_longvideos.py | 11 +++++++++++ 2 files changed, 26 insertions(+), 12 deletions(-) diff --git a/dumas_h3_longvideos.py b/dumas_h3_longvideos.py index e900518..ef4b79f 100644 --- a/dumas_h3_longvideos.py +++ b/dumas_h3_longvideos.py @@ -5448,7 +5448,7 @@ def _evict_all_but(keep_model): class H3LongVideos: CATEGORY = "Dumas/MiniMax" - FUNCTION = "run" + FUNCTION = "run" # fps is emitted as BOTH types on purpose: ComfyUI does not coerce between them, # and the nodes that want a frame rate are split -- CreateVideo / SaveWEBM / # VHS Video Combine take a FLOAT, while plenty of utility nodes take an INT. @@ -5458,11 +5458,13 @@ class H3LongVideos: # output. Inserting mid-list would silently re-target them. # APPEND to these, never insert. A workflow stores an output link by SLOT INDEX, # so a new type in the middle silently re-points every link after it. - RETURN_TYPES = ("IMAGE", "AUDIO", "STRING", "STRING", "INT", "INT", "INT", "FLOAT", "FLOAT", "INT", - "LATENT", "STRING") - RETURN_NAMES = ("images", "audio", "info", "script", "frames_per_shot", "total_frames", - "shots", "video_seconds", "fps", "fps_int", - "latent", "soundscape") + RETURN_TYPES = ("IMAGE", "AUDIO", "STRING", "STRING", "INT", "INT", "INT", "FLOAT", "FLOAT", "INT", + "LATENT", "STRING", "IMAGE", "AUDIO") + RETURN_NAMES = ("images", "audio", "info", "script", "frames_per_shot", "total_frames", + "shots", "video_seconds", "fps", "fps_int", + "latent", "soundscape", "beat_images", "beat_audio") + OUTPUT_IS_LIST = (False, False, False, False, False, False, False, False, False, False, + False, False, True, True) @classmethod def IS_CHANGED(cls, plan_only=False, **kwargs): @@ -6413,9 +6415,9 @@ class H3LongVideos: # correctly-SHAPED empty one rather than None: a downstream LATENT input # would choke on None, and this keeps the preview wireable exactly like # a real run. - return (ph_img, ph_audio, plan, "\n---\n".join(gens), max(plan_lens), - sum(plan_lens), shots, total, float(fps), int(fps), - _empty_av_latent(w, h, 5, fps)[0], global_soundscape) + return (ph_img, ph_audio, plan, "\n---\n".join(gens), max(plan_lens), + sum(plan_lens), shots, total, float(fps), int(fps), + _empty_av_latent(w, h, 5, fps)[0], global_soundscape, [], []) spk = speech_flags(beats) # which shots have real (quoted) dialogue vram_trace = [] # free VRAM after each shot @@ -6841,9 +6843,10 @@ class H3LongVideos: # reassigns it, so this is the derived bed when auto_soundscape fired and # your own text when it did not. Emitting it means you can read what was # generated, and feed it straight back into the widget-input to pin it. - return (all_frames, {"waveform": all_audio, "sample_rate": sr}, info, script, - max(shot_lens), all_frames.shape[0], len(gens), round(actual, 2), - float(fps), int(fps), latent_out, global_soundscape) + return (all_frames, {"waveform": all_audio, "sample_rate": sr}, info, script, + max(shot_lens), all_frames.shape[0], len(gens), round(actual, 2), + float(fps), int(fps), latent_out, global_soundscape, + video_chunks, [{"waveform": chunk, "sample_rate": sr} for chunk in audio_chunks]) # The old FL2VA / REF2VA aliases were only alternate menu entries for the same class. diff --git a/tests/test_dumas_h3_longvideos.py b/tests/test_dumas_h3_longvideos.py index 8411742..f05722d 100644 --- a/tests/test_dumas_h3_longvideos.py +++ b/tests/test_dumas_h3_longvideos.py @@ -276,6 +276,16 @@ class DumasH3LongVideosHelperTests(unittest.TestCase): for index in range(1, 10): self.assertIn(f"ref_{index}", optional) + def test_node_appends_per_beat_list_outputs_without_reordering_existing_slots(self): + self.assertEqual( + self.module.H3LongVideos.RETURN_NAMES[-2:], + ("beat_images", "beat_audio"), + ) + self.assertEqual( + self.module.H3LongVideos.OUTPUT_IS_LIST[-2:], + (True, True), + ) + def test_reference_context_matches_character_names_and_location_tags(self): refs = [ {"kind": "character", "image": "img1", "name": "Mara", "description": "silver hair", "wardrobe": "red jacket"}, @@ -456,6 +466,7 @@ class DumasH3LongVideosHelperTests(unittest.TestCase): ) self.assertIn("2 beat(s)", result[2]) self.assertIn("2 shot(s)", result[2]) + self.assertEqual(result[-2:], ([], [])) finally: module.torch = original_torch module.vram_gb = original_vram_gb