diff --git a/H3_LONG_VIDEOS_GUIDE.md b/H3_LONG_VIDEOS_GUIDE.md index e2847b0..80b78d2 100644 --- a/H3_LONG_VIDEOS_GUIDE.md +++ b/H3_LONG_VIDEOS_GUIDE.md @@ -944,23 +944,23 @@ Character wardrobe seeding: What the long-videos code actually uses: - `image`: the real location reference image H3 sees when that slot is selected for a shot -- `description`: injected as location context when that slot is explicitly tagged -- `general`: injected as location notes when that slot is explicitly tagged -- `name`, `location_id`, `alias`: used mainly as labels in summaries and injected sentences, not for automatic prose matching +- `description`: injected as location context when that location ref is matched +- `general`: injected as location notes when that location ref is matched +- `name`, `location_id`, `alias`: used for location-name matching in beat text and as labels in injected sentences Important difference from character refs: -- location refs are **not** auto-matched from location names in beat prose -- if you write `in the cafe` and do not tag ``, the location image may still be in the shot because of `ref_mode`, but the location text metadata is not auto-injected just from that name +- location refs can now be matched from location names or aliases in beat prose +- explicit `` tags are still the strongest way to force an exact slot on an exact beat ### When location data gets used Location node data has two separate paths: 1. The location image can be attached to shots through normal ref routing. -2. The location text metadata is only injected when you explicitly tag that slot with ``. +2. The location text metadata is injected when the beat tags that slot with `` or names the location by `name`, `id`, or `alias`. -So location references are more tag-driven than character references. +So location references are no longer tag-only, but tags are still the safest exact-routing tool. Simple usage patterns: @@ -972,7 +972,8 @@ What to remember: - tags control explicit slot placement - names can pull matching character refs into the real image-conditioning list -- location refs are often tagged because they are about where the shot happens +- names can now pull matching location refs into the real image-conditioning list too +- location refs are still often tagged because they are about where the shot happens - character refs are often named because they are about who is in the shot ### Scenarios @@ -1021,12 +1022,13 @@ Setup: What happens: -- with `auto ref2v` or `every shot`, the cafe image may still be conditioning the shot -- but the location `description` and `general` fields are not auto-injected from the word `cafe` +- the word `cafe` can now match the location ref by name or alias +- that means the location image and location text context can both be pulled in even without a tag +- `` is still better if you want exact manual slot routing on that specific beat What to do: -- add `` on the beat where the location wording matters +- add `` when you want to force that exact location slot on that exact beat ### Scenario: character face is partly right, but identity is weak diff --git a/dumas_h3_longvideos.py b/dumas_h3_longvideos.py index 2220eaa..125f789 100644 --- a/dumas_h3_longvideos.py +++ b/dumas_h3_longvideos.py @@ -3288,9 +3288,10 @@ def annotate_script_debug(gens, anatomy_shots, anatomy_mode, ref_slots): def annotate_script_refs(gens, ref_slots): """Per-shot reference routing summary for the script socket.""" lines = [] + normalized_slots = _normalized_ref_slots(ref_slots) for shot_index, gen in enumerate(gens or [], 1): - tagged = _slot_refs_for_text(gen, ref_slots) - named = _named_character_refs_for_text(gen, ref_slots) + tagged = _slot_refs_for_text(gen, normalized_slots) + named = _named_refs_for_text(gen, normalized_slots, kinds=("character", "location")) seen = set() merged = [] for slot_number, ref in tagged: @@ -4282,7 +4283,10 @@ def _reference_slot(ref, slot_index=None): def _reference_image(ref): try: - normalized = _reference_slot(ref) + if isinstance(ref, dict): + normalized = _image_nodes.normalize_reference(ref, allow_image_fallback=False) + else: + normalized = _reference_slot(ref) except Exception: return None return normalized.get("image") @@ -4372,25 +4376,39 @@ def _reference_name_keys(ref): return out +def _normalized_ref_slots(ref_slots): + """Normalize every connected ref once so downstream helpers can reuse them.""" + out = [] + for slot_number, raw in enumerate(ref_slots or [], 1): + if raw is None: + out.append(None) + elif isinstance(raw, dict) and "image" in raw and raw.get("image") is None: + out.append(_image_nodes.normalize_reference(raw, picture_id=slot_number, allow_image_fallback=False)) + else: + out.append(_reference_slot(raw, slot_number)) + return tuple(out) + + def _slot_refs_for_text(text, ref_slots): refs = [] for slot_number in picture_tags(text): if not (1 <= slot_number <= len(ref_slots or [])): continue - raw = ref_slots[slot_number - 1] - ref = _reference_slot(raw, slot_number) + ref = ref_slots[slot_number - 1] if _reference_image(ref) is None: continue refs.append((slot_number, ref)) return refs -def _named_character_refs_for_text(text, ref_slots): +def _named_refs_for_text(text, ref_slots, kinds=None): haystack = str(text or "") matched = [] - for slot_number, raw in enumerate(ref_slots or [], 1): - ref = _reference_slot(raw, slot_number) - if ref.get("kind") != "character" or _reference_image(ref) is None: + wanted = {str(k).strip().lower() for k in (kinds or ()) if str(k).strip()} + for slot_number, ref in enumerate(ref_slots or [], 1): + if ref is None or _reference_image(ref) is None: + continue + if wanted and str(ref.get("kind") or "").strip().lower() not in wanted: continue for name in _reference_name_keys(ref): if re.search(r"\b" + re.escape(name) + r"\b", haystack, re.I): @@ -4402,7 +4420,11 @@ def _named_character_refs_for_text(text, ref_slots): def _matched_reference_slots(text, ref_slots): matched = [] seen = set() - for slot_number, ref in _slot_refs_for_text(text, ref_slots) + _named_character_refs_for_text(text, ref_slots): + normalized_slots = _normalized_ref_slots(ref_slots) + for slot_number, ref in ( + _slot_refs_for_text(text, normalized_slots) + + _named_refs_for_text(text, normalized_slots, kinds=("character", "location")) + ): if slot_number in seen: continue seen.add(slot_number) @@ -4456,8 +4478,9 @@ def _inject_reference_context(block, context): def _reference_character_memory(ref_slots): lines = [] seen = set() - for slot_number, raw in enumerate(ref_slots or [], 1): - ref = _reference_slot(raw, slot_number) + for slot_number, ref in enumerate(_normalized_ref_slots(ref_slots), 1): + if ref is None: + continue if ref.get("kind") != "character": continue wardrobe = _reference_text(ref.get("wardrobe")) @@ -4478,7 +4501,7 @@ def _reference_character_memory(ref_slots): def _connected_refs(ref_slots): """Connected refs only, preserving slot order and skipping empty sockets.""" - return [ref for ref in (ref_slots or []) if _reference_image(ref) is not None] + return [ref for ref in _normalized_ref_slots(ref_slots) if _reference_image(ref) is not None] def resolve_tagged_refs(text, ref_list): @@ -4493,7 +4516,8 @@ def resolve_tagged_refs(text, ref_list): A tag naming a slot with no image connected refers to nothing at all, so it is removed from the text rather than left to confuse the encoder, and reported.""" wanted = picture_tags(text) - live = [n for n in wanted if 1 <= n <= len(ref_list or []) and ref_list[n - 1] is not None] + normalized_refs = _normalized_ref_slots(ref_list) + live = [n for n in wanted if 1 <= n <= len(normalized_refs) and _reference_image(normalized_refs[n - 1]) is not None] dropped = [n for n in wanted if n not in live] renumber = {old: new for new, old in enumerate(live, 1)} @@ -4501,12 +4525,12 @@ def resolve_tagged_refs(text, ref_list): n = int(m.group(1)) return f"" if n in renumber else "" - out = _PICTURE_TAG.sub(sub, text or "") + out = _PICTURE_TAG.sub(sub, text or "") if dropped: # tidy the gap a removed tag leaves behind out = re.sub(r"\s+([,.;:])", r"\1", out) out = re.sub(r"(,\s*){2,}", ", ", out) out = re.sub(r"\s{2,}", " ", out) - return out.strip(), [ref_list[n - 1] for n in live], dropped + return out.strip(), [normalized_refs[n - 1] for n in live], dropped def resolve_prompt_refs(text, ref_list): @@ -4517,10 +4541,11 @@ def resolve_prompt_refs(text, ref_list): that split, a shot could inherit the facts/context for "Mara" and "Jon" while only carrying a tagged location image, which reads exactly like the names were understood but the faces were ignored.""" - rewritten, tagged_refs, dropped = resolve_tagged_refs(text, ref_list) + normalized_refs = _normalized_ref_slots(ref_list) + rewritten, tagged_refs, dropped = resolve_tagged_refs(text, normalized_refs) refs = list(tagged_refs) seen = {id(ref) for ref in refs} - for _slot_number, ref in _named_character_refs_for_text(rewritten, ref_list): + for _slot_number, ref in _named_refs_for_text(rewritten, normalized_refs, kinds=("character", "location")): marker = id(ref) if marker in seen: continue diff --git a/tests/test_dumas_h3_longvideos.py b/tests/test_dumas_h3_longvideos.py index 39dd474..e869935 100644 --- a/tests/test_dumas_h3_longvideos.py +++ b/tests/test_dumas_h3_longvideos.py @@ -509,6 +509,19 @@ class DumasH3LongVideosHelperTests(unittest.TestCase): self.assertIn("Persistent wardrobe/style for Mara: red jacket.", context) self.assertIn("Character notes for Mara: wears a long grey coat.", context) + def test_reference_context_matches_location_names_without_picture_tag(self): + refs = [ + {"kind": "location", "image": "img2", "name": "Hangar", "aliases": ["loading bay"], "description": "wet concrete floor", "general": "cold industrial lighting"}, + ] + + context = self.module._reference_context_for_text( + "[Generation 1] They argue in the hangar near the loading bay.", + refs, + ) + + self.assertIn("Location context for Hangar: wet concrete floor.", context) + self.assertIn("Location notes for Hangar: cold industrial lighting.", context) + def test_run_uses_legacy_ref_image_inputs_when_new_slots_are_empty(self): calls = {} original_parse_resolution = self.module.parse_resolution @@ -634,6 +647,37 @@ class DumasH3LongVideosHelperTests(unittest.TestCase): self.assertIn("Persistent appearance for Mara: silver hair.", context) self.assertNotIn("Persistent wardrobe/style for Mara: red jacket.", context) + def test_resolve_prompt_refs_adds_named_location_refs(self): + refs = [ + {"kind": "location", "image": "img2", "name": "Hangar", "description": "wet concrete floor"}, + ] + + rewritten, matched, dropped = self.module.resolve_prompt_refs( + "[Generation 1] They wait in the hangar.", + refs, + ) + + self.assertEqual(rewritten, "[Generation 1] They wait in the hangar.") + self.assertEqual(dropped, []) + self.assertEqual(len(matched), 1) + self.assertEqual(matched[0]["name"], "Hangar") + + def test_resolve_tagged_refs_drops_reference_without_image(self): + refs = [ + {"kind": "character", "image": None, "name": "Mara"}, + {"kind": "character", "image": "img2", "name": "Jon"}, + ] + + rewritten, matched, dropped = self.module.resolve_tagged_refs( + "[Generation 1] faces .", + refs, + ) + + self.assertEqual(rewritten, "[Generation 1] faces .") + self.assertEqual(dropped, [1]) + self.assertEqual(len(matched), 1) + self.assertEqual(matched[0]["name"], "Jon") + def test_reference_context_injects_immediately_after_generation_label(self): block = ( "[Generation 1] Classic sitcom lighting and staging. "