Files
2026-09-29 22:26:15 +02:00

2342 lines
119 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
# SPDX-License-Identifier: GPL-3.0-or-later
"""Structured MiniMax H3 prompt editor and CineLinX injection contract.
The browser editor stores only user-authored project data. The deterministic
path formats MiniMax prompt sections and carries an injection request through
CineLinX. The optional assistant is implemented locally in this module and can
call Ollama or a user-selected compatible provider without wrapping another
custom-node package.
"""
from __future__ import annotations
import copy
import asyncio
import json
import os
import re
import urllib.error
import urllib.parse
import urllib.request
from typing import Any
SUPERNODE_LINX_TYPE = "IAMCCS_SUPERNODE_LINX"
CATEGORY = "IAMCCS/MiniMax H3/Prompting"
PROJECT_SCHEMA = "iamccs.minimax_h3.prompter_project"
PROJECT_VERSION = 7
H3_ABSOLUTE_CHAR_LIMIT = 7000
AI_IMAGE_LIMIT = 4
AI_IMAGE_MAX_BYTES = 16 * 1024 * 1024
AUDIO_HANDOFF_AUTHORING_RULE = (
"Never carry dialogue or a new vocalisation across two independently generated chunks. "
"For every non-final chunk, finish all dialogue and shouts at least 1.00 second before the end; "
"reserve the final 1.00 second for only the ambience and sounds requested by the user. "
"Every following chunk must also reserve its first 1.00 second for that same ambience and continued physical action before any new line starts. "
"Do not impose this restriction on the final or only chunk."
)
EVOLVING_DEMO_TIMELINE = (
"0-5 seconds: the woman walks steadily through the city, looking ahead.\n"
"At 5 seconds she notices a red umbrella and slows down while turning her gaze toward it.\n"
"At 10 seconds she stops beside the umbrella, reaches for it, and smiles.\n"
"At 15 seconds she opens the umbrella and continues walking as the camera gently follows."
)
TASK_MODE_ALIASES = {
"v2v_object_swap": "v2va_object_swap",
"v2va": "v2va_object_swap",
"object_swap": "v2va_object_swap",
}
def _normalise_task_mode(value: Any) -> str:
mode = str(value or "t2va").strip().lower()
return TASK_MODE_ALIASES.get(mode, mode)
MODE_SECTIONS: dict[str, tuple[tuple[str, str], ...]] = {
"t2va": (
("scene", "SCENE"),
("shot_list", "SHOT LIST"),
("acting", "ACTING"),
("dialogue", "DIALOGUE"),
("light_and_image", "LIGHT AND IMAGE"),
("camera", "CAMERA"),
("production_sound", "PRODUCTION SOUND"),
("non_diegetic_music", "NON-DIEGETIC MUSIC"),
("negatives", "NEGATIVES"),
),
"i2va": (
("reference_use", "REFERENCE USE"),
("identity_continuity_locks", "IDENTITY / CONTINUITY LOCKS"),
("scene", "SCENE"),
("shot_list", "SHOT LIST"),
("acting", "ACTING"),
("dialogue", "DIALOGUE"),
("light_and_image", "LIGHT AND IMAGE"),
("camera", "CAMERA"),
("production_sound", "PRODUCTION SOUND"),
("non_diegetic_music", "NON-DIEGETIC MUSIC"),
("negatives", "NEGATIVES"),
),
"fl2va": (
("boundary_frames", "BOUNDARY FRAMES"),
("reference_use", "REFERENCE USE"),
("identity_continuity_locks", "IDENTITY / CONTINUITY LOCKS"),
("action", "ACTION"),
("shot_list", "SHOT LIST"),
("acting", "ACTING"),
("dialogue", "DIALOGUE"),
("light_and_image", "LIGHT AND IMAGE"),
("camera", "CAMERA"),
("production_sound", "PRODUCTION SOUND"),
("non_diegetic_music", "NON-DIEGETIC MUSIC"),
("negatives", "NEGATIVES"),
),
"ref2va": (
("subject_definitions", "subject_definitions"),
("summary", "summary"),
("retention_analysis", "retention_analysis"),
("detailed_description", "detailed_description"),
("overall_soundscape", "overall_soundscape"),
("non_diegetic_music", "non_diegetic_music"),
),
"v2va_object_swap": (
("v2va_subject_definitions", "SUBJECT DEFINITIONS"),
("v2va_source_video_authority", "SOURCE VIDEO 1 AUTHORITY"),
("v2va_replacement_retention", "REPLACEMENT / RETENTION ANALYSIS"),
("v2va_interval_edits", "INTERVAL EDIT INSTRUCTIONS"),
("v2va_sound_policy", "SOUND POLICY"),
("v2va_exclusions", "EXCLUSIONS / CONTINUITY SAFEGUARDS"),
),
"audio_driven": (
("audio_drive_contract", "AUDIO DRIVE CONTRACT"),
("audio_subject_map", "SUBJECT / SPEAKER MAP"),
("audio_scene_intent", "SCENE INTENT"),
("audio_timed_performance", "TIMED PERFORMANCE"),
("audio_dialogue_map", "DIALOGUE MAP"),
("audio_visual_sync", "VISIBLE SYNC CUES"),
("audio_camera_sync", "CAMERA"),
("audio_environment", "ENVIRONMENT SOUND"),
("audio_continuity_locks", "CONTINUITY SAFEGUARDS"),
),
}
DEFAULT_SECTIONS = {
"scene": "A rain-polished railway platform before sunrise. <Subject 1>, a tired courier in a charcoal coat, waits beside a silver case while an empty train approaches through blue mist.",
"shot_list": "0.00-2.00s: hold a medium-wide profile. 2.00-4.50s: the train enters and throws moving reflections across the platform. 4.50s-end: <Subject 1> turns toward camera and grips the case.",
"acting": "Restrained performance: shoulders tense first, then the eyes react, then one deliberate turn. Preserve natural blinking and breathing.",
"dialogue": "<Subject 1> (S1): <d>[English] Not this train.</d>",
"light_and_image": "Cool dawn ambience, practical sodium lamps, wet reflections, restrained contrast, realistic skin texture, cinematic depth without artificial glow.",
"camera": "One slow lateral tracking move at chest height with mild foreground parallax; no cut and no change of lens language.",
"production_sound": "Distant rail vibration, light rain on metal roofing, one approaching brake squeal, coat movement, clear close dialogue with matching platform reverb.",
"non_diegetic_music": "A sparse low cello pulse enters only after the train becomes visible; keep it separate from the physical scene sound.",
"negatives": "No identity drift, no duplicate people, no wardrobe change, no warped hands, no sudden zoom, no jump cut, no subtitles, no logo.",
"reference_use": "Use <Picture 1> as the complete opening-frame authority for identity, wardrobe, composition, lens perspective, lighting direction and visible environment. Animate from it rather than redesigning it.",
"identity_continuity_locks": "Keep <Subject 1>'s face, hairline, coat, silver case, body proportions and screen side unchanged. Preserve the platform geometry and time of day.",
"boundary_frames": "Open exactly on <Picture 1> and arrive naturally at <Picture 2> as the final composition. Treat both pictures as full-frame boundaries, not loose style references.",
"action": "The character crosses the connected space in one continuous action. Movement should develop physically toward the final pose with stable identity and coherent screen direction.",
"subject_definitions": "<Subject 1>: the principal performer shown in <Picture 1>; preserve face, body proportions, wardrobe and signature accessories.\n<Subject 2>: the compact silver case; preserve its shape, scale, surface marks and position relative to <Subject 1>.",
"summary": "A tense cinematic beat in which <Subject 1> notices an approaching threat while protecting <Subject 2>. The result should feel observational, grounded and continuous.",
"retention_analysis": "Retain identity and wardrobe from <Picture 1>. Retain the physical timing and camera rhythm from <Video 1> only where supplied. Use <Audio 1> for voice character or cadence only when it is connected; do not invent an unseen speaker.",
"detailed_description": "Begin with the supplied reference composition. <Subject 1> hears the approaching train, tightens one hand around <Subject 2>, then turns with a controlled breath. Use a single lateral camera move and preserve spatial geography. If dialogue is desired: <Subject 1> (S1): <d>[English] Not this train.</d>",
"overall_soundscape": "Layer the location ambience, contact sounds, movement and dialogue in chronological order. Keep perspective and reverberation consistent with camera distance; avoid wall-to-wall effects.",
# V2VA Object Swap is a text/reference contract. It deliberately makes no
# ControlNet, mask, depth, pose, segmentation, or tracker claim.
"v2va_subject_definitions": (
"<Picture 1> defines the replacement <Subject 1>: [write only visible identity, body, wardrobe, material or object facts that must be preserved].\n"
"<Video 1> contains source <Subject 2>: [identify exactly what is being replaced]. Each additional <Picture N> may define only the named <Subject N> or continuity attribute."
),
"v2va_source_video_authority": (
"<Video 1> is the temporal source authority for duration, action timing, body or object motion, camera path, framing, occlusion order, environment and edit rhythm. "
"Preserve those source relationships unless an interval instruction below explicitly changes one."
),
"v2va_replacement_retention": (
"Replace source <Subject 2> from <Video 1> with replacement <Subject 1> from <Picture 1>. "
"Retain [list source environment, secondary subjects, interactions, contact points, lighting response and camera behavior]. "
"Change only [list the requested identity, object, clothing or appearance attributes]."
),
"v2va_interval_edits": (
"[Define source-time intervals from <Video 1>, for example 00:00.00-00:02.50, and state the visible replacement action or retained event in each interval. "
"Leave this field untimed when the edit applies uniformly to the complete source video.]"
),
"v2va_sound_policy": (
"[State whether connected source-video audio is retained, replaced, muted or supplemented. Name <Audio 1> only when an audio reference is actually connected. "
"Keep dialogue wording, lip timing, contact sounds and ambience consistent with the chosen policy.]"
),
"v2va_exclusions": (
"Do not change unselected subjects, source environment, camera trajectory, duration, occlusion order or interactions. "
"No identity blending between <Subject 1> and <Subject 2>, duplicate replacement, geometry drift, temporal jump, subtitle, logo or invented reference."
),
# R21 audio-drive fields are deliberately content-free. They are a
# reusable authoring scaffold, not a hardcoded character, story or line.
"audio_drive_contract": "Treat the connected custom audio as the timing authority. Preserve its order, pauses, breaths and duration; do not invent, remove or reorder speech.",
"audio_subject_map": "<Subject 1> (S1): [describe the visible speaker and the identity/reference facts that must remain stable].",
"audio_scene_intent": "[Describe location, time, dramatic purpose and the visible starting situation.]",
"audio_timed_performance": "[Map audible phrases, pauses and breaths to chronological facial expression, gaze, gesture and body action.]",
"audio_dialogue_map": "<Subject 1> (S1): <d>[Language] ...</d>",
"audio_visual_sync": "[Describe visible mouth articulation, breath, contact or musical actions that must synchronize with the connected audio.]",
"audio_camera_sync": "[Describe one coherent framing and camera move that supports the timed performance without hiding the speaker.]",
"audio_environment": "[Describe only environmental ambience and contact sounds not already fixed by the custom audio.]",
"audio_continuity_locks": "[List identity, wardrobe, anatomy, prop, geography, eyeline and lip-visibility facts that cannot drift.]",
}
HEADING_ALIASES = {
"identity_continuity_locks": "identity_continuity_locks",
"identity_locks": "identity_continuity_locks",
"continuity_locks": "identity_continuity_locks",
"light_image": "light_and_image",
"light_and_image": "light_and_image",
"camera_and_sound": "camera",
"sound": "production_sound",
**{key: key for key in DEFAULT_SECTIONS},
}
def default_project() -> dict[str, Any]:
return {
"schema": PROJECT_SCHEMA,
"schema_version": PROJECT_VERSION,
"project_name": "Untitled H3 Prompt",
"task_mode": "t2va",
"injection_target": "global",
"writing_mode": "guided",
"merge_policy": "replace",
"extended_conditioning_policy": "default",
"conditioning_mode_explicit": True,
"evolving_timeline": "",
"ai_direction": "",
"ai_scope": "active_field",
"ai_visual_roles": {},
"final_prompt_override_enabled": False,
"final_prompt_override": "",
"final_local_prompt_override_enabled": False,
"final_local_prompt_override": "",
"audio_transcript": "",
"audio_dialogue_tag": "",
# Examples remain available through Load Example, but a newly added
# Prompter must never carry a character/story into Shotboard merely
# because its CineLinX socket is connected.
"sections": {key: "" for key in DEFAULT_SECTIONS},
}
def _safe_project(value: Any) -> dict[str, Any]:
if isinstance(value, dict):
source = copy.deepcopy(value)
else:
raw = str(value or "").strip()
if not raw:
source = default_project()
else:
try:
parsed = json.loads(raw)
except json.JSONDecodeError as exc:
raise ValueError(f"IAMCCS_Prompter project JSON non valido: {exc}") from exc
if not isinstance(parsed, dict):
raise ValueError("IAMCCS_Prompter project_data deve essere un oggetto JSON")
source = parsed
project = default_project()
project.update({key: value for key, value in source.items() if key != "sections"})
sections = source.get("sections")
if isinstance(sections, dict):
project["sections"].update({str(key): str(value or "") for key, value in sections.items()})
# R21 audio-drive projects created before the Final Draft field rename
# used the aliases below. Preserve their authored content instead of
# silently falling back to the current bracketed example placeholders.
def _legacy_text(*keys: str) -> str:
return "\n".join(
str(sections.get(key) or "").strip()
for key in keys
if str(sections.get(key) or "").strip()
)
legacy_audio_aliases = {
"audio_timed_performance": _legacy_text("audio_timing_map", "audio_performance"),
"audio_visual_sync": _legacy_text("audio_visible_sync"),
"audio_camera_sync": _legacy_text("audio_camera"),
"audio_environment": _legacy_text("audio_ambience"),
"audio_continuity_locks": _legacy_text("audio_continuity"),
}
for current_key, migrated_value in legacy_audio_aliases.items():
if migrated_value and not str(sections.get(current_key) or "").strip():
project["sections"][current_key] = migrated_value
project["ai_direction"] = str(project.get("ai_direction") or "")
project["ai_scope"] = str(project.get("ai_scope") or "active_field")
visual_roles = project.get("ai_visual_roles")
project["ai_visual_roles"] = visual_roles if isinstance(visual_roles, dict) else {}
source_version = int(source.get("schema_version") or 0)
project["schema"] = PROJECT_SCHEMA
project["schema_version"] = PROJECT_VERSION
project["task_mode"] = _normalise_task_mode(project.get("task_mode"))
policy = str(project.get("extended_conditioning_policy") or "default").strip().lower()
if policy == "evolving":
project["extended_conditioning_policy"] = "evolving"
elif policy == "continuous" and (source_version >= PROJECT_VERSION or project.get("conditioning_mode_explicit") is True):
project["extended_conditioning_policy"] = "continuous"
else:
project["extended_conditioning_policy"] = "default"
project["conditioning_mode_explicit"] = source_version >= PROJECT_VERSION or project.get("conditioning_mode_explicit") is True
project["evolving_timeline"] = str(project.get("evolving_timeline") or "")
project["final_prompt_override_enabled"] = bool(project.get("final_prompt_override_enabled"))
project["final_prompt_override"] = str(project.get("final_prompt_override") or "")
project["final_local_prompt_override_enabled"] = bool(project.get("final_local_prompt_override_enabled"))
project["final_local_prompt_override"] = str(project.get("final_local_prompt_override") or "")
return project
def _parse_seconds_token(value: str) -> float:
token = str(value or "").strip().replace(",", ".")
if ":" in token:
minutes, seconds = token.rsplit(":", 1)
return float(minutes) * 60.0 + float(seconds)
return float(token)
def parse_evolving_timeline(value: Any, *, duration_seconds: float | None = None, fps: int = 24) -> list[dict[str, Any]]:
"""Parse readable timed actions into the H3 evolving beat contract."""
text = str(value or "").strip()
if not text:
return []
number = r"(?:\d+(?::\d+(?:[.,]\d+)?)?|\d+(?:[.,]\d+)?)"
range_re = re.compile(
rf"^\s*(?:(?:from|da)\s+)?(?P<start>{number})\s*(?:sec(?:ond(?:s|i)?)?|s)?\s*"
rf"(?:-|–|—|to|a|fino\s+a)\s*(?P<end>{number})\s*(?:sec(?:ond(?:s|i)?)?|s)?\s*[:;,\-]?\s*(?P<action>.+)$",
re.IGNORECASE,
)
point_re = re.compile(
rf"^\s*(?:(?:at|a|from|da|dal\s+secondo)\s+)?(?P<start>{number})\s*"
rf"(?:sec(?:ond(?:s|i)?)?|s)\s*[:;,\-]?\s*(?P<action>.+)$",
re.IGNORECASE,
)
parsed: list[dict[str, Any]] = []
for line_number, raw_line in enumerate(text.splitlines(), 1):
line = re.sub(r"^\s*(?:[-*•]|\d+[.)])\s*", "", raw_line).strip()
if not line:
continue
match = range_re.match(line) or point_re.match(line)
if not match:
raise ValueError(
f"Evolving timeline line {line_number} has no readable second marker. "
"Use '0-5 seconds: action' or 'At 5 seconds action'."
)
start = _parse_seconds_token(match.group("start"))
end_token = match.groupdict().get("end")
end = _parse_seconds_token(end_token) if end_token else None
action = str(match.group("action") or "").strip(" .:-")
if start < 0 or (end is not None and end <= start) or not action:
raise ValueError(f"Invalid evolving event at line {line_number}.")
parsed.append({"start_seconds": start, "end_seconds": end, "action": action, "line": line_number})
parsed.sort(key=lambda item: (item["start_seconds"], item["line"]))
limit = float(duration_seconds) if duration_seconds is not None else None
beats: list[dict[str, Any]] = []
for index, item in enumerate(parsed):
start = float(item["start_seconds"])
following = float(parsed[index + 1]["start_seconds"]) if index + 1 < len(parsed) else None
end = item["end_seconds"] if item["end_seconds"] is not None else following
if end is None:
end = limit
if end is None:
raise ValueError("The final Evolving event needs an end second or the Shotboard duration.")
if limit is not None:
if start >= limit:
raise ValueError(f"Evolving event at {start:g}s starts outside the {limit:g}s Shotboard duration.")
end = min(float(end), limit)
if float(end) <= start:
raise ValueError(f"Evolving event at {start:g}s has no positive duration.")
beats.append({
"id": f"beat_{index + 1}", "start_frame": round(start * fps),
"end_frame": round(float(end) * fps), "action": item["action"], "source_line": item["line"],
})
return beats
def _normalise_heading(value: str) -> str:
clean = re.sub(r"[^a-z0-9]+", "_", str(value or "").strip().lower()).strip("_")
return HEADING_ALIASES.get(clean, clean)
def _parse_assistant_draft(value: str) -> dict[str, str]:
"""Parse the headings emitted by common H3 prompting assistants.
Manual fields always win. This parser therefore only needs to recover
recognisable sections so an optional assistant can fill blank boxes.
"""
text = str(value or "").strip()
if not text:
return {}
matches = list(
re.finditer(
r"(?m)^\s*(?:\[([^\]\r\n]+)\]|([A-Za-z][A-Za-z0-9 _/\-]{1,60})\s*:)\s*$",
text,
)
)
sections: dict[str, str] = {}
for index, match in enumerate(matches):
key = _normalise_heading(match.group(1) or match.group(2) or "")
start = match.end()
end = matches[index + 1].start() if index + 1 < len(matches) else len(text)
body = text[start:end].strip()
if key and body:
sections[key] = body
if not sections:
sections["detailed_description"] = text
sections["scene"] = text
return sections
def _canonical_evolving_tag_text(value: Any) -> str:
text = str(value or "")
text = re.sub(r"\[ONSET(?:_|\s+)ONCE\]", "[ONSET ONCE]", text, flags=re.I)
text = re.sub(r"\[RESOLVED(?:_|\s+)STATE\]", "[RESOLVED STATE]", text, flags=re.I)
text = re.sub(r"\[(?:THEN(?:_|\s+))?SUSTAIN\]", "[THEN SUSTAIN]", text, flags=re.I)
return text
def _evolving_gerund(verb: str) -> str:
word = str(verb or "").strip().lower()
if not word:
return ""
if word.endswith("ie"):
return f"{word[:-2]}ying"
if word.endswith("e") and not word.endswith("ee"):
return f"{word[:-1]}ing"
if word in {"run", "sit", "stop", "swim"}:
return f"{word}{word[-1]}ing"
return f"{word}ing"
def _positive_evolving_sustain(value: Any) -> str:
text = str(value or "").strip()
patterns = (
r"^after\s+[^,.;:]+?\s+(?:has|have)\s+finished[,;:\-]*\s*",
r"^once\s+[^,.;:]+?\s+(?:ends?|finishes?)[,;:\-]*\s*",
r"\bwithout\s+repeating\s+[^,.;:]+(?:\s+before\s+\d+(?:[.,]\d+)?\s*seconds?)?",
r"\bdo\s+not\s+repeat\s+[^,.;:]+",
r"\bno\s+more\s+[^,.;:]+",
r"\bbefore\s+\d+(?:[.,]\d+)?\s*seconds?\b",
)
for pattern in patterns:
text = re.sub(pattern, " ", text, flags=re.I)
text = re.sub(r"\s{2,}", " ", text)
text = re.sub(r"\s+([.,;:!?])", r"\1", text)
return re.sub(r"^[,;:\-\s]+|[,;:\-\s]+$", "", text).strip()
def _auto_structure_evolving_action(value: Any) -> str:
action = str(value or "").strip()
if not action or re.search(r"\[(?:ONSET ONCE|RESOLVED STATE|THEN SUSTAIN)\]", action, flags=re.I):
return action
match = re.match(r"^(.+?)\s+(takes?\s+(?:one|a)\s+(?:deep\s+)?breath)\s+and\s+(.+)$", action, flags=re.I)
if match:
subject = match.group(1).strip()
return f"[ONSET ONCE] {subject} {match.group(2).strip()}; [THEN SUSTAIN] {subject} {match.group(3).strip()}"
match = re.match(r"^(.+?)\s+(stops?(?:\s+(?:advancing|walking|moving|marching))?)\s+and\s+(?:then\s+)?starts?\s+to\s+([a-z]+)([\s\S]*)$", action, flags=re.I)
if match:
subject = match.group(1).strip()
onset = f"{subject} {match.group(2).strip()}"
rest = re.sub(r"\bstarts?\s+to\s+([a-z]+)", lambda m: f"continues {_evolving_gerund(m.group(1))}", match.group(4), flags=re.I)
sustain = re.sub(r"\s{2,}", " ", f"{subject} continues {_evolving_gerund(match.group(3))}{rest}").strip()
return f"[ONSET ONCE] {onset}; [THEN SUSTAIN] {sustain}"
return action
def _canonicalize_evolving_action(value: Any) -> str:
action = re.sub(r"\s{2,}", " ", _canonical_evolving_tag_text(value)).strip()
action = _auto_structure_evolving_action(action)
matches = list(re.finditer(r"\[(ONSET ONCE|RESOLVED STATE|THEN SUSTAIN)\]", action, flags=re.I))
if not matches:
return action
values = {"onset": "", "resolved": "", "sustain": ""}
prefix = re.sub(r"^[;,:\-\s]+|[;,:\-\s]+$", "", action[:matches[0].start()]).strip()
for index, match in enumerate(matches):
end = matches[index + 1].start() if index + 1 < len(matches) else len(action)
body = re.sub(r"^[;,:\-\s]+|[;,:\-\s]+$", "", action[match.end():end]).strip()
kind = match.group(1).upper()
key = "onset" if kind == "ONSET ONCE" else "resolved" if kind == "RESOLVED STATE" else "sustain"
if body:
values[key] = body
if prefix and not values["sustain"]:
values["sustain"] = prefix
values["resolved"] = _positive_evolving_sustain(values["resolved"])
values["sustain"] = _positive_evolving_sustain(values["sustain"])
out = []
if values["onset"]:
out.append(f"[ONSET ONCE] {values['onset']}")
if values["resolved"]:
out.append(f"[RESOLVED STATE] {values['resolved']}")
if values["sustain"]:
out.append(f"[THEN SUSTAIN] {values['sustain']}")
return "; ".join(out) or action
def _canonicalize_evolving_timeline(value: Any) -> str:
number = r"(?:\d+(?::\d+(?:[.,]\d+)?)?|\d+(?:[.,]\d+)?)"
unit = r"(?:sec(?:ond(?:s|i)?)?|s)"
range_text = rf"{number}\s*{unit}?\s*(?:-|–|—|to|a|fino\s+a)\s*{number}\s*{unit}\b\s*[:;,\-]?"
point_text = rf"(?:at|a|from|da|dal\s+secondo)\s+{number}\s*{unit}\b\s*[:;,\-]?"
text = re.sub(r"\s{2,}", " ", _canonical_evolving_tag_text(value).replace("\r", " ").replace("\n", " ")).strip()
text = re.sub(rf"(\[(?:ONSET ONCE|RESOLVED STATE|THEN SUSTAIN)\])\s*({range_text})", r"\2 \1 ", text, flags=re.I)
text = re.sub(rf"\s+(?={range_text})", "\n", text, flags=re.I)
text = re.sub(rf"\s+(?={point_text})", "\n", text, flags=re.I)
range_re = re.compile(rf"^\s*({number})\s*{unit}?\s*(?:-|–|—|to|a|fino\s+a)\s*({number})\s*{unit}\b\s*[:;,\-]?\s*(.+)$", flags=re.I)
point_re = re.compile(rf"^\s*((?:at|a|from|da|dal\s+secondo)\s+{number}\s*{unit})\b\s*[:;,\-]?\s*(.+)$", flags=re.I)
lines = []
for raw in text.splitlines():
line = raw.strip()
if not line:
continue
match = range_re.match(line)
if match:
lines.append(f"{match.group(1)}-{match.group(2)} seconds: {_canonicalize_evolving_action(match.group(3))}")
continue
match = point_re.match(line)
if match:
lines.append(f"{match.group(1)}: {_canonicalize_evolving_action(match.group(2))}")
continue
lines.append(line)
return "\n".join(lines)
def _validate_canonical_evolving_timeline(value: Any) -> str:
text = _canonicalize_evolving_timeline(value)
if not text.strip():
raise ValueError("Evolving timeline is empty")
for raw in text.splitlines():
line = raw.strip()
if not line:
continue
match = re.match(r"^\s*(?:(?:at|a|from|da|dal\s+secondo)\s+)?(?:\d+(?::\d+(?:[.,]\d+)?)?|\d+(?:[.,]\d+)?)(?:\s*(?:sec(?:ond(?:s|i)?)?|s))?(?:\s*(?:-|–|—|to|a|fino\s+a)\s*(?:\d+(?::\d+(?:[.,]\d+)?)?|\d+(?:[.,]\d+)?)\s*(?:sec(?:ond(?:s|i)?)?|s))?\s*:\s*(.+)$", line, flags=re.I)
if not match:
raise ValueError(f"Evolving phase must start with its timestamp: {line[:120]}")
action = match.group(1)
onset = bool(re.search(r"\[ONSET ONCE\]", action, flags=re.I))
sustain = bool(re.search(r"\[THEN SUSTAIN\]", action, flags=re.I))
if onset and not sustain:
raise ValueError("Evolving ONSET ONCE requires a positive THEN SUSTAIN state")
carried = re.split(r"\[ONSET ONCE\]", action, maxsplit=1, flags=re.I)[-1] if not onset else re.split(r"\[(?:RESOLVED STATE|THEN SUSTAIN)\]", action, maxsplit=1, flags=re.I)[-1]
if re.search(r"\b(?:do\s+not|don't|never|without\s+repeating|no\s+more|avoid)\b", carried, flags=re.I) or re.match(r"^\s*after\s+.+?\s+(?:has|have)\s+finished", carried, flags=re.I):
raise ValueError("Evolving carried state must be positive and must not refer back to a completed onset")
return text
def _validate_evolving_global_prompt(value: Any) -> str:
text = str(value or "").strip()
if not text:
return text
if re.search(r"(?:^|\n)\s*(?:(?:at|from)\s+)?\d+(?::\d+(?:[.,]\d+)?)?\s*(?:sec(?:ond(?:s|i)?)?|s)?\s*(?:-|–|—|to|a|:)|\[(?:ONSET|RESOLVED|THEN|SUSTAIN)", text, flags=re.I):
raise ValueError("FL2VA Evolving GLOBAL contains timed/action syntax; actions belong only to LOCAL/TIMELINE")
without_music = re.sub(r"non_diegetic_music:\s*[\s\S]*$", "", text, flags=re.I)
if re.search(r"(?:^|[.!?]\s+|\n)\s*(?:No\b|Do\s+not\b|Don't\b|Never\b|Without\b|Avoid\b)", without_music, flags=re.I):
raise ValueError("FL2VA Evolving GLOBAL contains negative H3 instructions; describe the positive stable visual/camera state")
return text
def _positive_h3_text(value: Any) -> str:
text = str(value or "")
lines: list[str] = []
for raw_line in text.splitlines():
parts = [
part.strip()
for part in re.split(r";\s*", raw_line)
if part.strip() and not re.match(r"^(?:no\b|do\s+not\b|don't\b|never\b|without\b|avoid\b)", part.strip(), flags=re.I)
]
if parts:
lines.append("; ".join(parts))
result = "\n".join(lines)
result = re.sub(r"(?:^|[.!?]\s+)(?:No\b|Do\s+not\b|Never\b|Without\b|Avoid\b)[^.!?]*(?=[.!?]|$)", " ", result, flags=re.I)
return re.sub(r"\s{2,}", " ", result).strip()
def _unique_prompt_parts(values: Any) -> list[str]:
seen: set[str] = set()
out: list[str] = []
for value in values:
item = str(value or "").strip()
if not item:
continue
key = re.sub(r"\s+", " ", item).strip().lower()
if key in seen:
continue
seen.add(key)
out.append(item)
return out
def _compose_prompt(project: dict[str, Any], task_mode: str, writing_mode: str, assistant_draft: str) -> tuple[str, dict[str, Any]]:
mode = _normalise_task_mode(task_mode or project.get("task_mode") or "t2va")
if mode not in MODE_SECTIONS:
mode = "t2va"
manual = project.get("sections") if isinstance(project.get("sections"), dict) else {}
assisted = _parse_assistant_draft(assistant_draft) if writing_mode == "assistant_fill" else {}
resolved: dict[str, str] = {}
assistant_fills: list[str] = []
for key, _label in MODE_SECTIONS[mode]:
value = str(manual.get(key, "") or "").strip()
if not value and str(assisted.get(key, "") or "").strip():
value = str(assisted[key]).strip()
assistant_fills.append(key)
resolved[key] = value
def join_fields(keys: tuple[str, ...]) -> str:
return "\n".join(resolved.get(key, "").strip() for key in keys if resolved.get(key, "").strip())
blocks: list[str] = []
if any(resolved.values()):
if mode == "ref2va":
# MiniMax full-reference mode has six mandatory headings in this
# exact order. Blank authoring fields become an explicit N/A rather
# than silently changing the prompt grammar.
for key, label in MODE_SECTIONS[mode]:
body = resolved.get(key, "").strip() or "N/A"
if key == "summary" and body != "N/A" and not body.lower().startswith("[reference"):
body = f"[reference generation] {body}"
blocks.append(f"{label}:\n{body}")
elif mode == "v2va_object_swap":
detail = join_fields(("v2va_interval_edits", "v2va_exclusions")) or "N/A"
summary = resolved.get("v2va_source_video_authority", "").strip() or "N/A"
if summary != "N/A" and not summary.lower().startswith("[reference"):
summary = f"[reference generation + video reference] {summary}"
blocks = [
f"subject_definitions:\n{resolved.get('v2va_subject_definitions', '').strip() or 'N/A'}",
f"summary:\n{summary}",
f"retention_analysis:\n{resolved.get('v2va_replacement_retention', '').strip() or 'N/A'}",
f"detailed_description:\n{detail}",
f"overall_soundscape:\n{resolved.get('v2va_sound_policy', '').strip() or 'N/A'}",
"non_diegetic_music:\nN/A",
]
else:
if mode == "t2va":
detail_keys = ("scene", "shot_list", "acting", "dialogue", "light_and_image", "camera")
alignment = ""
sound = resolved.get("production_sound", "").strip()
music = resolved.get("non_diegetic_music", "").strip()
elif mode == "i2va":
detail_keys = ("reference_use", "identity_continuity_locks", "scene", "shot_list", "acting", "dialogue", "light_and_image", "camera")
alignment = "For the target video, at 0.00 seconds into the target video, <Picture 1> (from [Shot 1]) is fully referenced."
sound = resolved.get("production_sound", "").strip()
music = resolved.get("non_diegetic_music", "").strip()
elif mode == "fl2va":
# FL2VA / first-last-frame / keyframe contract: GLOBAL carries
# stable image/reference/camera continuity only. All actions,
# performance events, dialogue and timed sound remain LOCAL.
detail_keys = ("reference_use", "identity_continuity_locks", "light_and_image", "camera")
alignment = _positive_h3_text(resolved.get("boundary_frames", ""))
sound = ""
music = resolved.get("non_diegetic_music", "").strip()
else: # audio_driven
detail_keys = (
"audio_drive_contract", "audio_subject_map", "audio_scene_intent",
"audio_timed_performance", "audio_dialogue_map", "audio_visual_sync",
"audio_camera_sync", "audio_continuity_locks",
)
alignment = ""
sound = resolved.get("audio_environment", "").strip()
music = "N/A"
if mode == "fl2va":
detail = "\n".join(_unique_prompt_parts(_positive_h3_text(resolved.get(key, "")) for key in detail_keys))
elif mode in {"t2va", "i2va"}:
detail = "\n".join(_unique_prompt_parts(
resolved.get(key, "") if key == "dialogue" else _positive_h3_text(resolved.get(key, ""))
for key in detail_keys
))
sound = _positive_h3_text(sound)
else:
detail = join_fields(detail_keys)
if detail and not re.match(r"^\s*\[Shot\s+1\]", detail, flags=re.I):
detail = f"[Shot 1] {detail}"
if alignment:
blocks.append(alignment)
blocks.extend([
f"integrated_multimodal_description:\n{detail or 'N/A'}",
f"overall_soundscape:\n{sound or 'N/A'}",
f"non_diegetic_music:\n{music or 'N/A'}",
])
prompt = "\n\n".join(blocks).strip()
local_prompt = ""
if mode == "fl2va":
local_prompt = "\n".join(_unique_prompt_parts((
resolved.get("action", ""),
resolved.get("shot_list", ""),
resolved.get("acting", ""),
resolved.get("dialogue", ""),
resolved.get("production_sound", ""),
)))
return prompt, {
"task_mode": mode,
"local_prompt": local_prompt,
"included_sections": [key for key, _label in MODE_SECTIONS[mode] if resolved.get(key)],
"missing_sections": [key for key, _label in MODE_SECTIONS[mode] if not resolved.get(key)],
"assistant_fills": assistant_fills,
}
def _merge_text(existing: str, incoming: str, policy: str) -> str:
old = str(existing or "").strip()
new = str(incoming or "").strip()
if not old or str(policy or "replace") == "replace":
return new
if not new:
return old
return f"{old}\n\n{new}"
def _continuous_global_prompt(global_prompt: Any, authored_action: Any) -> str:
"""Fold FL2VA action authority into one untimed GLOBAL prompt."""
time_prefix = re.compile(
r"^\s*(?:[-*•]\s*)?(?:timeline\s+)?(?:from\s+|at\s+)?"
r"\d+(?::\d+(?:[.,]\d+)?)?(?:\s*(?:sec(?:ond(?:s|i)?)?|s))?"
r"(?:\s*(?:-|–|—|to|a|fino\s+a)\s*\d+(?::\d+(?:[.,]\d+)?)?"
r"(?:\s*(?:sec(?:ond(?:s|i)?)?|s))?)?\s*[:;,\-]?\s*",
re.IGNORECASE,
)
action = "\n".join(
cleaned for cleaned in (
time_prefix.sub("", line).strip() for line in str(authored_action or "").splitlines()
) if cleaned
)
stable = str(global_prompt or "").strip()
if not action:
return stable
contract = (
"continuous_action:\nPerform only the following user-authored action as one uninterrupted "
"continuous action throughout the complete take. Preserve the same action, direction, identity, "
"environment and camera continuity across every technical generation boundary. Do not divide it "
f"into timed phases or local prompts.\n{action}"
)
return "\n\n".join(part for part in (stable, contract) if part)
def _normalise_ai_images(value: Any) -> list[dict[str, str]]:
images: list[dict[str, str]] = []
for item in value if isinstance(value, list) else []:
if not isinstance(item, dict) or len(images) >= AI_IMAGE_LIMIT:
continue
data = str(item.get("data") or "").strip()
if data.startswith("data:") and "," in data:
header, data = data.split(",", 1)
guessed = header[5:].split(";", 1)[0]
else:
guessed = ""
data = re.sub(r"\s+", "", data)
if not data:
continue
estimated_bytes = (len(data) * 3) // 4
if estimated_bytes > AI_IMAGE_MAX_BYTES:
raise ValueError(f"AI reference image exceeds {AI_IMAGE_MAX_BYTES // (1024 * 1024)} MB")
mime_type = str(item.get("mime_type") or guessed or "image/png").strip().lower()
if not mime_type.startswith("image/"):
mime_type = "image/png"
role = str(item.get("role") or "reference").strip().lower()
if role not in {"opening", "closing", "identity", "composition", "style", "reference"}:
role = "reference"
images.append({
"data": data,
"mime_type": mime_type,
"name": str(item.get("name") or f"Picture {len(images) + 1}").strip(),
"role": role,
"slot": str(item.get("slot") or len(images) + 1),
})
return images
def _assistant_instruction(
task_mode: str,
sections: dict[str, str],
user_direction: str = "",
target_keys: Any = None,
images: Any = None,
) -> tuple[str, str]:
mode = _normalise_task_mode(task_mode or "t2va")
if mode not in MODE_SECTIONS:
mode = "t2va"
allowed = [key for key, _label in MODE_SECTIONS[mode]]
rough = {key: str(sections.get(key, "") or "").strip() for key in allowed}
filled = {key: value for key, value in rough.items() if value}
selected = [str(key) for key in (target_keys if isinstance(target_keys, list) else []) if str(key) in allowed]
if not selected:
selected = list(filled)
if not selected:
raise ValueError("Select a MiniMax prompt section or write a rough idea before calling the AI")
visuals = _normalise_ai_images(images)
mode_rules = {
"t2va": "Build the requested event from text. Keep the action chronological, filmable and compatible with one continuous audiovisual clip.",
"i2va": "Treat <Picture 1> as the exact opening-frame authority. Animate from it without redesigning identity, wardrobe, composition or screen geography.",
"fl2va": "Treat the opening and closing pictures as exact boundary frames. Keep stable image/reference/camera description in global fields and put actions, performance events, dialogue and time-dependent sound in local/timed fields. For Extended Evolving timelines, every phase line starts with its timestamp/range, then uses only canonical [ONSET ONCE], optional [RESOLVED STATE], and [THEN SUSTAIN] tags. A completed onset is never mentioned again in resolved/sustain text. Describe one physically continuous path from the first frame to the last using positive observable language.",
"ref2va": "Use explicit <Picture N>, <Video N>, <Audio N> and <Subject N> references. State what each reference contributes and what must be ignored; preserve the lowercase REF2VA section semantics.",
"v2va_object_swap": (
"Write a MiniMax H3 video-to-video object/subject replacement contract. Use <Picture N> only for connected replacement/identity references, <Video 1> for the connected source video's temporal motion, camera and environment authority, and stable <Subject N> labels. "
"Separate what is replaced from what remains, then express user-supplied interval edits in source-video time. Do not claim a mask, tracker, ControlNet, depth, pose or segmentation signal unless the user's connected workflow explicitly provides and names it."
),
"audio_driven": (
"Treat the connected custom audio as immutable timing authority for visible performance. "
"Map the user-supplied transcript with stable speaker notation such as <Subject 1> (S1): "
"<d>[Language] ...</d>. Never invent words that are not supplied by the user, never call custom "
"drive audio <Audio 1> unless it is also explicitly connected as a REF2VA reference, and keep the "
"speaker's mouth visible when lip synchronization is requested."
),
}[mode]
system = (
"You are the autonomous IAMCCS MiniMax H3 prompt editor. Improve the user's own direction; do not replace it with a different story. "
"Return one JSON object only, with plain-string values and no markdown. Valid keys are "
f"{allowed}. Return only the selected keys {selected}; never create a blank or unselected section. "
"Write concise production-ready English optimized for MiniMax H3 audiovisual generation. Preserve exact identity facts, reference tags, requested timing, language and quoted dialogue unless the user explicitly asks to change them. "
"Use chronological visible action, realistic body mechanics, stable screen geography and one coherent camera language. Prefer one motivated camera move over a list of conflicting moves. "
"Return only the content of each selected authoring field, never repeat its JSON key as a heading. The IAMCCS composer will assemble base modes into MiniMax's official integrated_multimodal_description, overall_soundscape and non_diegetic_music grammar, and full-reference modes into the official six-section grammar. "
"When rewriting shot_list, do not add a second [Shot 1] marker because the composer supplies it; mark only a real later cut as [Shot N] At MM:SS.mmm, and do not invent cuts merely to make the description longer. "
"Separate diegetic ambience, dialogue and contact effects from non-diegetic score. Use <Subject N> consistently and keep dialogue inside <d>[Language] ...</d> with stable speaker labels such as (S1) when those tags are present. "
f"Chunk-boundary sound rule: {AUDIO_HANDOFF_AUTHORING_RULE} "
"Do not invent extra characters, products, dialogue, scene changes, cuts, subtitles or logos. Express generated H3 fields in positive observable language: describe the desired stable state instead of writing negative prompt lists or phrases such as no/do not/never/without/avoid. The normal exception is an explicit music absence such as 'No score' inside NON_DIEGETIC_MUSIC, or an explicit source-audio policy requested by the user. "
f"Mode rule: {mode_rules} "
"When images are attached, analyze only the contribution named by each image role. An opening image governs the first frame; a closing image governs the last frame; identity, composition and style images govern only those named attributes. "
"Never mention unavailable media or claim to have seen a detail that is not visible."
)
if len(system) > 24000:
raise RuntimeError("MiniMax assistant system prompt exceeds the 7000-token safety envelope")
user = json.dumps({
"task_mode": mode,
"selected_sections": selected,
"user_direction": str(user_direction or "").strip(),
"rough_sections": {key: rough[key] for key in selected},
"visual_context": [
{"slot": item["slot"], "name": item["name"], "role": item["role"]}
for item in visuals
],
}, ensure_ascii=False, indent=2)
return system, user
def _http_json(url: str, payload: dict[str, Any], headers: dict[str, str], timeout: float) -> dict[str, Any]:
data = json.dumps(payload, ensure_ascii=False).encode("utf-8")
request = urllib.request.Request(
str(url),
data=data,
headers={"Content-Type": "application/json", **headers},
method="POST",
)
try:
with urllib.request.urlopen(request, timeout=max(5.0, min(300.0, float(timeout)))) as response:
raw = response.read().decode("utf-8", errors="replace")
except urllib.error.HTTPError as exc:
detail = exc.read().decode("utf-8", errors="replace")[:1200]
raise RuntimeError(f"AI provider HTTP {exc.code}: {detail}") from exc
except urllib.error.URLError as exc:
raise RuntimeError(f"AI provider connection failed: {exc.reason}") from exc
try:
parsed = json.loads(raw)
except json.JSONDecodeError as exc:
raise RuntimeError("AI provider returned invalid JSON") from exc
if not isinstance(parsed, dict):
raise RuntimeError("AI provider returned an unsupported response")
return parsed
def _http_get_json(url: str, timeout: float = 10.0) -> dict[str, Any]:
request = urllib.request.Request(str(url), headers={"Accept": "application/json"}, method="GET")
try:
with urllib.request.urlopen(request, timeout=max(2.0, min(30.0, float(timeout)))) as response:
raw = response.read().decode("utf-8", errors="replace")
except urllib.error.HTTPError as exc:
detail = exc.read().decode("utf-8", errors="replace")[:1200]
raise RuntimeError(f"Ollama HTTP {exc.code}: {detail}") from exc
except urllib.error.URLError as exc:
raise RuntimeError(f"Ollama connection failed: {exc.reason}") from exc
try:
parsed = json.loads(raw)
except json.JSONDecodeError as exc:
raise RuntimeError("Ollama returned invalid JSON") from exc
if not isinstance(parsed, dict):
raise RuntimeError("Ollama returned an unsupported response")
return parsed
def _ollama_native_base(value: Any) -> str:
"""Return the Ollama native API root accepted by /api/chat and /api/tags.
Ollama also exposes an OpenAI-compatible ``/v1`` surface. Users commonly
paste that URL after using another local client, but native endpoints must
be addressed from the server root. Normalising here keeps model discovery
and rewrite requests on the same endpoint contract.
"""
root = str(value or "http://127.0.0.1:11434").strip().rstrip("/")
if root.lower().endswith("/v1"):
root = root[:-3].rstrip("/")
return root or "http://127.0.0.1:11434"
def _extract_json_payload(text: str) -> dict[str, Any]:
clean = re.sub(r"^\s*```(?:json)?\s*|\s*```\s*$", "", str(text or "").strip(), flags=re.I | re.S)
start = clean.find("{")
end = clean.rfind("}")
if start < 0 or end <= start:
raise RuntimeError("The AI response did not contain a JSON object")
try:
value = json.loads(clean[start:end + 1])
except json.JSONDecodeError as exc:
raise RuntimeError(f"The AI response JSON is invalid: {exc}") from exc
if not isinstance(value, dict):
raise RuntimeError("The AI response must be a JSON object")
return value
def _extract_json_object(text: str) -> dict[str, str]:
value = _extract_json_payload(text)
def section_text(item: Any) -> str:
if item is None:
return ""
if isinstance(item, str):
return item.strip()
if isinstance(item, (int, float)):
return str(item).strip()
if isinstance(item, list):
return "\n".join(part for part in (section_text(entry) for entry in item) if part).strip()
if isinstance(item, dict):
parts = []
for key, nested in item.items():
body = section_text(nested)
if body:
parts.append(f"{str(key).replace('_', ' ').strip()}: {body}")
return "; ".join(parts).strip()
return ""
return {
str(key): rendered
for key, item in value.items()
if (rendered := section_text(item))
}
def _compact_assistant_instruction(task_mode: str, sections: dict[str, str], user_direction: str,
target_keys: Any) -> tuple[str, str]:
"""Small local-model fallback used only after Ollama aborts for repetition."""
mode = _normalise_task_mode(task_mode or "t2va")
allowed = [key for key, _label in MODE_SECTIONS.get(mode, MODE_SECTIONS["t2va"])]
selected = [str(key) for key in (target_keys if isinstance(target_keys, list) else []) if str(key) in allowed]
if not selected:
selected = [key for key in allowed if str(sections.get(key, "") or "").strip()]
if not selected:
raise ValueError("Select a MiniMax prompt section or write a rough idea before calling the AI")
evolving = mode == "fl2va" and "extended evolving" in str(user_direction or "").lower()
continuous = mode == "fl2va" and "continuous" in str(user_direction or "").lower()
special = (
"For ACTION and SHOT_LIST, return identical lines. Every line begins with a user-supplied timestamp or range. "
"Use [ONSET ONCE] only with a positive [THEN SUSTAIN] state."
if evolving else
"Return one uninterrupted untimed action in ACTION only."
if continuous else
"Use chronological visible action and one coherent camera movement."
)
system = (
"Convert the user's natural-language request into concise MiniMax H3 production fields. "
"Return one JSON object only, without markdown. Every value must be one plain string. "
f"Use only these keys: {selected}. Omit a key when the user supplied no relevant content. "
"Preserve requested identity, action, camera direction, timestamps, language and quoted dialogue. "
f"{special}"
)
user = json.dumps({
"task_mode": mode,
"request": str(user_direction or "").strip(),
"existing_fields": {key: str(sections.get(key, "") or "").strip() for key in selected if str(sections.get(key, "") or "").strip()},
}, ensure_ascii=False)
return system, user
def _evolving_timeline_from_request(user_direction: Any) -> str:
"""Recover immutable user-authored phase boundaries when a local LLM misformats them."""
text = re.split(r"\bREQUEST\s*:\s*", str(user_direction or ""), flags=re.I)[-1].strip()
if not text:
return ""
try:
return _validate_canonical_evolving_timeline(text)
except ValueError:
pass
number = r"(?:\d+(?::\d+(?:[.,]\d+)?)?|\d+(?:[.,]\d+)?)"
point_re = re.compile(
rf"\b(?:at|a|from|da|dal\s+secondo)\s+(?P<start>{number})\s*(?:sec(?:ond(?:s|i)?)?|s)\b\s*[:;,\-]?\s*",
flags=re.I,
)
matches = list(point_re.finditer(text))
if not matches:
return ""
lines: list[str] = []
opening = re.sub(r"^[\s,;:.\-]+|[\s,;:.\-]+$", "", text[:matches[0].start()]).strip()
first_seconds = _parse_seconds_token(matches[0].group("start"))
if opening and first_seconds > 0:
lines.append(f"0-{matches[0].group('start')} seconds: {_canonicalize_evolving_action(opening)}")
for index, match in enumerate(matches):
end = matches[index + 1].start() if index + 1 < len(matches) else len(text)
action = re.sub(r"^[\s,;:.\-]+|[\s,;:.\-]+$", "", text[match.end():end]).strip()
if action:
lines.append(f"At {match.group('start')} seconds: {_canonicalize_evolving_action(action)}")
return _validate_canonical_evolving_timeline("\n".join(lines)) if lines else ""
def rewrite_sections_with_ai(
provider: str,
base_url: str,
model: str,
api_key: str,
task_mode: str,
sections: dict[str, str],
temperature: float = 0.35,
timeout: float = 120.0,
user_direction: str = "",
target_keys: Any = None,
images: Any = None,
) -> tuple[dict[str, str], dict[str, Any]]:
provider = str(provider or "ollama").strip().lower()
if provider == "lm_studio":
base_url = str(base_url or "http://localhost:1234/v1").rstrip("/")
if not base_url.endswith("/v1") and not base_url.endswith("/chat/completions"):
base_url += "/v1"
model = str(model or "").strip()
if not model:
raise ValueError("Select an AI model before rewriting")
visual_inputs = _normalise_ai_images(images)
system, user = _assistant_instruction(task_mode, sections, user_direction, target_keys, visual_inputs)
api_key = str(api_key or "").strip()
if not api_key:
api_key = {
"openai_compatible": os.environ.get("OPENAI_API_KEY", ""),
"gemini": os.environ.get("GEMINI_API_KEY", ""),
"anthropic": os.environ.get("ANTHROPIC_API_KEY", ""),
}.get(provider, "")
content = ""
transport_retry = ""
if provider == "ollama":
root = _ollama_native_base(base_url)
def ollama_payload(system_text: str, user_text: str, *, compact: bool = False) -> dict[str, Any]:
return {
"model": model,
"stream": False,
"format": "json",
"messages": [
{"role": "system", "content": system_text},
{
"role": "user",
"content": user_text,
**({"images": [item["data"] for item in visual_inputs]} if visual_inputs else {}),
},
],
"options": {
"temperature": min(float(temperature), 0.25) if compact else float(temperature),
**({"num_predict": 1600, "repeat_penalty": 1.1} if compact else {}),
},
}
try:
result = _http_json(f"{root}/api/chat", ollama_payload(system, user), {}, timeout)
except RuntimeError as exc:
if "token repeat limit reached" not in str(exc).lower():
raise
compact_system, compact_user = _compact_assistant_instruction(
task_mode, sections, user_direction, target_keys,
)
result = _http_json(
f"{root}/api/chat", ollama_payload(compact_system, compact_user, compact=True), {}, timeout,
)
transport_retry = "ollama_compact_after_repeat_limit"
content = str((result.get("message") or {}).get("content") or "")
elif provider in {"openai_compatible", "lm_studio"}:
root = str(base_url or "https://api.openai.com/v1").rstrip("/")
url = root if root.endswith("/chat/completions") else f"{root}/chat/completions"
headers = {"Authorization": f"Bearer {api_key}"} if api_key else {}
openai_user: Any = user
if visual_inputs:
openai_user = [{"type": "text", "text": user}] + [
{
"type": "image_url",
"image_url": {"url": f"data:{item['mime_type']};base64,{item['data']}"},
}
for item in visual_inputs
]
result = _http_json(
url,
{
"model": model,
"temperature": float(temperature),
"response_format": {"type": "json_object"},
"messages": [{"role": "system", "content": system}, {"role": "user", "content": openai_user}],
},
headers,
timeout,
)
choices = result.get("choices") or []
content = str(((choices[0] if choices else {}).get("message") or {}).get("content") or "")
elif provider == "gemini":
root = str(base_url or "https://generativelanguage.googleapis.com/v1beta").rstrip("/")
encoded_model = urllib.parse.quote(model, safe="-._")
suffix = f"/models/{encoded_model}:generateContent"
url = f"{root}{suffix}?key={urllib.parse.quote(api_key)}"
gemini_parts: list[dict[str, Any]] = [{"text": user}]
gemini_parts.extend(
{"inlineData": {"mimeType": item["mime_type"], "data": item["data"]}}
for item in visual_inputs
)
result = _http_json(
url,
{
"systemInstruction": {"parts": [{"text": system}]},
"contents": [{"role": "user", "parts": gemini_parts}],
"generationConfig": {"temperature": float(temperature), "responseMimeType": "application/json"},
},
{},
timeout,
)
candidates = result.get("candidates") or []
parts = (((candidates[0] if candidates else {}).get("content") or {}).get("parts") or [])
content = "".join(str(part.get("text") or "") for part in parts if isinstance(part, dict))
elif provider == "anthropic":
root = str(base_url or "https://api.anthropic.com/v1").rstrip("/")
url = root if root.endswith("/messages") else f"{root}/messages"
anthropic_user: Any = user
if visual_inputs:
anthropic_user = [
{
"type": "image",
"source": {
"type": "base64",
"media_type": item["mime_type"],
"data": item["data"],
},
}
for item in visual_inputs
] + [{"type": "text", "text": user}]
result = _http_json(
url,
{
"model": model,
"max_tokens": 4096,
"temperature": float(temperature),
"system": system,
"messages": [{"role": "user", "content": anthropic_user}],
},
{"x-api-key": api_key, "anthropic-version": "2023-06-01"},
timeout,
)
content = "".join(str(item.get("text") or "") for item in (result.get("content") or []) if isinstance(item, dict))
else:
raise ValueError(f"Unsupported AI provider: {provider}")
rewritten = _extract_json_object(content)
allowed = {key for key, _label in MODE_SECTIONS.get(_normalise_task_mode(task_mode), MODE_SECTIONS["t2va"])}
supplied = {key for key, value in sections.items() if key in allowed and str(value or "").strip()}
requested = {str(key) for key in target_keys} if isinstance(target_keys, list) else supplied
requested = requested & allowed
if not requested:
requested = supplied
filtered = {key: value for key, value in rewritten.items() if key in requested and value}
if not filtered:
raise RuntimeError("The AI did not return any valid filled MiniMax section")
mode_name = _normalise_task_mode(task_mode)
if mode_name == "fl2va" and "extended evolving" in str(user_direction or "").lower():
raw_timeline = str(filtered.get("action") or filtered.get("shot_list") or "").strip()
if not raw_timeline:
raise RuntimeError("Extended Evolving AI response did not return ACTION/SHOT_LIST timeline text")
try:
canonical_timeline = _validate_canonical_evolving_timeline(raw_timeline)
except ValueError:
canonical_timeline = _evolving_timeline_from_request(user_direction)
if not canonical_timeline:
raise
if "action" in requested:
filtered["action"] = canonical_timeline
if "shot_list" in requested:
filtered["shot_list"] = canonical_timeline
# An AI rewrite may paraphrase a Whisper transcript even when told not to.
# Restore the exact H3 dialogue line supplied by the user after parsing.
authored = "\n".join([str(user_direction or ""), *(str(value or "") for value in sections.values())])
dialogue_lines = list(dict.fromkeys(re.findall(
r"<Subject\s+\d+>\s*\(S\d+\):\s*<d>\[[^\]]+\][\s\S]*?</d>",
authored,
flags=re.IGNORECASE,
)))
if dialogue_lines:
target = "audio_dialogue_map" if "audio_dialogue_map" in requested else next(
(key for key in target_keys if key in requested), next(iter(requested))
) if isinstance(target_keys, list) else next(iter(requested))
for line in dialogue_lines:
if line not in "\n".join(filtered.values()):
filtered[target] = "\n".join(part for part in (filtered.get(target, ""), line) if part)
return filtered, {
"provider": provider,
"model": model,
"rewritten_sections": sorted(filtered),
"preserved_blank_sections": sorted(allowed - supplied),
"selected_sections": sorted(requested),
"visual_references": [
{"slot": item["slot"], "name": item["name"], "role": item["role"]}
for item in visual_inputs
],
"system_prompt_characters": len(system),
"transport_retry": transport_retry,
"audio_handoff_authoring_rule": AUDIO_HANDOFF_AUTHORING_RULE,
}
def _multimodal_json_chat(provider: str, base_url: str, model: str, api_key: str,
system: str, user: str, images: Any = None,
temperature: float = 0.3, timeout: float = 120.0) -> dict[str, Any]:
"""Shared vision/JSON transport for IAMCCS planning tools."""
provider = str(provider or "ollama").strip().lower()
if provider == "lm_studio":
base_url = str(base_url or "http://localhost:1234/v1").rstrip("/")
if not base_url.endswith("/v1") and not base_url.endswith("/chat/completions"):
base_url += "/v1"
model = str(model or "").strip()
if not model:
raise ValueError("Select an AI model first")
visual_inputs = _normalise_ai_images(images)
api_key = str(api_key or "").strip() or {
"openai_compatible": os.environ.get("OPENAI_API_KEY", ""),
"gemini": os.environ.get("GEMINI_API_KEY", ""),
"anthropic": os.environ.get("ANTHROPIC_API_KEY", ""),
}.get(provider, "")
if provider == "ollama":
result = _http_json(
f"{_ollama_native_base(base_url)}/api/chat",
{"model": model, "stream": False, "format": "json",
"messages": [{"role": "system", "content": system}, {
"role": "user", "content": user,
**({"images": [item["data"] for item in visual_inputs]} if visual_inputs else {}),
}], "options": {"temperature": float(temperature)}}, {}, timeout)
content = str((result.get("message") or {}).get("content") or "")
elif provider in {"openai_compatible", "lm_studio"}:
root = str(base_url or "https://api.openai.com/v1").rstrip("/")
url = root if root.endswith("/chat/completions") else f"{root}/chat/completions"
body: Any = user
if visual_inputs:
body = [{"type": "text", "text": user}] + [{
"type": "image_url", "image_url": {"url": f"data:{item['mime_type']};base64,{item['data']}"},
} for item in visual_inputs]
result = _http_json(url, {"model": model, "temperature": float(temperature),
"response_format": {"type": "json_object"},
"messages": [{"role": "system", "content": system}, {"role": "user", "content": body}]},
{"Authorization": f"Bearer {api_key}"} if api_key else {}, timeout)
choices = result.get("choices") or []
content = str(((choices[0] if choices else {}).get("message") or {}).get("content") or "")
elif provider == "gemini":
root = str(base_url or "https://generativelanguage.googleapis.com/v1beta").rstrip("/")
encoded_model = urllib.parse.quote(model, safe="-._")
parts: list[dict[str, Any]] = [{"text": user}]
parts.extend({"inlineData": {"mimeType": item["mime_type"], "data": item["data"]}} for item in visual_inputs)
result = _http_json(f"{root}/models/{encoded_model}:generateContent?key={urllib.parse.quote(api_key)}",
{"systemInstruction": {"parts": [{"text": system}]},
"contents": [{"role": "user", "parts": parts}],
"generationConfig": {"temperature": float(temperature), "responseMimeType": "application/json"}}, {}, timeout)
candidates = result.get("candidates") or []
response_parts = (((candidates[0] if candidates else {}).get("content") or {}).get("parts") or [])
content = "".join(str(part.get("text") or "") for part in response_parts if isinstance(part, dict))
elif provider == "anthropic":
if not api_key:
raise ValueError("Claude requires an API key or ANTHROPIC_API_KEY")
root = str(base_url or "https://api.anthropic.com/v1").rstrip("/")
url = root if root.endswith("/messages") else f"{root}/messages"
body = [{"type": "image", "source": {"type": "base64", "media_type": item["mime_type"], "data": item["data"]}}
for item in visual_inputs] + [{"type": "text", "text": user}]
result = _http_json(url, {"model": model, "max_tokens": 4096,
"temperature": float(temperature), "system": system,
"messages": [{"role": "user", "content": body}]},
{"x-api-key": api_key, "anthropic-version": "2023-06-01"}, timeout)
content = "".join(str(item.get("text") or "") for item in (result.get("content") or []) if isinstance(item, dict))
else:
raise ValueError(f"Unsupported AI provider: {provider}")
return _extract_json_payload(content)
VISUAL_STORY_SYSTEM_PROMPT = """You are IAMCCS Visual Story Planner for MiniMax H3.
Read every supplied image in exact Picture/Shotboard slot order and obey the user's action idea. Return JSON only:
{"global_prompt":"...","global_direction":"...","recommended_mode":"auto|i2va|fl2va|longvid_guides","continuity_locks":"...","shots":[{"slot":1,"local_prompt":"...","h3_transition_prompt":"..."}]}
global_prompt must be a directly usable MiniMax H3 GLOBAL prompt. In FL2VA / first-last-frame / keyframe work, GLOBAL contains only stable visual authority: subject identity, reference/guide roles, environment, wardrobe, props, geography, lighting, image texture, screen direction and the persistent camera/lens language. Keep chronological actions, performance changes, dialogue, transient vocal events and time-dependent sound out of GLOBAL; they belong in local_prompt / h3_transition_prompt. For other modes, keep global_prompt limited to information shared by every local shot. Treat pictures as ordered visual guide states, never as a collage. Preserve user-supplied timing exactly. Use positive observable H3 language in generated prompt strings rather than negative reminder lists; an explicit music absence such as “No score” is allowed in the music field.
Create one shot object per image in slot order. local_prompt is the filmable H3 action authority for that guide: active movement, performance, interval-specific camera behavior and audible visible events. h3_transition_prompt describes the causal action/camera hand-off from the preceding guide into this guide; slot 1 describes how motion begins from <Picture 1>. Preserve identity, wardrobe, anatomy, props, screen direction, geography, lighting, lens logic and action state. English only; no Markdown outside the plain prompt strings."""
def build_visual_story_plan_with_ai(provider: str, base_url: str, model: str, api_key: str,
relationship: str, task_mode: str, images: Any,
temperature: float = 0.3, timeout: float = 150.0):
visuals = _normalise_ai_images(images)
relation = str(relationship or "").strip()
if not relation:
raise ValueError("Describe the relationship between the images")
visual_map = "\n".join(f"Picture {item['slot']}: {item['name']} · role={item['role']}" for item in visuals)
payload = _multimodal_json_chat(provider, base_url, model, api_key,
VISUAL_STORY_SYSTEM_PROMPT + ("" if visuals else
"\nNo images supplied. Do not invent Picture references. Create one local shot per explicitly numbered prompt in the brief. Preserve those slot numbers and their individual instructions; do not collapse them into the global prompt. If no slots are numbered, choose a short consecutive shot sequence. Global defines shared identity, location and atmosphere; local prompts carry the chronological development."),
f"CURRENT IAMCCS MODE: {task_mode}\nVISUAL MAP:\n{visual_map}\n\nRELATIONSHIP / STORY INTENT:\n{relation}",
visuals, temperature, timeout)
shots = []
for raw in payload.get("shots") if isinstance(payload.get("shots"), list) else []:
if not isinstance(raw, dict):
continue
local = str(raw.get("local_prompt") or "").strip()
transition = str(raw.get("h3_transition_prompt") or "").strip()
if local:
shots.append({"slot": max(1, int(raw.get("slot") or len(shots) + 1)),
"local_prompt": local, "h3_transition_prompt": transition})
if not shots:
raise RuntimeError("The AI returned no usable H3 shots")
slots = [shot["slot"] for shot in shots]
if len(slots) != len(set(slots)):
raise RuntimeError("The AI returned duplicate local prompt slots; retry the request.")
if not visuals:
requested_slots = {int(n) for n in re.findall(r"\b(?:prompt|shot|slot)\s*(\d+)\b", relation, flags=re.I)}
if requested_slots and set(slots) != requested_slots:
raise RuntimeError("The AI did not preserve the requested local prompt numbers; retry the request.")
shots.sort(key=lambda shot: shot["slot"])
mode = str(payload.get("recommended_mode") or "auto").strip().lower()
if mode not in {"auto", "i2va", "fl2va", "longvid_guides"}:
mode = "auto"
global_direction = str(payload.get("global_direction") or "").strip()
continuity_locks = str(payload.get("continuity_locks") or "").strip()
global_prompt = str(payload.get("global_prompt") or "").strip()
if not global_prompt:
shot_plan = "\n".join(
f"[Shot {shot['slot']}] <Picture {shot['slot']}>: {shot['local_prompt']}"
+ (f" Transition: {shot['h3_transition_prompt']}" if shot['h3_transition_prompt'] else "")
for shot in shots
)
global_prompt = "\n\n".join(part for part in (
global_direction,
"Reference and guide authority:\n" + "\n".join(
f"<Picture {item['slot']}> is the ordered visual guide for [Shot {item['slot']}]." for item in visuals
),
"Integrated chronological action:\n" + shot_plan,
f"Continuity locks:\n{continuity_locks}" if continuity_locks else "",
) if part)
# REQUEST → GLOBAL + LOCALS has a separate AI endpoint. Keep verbatim
# Whisper dialogue here too, even when the model paraphrases its output.
for line in dict.fromkeys(re.findall(
r"<Subject\s+\d+>\s*\(S\d+\):\s*<d>\[[^\]]+\][\s\S]*?</d>",
relation, flags=re.IGNORECASE,
)):
if line not in global_prompt and not any(line in shot["local_prompt"] for shot in shots):
global_prompt = "\n\n".join(part for part in (global_prompt, line) if part)
return {"global_prompt": global_prompt,
"global_direction": global_direction,
"recommended_mode": mode,
"continuity_locks": continuity_locks,
"shots": shots}, {"provider": provider, "model": model, "visual_references": len(visuals)}
NEXTFRAME_H3_SYSTEM_PROMPT = """Convert a Qwen Next Scene still-image prompt and its visual references into MiniMax H3 video prompting. Return JSON only:
{"h3_local_prompt":"...","h3_transition_prompt":"...","h3_continuity_locks":"...","recommended_mode":"auto|i2va|fl2va|longvid_guides"}
The local prompt describes moving action, acting, camera, timing and audible visible events, not a static result. The transition prompt explains how the source image reaches the target while preserving identity, wardrobe, props, screen direction, geography, lighting and style. Choose FL2VA for start+end frames, I2VA for one source, LongVid only for a true ordered multi-shot sequence. English only, no Next Scene trigger, no Markdown."""
def rewrite_nextframe_h3_with_ai(provider: str, base_url: str, model: str, api_key: str,
qwen_prompt: str, requested_mode: str = "auto", images: Any = None,
temperature: float = 0.25, timeout: float = 150.0):
prompt = str(qwen_prompt or "").strip()
if not prompt:
raise ValueError("Write or generate a NextFrame prompt first")
visuals = _normalise_ai_images(images)
payload = _multimodal_json_chat(provider, base_url, model, api_key,
NEXTFRAME_H3_SYSTEM_PROMPT,
f"REQUESTED MODE: {requested_mode}\nQWEN NEXT-FRAME PROMPT:\n{prompt}\n\nImages are ordered source first, target/extra references next.",
visuals, temperature, timeout)
local = str(payload.get("h3_local_prompt") or "").strip()
if not local:
raise RuntimeError("The AI returned no H3 local prompt")
mode = str(payload.get("recommended_mode") or requested_mode or "auto").strip().lower()
if mode not in {"auto", "i2va", "fl2va", "longvid_guides"}:
mode = "auto"
return {"h3_local_prompt": local,
"h3_transition_prompt": str(payload.get("h3_transition_prompt") or "").strip(),
"h3_continuity_locks": str(payload.get("h3_continuity_locks") or "").strip(),
"recommended_mode": mode}, {"provider": provider, "model": model, "visual_references": len(visuals)}
NEXTFRAME_ASSISTANT_SYSTEM_PROMPT = """You are IAMCCS NextFrame Prompt Director, a specialist in Qwen-Image-Edit-2511 and the Next Scene LoRA.
Return JSON only, with exactly this shape: {\"prompt\":\"...\"}.
Write one production-ready English image-edit prompt of roughly 70-140 words. It MUST begin exactly with the LoRA trigger `Next Scene:`. Preserve the user's intent and describe only the immediately following visible storyboard beat.
Use this order:
1. camera movement;
2. shot size, angle, and composition;
3. one clear subject action or environmental change;
4. continuity locks for Image 1: exact identity, face, hairstyle, wardrobe, body proportions, props, location geometry, spatial relationships, color palette, and cinematic style;
5. lighting direction, atmosphere, depth, and physically plausible detail.
Be direct, specific, concise, and visually observable. Treat Image 1 as the primary source frame. Do not invent new people, dialogue, captions, logos, or unrelated events. Do not include a negative prompt, Markdown, notes, alternatives, explanations, or camera metadata outside the prompt."""
def rewrite_nextframe_prompt_with_ai(
provider: str,
base_url: str,
model: str,
api_key: str,
user_prompt: str,
current_prompt: str = "",
temperature: float = 0.25,
timeout: float = 120.0,
) -> tuple[str, dict[str, Any]]:
"""Turn a rough scene direction into a Qwen 2511 Next Scene prompt."""
provider = str(provider or "ollama").strip().lower()
if provider == "lm_studio":
base_url = str(base_url or "http://localhost:1234/v1").rstrip("/")
if not base_url.endswith("/v1") and not base_url.endswith("/chat/completions"):
base_url += "/v1"
model = str(model or "").strip()
if not model:
raise ValueError("Select an AI model before using AI Assistance")
api_key = str(api_key or "").strip()
if not api_key:
api_key = {
"openai_compatible": os.environ.get("OPENAI_API_KEY", ""),
"anthropic": os.environ.get("ANTHROPIC_API_KEY", ""),
}.get(provider, "")
direction = str(user_prompt or current_prompt or "").strip()
if not direction:
raise ValueError("Write a rough next-scene direction first")
current = str(current_prompt or "").strip()
user = (
"Transform this user direction into the final prompt.\n"
f"USER DIRECTION:\n{direction}\n\n"
f"CURRENT DRAFT (use only when helpful):\n{current or '[none]'}"
)
content = ""
if provider == "ollama":
root = _ollama_native_base(base_url)
result = _http_json(
f"{root}/api/chat",
{
"model": model,
"stream": False,
"format": "json",
"messages": [
{"role": "system", "content": NEXTFRAME_ASSISTANT_SYSTEM_PROMPT},
{"role": "user", "content": user},
],
"options": {"temperature": float(temperature)},
},
{},
timeout,
)
content = str((result.get("message") or {}).get("content") or "")
elif provider in {"openai_compatible", "lm_studio"}:
root = str(base_url or "https://api.openai.com/v1").rstrip("/")
url = root if root.endswith("/chat/completions") else f"{root}/chat/completions"
result = _http_json(
url,
{
"model": model,
"temperature": float(temperature),
"response_format": {"type": "json_object"},
"messages": [
{"role": "system", "content": NEXTFRAME_ASSISTANT_SYSTEM_PROMPT},
{"role": "user", "content": user},
],
},
{"Authorization": f"Bearer {api_key}"} if api_key else {},
timeout,
)
choices = result.get("choices") or []
content = str(((choices[0] if choices else {}).get("message") or {}).get("content") or "")
elif provider == "anthropic":
root = str(base_url or "https://api.anthropic.com/v1").rstrip("/")
url = root if root.endswith("/messages") else f"{root}/messages"
if not api_key:
raise ValueError("Claude requires an API key or ANTHROPIC_API_KEY")
result = _http_json(
url,
{
"model": model,
"max_tokens": 900,
"temperature": float(temperature),
"system": NEXTFRAME_ASSISTANT_SYSTEM_PROMPT,
"messages": [{"role": "user", "content": user}],
},
{"x-api-key": api_key, "anthropic-version": "2023-06-01"},
timeout,
)
content = "".join(
str(item.get("text") or "")
for item in (result.get("content") or [])
if isinstance(item, dict)
)
else:
raise ValueError(f"Unsupported AI provider: {provider}")
prompt = str(_extract_json_object(content).get("prompt", "") or "").strip()
prompt = re.sub(r"^next\s+scene\s*:\s*", "", prompt, flags=re.I).strip()
if not prompt:
raise RuntimeError("The AI did not return a usable Next Scene prompt")
prompt = f"Next Scene: {prompt}"
return prompt, {
"provider": provider,
"model": model,
"trigger": "Next Scene:",
"system_prompt_characters": len(NEXTFRAME_ASSISTANT_SYSTEM_PROMPT),
}
NEXTFRAME_IDEA_SYSTEM_PROMPT = """You are IAMCCS Story Idea Director, a visual storyteller for Qwen-Image-Edit-2511 storyboard continuation.
Return JSON only, with exactly this shape:
{"ideas":[{"title":"...","beat":"...","prompt":"Next Scene: ..."}]}
Invent the requested number of distinct, plausible immediately-following storyboard frames. The LOG_LINE is the long-range story direction, not permission to jump to the ending. Read the supplied images when available: Image 1 is the current-frame continuity authority; additional images contribute only the roles stated in the reference map. Preserve identities, wardrobe, props, screen geography, visual style and lighting continuity unless the logline explicitly requires a visible change.
Each idea must:
- advance the story by one clear, filmable visual beat;
- vary action, camera movement, shot size or composition meaningfully;
- avoid dialogue, captions, logos, montage, cuts and events that cannot be shown in one frame;
- use no more than one newly invented story element;
- include a short title, a one-sentence beat, and a 70-140 word English Qwen edit prompt;
- begin its prompt exactly with `Next Scene:`.
Do not repeat ideas, explain your reasoning, or return Markdown. Surprise the user while remaining causally coherent with the logline and visible references."""
def invent_nextframe_ideas_with_ai(
provider: str, base_url: str, model: str, api_key: str, logline: str,
current_prompt: str = "", reference_context: str = "", count: int = 4,
temperature: float = 0.9, timeout: float = 120.0, images: Any = None,
nonce: str = "",
) -> tuple[list[dict[str, str]], dict[str, Any]]:
"""Invent several next-frame alternatives from a story logline and visual references."""
provider = str(provider or "ollama").strip().lower()
if provider == "lm_studio":
base_url = str(base_url or "http://localhost:1234/v1").rstrip("/")
if not base_url.endswith("/v1") and not base_url.endswith("/chat/completions"):
base_url += "/v1"
model = str(model or "").strip()
if not model:
raise ValueError("Select an AI model before using Idea AI")
story = str(logline or "").strip()
if not story:
raise ValueError("Write the story logline before using Idea AI")
count = max(2, min(6, int(count or 4)))
temperature = max(0.4, min(1.2, float(temperature)))
timeout = max(10.0, min(300.0, float(timeout)))
visual_inputs = _normalise_ai_images(images)
api_key = str(api_key or "").strip()
if not api_key:
api_key = {"openai_compatible": os.environ.get("OPENAI_API_KEY", ""), "anthropic": os.environ.get("ANTHROPIC_API_KEY", "")}.get(provider, "")
user = (
f"Create exactly {count} alternative next-frame ideas.\nLOG_LINE:\n{story}\n\n"
f"CURRENT NEXT-SCENE DRAFT (context only):\n{str(current_prompt or '').strip() or '[none]'}\n\n"
f"REFERENCE MAP:\n{str(reference_context or '').strip() or 'Image 1 is the current frame.'}\n\n"
f"RANDOMIZATION NONCE: {str(nonce or '')}"
)
content = ""
if provider == "ollama":
root = _ollama_native_base(base_url)
result = _http_json(f"{root}/api/chat", {
"model": model, "stream": False, "format": "json",
"messages": [
{"role": "system", "content": NEXTFRAME_IDEA_SYSTEM_PROMPT},
{"role": "user", "content": user, **({"images": [item["data"] for item in visual_inputs]} if visual_inputs else {})},
], "options": {"temperature": temperature},
}, {}, timeout)
content = str((result.get("message") or {}).get("content") or "")
elif provider in {"openai_compatible", "lm_studio"}:
root = str(base_url or "https://api.openai.com/v1").rstrip("/")
url = root if root.endswith("/chat/completions") else f"{root}/chat/completions"
openai_user: Any = user
if visual_inputs:
openai_user = [{"type": "text", "text": user}] + [
{"type": "image_url", "image_url": {"url": f"data:{item['mime_type']};base64,{item['data']}"}}
for item in visual_inputs
]
result = _http_json(url, {
"model": model, "temperature": temperature, "response_format": {"type": "json_object"},
"messages": [{"role": "system", "content": NEXTFRAME_IDEA_SYSTEM_PROMPT}, {"role": "user", "content": openai_user}],
}, {"Authorization": f"Bearer {api_key}"} if api_key else {}, timeout)
choices = result.get("choices") or []
content = str(((choices[0] if choices else {}).get("message") or {}).get("content") or "")
elif provider == "anthropic":
root = str(base_url or "https://api.anthropic.com/v1").rstrip("/")
url = root if root.endswith("/messages") else f"{root}/messages"
if not api_key:
raise ValueError("Claude requires an API key or ANTHROPIC_API_KEY")
anthropic_user: Any = user
if visual_inputs:
anthropic_user = [
{"type": "image", "source": {"type": "base64", "media_type": item["mime_type"], "data": item["data"]}}
for item in visual_inputs
] + [{"type": "text", "text": user}]
result = _http_json(url, {
"model": model, "max_tokens": 3600, "temperature": temperature,
"system": NEXTFRAME_IDEA_SYSTEM_PROMPT, "messages": [{"role": "user", "content": anthropic_user}],
}, {"x-api-key": api_key, "anthropic-version": "2023-06-01"}, timeout)
content = "".join(str(item.get("text") or "") for item in (result.get("content") or []) if isinstance(item, dict))
else:
raise ValueError(f"Unsupported AI provider: {provider}")
raw_ideas = _extract_json_payload(content).get("ideas")
if not isinstance(raw_ideas, list):
raise RuntimeError("Idea AI did not return an ideas list")
ideas: list[dict[str, str]] = []
for index, item in enumerate(raw_ideas[:count]):
if not isinstance(item, dict):
continue
prompt = re.sub(r"^next\s+scene\s*:\s*", "", str(item.get("prompt") or "").strip(), flags=re.I).strip()
if not prompt:
continue
ideas.append({
"title": str(item.get("title") or f"Scene idea {index + 1}").strip()[:120],
"beat": str(item.get("beat") or "").strip()[:500],
"prompt": f"Next Scene: {prompt}",
})
if not ideas:
raise RuntimeError("Idea AI did not return any usable scene idea")
return ideas, {
"provider": provider, "model": model, "requested": count, "returned": len(ideas),
"visual_references": len(visual_inputs), "temperature": temperature,
"system_prompt_characters": len(NEXTFRAME_IDEA_SYSTEM_PROMPT),
}
def _linx_resources(value: Any) -> dict[str, Any]:
if not isinstance(value, dict):
return {}
resources = value.get("resources")
return resources if isinstance(resources, dict) else {}
VISION_CONTEXT_SCHEMA = "iamccs.minimax_h3.vision_context"
VISION_CONTEXT_RESOURCE = "iamccs_h3_vision_context_by_target"
VISION_MANIFEST_RESOURCE = "iamccs_h3_vision_manifest"
def _format_vision_context(task_mode: str, context: Any) -> str:
text = str(context or "").strip()
if not text:
return ""
if str(task_mode or "").strip().lower() == "ref2va":
return f"visual_reference_analysis:\n{text}"
return f"[VISUAL REFERENCE ANALYSIS]\n{text}"
def _vision_context_requests(
cine_linx: Any,
task_mode: str,
primary_target: str,
) -> tuple[dict[str, str], str, dict[str, Any]]:
resources = _linx_resources(cine_linx)
manifest = resources.get(VISION_MANIFEST_RESOURCE)
contexts = resources.get(VISION_CONTEXT_RESOURCE)
if not isinstance(manifest, dict) or manifest.get("schema") != VISION_CONTEXT_SCHEMA:
return {}, "append", {"consumed": False, "reason": "no_h3_vision_manifest"}
if not isinstance(contexts, dict):
contexts = manifest.get("target_contexts")
if not isinstance(contexts, dict):
return {}, "append", {"consumed": False, "reason": "no_h3_vision_target_contexts"}
allowed_targets = {"global", "local_auto", "local_1", "local_2", "local_3"}
clean: dict[str, str] = {}
for target, context in contexts.items():
target_name = str(target or "").strip().lower()
if target_name not in allowed_targets:
continue
formatted = _format_vision_context(task_mode, context)
if formatted:
clean[target_name] = formatted[:H3_ABSOLUTE_CHAR_LIMIT]
policy = str(
resources.get(
"iamccs_h3_vision_context_merge_policy",
manifest.get("context_merge_policy", "append"),
)
or "append"
).strip().lower()
policy = policy if policy in {"append", "replace"} else "append"
return clean, policy, {
"consumed": bool(clean),
"schema_version": manifest.get("schema_version"),
"status": manifest.get("status"),
"analysis_mode": manifest.get("analysis_mode"),
"available_targets": sorted(clean),
"primary_target": str(primary_target),
"merge_policy": policy,
"pictures": [
{
"slot": item.get("slot"),
"role": item.get("role"),
"target": item.get("target"),
}
for item in manifest.get("pictures", [])
if isinstance(item, dict)
],
}
def _append_prompter_stage(
upstream_linx: Any,
injection: dict[str, Any],
injections: list[dict[str, Any]],
final_prompt: str,
project_json: str,
report: str,
mode: str,
injection_target: str,
character_count: int,
) -> dict[str, Any]:
out = dict(upstream_linx) if isinstance(upstream_linx, dict) else {}
out["type"] = SUPERNODE_LINX_TYPE
out["mode"] = "iamccs_minimax_h3_prompter"
out["active_stage"] = "iamccs_prompter"
out["active_stage_kind"] = "prompt_authoring"
chain = [dict(item) for item in (out.get("chain") or []) if isinstance(item, dict)]
chain.append({"role": "prompt_author", "name": "IAMCCS_Prompter"})
out["chain"] = chain
stages = [dict(item) for item in (out.get("stages") or []) if isinstance(item, dict)]
stages.append({
"name": "iamccs_prompter",
"kind": "prompt_authoring",
"payload": {
"task_mode": mode,
"target": str(injection_target),
"characters": character_count,
"injection_count": len(injections),
},
})
out["stages"] = stages
out["stage_count"] = len(stages)
resources = dict(_linx_resources(out))
resources.update({
# Keep the singular contract for every older Shotboard/workflow.
"iamccs_prompter_injection": injection,
# New contract: deterministic ordered requests can address global and
# independent local prompt targets in one CineLinX pass.
"iamccs_prompter_injections": [dict(item) for item in injections],
"iamccs_prompter_prompt": final_prompt,
"iamccs_prompter_queue_authority": "shotboard_visible_fields",
"iamccs_prompter_project_json": project_json,
"iamccs_prompter_audio_handoff_rule": AUDIO_HANDOFF_AUTHORING_RULE,
"iamccs_prompter_audio_driven_dialogue_template": "<Subject 1> (S1): <d>[Language] ...</d>",
"cine_report": report,
})
out["resources"] = resources
outputs = dict(out.get("outputs") or {})
outputs.update({
"final_prompt": final_prompt,
"project_json": project_json,
"injection_target": str(injection_target),
"report": report,
})
out["outputs"] = outputs
out["resource_keys"] = sorted(resources)
out["resource_types"] = {key: type(value).__name__ for key, value in resources.items()}
return out
def _apply_one_prompter_request(
global_prompt: str,
timeline_data: Any,
request: dict[str, Any],
) -> tuple[str, str, dict[str, Any]]:
"""Apply one explicit request without reading mutable state from CineLinX."""
prompt = str(request.get("prompt", "") or "").strip()
target = str(request.get("target", "global") or "global").strip().lower()
policy = str(request.get("merge_policy", "replace") or "replace").strip().lower()
if not prompt:
return str(global_prompt or ""), str(timeline_data or ""), {"applied": False, "reason": "empty_prompter_prompt"}
if target == "global":
merged = _merge_text(str(global_prompt or ""), prompt, policy)
return merged, str(timeline_data or ""), {
"applied": True,
"requested_target": target,
"actual_target": "global",
"merge_policy": policy,
}
raw_timeline = str(timeline_data or "").strip()
try:
timeline = json.loads(raw_timeline) if raw_timeline else {}
except json.JSONDecodeError:
timeline = {}
if not isinstance(timeline, dict):
timeline = {}
rows: list[dict[str, Any]] | None = None
row_key = ""
# Match the Shotboard planner: live editor rows are Queue truth, while the
# segments field is only a legacy mirror that may contain deleted prompts.
for candidate in ("rows", "segments", "slots", "shots"):
value = timeline.get(candidate)
if isinstance(value, list):
rows = value
row_key = candidate
break
visual: list[tuple[int, dict[str, Any]]] = []
if isinstance(rows, list):
for row_index, row in enumerate(rows):
if not isinstance(row, dict) or bool(row.get("placeholder", False)):
continue
row_type = str(row.get("type", "image") or "image").strip().lower()
if row_type in {"audio", "motion", "video"}:
continue
visual.append((row_index, row))
if not visual:
merged = _merge_text(str(global_prompt or ""), prompt, "append" if str(global_prompt or "").strip() else "replace")
return merged, raw_timeline, {
"applied": True,
"requested_target": target,
"actual_target": "global_fallback_no_local_slots",
"merge_policy": "append" if str(global_prompt or "").strip() else "replace",
}
limit = min(3, len(visual))
selected_position: int | None = None
effective_policy = policy
if target == "local_auto":
for position in range(limit):
row = visual[position][1]
existing = str(row.get("prompt", row.get("local_prompt", row.get("relay_prompt", ""))) or "").strip()
if not existing:
selected_position = position
break
if selected_position is None:
selected_position = limit - 1
# Auto must never silently destroy three completed local prompts.
effective_policy = "append"
else:
match = re.fullmatch(r"local_([123])", target)
requested_position = int(match.group(1)) - 1 if match else 0
if requested_position < limit:
selected_position = requested_position
else:
# Detect what is actually present and select the first empty slot,
# otherwise the final available slot without creating fake timing.
selected_position = next(
(
position
for position in range(limit)
if not str(visual[position][1].get("prompt", visual[position][1].get("local_prompt", "")) or "").strip()
),
limit - 1,
)
assert selected_position is not None
actual_row_index, row = visual[selected_position]
existing = str(row.get("prompt", row.get("local_prompt", row.get("relay_prompt", ""))) or "")
merged = _merge_text(existing, prompt, effective_policy)
row["prompt"] = merged
row["local_prompt"] = merged
row["relay_prompt"] = merged
row["use_prompt"] = True
row["relay_manual_off"] = False
row["promptrelay_manual_off"] = False
timeline[row_key] = rows
return str(global_prompt or ""), json.dumps(timeline, ensure_ascii=False), {
"applied": True,
"requested_target": target,
"actual_target": f"local_{selected_position + 1}",
"timeline_row_index": actual_row_index,
"merge_policy": effective_policy,
"available_local_slots": len(visual),
}
def apply_prompter_to_minimax(
cine_linx: Any,
global_prompt: str,
timeline_data: Any,
) -> tuple[str, str, dict[str, Any]]:
"""Preserve Shotboard visible fields as the only queue-time prompt truth.
Prompter requests are authoring metadata. The UI's explicit INJECT action
writes the chosen text into Shotboard before Queue. Re-applying the
connected node's serialized project here made stale demo content override
later Shotboard edits, so queue-time auto-application is intentionally
disabled for both current and legacy Prompter payloads.
"""
resources = _linx_resources(cine_linx)
requests = resources.get("iamccs_prompter_injections")
if not isinstance(requests, list):
legacy = resources.get("iamccs_prompter_injection")
requests = [legacy] if isinstance(legacy, dict) else []
requests = [dict(item) for item in requests if isinstance(item, dict)]
if not requests:
return str(global_prompt or ""), str(timeline_data or ""), {
"applied": False,
"reason": "no_prompter_cine_linx",
}
return str(global_prompt or ""), str(timeline_data or ""), {
"applied": False,
"reason": "shotboard_visible_fields_are_queue_truth",
"requested_count": len(requests),
"ignored_stale_requests": len(requests),
"applied_count": 0,
"actual_target": "none",
"actual_targets": [],
"applications": [],
"multi_target_contract": len(requests) > 1,
"queue_authority": "shotboard_visible_fields",
}
class IAMCCS_Prompter:
@classmethod
def INPUT_TYPES(cls):
return {
"required": {
"project_data": (
"STRING",
{
"default": json.dumps(default_project(), ensure_ascii=False),
"multiline": True,
},
),
"task_mode": (
["t2va", "i2va", "fl2va", "ref2va", "v2va_object_swap", "audio_driven"],
{"default": "t2va"},
),
"injection_target": (
["global", "local_auto", "local_1", "local_2", "local_3"],
{"default": "global"},
),
"writing_mode": (
["manual", "guided", "assistant_fill"],
{"default": "guided"},
),
"merge_policy": (["replace", "append"], {"default": "replace"}),
"character_budget": ("INT", {"default": 6800, "min": 1000, "max": H3_ABSOLUTE_CHAR_LIMIT, "step": 100}),
"audio_transcription_model": (
["tiny", "base", "small", "medium", "medium.en", "large-v2", "large-v3", "large-v3-turbo"],
{"default": "tiny", "tooltip": "Whisper model used by comfy-mtb when an AUDIO input is connected. Tiny is the installed low-VRAM default; larger models are optional."},
),
"audio_transcription_language": (
["auto", "de", "en", "es", "fr", "it", "ja", "ko", "nl", "pt", "ru", "zh"],
{"default": "auto"},
),
"audio_dialogue_language": (
["English", "Italian", "French", "German", "Spanish", "Portuguese", "Arabic", "Chinese", "Japanese", "Korean", "Russian"],
{"default": "English", "tooltip": "Language label written inside the H3 <d> block. It does not translate the transcript."},
),
"audio_dialogue_subject": (
["1", "2", "3", "4"],
{"default": "1", "tooltip": "Stable H3 <Subject N> / (SN) identity used by the transcript insertion button."},
),
},
"optional": {
"cine_linx": (
SUPERNODE_LINX_TYPE,
{
"tooltip": (
"Optional upstream IAMCCS Cine H3 Vision Info. Its analyzed visual context is "
"routed to the declared global/local targets without embedding image tensors in this node."
),
},
),
"assistant_draft": (
"STRING",
{
"default": "",
"multiline": True,
"forceInput": True,
"tooltip": "Optional structured draft from any text source. In Assistant Fill mode it fills only empty structured boxes.",
},
),
"audio": (
"AUDIO",
{
"lazy": True,
"tooltip": (
"Loaded only for the explicit TRANSCRIBE button. Ordinary global Queue leaves this connected AUDIO branch inert."
),
},
),
},
}
RETURN_TYPES = (SUPERNODE_LINX_TYPE, "STRING", "STRING", "STRING", "STRING", "STRING")
RETURN_NAMES = ("cine_linx", "final_prompt", "project_json", "report", "audio_transcript", "h3_dialogue_tag")
FUNCTION = "compose"
CATEGORY = CATEGORY
# The UI's TRANSCRIBE button queues this node as a partial execution target.
# ComfyUI accepts partial targets only when the target class is an output node.
OUTPUT_NODE = True
@classmethod
def IS_CHANGED(cls, **kwargs):
return json.dumps(kwargs, ensure_ascii=False, sort_keys=True, default=str)
def check_lazy_status(self, project_data, audio=None, **kwargs):
# A connected Load Audio must not run for every normal H3 generation.
return ["audio"] if _safe_project(project_data).get("_transcribe_once") and audio is None else []
def compose(
self,
project_data,
task_mode,
injection_target,
writing_mode,
merge_policy,
character_budget,
audio_transcription_model="tiny",
audio_transcription_language="auto",
audio_dialogue_language="English",
audio_dialogue_subject="1",
assistant_draft="",
cine_linx=None,
audio=None,
):
project = _safe_project(project_data)
transcribe_once = bool(project.pop("_transcribe_once", False))
mode = _normalise_task_mode(task_mode or project.get("task_mode") or "t2va")
project["task_mode"] = mode
project["injection_target"] = str(injection_target)
project["writing_mode"] = str(writing_mode)
project["merge_policy"] = str(merge_policy)
generated_prompt, details = _compose_prompt(project, mode, str(writing_mode), str(assistant_draft or ""))
if mode == "fl2va" and project.get("extended_conditioning_policy") == "continuous":
generated_prompt = _continuous_global_prompt(generated_prompt, details.get("local_prompt"))
final_prompt = (
str(project.get("final_prompt_override") or "").strip()
if project.get("final_prompt_override_enabled")
else generated_prompt
)
local_prompt = str(details.get("local_prompt") or "").strip()
if mode == "fl2va" and project.get("extended_conditioning_policy") == "continuous":
local_prompt = ""
if mode == "fl2va" and project.get("extended_conditioning_policy") == "evolving":
local_prompt = str(project.get("evolving_timeline") or local_prompt).strip()
if mode == "fl2va" and project.get("final_local_prompt_override_enabled"):
local_prompt = str(project.get("final_local_prompt_override") or "").strip()
if mode == "fl2va" and project.get("extended_conditioning_policy") == "evolving":
final_prompt = _validate_evolving_global_prompt(final_prompt)
if local_prompt:
local_prompt = _validate_canonical_evolving_timeline(local_prompt)
project["evolving_timeline"] = local_prompt
project["sections"]["action"] = local_prompt
project["sections"]["shot_list"] = local_prompt
if not final_prompt and not local_prompt and not transcribe_once:
raise ValueError("IAMCCS_Prompter: compila almeno un box prima di accodare il workflow")
primary_target = str(injection_target or "global").strip().lower()
vision_contexts, vision_merge_policy, vision_report = _vision_context_requests(
cine_linx,
mode,
primary_target,
)
primary_vision_context = vision_contexts.pop(primary_target, "")
if primary_vision_context:
final_prompt = _merge_text(final_prompt, primary_vision_context, vision_merge_policy)
char_count = len(final_prompt) + (len(local_prompt) if mode == "fl2va" else 0)
budget = min(H3_ABSOLUTE_CHAR_LIMIT, max(1000, int(character_budget)))
if char_count > H3_ABSOLUTE_CHAR_LIMIT:
raise ValueError(
f"IAMCCS_Prompter: prompt di {char_count} caratteri; MiniMax H3 richiede massimo "
f"{H3_ABSOLUTE_CHAR_LIMIT}. Riduci i box di almeno {char_count - H3_ABSOLUTE_CHAR_LIMIT} caratteri."
)
effective_primary_target = "global" if mode == "fl2va" else str(injection_target)
injection = {
"schema": "iamccs.minimax_h3.prompt_injection",
"schema_version": 2,
"prompt": final_prompt,
"target": effective_primary_target,
"merge_policy": str(merge_policy),
"task_mode": mode,
"project_name": str(project.get("project_name") or "Untitled Prompt"),
"source": "iamccs_prompter",
"extended_conditioning_policy": project["extended_conditioning_policy"],
"evolving_timeline": project["evolving_timeline"],
}
injections = [injection] if final_prompt else []
if mode == "fl2va" and local_prompt and project.get("extended_conditioning_policy") == "default":
local_target = primary_target if re.match(r"^local_[1-9][0-9]*$", primary_target) else "local_1"
injections.append({
"schema": "iamccs.minimax_h3.prompt_injection",
"schema_version": 2,
"prompt": local_prompt,
"target": local_target,
"merge_policy": str(merge_policy),
"task_mode": mode,
"project_name": str(project.get("project_name") or "Untitled Prompt"),
"source": "iamccs_prompter_fl2va_local",
})
for target, context in vision_contexts.items():
if not str(context or "").strip():
continue
if len(context) > H3_ABSOLUTE_CHAR_LIMIT:
raise ValueError(
f"IAMCCS_Prompter: visual context for {target} contains {len(context)} characters; "
f"the MiniMax H3 request limit is {H3_ABSOLUTE_CHAR_LIMIT}."
)
injections.append({
"schema": "iamccs.minimax_h3.prompt_injection",
"schema_version": 1,
"prompt": context,
"target": target,
"merge_policy": vision_merge_policy,
"task_mode": mode,
"project_name": injection["project_name"],
"source": "iamccs_cine_h3_vision_info",
})
transcript = str(project.get("audio_transcript") or "")
dialogue_tag = str(project.get("audio_dialogue_tag") or "")
transcription_error = ""
if transcribe_once and audio is not None:
transcript = ""
dialogue_tag = ""
try:
try:
from .iamccs_cine_audio_dialogue import IAMCCS_CineAudioTranscriptPromptCompiler
except ImportError:
from iamccs_cine_audio_dialogue import IAMCCS_CineAudioTranscriptPromptCompiler
transcript = IAMCCS_CineAudioTranscriptPromptCompiler._clean(
IAMCCS_CineAudioTranscriptPromptCompiler._transcribe(
audio,
str(audio_transcription_model or "tiny"),
str(audio_transcription_language or "auto"),
False,
False,
)
)
if transcript:
subject_index = max(1, min(4, int(audio_dialogue_subject or 1)))
language_label = str(audio_dialogue_language or "English").strip() or "English"
dialogue_tag = f"<Subject {subject_index}> (S{subject_index}): <d>[{language_label}] {transcript}</d>"
except Exception as exc:
transcription_error = repr(exc)
project["audio_transcript"] = transcript
project["audio_dialogue_tag"] = dialogue_tag
project_json = json.dumps(project, ensure_ascii=False, indent=2)
report_data = {
"node": "IAMCCS_Prompter",
"project_name": injection["project_name"],
"task_mode": mode,
"writing_mode": str(writing_mode),
"requested_target": str(injection_target),
"merge_policy": str(merge_policy),
"characters": char_count,
"global_characters": len(final_prompt),
"local_characters": len(local_prompt),
"final_prompt_override": bool(project.get("final_prompt_override_enabled")),
"final_local_prompt_override": bool(project.get("final_local_prompt_override_enabled")),
"character_budget": budget,
"within_recommended_budget": char_count <= budget,
"injection_count": len(injections),
"injection_targets": [item["target"] for item in injections],
"vision_context": vision_report,
"audio_handoff_authoring_rule": AUDIO_HANDOFF_AUTHORING_RULE,
"extended_conditioning": {
"policy": project["extended_conditioning_policy"],
"evolving_event_count": len(parse_evolving_timeline(project["evolving_timeline"], duration_seconds=86400.0))
if project["extended_conditioning_policy"] == "evolving" else 0,
"required_shotboard_mode": "fl2va_extended_av",
},
"audio_driven_dialogue_template": "<Subject 1> (S1): <d>[Language] ...</d>",
"audio_transcription": {
"requested": transcribe_once,
"engine": "comfy-mtb Whisper",
"model": str(audio_transcription_model),
"source_language": str(audio_transcription_language),
"dialogue_language": str(audio_dialogue_language),
"subject": str(audio_dialogue_subject),
"characters": len(transcript),
"error": transcription_error,
"cursor_insertion_required": bool(dialogue_tag),
},
**details,
"truth": "The MiniMax Shotboard resolves local_auto only after reading its own timeline slots.",
}
report = json.dumps(report_data, ensure_ascii=False, indent=2)
out_linx = _append_prompter_stage(
cine_linx,
injection,
injections,
final_prompt,
project_json,
report,
mode,
str(injection_target),
char_count,
)
resources = out_linx.setdefault("resources", {})
resources["iamccs_prompter_audio_transcript"] = transcript
resources["iamccs_prompter_h3_dialogue_tag"] = dialogue_tag
resources["iamccs_prompter_local_prompt"] = local_prompt
out_linx.setdefault("outputs", {})["audio_transcript"] = transcript
out_linx["outputs"]["h3_dialogue_tag"] = dialogue_tag
ui_status = (
f"Whisper transcript ready ({len(transcript)} characters). Insert the H3 dialogue tag at the desired cursor."
if transcribe_once and dialogue_tag else
(f"Whisper transcription failed: {transcription_error}" if transcription_error else "No AUDIO input connected; transcript stage skipped.")
)
return {
"ui": {
"iamccs_audio_transcript": [transcript],
"iamccs_h3_dialogue_tag": [dialogue_tag],
"iamccs_audio_transcription_error": [transcription_error],
"text": [ui_status],
},
"result": (out_linx, final_prompt, project_json, report, transcript, dialogue_tag),
}
def _register_prompter_routes() -> None:
"""Register the interactive AI rewrite endpoint without adding a dependency."""
try:
from aiohttp import web
from server import PromptServer
routes = PromptServer.instance.routes
@routes.get("/iamccs/prompter/lmstudio/models")
async def iamccs_prompter_lmstudio_models(request):
try:
root = str(request.query.get("base_url") or "http://localhost:1234/v1").rstrip("/")
if not root.endswith("/v1"):
root += "/v1"
payload = await asyncio.to_thread(_http_get_json, f"{root}/models", 10.0)
models = [{"name": str(item["id"])} for item in payload.get("data", [])
if isinstance(item, dict) and item.get("id")]
return web.json_response({"ok": True, "models": models})
except Exception as exc:
return web.json_response({"ok": False, "error": str(exc)}, status=400)
@routes.get("/iamccs/prompter/ollama/models")
async def iamccs_prompter_ollama_models(request):
try:
base_url = _ollama_native_base(request.query.get("base_url"))
payload = await asyncio.to_thread(_http_get_json, f"{base_url}/api/tags", 10.0)
models = []
for item in payload.get("models") if isinstance(payload.get("models"), list) else []:
if not isinstance(item, dict):
continue
name = str(item.get("name") or item.get("model") or "").strip()
if name:
models.append({
"name": name,
"size": int(item.get("size") or 0),
"modified_at": str(item.get("modified_at") or ""),
})
return web.json_response({"ok": True, "models": models})
except Exception as exc:
return web.json_response({"ok": False, "error": str(exc)}, status=400)
@routes.post("/iamccs/prompter/rewrite")
async def iamccs_prompter_rewrite(request):
try:
payload = await request.json()
sections = payload.get("sections") if isinstance(payload, dict) else None
if not isinstance(sections, dict):
raise ValueError("sections must be a JSON object")
rewritten, report = await asyncio.to_thread(
rewrite_sections_with_ai,
str(payload.get("provider", "ollama")),
str(payload.get("base_url", "")),
str(payload.get("model", "")),
str(payload.get("api_key", "")),
str(payload.get("task_mode", "t2va")),
{str(key): str(value or "") for key, value in sections.items()},
float(payload.get("temperature", 0.35)),
float(payload.get("timeout", 120.0)),
str(payload.get("user_direction", "")),
payload.get("target_keys"),
payload.get("images"),
)
return web.json_response({"ok": True, "sections": rewritten, "report": report})
except Exception as exc:
safe_error = re.sub(
r"(?i)(api[_ -]?key|authorization)[^,;\n]*",
r"\1=[redacted]",
str(exc),
)
return web.json_response({"ok": False, "error": safe_error}, status=400)
@routes.post("/iamccs/nextframe/assist")
async def iamccs_nextframe_assist(request):
try:
payload = await request.json()
if not isinstance(payload, dict):
raise ValueError("Request body must be a JSON object")
prompt, report = await asyncio.to_thread(
rewrite_nextframe_prompt_with_ai,
str(payload.get("provider", "ollama")),
str(payload.get("base_url", "")),
str(payload.get("model", "")),
str(payload.get("api_key", "")),
str(payload.get("user_prompt", "")),
str(payload.get("current_prompt", "")),
float(payload.get("temperature", 0.25)),
float(payload.get("timeout", 120.0)),
)
return web.json_response({"ok": True, "prompt": prompt, "report": report})
except Exception as exc:
safe_error = re.sub(
r"(?i)(api[_ -]?key|authorization)[^,;\n]*",
r"\1=[redacted]",
str(exc),
)
return web.json_response({"ok": False, "error": safe_error}, status=400)
@routes.post("/iamccs/prompter/visual-story")
async def iamccs_prompter_visual_story(request):
try:
payload = await request.json()
plan, report = await asyncio.to_thread(
build_visual_story_plan_with_ai,
str(payload.get("provider", "ollama")), str(payload.get("base_url", "")),
str(payload.get("model", "")), str(payload.get("api_key", "")),
str(payload.get("relationship", "")), str(payload.get("task_mode", "i2va")),
payload.get("images"), float(payload.get("temperature", 0.3)),
float(payload.get("timeout", 150.0)),
)
return web.json_response({"ok": True, "plan": plan, "report": report})
except Exception as exc:
safe_error = re.sub(r"(?i)(api[_ -]?key|authorization)[^,;\n]*", r"\1=[redacted]", str(exc))
return web.json_response({"ok": False, "error": safe_error}, status=400)
@routes.post("/iamccs/nextframe/h3")
async def iamccs_nextframe_h3(request):
try:
payload = await request.json()
result, report = await asyncio.to_thread(
rewrite_nextframe_h3_with_ai,
str(payload.get("provider", "ollama")), str(payload.get("base_url", "")),
str(payload.get("model", "")), str(payload.get("api_key", "")),
str(payload.get("qwen_prompt", "")), str(payload.get("requested_mode", "auto")),
payload.get("images"), float(payload.get("temperature", 0.25)),
float(payload.get("timeout", 150.0)),
)
return web.json_response({"ok": True, **result, "report": report})
except Exception as exc:
safe_error = re.sub(r"(?i)(api[_ -]?key|authorization)[^,;\n]*", r"\1=[redacted]", str(exc))
return web.json_response({"ok": False, "error": safe_error}, status=400)
@routes.post("/iamccs/nextframe/ideas")
async def iamccs_nextframe_ideas(request):
try:
payload = await request.json()
if not isinstance(payload, dict):
raise ValueError("Request body must be a JSON object")
ideas, report = await asyncio.to_thread(
invent_nextframe_ideas_with_ai,
str(payload.get("provider", "ollama")), str(payload.get("base_url", "")),
str(payload.get("model", "")), str(payload.get("api_key", "")),
str(payload.get("logline", "")), str(payload.get("current_prompt", "")),
str(payload.get("reference_context", "")), int(payload.get("count", 4)),
float(payload.get("temperature", 0.9)), float(payload.get("timeout", 120.0)),
payload.get("images"), str(payload.get("nonce", "")),
)
return web.json_response({"ok": True, "ideas": ideas, "report": report})
except Exception as exc:
safe_error = re.sub(r"(?i)(api[_ -]?key|authorization)[^,;\n]*", r"\1=[redacted]", str(exc))
return web.json_response({"ok": False, "error": safe_error}, status=400)
except Exception:
# Schema discovery and headless tests can import before PromptServer.
return
_register_prompter_routes()
NODE_CLASS_MAPPINGS = {"IAMCCS_Prompter": IAMCCS_Prompter}
NODE_DISPLAY_NAME_MAPPINGS = {"IAMCCS_Prompter": "IAMCCS Prompter — MiniMax H3 Screenplay"}