resolution.py now uses the vendor adapt_canvas rule, camera vocabulary and writer directives are updated, workflows audited

This commit is contained in:
Dag Thomas Olsen
2026-08-29 21:59:59 +02:00
parent 8b58c7a370
commit 61f87db161
6 changed files with 295 additions and 84 deletions
+10 -3
View File
@@ -92,8 +92,13 @@ class H3BasePromptWriter:
"tooltip": "Which H3 task the prompt targets. Anything other than T2VA emits the matching reference-alignment instruction line.",
}),
"duration_seconds": ("FLOAT", {
"default": 6.0, "min": 1.0, "max": 60.0, "step": 0.5,
"tooltip": "Effective video duration. Drives the cut times and the S.SS value in the alignment instruction.",
"default": 8.0, "min": 1.0, "max": 60.0, "step": 0.5,
"tooltip": (
"Effective video duration. Drives the cut times and the S.SS value in the "
"alignment instruction. The render snaps frames UP to the 17n+5 grid, so "
"prefer grid durations - 8.00s (192f) is the only common integer one; "
"the trained ceiling is 15.083s (362f)."
),
}),
"shot_plan": (SHOT_PLANS, {"default": AUTO}),
"visual_style": (VISUAL_STYLES, {
@@ -253,7 +258,9 @@ class H3BasePromptWriter:
"Replace N with the index of the actual final shot. Follow it with one blank "
"line, then the core fields. <Picture 1> is the final frame and belongs to the "
"last shot, not Shot 1: infer a plausible earlier state and converge onto the "
"image (preceding state -> transition path -> gradual convergence -> landing)."
"image (preceding state -> transition path -> gradual convergence -> landing). "
"If the final frame shows a closed mouth, finish all dialogue early enough for "
"the mouth to return to that closed position by the end."
)
def _reference_rule(self, image_count, references):
+6 -2
View File
@@ -66,8 +66,12 @@ class H3ClaudeCodeBaseWriter(H3BasePromptWriter):
"tooltip": "Which H3 task the prompt targets. Anything other than T2VA emits the matching reference-alignment instruction line.",
}),
"duration_seconds": ("FLOAT", {
"default": 6.0, "min": 1.0, "max": 60.0, "step": 0.5,
"tooltip": "Effective video duration. Drives the cut times and the S.SS value in the alignment instruction.",
"default": 8.0, "min": 1.0, "max": 60.0, "step": 0.5,
"tooltip": (
"Effective video duration. Drives the cut times and the S.SS value in the "
"alignment instruction. The render snaps frames UP to the 17n+5 grid, so "
"prefer grid durations - 8.00s (192f) is the only common integer one."
),
}),
"shot_plan": (SHOT_PLANS, {"default": AUTO}),
"visual_style": (VISUAL_STYLES, {
+5 -2
View File
@@ -130,8 +130,11 @@ class H3ClaudeCodeContinueWriter(H3BasePromptWriter):
),
}),
"duration_seconds": ("FLOAT", {
"default": 6.0, "min": 1.0, "max": 60.0, "step": 0.5,
"tooltip": "Length of the NEW clip.",
"default": 8.0, "min": 1.0, "max": 60.0, "step": 0.5,
"tooltip": (
"Length of the NEW clip. The render snaps frames UP to the 17n+5 grid, so "
"prefer grid durations - 8.00s (192f) is the only common integer one."
),
}),
"shot_plan": (SHOT_PLANS, {"default": AUTO}),
"visual_style": (VISUAL_STYLES, {
+82 -12
View File
@@ -97,14 +97,14 @@ _CINEMATIC_LOOKS = [
"Live-action, 35mm cinematic film aesthetic",
"Live-action, modern large-format digital cinema: Alexa 65 clarity, creamy shallow depth of field, natural HDR skin tones, smooth floating gimbal moves, neutral filmic grade",
"Live-action, 65mm epic scale: IMAX-format deep-focus vistas, slow deliberate crane and dolly moves, natural available light, cool desaturated grade with warm protected skin tones",
"Live-action, Wes Anderson style: perfectly symmetrical planimetric framing, deadpan centered portraits and 90-degree whip pans, pastel candy-box palette, meticulous dollhouse production design, flat 40mm perspective",
"Live-action, Wes Anderson style: perfectly symmetrical planimetric framing, deadpan centered portraits and sudden fast 90-degree pans, pastel candy-box palette, meticulous dollhouse production design, flat 40mm perspective",
"Live-action, David Fincher style: locked-down surgically precise camera, low-key tungsten and sodium-vapour practicals, cold teal-green shadow grade, crisp clinical digital sharpness",
"Live-action, Roger Deakins naturalism: motivated single-source lighting, silhouettes against windows and fire, clean geometric widescreen compositions, gentle drifting camera, restrained filmic grade",
"Live-action, Denis Villeneuve scale: monolithic minimalist compositions, tiny figures in vast spaces, atmospheric dust and haze diffusion, slow creeping push-ins, muted near-monochrome grade",
"Live-action, Stanley Kubrick style: one-point-perspective symmetry, slow menacing zooms, ultra-wide 18mm interiors, cold practical-lit spaces, immaculate composed stillness",
"Live-action, Terrence Malick golden hour: wide-angle handheld lyricism, backlit magic-hour sun flare, natural bounce light only, wandering steadicam through grass and doorways",
"Live-action, Wong Kar-wai style: step-printed motion smear, saturated neon greens and reds, handheld intimacy in cramped interiors, rain-streaked glass and mirrors, romantic halation glow",
"Live-action, Quentin Tarantino grindhouse: punchy 35mm Kodak saturation, crash zooms and low trunk-shot angles, long unbroken dialogue two-shots, warm 70s amber cast",
"Live-action, Quentin Tarantino grindhouse: punchy 35mm Kodak saturation, sudden fast zoom-ins and low trunk-shot angles, long unbroken dialogue two-shots, warm 70s amber cast",
"Live-action, 1970s paranoid thriller: 2.39 anamorphic Panavision, zoom-heavy long-lens surveillance framing, grimy urban browns and greens, grainy push-processed stock",
"Live-action, A24 indie realism: 35mm grain, soft natural window light, muted pastel grade with lifted blacks, static tableaux broken by intimate handheld close-ups",
"Live-action, neon noir: rain-slick night streets, cyan and magenta neon reflections, volumetric smoke and searchlights, anamorphic blue-streak flares, deep crushed blacks",
@@ -112,7 +112,7 @@ _CINEMATIC_LOOKS = [
"Live-action, golden-age Technicolor: three-strip saturated primaries, glamour key lighting with crisp eye-lights, stately dolly moves, painted-backdrop soundstage depth",
"Live-action, 16mm vérité documentary: handheld reportage framing, visible film grain and gate weave, available light only, quick reactive zooms and refocuses",
"Live-action, Super 8 home movie: heavy grain and light leaks, warm faded Kodachrome colours, jittery handheld framing, soft vignetted frame edges",
"Live-action, late-80s VHS camcorder: smeared lo-fi video texture, auto-exposure pumping, bleeding oversaturated colour, abrupt handheld pans and snap zooms",
"Live-action, late-80s VHS camcorder: smeared lo-fi video texture, auto-exposure pumping, bleeding oversaturated colour, abrupt handheld pans and fast zoom-ins",
"Live-action, BBC nature documentary: long-lens telephoto compression, patient locked-off observation, golden dawn haze, pristine 8K clarity, sweeping aerial establishing shots",
"Live-action, glossy blockbuster: sweeping 360-degree arc shots circling the hero at chest height, teal-and-orange grade, horizontal lens flares, low-angle wide shots, crisp high-shutter action",
"Live-action, high-fashion editorial: glossy beauty lighting, bold saturated gel colours, slow-motion hair and fabric, macro texture inserts, high-contrast punchy grade",
@@ -308,6 +308,30 @@ CUT_STYLES = [
]
# Camera language is a CLOSED vocabulary (guide 4.3) - and it matters
# empirically: fixing exactly four off-vocabulary phrases in a default prompt
# ("tracks left", "medium amplitude and moderate speed", "whip pan", a
# misplaced speaker intro) turned a same-seed render from "looks and sounds
# wrong" to "fantastic". Every writer gets this rule with the camera line.
CAMERA_VOCABULARY_RULE = (
"Camera language is a CLOSED vocabulary - the only motion types are Zoom "
"In/Out, Push In, Pull Out, Pan Left/Right, Truck Left/Right, Tilt Up/Down, "
"Pedestal Up/Down, Arc Shot, Tracking Shot, Static Shot, Shake "
"Slightly/Strongly, POV and Roll Clockwise/Counterclockwise; the only "
"amplitudes `with small amplitude` / `with large amplitude`; the only "
"speeds `at slow speed` / `at fast speed`. Medium and normal are OMITTED, "
"never written - `medium amplitude`, `moderate speed` and `normal speed` "
"do not exist. Never write `whip pan`, `crash zoom`, `snap zoom`, or "
"`tracks left/right` (which conflates Truck Left with Tracking Shot); "
"re-express such moves inside the vocabulary (`a fast Pan Right with large "
"amplitude`, `a fast Zoom In`). One primary camera change per beat - two "
"simultaneous changes collapse into whichever is easier to render. A "
"camera move lands more reliably when its visible consequences are written "
"into the scene (what shifts against what, where the light moves) than as "
"a bare asserted sentence."
)
# The video model takes every word at face value: `the camera orbits her`
# can render outer space, not an arc shot. Every writer appends this so the
# scene text spells out what is physically meant.
@@ -335,7 +359,7 @@ def camera_directive(motion, amplitude, speed):
"Camera motion: choose motion types that suit the action, and write them as "
"natural English inside the shot (motion type, plus amplitude and speed only "
"when meaningful). Do not stack them as labels at the end of a sentence. "
+ LITERAL_CAMERA_DIRECTIVE
+ CAMERA_VOCABULARY_RULE + " " + LITERAL_CAMERA_DIRECTIVE
)
parts = [motion]
@@ -345,11 +369,20 @@ def camera_directive(motion, amplitude, speed):
parts.append(speed)
phrase = " ".join(parts)
# H3 reframes on its own by default: a bare "static shot" is not enough to
# actually hold a composition - the moves that do NOT happen must be named.
hold = (
"H3 reframes by default, so to truly hold the composition state that the "
"camera does not pan, zoom, push or drift in addition to naming the "
"static shot. "
if motion == "Static Shot"
else ""
)
return (
f"Camera motion: the primary camera movement is `{phrase}`. Express it as natural "
"English action inside the shot rather than as a trailing label. Additional shots "
"may use other motion types when the action calls for it. "
+ LITERAL_CAMERA_DIRECTIVE
f"may use other motion types when the action calls for it. {hold}"
+ CAMERA_VOCABULARY_RULE + " " + LITERAL_CAMERA_DIRECTIVE
)
@@ -357,10 +390,27 @@ def shot_directive(shot_plan, duration_seconds):
"""Instruction covering shot count and cut-time formatting."""
duration = f"{duration_seconds:.2f}"
# The render snaps frame counts UP onto the 17n+5 grid, so an off-grid
# duration means the prompt narrates a shorter video than gets made.
# 192 frames = 8.000s is the only common integer duration on the grid.
frames24 = round(duration_seconds * 24)
if (frames24 - 5) % 17:
lo = 17 * max(1, (frames24 - 5) // 17) + 5
hi = lo + 17
print(
f"ℹ️ H3 duration {duration_seconds:.2f}s ({frames24}f) is off the 17n+5 frame grid "
f"the render snaps up to - the video will run longer than the prompt narrates. "
f"Nearest grid durations: {lo / 24:.3f}s ({lo}f) and {hi / 24:.3f}s ({hi}f); "
f"8.00s (192f) is the only common integer duration that lands exactly."
)
if shot_plan == AUTO:
count_rule = (
"Choose the shot count that fits the action. Prefer a single shot unless a cut "
"genuinely introduces new information about the subject, space, state, viewpoint or time."
"genuinely introduces new information about the subject, space, state, viewpoint "
"or time, and avoid packing three or more distinct setups (different locations "
"or crowded wide compositions) into one clip - detail demanded per scene is what "
"degrades renders, especially with few sampling steps."
)
else:
count = _SHOT_COUNTS[shot_plan]
@@ -405,8 +455,19 @@ def toggle_directives(
)
lines.append(
f"{spoken} Give each vocal source a stable (S1)/(S2) ID, and keep the identifying "
"phrase, action and delivery outside the <d> block and in English."
f"{spoken} Give each vocal source a stable (S1)/(S2) ID - always, even for a single "
"unambiguous speaker - with the identity described where the speaker first APPEARS "
"on screen, not where they first speak. Keep the identifying phrase, action and "
"delivery outside the <d> block and in English. At the end of each spoken line "
"close the mouth: describe the lips closing or the jaw coming to rest, or the "
"mouth keeps moving past the audio. Every silent on-screen character gets an "
"explicit statement that they produce no vocal sound, or the model may voice "
"them. A voiceover uses the exact phrase `says in an off-screen voiceover`, "
"immediately followed by a statement that the speaker's lips remain closed. "
"Keep a speaking or singing character on screen across any cut their vocal "
"continues through - H3 readily hands a continuing vocal to whoever appears "
"after a cut. As a rule of thumb, budget spoken words to about 2.5 x (shot "
"seconds - 1) per shot so every line fits its timeline."
)
else:
lines.append(
@@ -417,7 +478,9 @@ def toggle_directives(
if include_on_screen_text:
lines.append(
'On-screen text: signs, banners, labels or subtitles that are actually visible go in '
'English double quotation marks, verbatim and untranslated.'
'English double quotation marks, verbatim and untranslated. Name the typography and '
'where in the frame each string sits - a visible text element left unspecified '
'renders as letter-shaped noise.'
)
else:
lines.append("On-screen text: keep the frame free of readable signs, banners, labels or subtitles.")
@@ -425,8 +488,9 @@ def toggle_directives(
if include_soundscape:
lines.append(
"overall_soundscape: 1-4 English sentences in one paragraph covering ambience, "
"physical action sounds and non-verbal human sounds. Do not repeat dialogue or "
"singing here."
"physical action sounds and non-verbal human sounds. Dialogue, singing, diegetic "
"music and any description of voices are out of scope here - do not even mention "
"that people are speaking."
)
else:
lines.append("overall_soundscape: output exactly `N/A`.")
@@ -847,6 +911,12 @@ def scale_reference_image(tensor):
f"⚠️ H3 reference image {w}x{h} is outside the 1:4..4:1 aspect range "
"the reference pipeline accepts - expect degraded identity transfer."
)
if min(h, w) < 256:
print(
f"⚠️ H3 reference image {w}x{h} is tiny (short edge under 256px). Upscaling "
"adds tokens, not detail - identity signal will be weak; use a larger source "
"crop if one exists."
)
s = REFERENCE_SHORT_EDGE / float(min(h, w))
nw = max(REFERENCE_MULTIPLE, round(w * s / REFERENCE_MULTIPLE) * REFERENCE_MULTIPLE)
nh = max(REFERENCE_MULTIPLE, round(h * s / REFERENCE_MULTIPLE) * REFERENCE_MULTIPLE)
+15
View File
@@ -344,6 +344,21 @@ class H3RefPromptWriter:
"distributing detail across the shots by information load. Fitting a complete "
"spoken timeline matters more than hitting the number exactly."
)
# Measured authority hierarchy: a wrong word in subject_definitions
# overrides the reference image itself (a brunette described as blonde
# renders blonde despite fully_preserved), and only a specific
# detailed_description can override the definition. Silence downstream
# leaves the definition in charge.
directives.append(
"Authority: subject_definitions is binding - a wrong or invented attribute "
"there overrides the reference image itself, and retention_analysis cannot "
"correct it - so define ONLY what a reference actually shows, never a generic "
"template line. Silence is not neutral: whatever detailed_description leaves "
"unsaid is filled from the definitions, so re-describe the environment, "
"palette and lighting in detailed_description even when a reference image "
"already supplies them, and relight referenced subjects into the scene's own "
"light rather than leaving their source lighting unstated."
)
directives.extend(
toggle_directives(
include_dialogue,
+177 -65
View File
@@ -1,60 +1,104 @@
# APNext H3 Resolution - frame sizes with the trained frame as the unit
# APNext H3 Resolution - the vendor's adapt_canvas rule, per aspect ratio
#
# H3 is trained at 1344x768: 16:9, 1,032,192 pixels. Core's Resolution
# Selector counts megapixels from 1 MP = 1024x1024, so its "1.0" is 3% under
# the trained frame and 16:9 lands on 1344x768 only by rounding luck; nudge
# the slider and the aspect drifts. Here 1.0 MP IS 1344x768. Every other
# aspect gets the same pixel count, the slider scales all of them together,
# and every side is a multiple of 32 - the grid the H3 nodes take.
# H3's released pipelines size every canvas with one rule (adapt_canvas):
#
# The size is chosen on that grid, not rounded onto it: of the four grid
# corners around the ideal (float) size, the one with the smallest error
# wins, the aspect weighing twice the pixel count - a user who asks for 4:3
# wants 4:3 framing, the megapixels are a slider anyway. 1.0 therefore gives
# 1344x768 / 768x1344 / 1024x1024 / 1248x832 / 832x1248 / 1152x864 / 864x1152.
# 1. scale so the SHORT edge is 768
# 2. if the area exceeds 1344x768 = 1,032,192 px, scale down by
# sqrt(cap / area)
# 3. round each side to the nearest multiple of 32
#
# That yields the 95 canvases H3 was trained on - and it means aspects do NOT
# share a pixel count: 16:9 is 1344x768 (1008 tokens/frame) while 1:1 is
# 768x768 (576 tokens/frame, a third of the attention cost). An equal-pixels
# model hands H3 off-distribution canvases (1:1 at 1024x1024) that no vendor
# pipeline would ever produce. Only the widest aspects hit the area cap; from
# ~3:4 through 7:4 the short edge stays a full 768.
#
# 1344x768 is exactly 7:4 (1.750), not 16:9 (1.778) - no H3 canvas is a true
# 16:9; a 16:9 request lands on the trained 1344x768 with a 1.6% squeeze.
#
# The frames input snaps onto the VAE's 17n+5 frame grid (latent frames
# 5n+2; trained range ~124-362, ceiling 362) and reports the video token
# count: (5n+2) x (W/32) x (H/32). At 1344x768 the fused-qkv int32 kernel
# crossing (99,864 tokens) sits between 311 and 328 frames - sage/quant
# builds without int64 offsets overflow past it, and references on top of
# the target only bring it closer.
import math
import re
from ...utils.constants import CUSTOM_CATEGORY
BASE_W, BASE_H = 1344, 768
BASE_PIXELS = BASE_W * BASE_H # 1,032,192 - what "1.0" means here
GRID = 32 # the H3 nodes' width / height step
SHORT_EDGE = 768 # adapt_canvas step 1
AREA_CAP = 1344 * 768 # 1,032,192 - adapt_canvas step 2
GRID = 32 # VAE 16x spatial x DiT patchify 2
TOKENS_16_9 = (1344 // 32) * (768 // 32) # 1008 - the cost yardstick
FRAME_GRID = 17 # frame counts are 17n + 5
FRAME_GRID_OFFSET = 5
MAX_FRAMES = 362 # longest H3 was trained on
MIN_TRAINED_FRAMES = 124
INT32_TOKEN_CROSSING = 99_864 # fused-qkv stride 21504: 2^31 / 21504
def adapt_canvas(aspect_w, aspect_h, scale=1.0, multiple=GRID):
"""
(width, height) for aspect w:h under the vendor's adapt_canvas rule,
optionally scaled: `scale` multiplies the AREA (linear sides by sqrt),
with 1.0 the exact vendor recipe.
"""
a = float(aspect_w) / float(aspect_h)
lin = math.sqrt(max(scale, 1e-6))
short = SHORT_EDGE * lin
if a >= 1.0:
h, w = short, short * a
else:
w, h = short, short / a
cap = AREA_CAP * scale
if w * h > cap:
k = math.sqrt(cap / (w * h))
w, h = w * k, h * k
m = int(multiple)
return (max(m, round(w / m) * m), max(m, round(h / m) * m))
def snap_frames(frames):
"""`frames` snapped UP onto the 17n+5 grid, like the H3 nodes do."""
n = max(1, math.ceil((int(frames) - FRAME_GRID_OFFSET) / FRAME_GRID))
return FRAME_GRID * n + FRAME_GRID_OFFSET
# (label prefix, aspect_w, aspect_h) - the label shows the real 1.0 canvas.
_ASPECT_DEFS = [
("16:9 widescreen - the trained canvas", 16, 9),
("9:16 portrait widescreen", 9, 16),
("21:9 ultrawide", 21, 9),
("9:21 portrait ultrawide", 9, 21),
("1:1 square - a third of 16:9's attention cost", 1, 1),
("3:2 photo", 3, 2),
("2:3 portrait photo", 2, 3),
("4:3 standard", 4, 3),
("3:4 portrait standard", 3, 4),
("5:4 near-square", 5, 4),
("4:5 portrait near-square", 4, 5),
]
ASPECTS = [
("16:9 widescreen - what H3 is trained on (1344x768 at 1.0)", 16, 9),
("9:16 portrait widescreen (768x1344 at 1.0)", 9, 16),
("1:1 square", 1, 1),
("2:3 portrait photo", 2, 3),
("3:2 photo", 3, 2),
("3:4 portrait standard", 3, 4),
("4:3 standard", 4, 3),
(f"{prefix} ({adapt_canvas(w, h)[0]}x{adapt_canvas(w, h)[1]})", w, h)
for prefix, w, h in _ASPECT_DEFS
]
_ASPECT_BY_LABEL = {label: (w, h) for label, w, h in ASPECTS}
_ASPECT_RE = re.compile(r"(\d+)\s*:\s*(\d+)")
def pick_size(pixels, aspect_w, aspect_h, multiple=GRID):
"""(width, height) on the `multiple` grid nearest `pixels` at aspect w:h.
Candidates are the grid corners around the ideal size; the score is the
log error of the pixel count plus twice the log error of the aspect.
"""
pixels = max(float(pixels), float(multiple * multiple))
a = float(aspect_w) / float(aspect_h)
w0 = math.sqrt(pixels * a)
h0 = w0 / a
m = int(multiple)
cands = set()
for w in (math.floor(w0 / m) * m, math.ceil(w0 / m) * m):
for h in (math.floor(h0 / m) * m, math.ceil(h0 / m) * m):
if w >= m and h >= m:
cands.add((int(w), int(h)))
def score(wh):
w, h = wh
return abs(math.log((w * h) / pixels)) + 2.0 * abs(math.log((w / h) / a))
return min(cands, key=score)
def _parse_aspect(label):
"""w:h from any current or legacy dropdown label; 16:9 when unreadable."""
match = _ASPECT_RE.search(str(label or ""))
if match:
w, h = int(match.group(1)), int(match.group(2))
if w > 0 and h > 0:
return w, h
return 16, 9
class H3ResolutionSelector:
@@ -65,50 +109,118 @@ class H3ResolutionSelector:
"aspect_ratio": ([label for label, _, _ in ASPECTS], {
"default": ASPECTS[0][0],
"tooltip": (
"The frame's shape. 16:9 is what H3 is trained on; the others keep the same "
"pixel count at that shape, so the render costs the same."
"The frame's shape. Each label shows the trained canvas the vendor's "
"adapt_canvas rule gives that aspect (short edge 768, area capped at "
"1344x768). Aspects do NOT cost the same: squarer is cheaper - 1:1 is "
"a third of 16:9's attention."
),
}),
"megapixels": ("FLOAT", {
"default": 1.0, "min": 0.25, "max": 2.0, "step": 0.05, "round": 0.01,
"display": "slider",
"tooltip": (
"Size, in units of the trained frame: 1.0 = 1344x768 (1.03 real megapixels), "
"0.5 = half the pixels (960x544 at 16:9), 2.0 = twice (1920x1088). The aspect "
"is kept while the slider scales the frame; sides stay multiples of 32."
"Canvas scale. 1.0 is the exact vendor recipe (the trained canvas for "
"the aspect) - leave it there for renders. Below 1.0 shrinks the area "
"for cheap previews; above 1.0 exceeds the vendor's area cap, which no "
"released H3 pipeline ever does."
),
}),
},
"optional": {
"multiple": ("INT", {
"default": GRID, "min": 16, "max": 128, "step": 16,
"tooltip": "Grid every side must sit on. The H3 nodes take multiples of 32.",
"tooltip": (
"Grid every side must sit on. H3 needs 32; 128 also aligns the token "
"grid for Morton-ordered sparse attention."
),
}),
"frames": ("INT", {
"default": 192, "min": 0, "max": 1000, "step": 1,
"tooltip": (
"Target frame count, snapped UP onto H3's 17n+5 grid for the frames / "
"duration outputs and the token report. 192 = exactly 8.000s (the only "
"common integer duration on the grid); trained range 124-362. 0 skips "
"the report."
),
}),
},
}
RETURN_TYPES = ("INT", "INT", "FLOAT", "STRING")
RETURN_NAMES = ("width", "height", "megapixels", "info")
RETURN_TYPES = ("INT", "INT", "FLOAT", "STRING", "INT", "FLOAT")
RETURN_NAMES = ("width", "height", "megapixels", "info", "frames", "duration_seconds")
OUTPUT_TOOLTIPS = (
"Frame width - wire into the render's `width`.",
"Frame height - wire into the render's `height`.",
"The real megapixel count of the chosen size (1344x768 = 1.03).",
"`1344x768 | 16:9 | 1.03 MP (1.00 of the trained frame)`.",
"The real megapixel count of the chosen canvas (1344x768 = 1.03).",
"Canvas, tokens/frame, attention cost vs 16:9, snapped length and video tokens.",
"The frame count snapped up onto the 17n+5 grid - wire into the render's `length`.",
"The snapped length in seconds at 24fps - wire into a writer's `duration_seconds`.",
)
FUNCTION = "pick"
CATEGORY = f"{CUSTOM_CATEGORY}/H3"
DESCRIPTION = (
"Frame size for the H3 renders, with the trained 1344x768 frame as the unit: "
"megapixels 1.0 is exactly 1344x768, every aspect gets the same pixel count, the "
"slider scales them together and every side is a multiple of 32. Drop-in for "
"core's Resolution Selector (same width / height outputs)."
"Frame size for the H3 renders using the vendor's adapt_canvas rule: short edge "
"768, area capped at 1344x768, sides rounded to 32 - the canvases H3 was trained "
"on, so 1:1 is 768x768 at a third of 16:9's attention cost. Also snaps the frame "
"count onto the 17n+5 grid and reports the video token budget."
)
def pick(self, aspect_ratio, megapixels, multiple=GRID):
aw, ah = _ASPECT_BY_LABEL.get(aspect_ratio, (BASE_W, BASE_H))
width, height = pick_size(float(megapixels) * BASE_PIXELS, aw, ah, multiple)
@classmethod
def VALIDATE_INPUTS(cls, aspect_ratio=None):
# Legacy labels (older equal-pixel builds of this node) parse fine and
# anything unreadable falls back to 16:9 - never fail a saved workflow.
return True
def pick(self, aspect_ratio, megapixels, multiple=GRID, frames=192):
aw, ah = _parse_aspect(aspect_ratio)
scale = float(megapixels)
width, height = adapt_canvas(aw, ah, scale, multiple)
if scale > 1.001:
print(
f"⚠️ H3 Resolution | scale {scale:.2f} exceeds the vendor's area cap "
f"(1,032,192 px) - no released H3 pipeline renders above it; expect "
"off-distribution output."
)
real_mp = width * height / 1_000_000.0
units = width * height / float(BASE_PIXELS)
info = f"{width}x{height} | {aw}:{ah} | {real_mp:.2f} MP ({units:.2f} of the trained frame)"
tokens_per_frame = (width // 32) * (height // 32)
attention = (tokens_per_frame / TOKENS_16_9) ** 2
info = (
f"{width}x{height} | {aw}:{ah} | {real_mp:.2f} MP | "
f"{tokens_per_frame} tok/frame | attention {attention:.2f}x vs 16:9"
)
snapped, duration = 0, 0.0
if int(frames) > 0:
snapped = snap_frames(frames)
duration = snapped / 24.0
latent_frames = 5 * ((snapped - FRAME_GRID_OFFSET) // FRAME_GRID) + 2
video_tokens = latent_frames * tokens_per_frame
info += f" | {snapped}f = {duration:.3f}s | {video_tokens:,} video tokens"
if snapped != int(frames):
info += f" (snapped from {int(frames)})"
if snapped > MAX_FRAMES:
print(
f"⚠️ H3 Resolution | {snapped} frames is past the trained ceiling of "
f"{MAX_FRAMES} (15.083s) - step down to 362 or less."
)
elif snapped < MIN_TRAINED_FRAMES:
print(
f"ℹ️ H3 Resolution | {snapped} frames is below the ~{MIN_TRAINED_FRAMES}-"
"frame start of the trained range."
)
# The whole packed sequence counts against the kernel limit, not just
# video: target audio is ~80 rows/second and text ~1,000 tokens, which
# is what puts the crossing between 311f and 328f at 1344x768.
est_sequence = video_tokens + round(duration * 80) + 1000
if est_sequence > INT32_TOKEN_CROSSING:
print(
f"⚠️ H3 Resolution | est. packed sequence ~{est_sequence:,} tokens "
f"({video_tokens:,} video + audio + text, before any reference rows) "
f"crosses the fused-qkv int32 kernel limit ({INT32_TOKEN_CROSSING:,}) - "
"sage/quant kernels without int64 offsets overflow here."
)
print(f"🖼️ H3 Resolution | {info}")
return (width, height, round(real_mp, 3), info)
return (width, height, round(real_mp, 3), info, snapped, round(duration, 5))