From 61f87db161aa8a136c3fb40ac3dbcf1637ec1de1 Mon Sep 17 00:00:00 2001 From: Dag Thomas Olsen Date: Sat, 29 Aug 2026 21:59:59 +0200 Subject: [PATCH] resolution.py now uses the vendor adapt_canvas rule, camera vocabulary and writer directives are updated, workflows audited --- nodes/h3/base_prompt_writer.py | 13 +- nodes/h3/claude_code_base_writer.py | 8 +- nodes/h3/claude_code_continue_writer.py | 7 +- nodes/h3/common.py | 94 +++++++-- nodes/h3/ref_prompt_writer.py | 15 ++ nodes/h3/resolution.py | 242 +++++++++++++++++------- 6 files changed, 295 insertions(+), 84 deletions(-) diff --git a/nodes/h3/base_prompt_writer.py b/nodes/h3/base_prompt_writer.py index bcdbcf1..ed13a53 100644 --- a/nodes/h3/base_prompt_writer.py +++ b/nodes/h3/base_prompt_writer.py @@ -92,8 +92,13 @@ class H3BasePromptWriter: "tooltip": "Which H3 task the prompt targets. Anything other than T2VA emits the matching reference-alignment instruction line.", }), "duration_seconds": ("FLOAT", { - "default": 6.0, "min": 1.0, "max": 60.0, "step": 0.5, - "tooltip": "Effective video duration. Drives the cut times and the S.SS value in the alignment instruction.", + "default": 8.0, "min": 1.0, "max": 60.0, "step": 0.5, + "tooltip": ( + "Effective video duration. Drives the cut times and the S.SS value in the " + "alignment instruction. The render snaps frames UP to the 17n+5 grid, so " + "prefer grid durations - 8.00s (192f) is the only common integer one; " + "the trained ceiling is 15.083s (362f)." + ), }), "shot_plan": (SHOT_PLANS, {"default": AUTO}), "visual_style": (VISUAL_STYLES, { @@ -253,7 +258,9 @@ class H3BasePromptWriter: "Replace N with the index of the actual final shot. Follow it with one blank " "line, then the core fields. is the final frame and belongs to the " "last shot, not Shot 1: infer a plausible earlier state and converge onto the " - "image (preceding state -> transition path -> gradual convergence -> landing)." + "image (preceding state -> transition path -> gradual convergence -> landing). " + "If the final frame shows a closed mouth, finish all dialogue early enough for " + "the mouth to return to that closed position by the end." ) def _reference_rule(self, image_count, references): diff --git a/nodes/h3/claude_code_base_writer.py b/nodes/h3/claude_code_base_writer.py index bace983..8481584 100644 --- a/nodes/h3/claude_code_base_writer.py +++ b/nodes/h3/claude_code_base_writer.py @@ -66,8 +66,12 @@ class H3ClaudeCodeBaseWriter(H3BasePromptWriter): "tooltip": "Which H3 task the prompt targets. Anything other than T2VA emits the matching reference-alignment instruction line.", }), "duration_seconds": ("FLOAT", { - "default": 6.0, "min": 1.0, "max": 60.0, "step": 0.5, - "tooltip": "Effective video duration. Drives the cut times and the S.SS value in the alignment instruction.", + "default": 8.0, "min": 1.0, "max": 60.0, "step": 0.5, + "tooltip": ( + "Effective video duration. Drives the cut times and the S.SS value in the " + "alignment instruction. The render snaps frames UP to the 17n+5 grid, so " + "prefer grid durations - 8.00s (192f) is the only common integer one." + ), }), "shot_plan": (SHOT_PLANS, {"default": AUTO}), "visual_style": (VISUAL_STYLES, { diff --git a/nodes/h3/claude_code_continue_writer.py b/nodes/h3/claude_code_continue_writer.py index fa32022..5633088 100644 --- a/nodes/h3/claude_code_continue_writer.py +++ b/nodes/h3/claude_code_continue_writer.py @@ -130,8 +130,11 @@ class H3ClaudeCodeContinueWriter(H3BasePromptWriter): ), }), "duration_seconds": ("FLOAT", { - "default": 6.0, "min": 1.0, "max": 60.0, "step": 0.5, - "tooltip": "Length of the NEW clip.", + "default": 8.0, "min": 1.0, "max": 60.0, "step": 0.5, + "tooltip": ( + "Length of the NEW clip. The render snaps frames UP to the 17n+5 grid, so " + "prefer grid durations - 8.00s (192f) is the only common integer one." + ), }), "shot_plan": (SHOT_PLANS, {"default": AUTO}), "visual_style": (VISUAL_STYLES, { diff --git a/nodes/h3/common.py b/nodes/h3/common.py index d2df19d..ce97d85 100644 --- a/nodes/h3/common.py +++ b/nodes/h3/common.py @@ -97,14 +97,14 @@ _CINEMATIC_LOOKS = [ "Live-action, 35mm cinematic film aesthetic", "Live-action, modern large-format digital cinema: Alexa 65 clarity, creamy shallow depth of field, natural HDR skin tones, smooth floating gimbal moves, neutral filmic grade", "Live-action, 65mm epic scale: IMAX-format deep-focus vistas, slow deliberate crane and dolly moves, natural available light, cool desaturated grade with warm protected skin tones", - "Live-action, Wes Anderson style: perfectly symmetrical planimetric framing, deadpan centered portraits and 90-degree whip pans, pastel candy-box palette, meticulous dollhouse production design, flat 40mm perspective", + "Live-action, Wes Anderson style: perfectly symmetrical planimetric framing, deadpan centered portraits and sudden fast 90-degree pans, pastel candy-box palette, meticulous dollhouse production design, flat 40mm perspective", "Live-action, David Fincher style: locked-down surgically precise camera, low-key tungsten and sodium-vapour practicals, cold teal-green shadow grade, crisp clinical digital sharpness", "Live-action, Roger Deakins naturalism: motivated single-source lighting, silhouettes against windows and fire, clean geometric widescreen compositions, gentle drifting camera, restrained filmic grade", "Live-action, Denis Villeneuve scale: monolithic minimalist compositions, tiny figures in vast spaces, atmospheric dust and haze diffusion, slow creeping push-ins, muted near-monochrome grade", "Live-action, Stanley Kubrick style: one-point-perspective symmetry, slow menacing zooms, ultra-wide 18mm interiors, cold practical-lit spaces, immaculate composed stillness", "Live-action, Terrence Malick golden hour: wide-angle handheld lyricism, backlit magic-hour sun flare, natural bounce light only, wandering steadicam through grass and doorways", "Live-action, Wong Kar-wai style: step-printed motion smear, saturated neon greens and reds, handheld intimacy in cramped interiors, rain-streaked glass and mirrors, romantic halation glow", - "Live-action, Quentin Tarantino grindhouse: punchy 35mm Kodak saturation, crash zooms and low trunk-shot angles, long unbroken dialogue two-shots, warm 70s amber cast", + "Live-action, Quentin Tarantino grindhouse: punchy 35mm Kodak saturation, sudden fast zoom-ins and low trunk-shot angles, long unbroken dialogue two-shots, warm 70s amber cast", "Live-action, 1970s paranoid thriller: 2.39 anamorphic Panavision, zoom-heavy long-lens surveillance framing, grimy urban browns and greens, grainy push-processed stock", "Live-action, A24 indie realism: 35mm grain, soft natural window light, muted pastel grade with lifted blacks, static tableaux broken by intimate handheld close-ups", "Live-action, neon noir: rain-slick night streets, cyan and magenta neon reflections, volumetric smoke and searchlights, anamorphic blue-streak flares, deep crushed blacks", @@ -112,7 +112,7 @@ _CINEMATIC_LOOKS = [ "Live-action, golden-age Technicolor: three-strip saturated primaries, glamour key lighting with crisp eye-lights, stately dolly moves, painted-backdrop soundstage depth", "Live-action, 16mm vérité documentary: handheld reportage framing, visible film grain and gate weave, available light only, quick reactive zooms and refocuses", "Live-action, Super 8 home movie: heavy grain and light leaks, warm faded Kodachrome colours, jittery handheld framing, soft vignetted frame edges", - "Live-action, late-80s VHS camcorder: smeared lo-fi video texture, auto-exposure pumping, bleeding oversaturated colour, abrupt handheld pans and snap zooms", + "Live-action, late-80s VHS camcorder: smeared lo-fi video texture, auto-exposure pumping, bleeding oversaturated colour, abrupt handheld pans and fast zoom-ins", "Live-action, BBC nature documentary: long-lens telephoto compression, patient locked-off observation, golden dawn haze, pristine 8K clarity, sweeping aerial establishing shots", "Live-action, glossy blockbuster: sweeping 360-degree arc shots circling the hero at chest height, teal-and-orange grade, horizontal lens flares, low-angle wide shots, crisp high-shutter action", "Live-action, high-fashion editorial: glossy beauty lighting, bold saturated gel colours, slow-motion hair and fabric, macro texture inserts, high-contrast punchy grade", @@ -308,6 +308,30 @@ CUT_STYLES = [ ] +# Camera language is a CLOSED vocabulary (guide 4.3) - and it matters +# empirically: fixing exactly four off-vocabulary phrases in a default prompt +# ("tracks left", "medium amplitude and moderate speed", "whip pan", a +# misplaced speaker intro) turned a same-seed render from "looks and sounds +# wrong" to "fantastic". Every writer gets this rule with the camera line. +CAMERA_VOCABULARY_RULE = ( + "Camera language is a CLOSED vocabulary - the only motion types are Zoom " + "In/Out, Push In, Pull Out, Pan Left/Right, Truck Left/Right, Tilt Up/Down, " + "Pedestal Up/Down, Arc Shot, Tracking Shot, Static Shot, Shake " + "Slightly/Strongly, POV and Roll Clockwise/Counterclockwise; the only " + "amplitudes `with small amplitude` / `with large amplitude`; the only " + "speeds `at slow speed` / `at fast speed`. Medium and normal are OMITTED, " + "never written - `medium amplitude`, `moderate speed` and `normal speed` " + "do not exist. Never write `whip pan`, `crash zoom`, `snap zoom`, or " + "`tracks left/right` (which conflates Truck Left with Tracking Shot); " + "re-express such moves inside the vocabulary (`a fast Pan Right with large " + "amplitude`, `a fast Zoom In`). One primary camera change per beat - two " + "simultaneous changes collapse into whichever is easier to render. A " + "camera move lands more reliably when its visible consequences are written " + "into the scene (what shifts against what, where the light moves) than as " + "a bare asserted sentence." +) + + # The video model takes every word at face value: `the camera orbits her` # can render outer space, not an arc shot. Every writer appends this so the # scene text spells out what is physically meant. @@ -335,7 +359,7 @@ def camera_directive(motion, amplitude, speed): "Camera motion: choose motion types that suit the action, and write them as " "natural English inside the shot (motion type, plus amplitude and speed only " "when meaningful). Do not stack them as labels at the end of a sentence. " - + LITERAL_CAMERA_DIRECTIVE + + CAMERA_VOCABULARY_RULE + " " + LITERAL_CAMERA_DIRECTIVE ) parts = [motion] @@ -345,11 +369,20 @@ def camera_directive(motion, amplitude, speed): parts.append(speed) phrase = " ".join(parts) + # H3 reframes on its own by default: a bare "static shot" is not enough to + # actually hold a composition - the moves that do NOT happen must be named. + hold = ( + "H3 reframes by default, so to truly hold the composition state that the " + "camera does not pan, zoom, push or drift in addition to naming the " + "static shot. " + if motion == "Static Shot" + else "" + ) return ( f"Camera motion: the primary camera movement is `{phrase}`. Express it as natural " "English action inside the shot rather than as a trailing label. Additional shots " - "may use other motion types when the action calls for it. " - + LITERAL_CAMERA_DIRECTIVE + f"may use other motion types when the action calls for it. {hold}" + + CAMERA_VOCABULARY_RULE + " " + LITERAL_CAMERA_DIRECTIVE ) @@ -357,10 +390,27 @@ def shot_directive(shot_plan, duration_seconds): """Instruction covering shot count and cut-time formatting.""" duration = f"{duration_seconds:.2f}" + # The render snaps frame counts UP onto the 17n+5 grid, so an off-grid + # duration means the prompt narrates a shorter video than gets made. + # 192 frames = 8.000s is the only common integer duration on the grid. + frames24 = round(duration_seconds * 24) + if (frames24 - 5) % 17: + lo = 17 * max(1, (frames24 - 5) // 17) + 5 + hi = lo + 17 + print( + f"ℹ️ H3 duration {duration_seconds:.2f}s ({frames24}f) is off the 17n+5 frame grid " + f"the render snaps up to - the video will run longer than the prompt narrates. " + f"Nearest grid durations: {lo / 24:.3f}s ({lo}f) and {hi / 24:.3f}s ({hi}f); " + f"8.00s (192f) is the only common integer duration that lands exactly." + ) + if shot_plan == AUTO: count_rule = ( "Choose the shot count that fits the action. Prefer a single shot unless a cut " - "genuinely introduces new information about the subject, space, state, viewpoint or time." + "genuinely introduces new information about the subject, space, state, viewpoint " + "or time, and avoid packing three or more distinct setups (different locations " + "or crowded wide compositions) into one clip - detail demanded per scene is what " + "degrades renders, especially with few sampling steps." ) else: count = _SHOT_COUNTS[shot_plan] @@ -405,8 +455,19 @@ def toggle_directives( ) lines.append( - f"{spoken} Give each vocal source a stable (S1)/(S2) ID, and keep the identifying " - "phrase, action and delivery outside the block and in English." + f"{spoken} Give each vocal source a stable (S1)/(S2) ID - always, even for a single " + "unambiguous speaker - with the identity described where the speaker first APPEARS " + "on screen, not where they first speak. Keep the identifying phrase, action and " + "delivery outside the block and in English. At the end of each spoken line " + "close the mouth: describe the lips closing or the jaw coming to rest, or the " + "mouth keeps moving past the audio. Every silent on-screen character gets an " + "explicit statement that they produce no vocal sound, or the model may voice " + "them. A voiceover uses the exact phrase `says in an off-screen voiceover`, " + "immediately followed by a statement that the speaker's lips remain closed. " + "Keep a speaking or singing character on screen across any cut their vocal " + "continues through - H3 readily hands a continuing vocal to whoever appears " + "after a cut. As a rule of thumb, budget spoken words to about 2.5 x (shot " + "seconds - 1) per shot so every line fits its timeline." ) else: lines.append( @@ -417,7 +478,9 @@ def toggle_directives( if include_on_screen_text: lines.append( 'On-screen text: signs, banners, labels or subtitles that are actually visible go in ' - 'English double quotation marks, verbatim and untranslated.' + 'English double quotation marks, verbatim and untranslated. Name the typography and ' + 'where in the frame each string sits - a visible text element left unspecified ' + 'renders as letter-shaped noise.' ) else: lines.append("On-screen text: keep the frame free of readable signs, banners, labels or subtitles.") @@ -425,8 +488,9 @@ def toggle_directives( if include_soundscape: lines.append( "overall_soundscape: 1-4 English sentences in one paragraph covering ambience, " - "physical action sounds and non-verbal human sounds. Do not repeat dialogue or " - "singing here." + "physical action sounds and non-verbal human sounds. Dialogue, singing, diegetic " + "music and any description of voices are out of scope here - do not even mention " + "that people are speaking." ) else: lines.append("overall_soundscape: output exactly `N/A`.") @@ -847,6 +911,12 @@ def scale_reference_image(tensor): f"⚠️ H3 reference image {w}x{h} is outside the 1:4..4:1 aspect range " "the reference pipeline accepts - expect degraded identity transfer." ) + if min(h, w) < 256: + print( + f"⚠️ H3 reference image {w}x{h} is tiny (short edge under 256px). Upscaling " + "adds tokens, not detail - identity signal will be weak; use a larger source " + "crop if one exists." + ) s = REFERENCE_SHORT_EDGE / float(min(h, w)) nw = max(REFERENCE_MULTIPLE, round(w * s / REFERENCE_MULTIPLE) * REFERENCE_MULTIPLE) nh = max(REFERENCE_MULTIPLE, round(h * s / REFERENCE_MULTIPLE) * REFERENCE_MULTIPLE) diff --git a/nodes/h3/ref_prompt_writer.py b/nodes/h3/ref_prompt_writer.py index 05a6aeb..8aaf0cd 100644 --- a/nodes/h3/ref_prompt_writer.py +++ b/nodes/h3/ref_prompt_writer.py @@ -344,6 +344,21 @@ class H3RefPromptWriter: "distributing detail across the shots by information load. Fitting a complete " "spoken timeline matters more than hitting the number exactly." ) + # Measured authority hierarchy: a wrong word in subject_definitions + # overrides the reference image itself (a brunette described as blonde + # renders blonde despite fully_preserved), and only a specific + # detailed_description can override the definition. Silence downstream + # leaves the definition in charge. + directives.append( + "Authority: subject_definitions is binding - a wrong or invented attribute " + "there overrides the reference image itself, and retention_analysis cannot " + "correct it - so define ONLY what a reference actually shows, never a generic " + "template line. Silence is not neutral: whatever detailed_description leaves " + "unsaid is filled from the definitions, so re-describe the environment, " + "palette and lighting in detailed_description even when a reference image " + "already supplies them, and relight referenced subjects into the scene's own " + "light rather than leaving their source lighting unstated." + ) directives.extend( toggle_directives( include_dialogue, diff --git a/nodes/h3/resolution.py b/nodes/h3/resolution.py index abd5fc9..a8a16ed 100644 --- a/nodes/h3/resolution.py +++ b/nodes/h3/resolution.py @@ -1,60 +1,104 @@ -# APNext H3 Resolution - frame sizes with the trained frame as the unit +# APNext H3 Resolution - the vendor's adapt_canvas rule, per aspect ratio # -# H3 is trained at 1344x768: 16:9, 1,032,192 pixels. Core's Resolution -# Selector counts megapixels from 1 MP = 1024x1024, so its "1.0" is 3% under -# the trained frame and 16:9 lands on 1344x768 only by rounding luck; nudge -# the slider and the aspect drifts. Here 1.0 MP IS 1344x768. Every other -# aspect gets the same pixel count, the slider scales all of them together, -# and every side is a multiple of 32 - the grid the H3 nodes take. +# H3's released pipelines size every canvas with one rule (adapt_canvas): # -# The size is chosen on that grid, not rounded onto it: of the four grid -# corners around the ideal (float) size, the one with the smallest error -# wins, the aspect weighing twice the pixel count - a user who asks for 4:3 -# wants 4:3 framing, the megapixels are a slider anyway. 1.0 therefore gives -# 1344x768 / 768x1344 / 1024x1024 / 1248x832 / 832x1248 / 1152x864 / 864x1152. +# 1. scale so the SHORT edge is 768 +# 2. if the area exceeds 1344x768 = 1,032,192 px, scale down by +# sqrt(cap / area) +# 3. round each side to the nearest multiple of 32 +# +# That yields the 95 canvases H3 was trained on - and it means aspects do NOT +# share a pixel count: 16:9 is 1344x768 (1008 tokens/frame) while 1:1 is +# 768x768 (576 tokens/frame, a third of the attention cost). An equal-pixels +# model hands H3 off-distribution canvases (1:1 at 1024x1024) that no vendor +# pipeline would ever produce. Only the widest aspects hit the area cap; from +# ~3:4 through 7:4 the short edge stays a full 768. +# +# 1344x768 is exactly 7:4 (1.750), not 16:9 (1.778) - no H3 canvas is a true +# 16:9; a 16:9 request lands on the trained 1344x768 with a 1.6% squeeze. +# +# The frames input snaps onto the VAE's 17n+5 frame grid (latent frames +# 5n+2; trained range ~124-362, ceiling 362) and reports the video token +# count: (5n+2) x (W/32) x (H/32). At 1344x768 the fused-qkv int32 kernel +# crossing (99,864 tokens) sits between 311 and 328 frames - sage/quant +# builds without int64 offsets overflow past it, and references on top of +# the target only bring it closer. import math +import re from ...utils.constants import CUSTOM_CATEGORY -BASE_W, BASE_H = 1344, 768 -BASE_PIXELS = BASE_W * BASE_H # 1,032,192 - what "1.0" means here -GRID = 32 # the H3 nodes' width / height step +SHORT_EDGE = 768 # adapt_canvas step 1 +AREA_CAP = 1344 * 768 # 1,032,192 - adapt_canvas step 2 +GRID = 32 # VAE 16x spatial x DiT patchify 2 +TOKENS_16_9 = (1344 // 32) * (768 // 32) # 1008 - the cost yardstick + +FRAME_GRID = 17 # frame counts are 17n + 5 +FRAME_GRID_OFFSET = 5 +MAX_FRAMES = 362 # longest H3 was trained on +MIN_TRAINED_FRAMES = 124 +INT32_TOKEN_CROSSING = 99_864 # fused-qkv stride 21504: 2^31 / 21504 + + +def adapt_canvas(aspect_w, aspect_h, scale=1.0, multiple=GRID): + """ + (width, height) for aspect w:h under the vendor's adapt_canvas rule, + optionally scaled: `scale` multiplies the AREA (linear sides by sqrt), + with 1.0 the exact vendor recipe. + """ + a = float(aspect_w) / float(aspect_h) + lin = math.sqrt(max(scale, 1e-6)) + short = SHORT_EDGE * lin + if a >= 1.0: + h, w = short, short * a + else: + w, h = short, short / a + cap = AREA_CAP * scale + if w * h > cap: + k = math.sqrt(cap / (w * h)) + w, h = w * k, h * k + m = int(multiple) + return (max(m, round(w / m) * m), max(m, round(h / m) * m)) + + +def snap_frames(frames): + """`frames` snapped UP onto the 17n+5 grid, like the H3 nodes do.""" + n = max(1, math.ceil((int(frames) - FRAME_GRID_OFFSET) / FRAME_GRID)) + return FRAME_GRID * n + FRAME_GRID_OFFSET + + +# (label prefix, aspect_w, aspect_h) - the label shows the real 1.0 canvas. +_ASPECT_DEFS = [ + ("16:9 widescreen - the trained canvas", 16, 9), + ("9:16 portrait widescreen", 9, 16), + ("21:9 ultrawide", 21, 9), + ("9:21 portrait ultrawide", 9, 21), + ("1:1 square - a third of 16:9's attention cost", 1, 1), + ("3:2 photo", 3, 2), + ("2:3 portrait photo", 2, 3), + ("4:3 standard", 4, 3), + ("3:4 portrait standard", 3, 4), + ("5:4 near-square", 5, 4), + ("4:5 portrait near-square", 4, 5), +] ASPECTS = [ - ("16:9 widescreen - what H3 is trained on (1344x768 at 1.0)", 16, 9), - ("9:16 portrait widescreen (768x1344 at 1.0)", 9, 16), - ("1:1 square", 1, 1), - ("2:3 portrait photo", 2, 3), - ("3:2 photo", 3, 2), - ("3:4 portrait standard", 3, 4), - ("4:3 standard", 4, 3), + (f"{prefix} ({adapt_canvas(w, h)[0]}x{adapt_canvas(w, h)[1]})", w, h) + for prefix, w, h in _ASPECT_DEFS ] -_ASPECT_BY_LABEL = {label: (w, h) for label, w, h in ASPECTS} + +_ASPECT_RE = re.compile(r"(\d+)\s*:\s*(\d+)") -def pick_size(pixels, aspect_w, aspect_h, multiple=GRID): - """(width, height) on the `multiple` grid nearest `pixels` at aspect w:h. - - Candidates are the grid corners around the ideal size; the score is the - log error of the pixel count plus twice the log error of the aspect. - """ - pixels = max(float(pixels), float(multiple * multiple)) - a = float(aspect_w) / float(aspect_h) - w0 = math.sqrt(pixels * a) - h0 = w0 / a - m = int(multiple) - cands = set() - for w in (math.floor(w0 / m) * m, math.ceil(w0 / m) * m): - for h in (math.floor(h0 / m) * m, math.ceil(h0 / m) * m): - if w >= m and h >= m: - cands.add((int(w), int(h))) - - def score(wh): - w, h = wh - return abs(math.log((w * h) / pixels)) + 2.0 * abs(math.log((w / h) / a)) - - return min(cands, key=score) +def _parse_aspect(label): + """w:h from any current or legacy dropdown label; 16:9 when unreadable.""" + match = _ASPECT_RE.search(str(label or "")) + if match: + w, h = int(match.group(1)), int(match.group(2)) + if w > 0 and h > 0: + return w, h + return 16, 9 class H3ResolutionSelector: @@ -65,50 +109,118 @@ class H3ResolutionSelector: "aspect_ratio": ([label for label, _, _ in ASPECTS], { "default": ASPECTS[0][0], "tooltip": ( - "The frame's shape. 16:9 is what H3 is trained on; the others keep the same " - "pixel count at that shape, so the render costs the same." + "The frame's shape. Each label shows the trained canvas the vendor's " + "adapt_canvas rule gives that aspect (short edge 768, area capped at " + "1344x768). Aspects do NOT cost the same: squarer is cheaper - 1:1 is " + "a third of 16:9's attention." ), }), "megapixels": ("FLOAT", { "default": 1.0, "min": 0.25, "max": 2.0, "step": 0.05, "round": 0.01, "display": "slider", "tooltip": ( - "Size, in units of the trained frame: 1.0 = 1344x768 (1.03 real megapixels), " - "0.5 = half the pixels (960x544 at 16:9), 2.0 = twice (1920x1088). The aspect " - "is kept while the slider scales the frame; sides stay multiples of 32." + "Canvas scale. 1.0 is the exact vendor recipe (the trained canvas for " + "the aspect) - leave it there for renders. Below 1.0 shrinks the area " + "for cheap previews; above 1.0 exceeds the vendor's area cap, which no " + "released H3 pipeline ever does." ), }), }, "optional": { "multiple": ("INT", { "default": GRID, "min": 16, "max": 128, "step": 16, - "tooltip": "Grid every side must sit on. The H3 nodes take multiples of 32.", + "tooltip": ( + "Grid every side must sit on. H3 needs 32; 128 also aligns the token " + "grid for Morton-ordered sparse attention." + ), + }), + "frames": ("INT", { + "default": 192, "min": 0, "max": 1000, "step": 1, + "tooltip": ( + "Target frame count, snapped UP onto H3's 17n+5 grid for the frames / " + "duration outputs and the token report. 192 = exactly 8.000s (the only " + "common integer duration on the grid); trained range 124-362. 0 skips " + "the report." + ), }), }, } - RETURN_TYPES = ("INT", "INT", "FLOAT", "STRING") - RETURN_NAMES = ("width", "height", "megapixels", "info") + RETURN_TYPES = ("INT", "INT", "FLOAT", "STRING", "INT", "FLOAT") + RETURN_NAMES = ("width", "height", "megapixels", "info", "frames", "duration_seconds") OUTPUT_TOOLTIPS = ( "Frame width - wire into the render's `width`.", "Frame height - wire into the render's `height`.", - "The real megapixel count of the chosen size (1344x768 = 1.03).", - "`1344x768 | 16:9 | 1.03 MP (1.00 of the trained frame)`.", + "The real megapixel count of the chosen canvas (1344x768 = 1.03).", + "Canvas, tokens/frame, attention cost vs 16:9, snapped length and video tokens.", + "The frame count snapped up onto the 17n+5 grid - wire into the render's `length`.", + "The snapped length in seconds at 24fps - wire into a writer's `duration_seconds`.", ) FUNCTION = "pick" CATEGORY = f"{CUSTOM_CATEGORY}/H3" DESCRIPTION = ( - "Frame size for the H3 renders, with the trained 1344x768 frame as the unit: " - "megapixels 1.0 is exactly 1344x768, every aspect gets the same pixel count, the " - "slider scales them together and every side is a multiple of 32. Drop-in for " - "core's Resolution Selector (same width / height outputs)." + "Frame size for the H3 renders using the vendor's adapt_canvas rule: short edge " + "768, area capped at 1344x768, sides rounded to 32 - the canvases H3 was trained " + "on, so 1:1 is 768x768 at a third of 16:9's attention cost. Also snaps the frame " + "count onto the 17n+5 grid and reports the video token budget." ) - def pick(self, aspect_ratio, megapixels, multiple=GRID): - aw, ah = _ASPECT_BY_LABEL.get(aspect_ratio, (BASE_W, BASE_H)) - width, height = pick_size(float(megapixels) * BASE_PIXELS, aw, ah, multiple) + @classmethod + def VALIDATE_INPUTS(cls, aspect_ratio=None): + # Legacy labels (older equal-pixel builds of this node) parse fine and + # anything unreadable falls back to 16:9 - never fail a saved workflow. + return True + + def pick(self, aspect_ratio, megapixels, multiple=GRID, frames=192): + aw, ah = _parse_aspect(aspect_ratio) + scale = float(megapixels) + width, height = adapt_canvas(aw, ah, scale, multiple) + + if scale > 1.001: + print( + f"⚠️ H3 Resolution | scale {scale:.2f} exceeds the vendor's area cap " + f"(1,032,192 px) - no released H3 pipeline renders above it; expect " + "off-distribution output." + ) + real_mp = width * height / 1_000_000.0 - units = width * height / float(BASE_PIXELS) - info = f"{width}x{height} | {aw}:{ah} | {real_mp:.2f} MP ({units:.2f} of the trained frame)" + tokens_per_frame = (width // 32) * (height // 32) + attention = (tokens_per_frame / TOKENS_16_9) ** 2 + info = ( + f"{width}x{height} | {aw}:{ah} | {real_mp:.2f} MP | " + f"{tokens_per_frame} tok/frame | attention {attention:.2f}x vs 16:9" + ) + + snapped, duration = 0, 0.0 + if int(frames) > 0: + snapped = snap_frames(frames) + duration = snapped / 24.0 + latent_frames = 5 * ((snapped - FRAME_GRID_OFFSET) // FRAME_GRID) + 2 + video_tokens = latent_frames * tokens_per_frame + info += f" | {snapped}f = {duration:.3f}s | {video_tokens:,} video tokens" + if snapped != int(frames): + info += f" (snapped from {int(frames)})" + if snapped > MAX_FRAMES: + print( + f"⚠️ H3 Resolution | {snapped} frames is past the trained ceiling of " + f"{MAX_FRAMES} (15.083s) - step down to 362 or less." + ) + elif snapped < MIN_TRAINED_FRAMES: + print( + f"ℹ️ H3 Resolution | {snapped} frames is below the ~{MIN_TRAINED_FRAMES}-" + "frame start of the trained range." + ) + # The whole packed sequence counts against the kernel limit, not just + # video: target audio is ~80 rows/second and text ~1,000 tokens, which + # is what puts the crossing between 311f and 328f at 1344x768. + est_sequence = video_tokens + round(duration * 80) + 1000 + if est_sequence > INT32_TOKEN_CROSSING: + print( + f"⚠️ H3 Resolution | est. packed sequence ~{est_sequence:,} tokens " + f"({video_tokens:,} video + audio + text, before any reference rows) " + f"crosses the fused-qkv int32 kernel limit ({INT32_TOKEN_CROSSING:,}) - " + "sage/quant kernels without int64 offsets overflow here." + ) + print(f"🖼️ H3 Resolution | {info}") - return (width, height, round(real_mp, 3), info) + return (width, height, round(real_mp, 3), info, snapped, round(duration, 5))