resolution.py now uses the vendor adapt_canvas rule, camera vocabulary and writer directives are updated, workflows audited
This commit is contained in:
@@ -92,8 +92,13 @@ class H3BasePromptWriter:
|
||||
"tooltip": "Which H3 task the prompt targets. Anything other than T2VA emits the matching reference-alignment instruction line.",
|
||||
}),
|
||||
"duration_seconds": ("FLOAT", {
|
||||
"default": 6.0, "min": 1.0, "max": 60.0, "step": 0.5,
|
||||
"tooltip": "Effective video duration. Drives the cut times and the S.SS value in the alignment instruction.",
|
||||
"default": 8.0, "min": 1.0, "max": 60.0, "step": 0.5,
|
||||
"tooltip": (
|
||||
"Effective video duration. Drives the cut times and the S.SS value in the "
|
||||
"alignment instruction. The render snaps frames UP to the 17n+5 grid, so "
|
||||
"prefer grid durations - 8.00s (192f) is the only common integer one; "
|
||||
"the trained ceiling is 15.083s (362f)."
|
||||
),
|
||||
}),
|
||||
"shot_plan": (SHOT_PLANS, {"default": AUTO}),
|
||||
"visual_style": (VISUAL_STYLES, {
|
||||
@@ -253,7 +258,9 @@ class H3BasePromptWriter:
|
||||
"Replace N with the index of the actual final shot. Follow it with one blank "
|
||||
"line, then the core fields. <Picture 1> is the final frame and belongs to the "
|
||||
"last shot, not Shot 1: infer a plausible earlier state and converge onto the "
|
||||
"image (preceding state -> transition path -> gradual convergence -> landing)."
|
||||
"image (preceding state -> transition path -> gradual convergence -> landing). "
|
||||
"If the final frame shows a closed mouth, finish all dialogue early enough for "
|
||||
"the mouth to return to that closed position by the end."
|
||||
)
|
||||
|
||||
def _reference_rule(self, image_count, references):
|
||||
|
||||
@@ -66,8 +66,12 @@ class H3ClaudeCodeBaseWriter(H3BasePromptWriter):
|
||||
"tooltip": "Which H3 task the prompt targets. Anything other than T2VA emits the matching reference-alignment instruction line.",
|
||||
}),
|
||||
"duration_seconds": ("FLOAT", {
|
||||
"default": 6.0, "min": 1.0, "max": 60.0, "step": 0.5,
|
||||
"tooltip": "Effective video duration. Drives the cut times and the S.SS value in the alignment instruction.",
|
||||
"default": 8.0, "min": 1.0, "max": 60.0, "step": 0.5,
|
||||
"tooltip": (
|
||||
"Effective video duration. Drives the cut times and the S.SS value in the "
|
||||
"alignment instruction. The render snaps frames UP to the 17n+5 grid, so "
|
||||
"prefer grid durations - 8.00s (192f) is the only common integer one."
|
||||
),
|
||||
}),
|
||||
"shot_plan": (SHOT_PLANS, {"default": AUTO}),
|
||||
"visual_style": (VISUAL_STYLES, {
|
||||
|
||||
@@ -130,8 +130,11 @@ class H3ClaudeCodeContinueWriter(H3BasePromptWriter):
|
||||
),
|
||||
}),
|
||||
"duration_seconds": ("FLOAT", {
|
||||
"default": 6.0, "min": 1.0, "max": 60.0, "step": 0.5,
|
||||
"tooltip": "Length of the NEW clip.",
|
||||
"default": 8.0, "min": 1.0, "max": 60.0, "step": 0.5,
|
||||
"tooltip": (
|
||||
"Length of the NEW clip. The render snaps frames UP to the 17n+5 grid, so "
|
||||
"prefer grid durations - 8.00s (192f) is the only common integer one."
|
||||
),
|
||||
}),
|
||||
"shot_plan": (SHOT_PLANS, {"default": AUTO}),
|
||||
"visual_style": (VISUAL_STYLES, {
|
||||
|
||||
+82
-12
@@ -97,14 +97,14 @@ _CINEMATIC_LOOKS = [
|
||||
"Live-action, 35mm cinematic film aesthetic",
|
||||
"Live-action, modern large-format digital cinema: Alexa 65 clarity, creamy shallow depth of field, natural HDR skin tones, smooth floating gimbal moves, neutral filmic grade",
|
||||
"Live-action, 65mm epic scale: IMAX-format deep-focus vistas, slow deliberate crane and dolly moves, natural available light, cool desaturated grade with warm protected skin tones",
|
||||
"Live-action, Wes Anderson style: perfectly symmetrical planimetric framing, deadpan centered portraits and 90-degree whip pans, pastel candy-box palette, meticulous dollhouse production design, flat 40mm perspective",
|
||||
"Live-action, Wes Anderson style: perfectly symmetrical planimetric framing, deadpan centered portraits and sudden fast 90-degree pans, pastel candy-box palette, meticulous dollhouse production design, flat 40mm perspective",
|
||||
"Live-action, David Fincher style: locked-down surgically precise camera, low-key tungsten and sodium-vapour practicals, cold teal-green shadow grade, crisp clinical digital sharpness",
|
||||
"Live-action, Roger Deakins naturalism: motivated single-source lighting, silhouettes against windows and fire, clean geometric widescreen compositions, gentle drifting camera, restrained filmic grade",
|
||||
"Live-action, Denis Villeneuve scale: monolithic minimalist compositions, tiny figures in vast spaces, atmospheric dust and haze diffusion, slow creeping push-ins, muted near-monochrome grade",
|
||||
"Live-action, Stanley Kubrick style: one-point-perspective symmetry, slow menacing zooms, ultra-wide 18mm interiors, cold practical-lit spaces, immaculate composed stillness",
|
||||
"Live-action, Terrence Malick golden hour: wide-angle handheld lyricism, backlit magic-hour sun flare, natural bounce light only, wandering steadicam through grass and doorways",
|
||||
"Live-action, Wong Kar-wai style: step-printed motion smear, saturated neon greens and reds, handheld intimacy in cramped interiors, rain-streaked glass and mirrors, romantic halation glow",
|
||||
"Live-action, Quentin Tarantino grindhouse: punchy 35mm Kodak saturation, crash zooms and low trunk-shot angles, long unbroken dialogue two-shots, warm 70s amber cast",
|
||||
"Live-action, Quentin Tarantino grindhouse: punchy 35mm Kodak saturation, sudden fast zoom-ins and low trunk-shot angles, long unbroken dialogue two-shots, warm 70s amber cast",
|
||||
"Live-action, 1970s paranoid thriller: 2.39 anamorphic Panavision, zoom-heavy long-lens surveillance framing, grimy urban browns and greens, grainy push-processed stock",
|
||||
"Live-action, A24 indie realism: 35mm grain, soft natural window light, muted pastel grade with lifted blacks, static tableaux broken by intimate handheld close-ups",
|
||||
"Live-action, neon noir: rain-slick night streets, cyan and magenta neon reflections, volumetric smoke and searchlights, anamorphic blue-streak flares, deep crushed blacks",
|
||||
@@ -112,7 +112,7 @@ _CINEMATIC_LOOKS = [
|
||||
"Live-action, golden-age Technicolor: three-strip saturated primaries, glamour key lighting with crisp eye-lights, stately dolly moves, painted-backdrop soundstage depth",
|
||||
"Live-action, 16mm vérité documentary: handheld reportage framing, visible film grain and gate weave, available light only, quick reactive zooms and refocuses",
|
||||
"Live-action, Super 8 home movie: heavy grain and light leaks, warm faded Kodachrome colours, jittery handheld framing, soft vignetted frame edges",
|
||||
"Live-action, late-80s VHS camcorder: smeared lo-fi video texture, auto-exposure pumping, bleeding oversaturated colour, abrupt handheld pans and snap zooms",
|
||||
"Live-action, late-80s VHS camcorder: smeared lo-fi video texture, auto-exposure pumping, bleeding oversaturated colour, abrupt handheld pans and fast zoom-ins",
|
||||
"Live-action, BBC nature documentary: long-lens telephoto compression, patient locked-off observation, golden dawn haze, pristine 8K clarity, sweeping aerial establishing shots",
|
||||
"Live-action, glossy blockbuster: sweeping 360-degree arc shots circling the hero at chest height, teal-and-orange grade, horizontal lens flares, low-angle wide shots, crisp high-shutter action",
|
||||
"Live-action, high-fashion editorial: glossy beauty lighting, bold saturated gel colours, slow-motion hair and fabric, macro texture inserts, high-contrast punchy grade",
|
||||
@@ -308,6 +308,30 @@ CUT_STYLES = [
|
||||
]
|
||||
|
||||
|
||||
# Camera language is a CLOSED vocabulary (guide 4.3) - and it matters
|
||||
# empirically: fixing exactly four off-vocabulary phrases in a default prompt
|
||||
# ("tracks left", "medium amplitude and moderate speed", "whip pan", a
|
||||
# misplaced speaker intro) turned a same-seed render from "looks and sounds
|
||||
# wrong" to "fantastic". Every writer gets this rule with the camera line.
|
||||
CAMERA_VOCABULARY_RULE = (
|
||||
"Camera language is a CLOSED vocabulary - the only motion types are Zoom "
|
||||
"In/Out, Push In, Pull Out, Pan Left/Right, Truck Left/Right, Tilt Up/Down, "
|
||||
"Pedestal Up/Down, Arc Shot, Tracking Shot, Static Shot, Shake "
|
||||
"Slightly/Strongly, POV and Roll Clockwise/Counterclockwise; the only "
|
||||
"amplitudes `with small amplitude` / `with large amplitude`; the only "
|
||||
"speeds `at slow speed` / `at fast speed`. Medium and normal are OMITTED, "
|
||||
"never written - `medium amplitude`, `moderate speed` and `normal speed` "
|
||||
"do not exist. Never write `whip pan`, `crash zoom`, `snap zoom`, or "
|
||||
"`tracks left/right` (which conflates Truck Left with Tracking Shot); "
|
||||
"re-express such moves inside the vocabulary (`a fast Pan Right with large "
|
||||
"amplitude`, `a fast Zoom In`). One primary camera change per beat - two "
|
||||
"simultaneous changes collapse into whichever is easier to render. A "
|
||||
"camera move lands more reliably when its visible consequences are written "
|
||||
"into the scene (what shifts against what, where the light moves) than as "
|
||||
"a bare asserted sentence."
|
||||
)
|
||||
|
||||
|
||||
# The video model takes every word at face value: `the camera orbits her`
|
||||
# can render outer space, not an arc shot. Every writer appends this so the
|
||||
# scene text spells out what is physically meant.
|
||||
@@ -335,7 +359,7 @@ def camera_directive(motion, amplitude, speed):
|
||||
"Camera motion: choose motion types that suit the action, and write them as "
|
||||
"natural English inside the shot (motion type, plus amplitude and speed only "
|
||||
"when meaningful). Do not stack them as labels at the end of a sentence. "
|
||||
+ LITERAL_CAMERA_DIRECTIVE
|
||||
+ CAMERA_VOCABULARY_RULE + " " + LITERAL_CAMERA_DIRECTIVE
|
||||
)
|
||||
|
||||
parts = [motion]
|
||||
@@ -345,11 +369,20 @@ def camera_directive(motion, amplitude, speed):
|
||||
parts.append(speed)
|
||||
|
||||
phrase = " ".join(parts)
|
||||
# H3 reframes on its own by default: a bare "static shot" is not enough to
|
||||
# actually hold a composition - the moves that do NOT happen must be named.
|
||||
hold = (
|
||||
"H3 reframes by default, so to truly hold the composition state that the "
|
||||
"camera does not pan, zoom, push or drift in addition to naming the "
|
||||
"static shot. "
|
||||
if motion == "Static Shot"
|
||||
else ""
|
||||
)
|
||||
return (
|
||||
f"Camera motion: the primary camera movement is `{phrase}`. Express it as natural "
|
||||
"English action inside the shot rather than as a trailing label. Additional shots "
|
||||
"may use other motion types when the action calls for it. "
|
||||
+ LITERAL_CAMERA_DIRECTIVE
|
||||
f"may use other motion types when the action calls for it. {hold}"
|
||||
+ CAMERA_VOCABULARY_RULE + " " + LITERAL_CAMERA_DIRECTIVE
|
||||
)
|
||||
|
||||
|
||||
@@ -357,10 +390,27 @@ def shot_directive(shot_plan, duration_seconds):
|
||||
"""Instruction covering shot count and cut-time formatting."""
|
||||
duration = f"{duration_seconds:.2f}"
|
||||
|
||||
# The render snaps frame counts UP onto the 17n+5 grid, so an off-grid
|
||||
# duration means the prompt narrates a shorter video than gets made.
|
||||
# 192 frames = 8.000s is the only common integer duration on the grid.
|
||||
frames24 = round(duration_seconds * 24)
|
||||
if (frames24 - 5) % 17:
|
||||
lo = 17 * max(1, (frames24 - 5) // 17) + 5
|
||||
hi = lo + 17
|
||||
print(
|
||||
f"ℹ️ H3 duration {duration_seconds:.2f}s ({frames24}f) is off the 17n+5 frame grid "
|
||||
f"the render snaps up to - the video will run longer than the prompt narrates. "
|
||||
f"Nearest grid durations: {lo / 24:.3f}s ({lo}f) and {hi / 24:.3f}s ({hi}f); "
|
||||
f"8.00s (192f) is the only common integer duration that lands exactly."
|
||||
)
|
||||
|
||||
if shot_plan == AUTO:
|
||||
count_rule = (
|
||||
"Choose the shot count that fits the action. Prefer a single shot unless a cut "
|
||||
"genuinely introduces new information about the subject, space, state, viewpoint or time."
|
||||
"genuinely introduces new information about the subject, space, state, viewpoint "
|
||||
"or time, and avoid packing three or more distinct setups (different locations "
|
||||
"or crowded wide compositions) into one clip - detail demanded per scene is what "
|
||||
"degrades renders, especially with few sampling steps."
|
||||
)
|
||||
else:
|
||||
count = _SHOT_COUNTS[shot_plan]
|
||||
@@ -405,8 +455,19 @@ def toggle_directives(
|
||||
)
|
||||
|
||||
lines.append(
|
||||
f"{spoken} Give each vocal source a stable (S1)/(S2) ID, and keep the identifying "
|
||||
"phrase, action and delivery outside the <d> block and in English."
|
||||
f"{spoken} Give each vocal source a stable (S1)/(S2) ID - always, even for a single "
|
||||
"unambiguous speaker - with the identity described where the speaker first APPEARS "
|
||||
"on screen, not where they first speak. Keep the identifying phrase, action and "
|
||||
"delivery outside the <d> block and in English. At the end of each spoken line "
|
||||
"close the mouth: describe the lips closing or the jaw coming to rest, or the "
|
||||
"mouth keeps moving past the audio. Every silent on-screen character gets an "
|
||||
"explicit statement that they produce no vocal sound, or the model may voice "
|
||||
"them. A voiceover uses the exact phrase `says in an off-screen voiceover`, "
|
||||
"immediately followed by a statement that the speaker's lips remain closed. "
|
||||
"Keep a speaking or singing character on screen across any cut their vocal "
|
||||
"continues through - H3 readily hands a continuing vocal to whoever appears "
|
||||
"after a cut. As a rule of thumb, budget spoken words to about 2.5 x (shot "
|
||||
"seconds - 1) per shot so every line fits its timeline."
|
||||
)
|
||||
else:
|
||||
lines.append(
|
||||
@@ -417,7 +478,9 @@ def toggle_directives(
|
||||
if include_on_screen_text:
|
||||
lines.append(
|
||||
'On-screen text: signs, banners, labels or subtitles that are actually visible go in '
|
||||
'English double quotation marks, verbatim and untranslated.'
|
||||
'English double quotation marks, verbatim and untranslated. Name the typography and '
|
||||
'where in the frame each string sits - a visible text element left unspecified '
|
||||
'renders as letter-shaped noise.'
|
||||
)
|
||||
else:
|
||||
lines.append("On-screen text: keep the frame free of readable signs, banners, labels or subtitles.")
|
||||
@@ -425,8 +488,9 @@ def toggle_directives(
|
||||
if include_soundscape:
|
||||
lines.append(
|
||||
"overall_soundscape: 1-4 English sentences in one paragraph covering ambience, "
|
||||
"physical action sounds and non-verbal human sounds. Do not repeat dialogue or "
|
||||
"singing here."
|
||||
"physical action sounds and non-verbal human sounds. Dialogue, singing, diegetic "
|
||||
"music and any description of voices are out of scope here - do not even mention "
|
||||
"that people are speaking."
|
||||
)
|
||||
else:
|
||||
lines.append("overall_soundscape: output exactly `N/A`.")
|
||||
@@ -847,6 +911,12 @@ def scale_reference_image(tensor):
|
||||
f"⚠️ H3 reference image {w}x{h} is outside the 1:4..4:1 aspect range "
|
||||
"the reference pipeline accepts - expect degraded identity transfer."
|
||||
)
|
||||
if min(h, w) < 256:
|
||||
print(
|
||||
f"⚠️ H3 reference image {w}x{h} is tiny (short edge under 256px). Upscaling "
|
||||
"adds tokens, not detail - identity signal will be weak; use a larger source "
|
||||
"crop if one exists."
|
||||
)
|
||||
s = REFERENCE_SHORT_EDGE / float(min(h, w))
|
||||
nw = max(REFERENCE_MULTIPLE, round(w * s / REFERENCE_MULTIPLE) * REFERENCE_MULTIPLE)
|
||||
nh = max(REFERENCE_MULTIPLE, round(h * s / REFERENCE_MULTIPLE) * REFERENCE_MULTIPLE)
|
||||
|
||||
@@ -344,6 +344,21 @@ class H3RefPromptWriter:
|
||||
"distributing detail across the shots by information load. Fitting a complete "
|
||||
"spoken timeline matters more than hitting the number exactly."
|
||||
)
|
||||
# Measured authority hierarchy: a wrong word in subject_definitions
|
||||
# overrides the reference image itself (a brunette described as blonde
|
||||
# renders blonde despite fully_preserved), and only a specific
|
||||
# detailed_description can override the definition. Silence downstream
|
||||
# leaves the definition in charge.
|
||||
directives.append(
|
||||
"Authority: subject_definitions is binding - a wrong or invented attribute "
|
||||
"there overrides the reference image itself, and retention_analysis cannot "
|
||||
"correct it - so define ONLY what a reference actually shows, never a generic "
|
||||
"template line. Silence is not neutral: whatever detailed_description leaves "
|
||||
"unsaid is filled from the definitions, so re-describe the environment, "
|
||||
"palette and lighting in detailed_description even when a reference image "
|
||||
"already supplies them, and relight referenced subjects into the scene's own "
|
||||
"light rather than leaving their source lighting unstated."
|
||||
)
|
||||
directives.extend(
|
||||
toggle_directives(
|
||||
include_dialogue,
|
||||
|
||||
+177
-65
@@ -1,60 +1,104 @@
|
||||
# APNext H3 Resolution - frame sizes with the trained frame as the unit
|
||||
# APNext H3 Resolution - the vendor's adapt_canvas rule, per aspect ratio
|
||||
#
|
||||
# H3 is trained at 1344x768: 16:9, 1,032,192 pixels. Core's Resolution
|
||||
# Selector counts megapixels from 1 MP = 1024x1024, so its "1.0" is 3% under
|
||||
# the trained frame and 16:9 lands on 1344x768 only by rounding luck; nudge
|
||||
# the slider and the aspect drifts. Here 1.0 MP IS 1344x768. Every other
|
||||
# aspect gets the same pixel count, the slider scales all of them together,
|
||||
# and every side is a multiple of 32 - the grid the H3 nodes take.
|
||||
# H3's released pipelines size every canvas with one rule (adapt_canvas):
|
||||
#
|
||||
# The size is chosen on that grid, not rounded onto it: of the four grid
|
||||
# corners around the ideal (float) size, the one with the smallest error
|
||||
# wins, the aspect weighing twice the pixel count - a user who asks for 4:3
|
||||
# wants 4:3 framing, the megapixels are a slider anyway. 1.0 therefore gives
|
||||
# 1344x768 / 768x1344 / 1024x1024 / 1248x832 / 832x1248 / 1152x864 / 864x1152.
|
||||
# 1. scale so the SHORT edge is 768
|
||||
# 2. if the area exceeds 1344x768 = 1,032,192 px, scale down by
|
||||
# sqrt(cap / area)
|
||||
# 3. round each side to the nearest multiple of 32
|
||||
#
|
||||
# That yields the 95 canvases H3 was trained on - and it means aspects do NOT
|
||||
# share a pixel count: 16:9 is 1344x768 (1008 tokens/frame) while 1:1 is
|
||||
# 768x768 (576 tokens/frame, a third of the attention cost). An equal-pixels
|
||||
# model hands H3 off-distribution canvases (1:1 at 1024x1024) that no vendor
|
||||
# pipeline would ever produce. Only the widest aspects hit the area cap; from
|
||||
# ~3:4 through 7:4 the short edge stays a full 768.
|
||||
#
|
||||
# 1344x768 is exactly 7:4 (1.750), not 16:9 (1.778) - no H3 canvas is a true
|
||||
# 16:9; a 16:9 request lands on the trained 1344x768 with a 1.6% squeeze.
|
||||
#
|
||||
# The frames input snaps onto the VAE's 17n+5 frame grid (latent frames
|
||||
# 5n+2; trained range ~124-362, ceiling 362) and reports the video token
|
||||
# count: (5n+2) x (W/32) x (H/32). At 1344x768 the fused-qkv int32 kernel
|
||||
# crossing (99,864 tokens) sits between 311 and 328 frames - sage/quant
|
||||
# builds without int64 offsets overflow past it, and references on top of
|
||||
# the target only bring it closer.
|
||||
|
||||
import math
|
||||
import re
|
||||
|
||||
from ...utils.constants import CUSTOM_CATEGORY
|
||||
|
||||
BASE_W, BASE_H = 1344, 768
|
||||
BASE_PIXELS = BASE_W * BASE_H # 1,032,192 - what "1.0" means here
|
||||
GRID = 32 # the H3 nodes' width / height step
|
||||
SHORT_EDGE = 768 # adapt_canvas step 1
|
||||
AREA_CAP = 1344 * 768 # 1,032,192 - adapt_canvas step 2
|
||||
GRID = 32 # VAE 16x spatial x DiT patchify 2
|
||||
TOKENS_16_9 = (1344 // 32) * (768 // 32) # 1008 - the cost yardstick
|
||||
|
||||
FRAME_GRID = 17 # frame counts are 17n + 5
|
||||
FRAME_GRID_OFFSET = 5
|
||||
MAX_FRAMES = 362 # longest H3 was trained on
|
||||
MIN_TRAINED_FRAMES = 124
|
||||
INT32_TOKEN_CROSSING = 99_864 # fused-qkv stride 21504: 2^31 / 21504
|
||||
|
||||
|
||||
def adapt_canvas(aspect_w, aspect_h, scale=1.0, multiple=GRID):
|
||||
"""
|
||||
(width, height) for aspect w:h under the vendor's adapt_canvas rule,
|
||||
optionally scaled: `scale` multiplies the AREA (linear sides by sqrt),
|
||||
with 1.0 the exact vendor recipe.
|
||||
"""
|
||||
a = float(aspect_w) / float(aspect_h)
|
||||
lin = math.sqrt(max(scale, 1e-6))
|
||||
short = SHORT_EDGE * lin
|
||||
if a >= 1.0:
|
||||
h, w = short, short * a
|
||||
else:
|
||||
w, h = short, short / a
|
||||
cap = AREA_CAP * scale
|
||||
if w * h > cap:
|
||||
k = math.sqrt(cap / (w * h))
|
||||
w, h = w * k, h * k
|
||||
m = int(multiple)
|
||||
return (max(m, round(w / m) * m), max(m, round(h / m) * m))
|
||||
|
||||
|
||||
def snap_frames(frames):
|
||||
"""`frames` snapped UP onto the 17n+5 grid, like the H3 nodes do."""
|
||||
n = max(1, math.ceil((int(frames) - FRAME_GRID_OFFSET) / FRAME_GRID))
|
||||
return FRAME_GRID * n + FRAME_GRID_OFFSET
|
||||
|
||||
|
||||
# (label prefix, aspect_w, aspect_h) - the label shows the real 1.0 canvas.
|
||||
_ASPECT_DEFS = [
|
||||
("16:9 widescreen - the trained canvas", 16, 9),
|
||||
("9:16 portrait widescreen", 9, 16),
|
||||
("21:9 ultrawide", 21, 9),
|
||||
("9:21 portrait ultrawide", 9, 21),
|
||||
("1:1 square - a third of 16:9's attention cost", 1, 1),
|
||||
("3:2 photo", 3, 2),
|
||||
("2:3 portrait photo", 2, 3),
|
||||
("4:3 standard", 4, 3),
|
||||
("3:4 portrait standard", 3, 4),
|
||||
("5:4 near-square", 5, 4),
|
||||
("4:5 portrait near-square", 4, 5),
|
||||
]
|
||||
|
||||
ASPECTS = [
|
||||
("16:9 widescreen - what H3 is trained on (1344x768 at 1.0)", 16, 9),
|
||||
("9:16 portrait widescreen (768x1344 at 1.0)", 9, 16),
|
||||
("1:1 square", 1, 1),
|
||||
("2:3 portrait photo", 2, 3),
|
||||
("3:2 photo", 3, 2),
|
||||
("3:4 portrait standard", 3, 4),
|
||||
("4:3 standard", 4, 3),
|
||||
(f"{prefix} ({adapt_canvas(w, h)[0]}x{adapt_canvas(w, h)[1]})", w, h)
|
||||
for prefix, w, h in _ASPECT_DEFS
|
||||
]
|
||||
_ASPECT_BY_LABEL = {label: (w, h) for label, w, h in ASPECTS}
|
||||
|
||||
_ASPECT_RE = re.compile(r"(\d+)\s*:\s*(\d+)")
|
||||
|
||||
|
||||
def pick_size(pixels, aspect_w, aspect_h, multiple=GRID):
|
||||
"""(width, height) on the `multiple` grid nearest `pixels` at aspect w:h.
|
||||
|
||||
Candidates are the grid corners around the ideal size; the score is the
|
||||
log error of the pixel count plus twice the log error of the aspect.
|
||||
"""
|
||||
pixels = max(float(pixels), float(multiple * multiple))
|
||||
a = float(aspect_w) / float(aspect_h)
|
||||
w0 = math.sqrt(pixels * a)
|
||||
h0 = w0 / a
|
||||
m = int(multiple)
|
||||
cands = set()
|
||||
for w in (math.floor(w0 / m) * m, math.ceil(w0 / m) * m):
|
||||
for h in (math.floor(h0 / m) * m, math.ceil(h0 / m) * m):
|
||||
if w >= m and h >= m:
|
||||
cands.add((int(w), int(h)))
|
||||
|
||||
def score(wh):
|
||||
w, h = wh
|
||||
return abs(math.log((w * h) / pixels)) + 2.0 * abs(math.log((w / h) / a))
|
||||
|
||||
return min(cands, key=score)
|
||||
def _parse_aspect(label):
|
||||
"""w:h from any current or legacy dropdown label; 16:9 when unreadable."""
|
||||
match = _ASPECT_RE.search(str(label or ""))
|
||||
if match:
|
||||
w, h = int(match.group(1)), int(match.group(2))
|
||||
if w > 0 and h > 0:
|
||||
return w, h
|
||||
return 16, 9
|
||||
|
||||
|
||||
class H3ResolutionSelector:
|
||||
@@ -65,50 +109,118 @@ class H3ResolutionSelector:
|
||||
"aspect_ratio": ([label for label, _, _ in ASPECTS], {
|
||||
"default": ASPECTS[0][0],
|
||||
"tooltip": (
|
||||
"The frame's shape. 16:9 is what H3 is trained on; the others keep the same "
|
||||
"pixel count at that shape, so the render costs the same."
|
||||
"The frame's shape. Each label shows the trained canvas the vendor's "
|
||||
"adapt_canvas rule gives that aspect (short edge 768, area capped at "
|
||||
"1344x768). Aspects do NOT cost the same: squarer is cheaper - 1:1 is "
|
||||
"a third of 16:9's attention."
|
||||
),
|
||||
}),
|
||||
"megapixels": ("FLOAT", {
|
||||
"default": 1.0, "min": 0.25, "max": 2.0, "step": 0.05, "round": 0.01,
|
||||
"display": "slider",
|
||||
"tooltip": (
|
||||
"Size, in units of the trained frame: 1.0 = 1344x768 (1.03 real megapixels), "
|
||||
"0.5 = half the pixels (960x544 at 16:9), 2.0 = twice (1920x1088). The aspect "
|
||||
"is kept while the slider scales the frame; sides stay multiples of 32."
|
||||
"Canvas scale. 1.0 is the exact vendor recipe (the trained canvas for "
|
||||
"the aspect) - leave it there for renders. Below 1.0 shrinks the area "
|
||||
"for cheap previews; above 1.0 exceeds the vendor's area cap, which no "
|
||||
"released H3 pipeline ever does."
|
||||
),
|
||||
}),
|
||||
},
|
||||
"optional": {
|
||||
"multiple": ("INT", {
|
||||
"default": GRID, "min": 16, "max": 128, "step": 16,
|
||||
"tooltip": "Grid every side must sit on. The H3 nodes take multiples of 32.",
|
||||
"tooltip": (
|
||||
"Grid every side must sit on. H3 needs 32; 128 also aligns the token "
|
||||
"grid for Morton-ordered sparse attention."
|
||||
),
|
||||
}),
|
||||
"frames": ("INT", {
|
||||
"default": 192, "min": 0, "max": 1000, "step": 1,
|
||||
"tooltip": (
|
||||
"Target frame count, snapped UP onto H3's 17n+5 grid for the frames / "
|
||||
"duration outputs and the token report. 192 = exactly 8.000s (the only "
|
||||
"common integer duration on the grid); trained range 124-362. 0 skips "
|
||||
"the report."
|
||||
),
|
||||
}),
|
||||
},
|
||||
}
|
||||
|
||||
RETURN_TYPES = ("INT", "INT", "FLOAT", "STRING")
|
||||
RETURN_NAMES = ("width", "height", "megapixels", "info")
|
||||
RETURN_TYPES = ("INT", "INT", "FLOAT", "STRING", "INT", "FLOAT")
|
||||
RETURN_NAMES = ("width", "height", "megapixels", "info", "frames", "duration_seconds")
|
||||
OUTPUT_TOOLTIPS = (
|
||||
"Frame width - wire into the render's `width`.",
|
||||
"Frame height - wire into the render's `height`.",
|
||||
"The real megapixel count of the chosen size (1344x768 = 1.03).",
|
||||
"`1344x768 | 16:9 | 1.03 MP (1.00 of the trained frame)`.",
|
||||
"The real megapixel count of the chosen canvas (1344x768 = 1.03).",
|
||||
"Canvas, tokens/frame, attention cost vs 16:9, snapped length and video tokens.",
|
||||
"The frame count snapped up onto the 17n+5 grid - wire into the render's `length`.",
|
||||
"The snapped length in seconds at 24fps - wire into a writer's `duration_seconds`.",
|
||||
)
|
||||
FUNCTION = "pick"
|
||||
CATEGORY = f"{CUSTOM_CATEGORY}/H3"
|
||||
DESCRIPTION = (
|
||||
"Frame size for the H3 renders, with the trained 1344x768 frame as the unit: "
|
||||
"megapixels 1.0 is exactly 1344x768, every aspect gets the same pixel count, the "
|
||||
"slider scales them together and every side is a multiple of 32. Drop-in for "
|
||||
"core's Resolution Selector (same width / height outputs)."
|
||||
"Frame size for the H3 renders using the vendor's adapt_canvas rule: short edge "
|
||||
"768, area capped at 1344x768, sides rounded to 32 - the canvases H3 was trained "
|
||||
"on, so 1:1 is 768x768 at a third of 16:9's attention cost. Also snaps the frame "
|
||||
"count onto the 17n+5 grid and reports the video token budget."
|
||||
)
|
||||
|
||||
def pick(self, aspect_ratio, megapixels, multiple=GRID):
|
||||
aw, ah = _ASPECT_BY_LABEL.get(aspect_ratio, (BASE_W, BASE_H))
|
||||
width, height = pick_size(float(megapixels) * BASE_PIXELS, aw, ah, multiple)
|
||||
@classmethod
|
||||
def VALIDATE_INPUTS(cls, aspect_ratio=None):
|
||||
# Legacy labels (older equal-pixel builds of this node) parse fine and
|
||||
# anything unreadable falls back to 16:9 - never fail a saved workflow.
|
||||
return True
|
||||
|
||||
def pick(self, aspect_ratio, megapixels, multiple=GRID, frames=192):
|
||||
aw, ah = _parse_aspect(aspect_ratio)
|
||||
scale = float(megapixels)
|
||||
width, height = adapt_canvas(aw, ah, scale, multiple)
|
||||
|
||||
if scale > 1.001:
|
||||
print(
|
||||
f"⚠️ H3 Resolution | scale {scale:.2f} exceeds the vendor's area cap "
|
||||
f"(1,032,192 px) - no released H3 pipeline renders above it; expect "
|
||||
"off-distribution output."
|
||||
)
|
||||
|
||||
real_mp = width * height / 1_000_000.0
|
||||
units = width * height / float(BASE_PIXELS)
|
||||
info = f"{width}x{height} | {aw}:{ah} | {real_mp:.2f} MP ({units:.2f} of the trained frame)"
|
||||
tokens_per_frame = (width // 32) * (height // 32)
|
||||
attention = (tokens_per_frame / TOKENS_16_9) ** 2
|
||||
info = (
|
||||
f"{width}x{height} | {aw}:{ah} | {real_mp:.2f} MP | "
|
||||
f"{tokens_per_frame} tok/frame | attention {attention:.2f}x vs 16:9"
|
||||
)
|
||||
|
||||
snapped, duration = 0, 0.0
|
||||
if int(frames) > 0:
|
||||
snapped = snap_frames(frames)
|
||||
duration = snapped / 24.0
|
||||
latent_frames = 5 * ((snapped - FRAME_GRID_OFFSET) // FRAME_GRID) + 2
|
||||
video_tokens = latent_frames * tokens_per_frame
|
||||
info += f" | {snapped}f = {duration:.3f}s | {video_tokens:,} video tokens"
|
||||
if snapped != int(frames):
|
||||
info += f" (snapped from {int(frames)})"
|
||||
if snapped > MAX_FRAMES:
|
||||
print(
|
||||
f"⚠️ H3 Resolution | {snapped} frames is past the trained ceiling of "
|
||||
f"{MAX_FRAMES} (15.083s) - step down to 362 or less."
|
||||
)
|
||||
elif snapped < MIN_TRAINED_FRAMES:
|
||||
print(
|
||||
f"ℹ️ H3 Resolution | {snapped} frames is below the ~{MIN_TRAINED_FRAMES}-"
|
||||
"frame start of the trained range."
|
||||
)
|
||||
# The whole packed sequence counts against the kernel limit, not just
|
||||
# video: target audio is ~80 rows/second and text ~1,000 tokens, which
|
||||
# is what puts the crossing between 311f and 328f at 1344x768.
|
||||
est_sequence = video_tokens + round(duration * 80) + 1000
|
||||
if est_sequence > INT32_TOKEN_CROSSING:
|
||||
print(
|
||||
f"⚠️ H3 Resolution | est. packed sequence ~{est_sequence:,} tokens "
|
||||
f"({video_tokens:,} video + audio + text, before any reference rows) "
|
||||
f"crosses the fused-qkv int32 kernel limit ({INT32_TOKEN_CROSSING:,}) - "
|
||||
"sage/quant kernels without int64 offsets overflow here."
|
||||
)
|
||||
|
||||
print(f"🖼️ H3 Resolution | {info}")
|
||||
return (width, height, round(real_mp, 3), info)
|
||||
return (width, height, round(real_mp, 3), info, snapped, round(duration, 5))
|
||||
|
||||
Reference in New Issue
Block a user