Files
WildAi 4d8686f9ad test: clean the suite to match the current architecture, and de-localise it
Suite goes from 1894 passed / 66 failed to 1876 passed / 0 failed.

The 66 failures were not 66 problems. Two mechanisms caused all of them:

  * comfy_stream._STREAMING_CONVERSION_ENABLED is False and nothing in
    production sets it, so convert_tree_for_streaming is a no-op. Every
    test asserting "this module was converted" was asserting on a sweep
    that never ran.
  * select_patcher_class ignores both its arguments and returns
    legacy_cls. Every test parametrized over a "dynamic" arm was
    exercising a configuration production cannot produce.

Deleted (assert wiring that no longer exists): test_dynamic_patcher_
selection.py, test_dynamic_vram_mechanism.py, test_ram_residual_
attribution.py, test_loader_streaming_integration.py (three of its five
tests were passing VACUOUSLY once conversion went), and
TestCensusOnRealDynamicRoute.

Corrected to the real contract, keeping the coverage: the conversion
tests now enable the gate explicitly and test the capability, which is
what they were written for; the TestP1 load tests now assert that the
LOADER places each tensor (it took a target_device) instead of asserting
the superseded CPU staging; the both-patcher-classes tests were
de-parametrized onto the selector's real answer.

Two production error messages had lost the hints their tests pinned, and
the tests were right: UnmappedKeyError no longer told a user with a
non-standard converter to place a sidecar config, and the GGUF block-size
error no longer said the layout could not be recovered.

Also fixes a latent bug: test_comfy_stream.py imported `modules.comfy_stream`
while everything else imported `ComfyUI_VibeVoice.modules.comfy_stream`.
Under pytest those are two distinct module objects with independent
globals, so a gate set through one was invisible to the other. All bare
`modules.*` imports in collected tests now use the package name.

Portability, verified by simulation rather than inspection:

  * conftest no longer falls back to a hardcoded C:\_Dev ComfyUI path. It
    used to put a nonexistent directory on sys.path off-box and surface
    as an opaque import error across 14 test files; now it names
    COMFYUI_ROOT and stops.
  * The 5.4 GB and 3.2 GB checkpoint fixtures are opt-in via env vars with
    no machine-specific default (VIBEVOICE_TEST_DENSE_CHECKPOINT,
    VIBEVOICE_TEST_GGUF).
  * comfy_kitchen guards added. It is genuinely optional; a driver
    script that blocks the import went from 9 failed to 37 passed /
    11 skipped / 0 errors.
  * comfy_aimdo import guarded, though it is in practice a hard
    dependency of the ComfyUI fork (core imports it unconditionally).
  * test_audio_backend writes to tmp_path instead of into the repo tree.
  * test_asr_loader no longer registers the real folder_paths.models_dir
    in an autouse fixture without restoring it.

The large-checkpoint and .dev-symlink tests skip cleanly off-box; that is
the intended shape for files that cannot ship with the repo.
2026-10-01 15:49:15 +03:00

235 lines
9.4 KiB
Python

"""Phase A tests: GGUF block dequantization kernels, bitwise vs gguf-py oracle.
gguf-py can QUANTIZE only F32/F16/BF16/Q8_0 (K-quant quantize_blocks raises
NotImplementedError), so K-quant test data comes from the conftest handcrafted
block builders with controlled fp16 scales; the oracle dequantizes whatever
bytes we craft.
"""
import os
import numpy as np
import pytest
import torch
import gguf
from gguf.constants import GGMLQuantizationType as T
from gguf.quants import dequantize as oracle_dequantize, quantize as oracle_quantize
from ComfyUI_VibeVoice.modules import gguf_quant as G
def _seeded_float(shape, seed=42):
rng = np.random.default_rng(seed)
return (rng.standard_normal(shape) * 0.05).astype(np.float32)
class TestBitwiseParityAgainstOracle:
"""dequantize_blocks must be BITWISE equal to gguf.dequantize."""
def _check(self, raw_bytes_u8, qtype, logical_shape):
ref = oracle_dequantize(raw_bytes_u8.view(np.uint8), qtype)
ours = G.dequantize_blocks(
torch.from_numpy(raw_bytes_u8.copy()), qtype,
torch.float32, tuple(ref.shape),
)
assert np.array_equal(ref, ours.numpy())
def _parity(self, raw_bytes, qtype, shape):
"""Assert our dequantizer is bitwise equal to gguf-py's.
``raw_bytes`` is whatever a ``GGUFReader`` tensor carries (the mmap is
read through here, not copied wholesale) and ``shape`` is the
TORCH-logical shape. The reader reports ggml (reversed) dims, so the
caller is responsible for having already flipped them.
"""
ref = oracle_dequantize(raw_bytes, qtype).reshape(shape)
ours = G.dequantize_blocks(
torch.from_numpy(np.ascontiguousarray(raw_bytes)).view(torch.uint8),
qtype, torch.float32, shape,
)
assert np.array_equal(ref, ours.numpy())
def test_q8_0_quantized_by_oracle(self):
for shape in [(64, 32), (37, 96), (1, 32)]:
x = _seeded_float(shape)
raw = oracle_quantize(x, T.Q8_0)
self._check(raw.reshape(-1), T.Q8_0, None)
def test_q4_k_crafted(self):
for nb in (1, 3, 17, 129):
blocks = craft_for_test("Q4_K", nb, seed=nb)
self._check(blocks.reshape(-1), T.Q4_K, None)
def test_q5_k_crafted(self):
for nb in (1, 9, 65):
blocks = craft_for_test("Q5_K", nb, seed=nb)
self._check(blocks.reshape(-1), T.Q5_K, None)
def test_q6_k_crafted(self):
for nb in (1, 5, 33):
blocks = craft_for_test("Q6_K", nb, seed=nb)
self._check(blocks.reshape(-1), T.Q6_K, None)
def test_synthetic_q8_0_tensor(self, make_gguf_file):
"""Bitwise parity on a Q8_0 tensor with a realistic block count.
This is the machine-independent half of the real-checkpoint check: the
synthetic file carries a 2048x256 Q8_0 weight (16 K blocks, the same
order as a real projection), so the oracle comparison below runs on
every machine and in CI instead of only where a 3.2 GB fixture happens
to live. See ``test_real_file_q8_0_tensor`` for the real checkpoint.
"""
path = make_gguf_file(
[("model.language_model.layers.0.self_attn.q_proj.weight",
"Q8_0", (2048, 256))],
tag="q8_oracle",
)
reader = gguf.GGUFReader(str(path))
t = max(reader.tensors, key=lambda x: int(np.prod(x.shape)))
assert t.tensor_type == T.Q8_0
shape = tuple(int(s) for s in t.shape)[::-1]
self._parity(t.data, t.tensor_type, shape)
def test_real_file_q8_0_tensor(self):
"""Bitwise parity on a REAL tensor from the user's VibeVoice GGUF.
The file is selected by TENSOR TYPE, not by name: this checkpoint
quantises 378 Q8_0 tensors but names none of them
`*.self_attn.q_proj.weight` (the vendored tree uses a different module
layout), so the old name filter raised StopIteration and the oracle
check never ran. The largest Q8_0 tensor is used so the comparison
exercises a realistic block count rather than a single block.
Optional by construction: the path is overridable via
``VIBEVOICE_TEST_GGUF`` and the test skips when no real checkpoint is
present. It must stay optional — the 3.2 GB file is not something a
unit suite can require, and nothing here may load a real model. The
always-runs coverage of the same code path is
``test_synthetic_q8_0_tensor`` above.
"""
path = os.environ.get("VIBEVOICE_TEST_GGUF", "")
if not path:
pytest.skip(
"set VIBEVOICE_TEST_GGUF to a real q8_0 VibeVoice GGUF to run "
"this; it stays optional because the file is 3.2 GB"
)
try:
reader = gguf.GGUFReader(path)
except Exception:
pytest.skip("real vibevoice gguf not present")
candidates = [t for t in reader.tensors if t.tensor_type == T.Q8_0]
assert candidates, (
f"{path} has no Q8_0 tensor (types: "
f"{sorted({t.tensor_type.name for t in reader.tensors})}); the "
f"fixture is not the q8_0 checkpoint this test is written for."
)
t = max(candidates, key=lambda x: int(np.prod(x.shape)))
shape = tuple(int(s) for s in t.shape)
self._parity(t.data, t.tensor_type, shape)
def craft_for_test(qtype_name: str, n_blocks: int, seed: int = 0):
"""Thin re-export so this module does not depend on conftest internals."""
from conftest import craft_kquant_blocks
block_size, type_size = {
"Q8_0": (32, 34), "Q4_K": (256, 144),
"Q5_K": (256, 176), "Q6_K": (256, 210),
}[qtype_name]
return craft_kquant_blocks(qtype_name, n_blocks * block_size, seed=seed)
class TestDenseFloatPaths:
def test_f32_zero_copy_view(self):
arr = _seeded_float((9, 17))
t = torch.from_numpy(arr)
out = G.dequantize_dense(t, T.F32)
assert out.data_ptr() == t.data_ptr()
assert torch.equal(out, t)
def test_f16_view_exact(self):
x = _seeded_float((7, 33)).astype(np.float16)
t = torch.from_numpy(np.ascontiguousarray(x))
out = G.dequantize_dense(t, T.F16)
assert out.dtype == torch.float16
assert torch.equal(out, torch.from_numpy(x.copy()))
def test_bf16_view_matches_oracle_fp32(self):
x = _seeded_float((4, 8))
bf_bytes = oracle_quantize(x, T.BF16) # uint8
ref = oracle_dequantize(bf_bytes.view(np.uint8), T.BF16).reshape(4, 8)
ours = G.dequantize_dense(torch.from_numpy(np.ascontiguousarray(bf_bytes)), T.BF16)
assert torch.equal(torch.from_numpy(ref.copy()), ours.float())
class TestExpectedNumel:
@pytest.mark.parametrize("kind,n_elem,expected", [
("Q8_0", 32, 34),
("Q8_0", 96, 102),
("Q4_K", 256, 144),
("Q5_K", 256, 176),
("Q6_K", 256, 210),
])
def test_formulas(self, kind, n_elem, expected):
assert G.expected_numel(n_elem, getattr(T, kind)) == expected
def test_rejects_indivisible(self):
with pytest.raises(ValueError, match="multiple of"):
G.expected_numel(100, T.Q4_K)
class TestGGUFTensor:
def test_from_reader_tensor_reverses_shape_and_clones_bytes(self, make_gguf_file):
spec = [("model.language_model.layers.0.self_attn.q_proj.weight", "Q8_0", (16, 32))]
path = make_gguf_file(spec, tag="gt")
reader = gguf.GGUFReader(str(path))
rt = reader.tensors[0]
gt = G.GGUFTensor.from_reader_tensor(rt)
# Reader reports REVERSED dims; GGUFTensor stores TORCH-logical shape.
assert gt.shape == (16, 32)
assert gt.ggml_type == T.Q8_0
assert gt.raw.dtype == torch.uint8
expected_bytes = int(rt.n_bytes)
assert gt.n_bytes == expected_bytes
assert np.array_equal(
gt.raw.numpy().reshape(-1),
np.ascontiguousarray(rt.data).reshape(-1),
)
def test_to_moves_raw_and_rejects_dtype(self, make_gguf_file):
spec = [("w", "Q8_0", (16, 32))]
path = make_gguf_file(spec, tag="gtto")
gt = G.GGUFTensor.from_reader_tensor(gguf.GGUFReader(str(path)).tensors[0])
moved = gt.to(torch.device("meta"))
assert moved.raw.device.type == "meta"
assert moved.shape == gt.shape
with pytest.raises(ValueError, match="raw bytes"):
gt.to(torch.device("cpu"), dtype=torch.float32)
def test_dequantize_method(self, make_gguf_file):
spec = [("w", "Q8_0", (16, 32))]
path = make_gguf_file(spec, tag="gtdq")
gt = G.GGUFTensor.from_reader_tensor(gguf.GGUFReader(str(path)).tensors[0])
out = gt.dequantize(torch.float32)
assert out.shape == (16, 32)
class TestErrorTaxonomy:
def test_unsupported_type_message_quality(self):
err = G.UnsupportedGGMLType(T.Q2_K, "blk.3.attn_v.weight")
msg = str(err)
assert "blk.3.attn_v.weight" in msg
assert "Q2_K" in msg
assert "Q8_0" in msg # supported list included
def test_dequantize_blocks_rejects_unsupported(self):
raw = torch.zeros(210, dtype=torch.uint8)
with pytest.raises(G.UnsupportedGGMLType):
G.dequantize_blocks(raw, T.Q2_K, torch.float32, (1, 256))
def test_dequantize_blocks_rejects_bad_byte_count(self):
with pytest.raises(ValueError, match="multiple"):
G.dequantize_blocks(torch.zeros(33, dtype=torch.uint8),
T.Q8_0, torch.float32, (1, 32))