chore(phase1): pin baseline toolchain and add bench harness scaffold

Phase 1 of the modernization plan: freeze the currently-working environment
so later refactors have a measured reference point.

- Pin python-coreml-stable-diffusion to commit e5d960c4 (the one already
  installed in the maintainer's apple_env), plus torch==2.0.1, coremltools==8.2
  and numpy<1.25 to match the only env that loads ComfyUI successfully
  (Comfy's checkpoint-safe-loading branch in utils.py is gated on torch>=2.4,
  so newer torch + numpy 1.23 breaks at import).
- Mirror the same pins in requirements.txt and commit uv.lock for
  reproducible installs.
- Add requires-comfyui pinning ComfyUI to ab541335 (the validated commit).
- Fix tests/unit/test_chunks.py fixture: get_model_config() now takes a
  ModelVersion argument; pass ModelVersion.SD15 (the previously-broken test
  was the only Phase 1 production-code change required).
- Add the Phase 1 baseline harness: bench/run.py (direct Core ML UNet
  latency, deterministic), bench/scripts/convert_sd15.py (one-command
  conversion bypassing the node graph), bench/scripts/smoke_image.py (POSTs
  the existing e2e workflow to a local ComfyUI server and saves the Core ML
  image), bench/env/capture.sh (env snapshot), bench/prompts.json (fixed
  prompt set).
- Ignore apple_env/, comfy_env/, and bench/scripts/*.log.

Tests: 20/20 unit pass (test_chunks + test_controlnet).
This commit is contained in:
aszc-dev
2026-05-22 15:06:24 +02:00
parent 7678a07ed5
commit ef2a18cff3
10 changed files with 2230 additions and 6 deletions
+8
View File
@@ -1,3 +1,11 @@
playground/ playground/
experiments/ experiments/
__pycache__/ __pycache__/
models/
.venv/
apple_env/
comfy_env/
coremlsuite-venv/
experiment_results/
test_results/
bench/scripts/*.log
Vendored Executable
+100
View File
@@ -0,0 +1,100 @@
#!/usr/bin/env bash
# Phase 1 baseline environment capture.
#
# Default mode: query the project's .venv via `uv pip` (matches how uv builds
# the venv — without a bundled pip executable). For a non-uv setup, point
# PYTHON_BIN at the right interpreter; pip freeze then falls back to
# `python -m pip` if available, otherwise `uv pip`.
#
# Usage:
# bash bench/env/capture.sh # uses .venv (uv-managed)
# PYTHON_BIN=apple_env/bin/python bash bench/env/capture.sh
#
# Writes bench/env/baseline-<gitsha>.txt with: this repo's git sha,
# ComfyUI's git sha (override path with COMFY_DIR=...), python + macOS
# versions, the resolved git commit of python-coreml-stable-diffusion as
# installed, and the full freeze of the active venv (for fresh-venv
# reproducibility).
set -euo pipefail
REPO_ROOT="$(git rev-parse --show-toplevel)"
cd "$REPO_ROOT"
SHA_SHORT="$(git rev-parse --short HEAD)"
OUT_DIR="bench/env"
OUT="${OUT_DIR}/baseline-${SHA_SHORT}.txt"
mkdir -p "$OUT_DIR"
COMFY_DIR="${COMFY_DIR:-$(cd ../.. && pwd)}"
PYTHON_BIN="${PYTHON_BIN:-$REPO_ROOT/.venv/bin/python}"
if [ ! -x "$PYTHON_BIN" ]; then
echo "PYTHON_BIN not executable: $PYTHON_BIN" >&2
exit 1
fi
# Prefer `python -m pip` (works in classic venvs). Fall back to `uv pip`,
# which queries any interpreter without needing pip installed in it.
run_freeze() {
if "$PYTHON_BIN" -m pip --version >/dev/null 2>&1; then
"$PYTHON_BIN" -m pip freeze
elif command -v uv >/dev/null 2>&1; then
uv pip freeze --python "$PYTHON_BIN"
else
echo "neither pip nor uv available for freeze"
fi
}
run_show_msd() {
if "$PYTHON_BIN" -m pip --version >/dev/null 2>&1; then
"$PYTHON_BIN" -m pip show python-coreml-stable-diffusion 2>/dev/null || echo "not-installed"
elif command -v uv >/dev/null 2>&1; then
uv pip show --python "$PYTHON_BIN" python-coreml-stable-diffusion 2>/dev/null || echo "not-installed"
else
echo "neither pip nor uv available for show"
fi
}
{
echo "# Phase 1 baseline environment capture"
echo "timestamp_utc: $(date -u +%Y-%m-%dT%H:%M:%SZ)"
echo "repo_sha: $(git rev-parse HEAD)"
echo "repo_branch: $(git rev-parse --abbrev-ref HEAD)"
echo "comfyui_dir: ${COMFY_DIR}"
echo "comfyui_sha: $(git -C "$COMFY_DIR" rev-parse HEAD 2>/dev/null || echo unknown)"
echo "python_bin: ${PYTHON_BIN}"
echo "python_version: $("$PYTHON_BIN" --version 2>&1)"
echo "platform_machine: $(uname -m)"
echo "platform_uname: $(uname -a)"
if command -v sw_vers >/dev/null 2>&1; then
echo "macos_product: $(sw_vers -productName)"
echo "macos_version: $(sw_vers -productVersion)"
echo "macos_build: $(sw_vers -buildVersion)"
fi
echo
echo "## python-coreml-stable-diffusion (resolved)"
run_show_msd
echo
echo "## python-coreml-stable-diffusion installed git metadata"
DIST_INFO="$("$PYTHON_BIN" -c "import importlib.metadata as m; d=m.distribution('python-coreml-stable-diffusion'); print(d._path)" 2>/dev/null || true)"
if [ -n "${DIST_INFO}" ] && [ -f "${DIST_INFO}/direct_url.json" ]; then
cat "${DIST_INFO}/direct_url.json"
echo
else
echo "direct_url.json not found"
fi
echo
echo "## coremltools version"
"$PYTHON_BIN" -c "import coremltools; print('coremltools', coremltools.__version__)" 2>&1 || true
echo
echo "## torch version"
"$PYTHON_BIN" -c "import torch; print('torch', torch.__version__)" 2>&1 || true
echo
echo "## numpy version"
"$PYTHON_BIN" -c "import numpy; print('numpy', numpy.__version__)" 2>&1 || true
echo
echo "## freeze"
run_freeze
} > "$OUT"
echo "wrote $OUT"
+19
View File
@@ -0,0 +1,19 @@
{
"_comment": "Fixed prompt set for Phase 1 baseline reproducibility. Do not edit without bumping the bench version: changing prompts invalidates cross-baseline PSNR comparisons.",
"seed": 0,
"negative_prompt": "blurry, low quality, text, watermark",
"prompts": [
{
"id": "p01_portrait",
"text": "a portrait of an astronaut in a sunflower field, soft natural light, 35mm film"
},
{
"id": "p02_cityscape",
"text": "a futuristic city skyline at sunset, glass towers, volumetric lighting, cinematic"
},
{
"id": "p03_still_life",
"text": "a still life of fruit on a wooden table, oil painting, dramatic chiaroscuro"
}
]
}
+417
View File
@@ -0,0 +1,417 @@
#!/usr/bin/env python3
"""
bench/run.py — Phase 1 baseline benchmark harness for ComfyUI-CoreMLSuite.
WHAT THIS MEASURES (the deterministic, easy-to-trust numbers)
-------------------------------------------------------------
For each converted Core ML UNet model in the matrix, in-process:
* model load time (includes mlpackage compile; mlmodelc is precompiled)
* model file size on disk
* warmup (first forward) time
* steady-state UNet forward latency (mean/std/min/median over N repeats)
* an end-to-end estimate (median forward * assumed sampler steps)
* peak process RSS during the run
It drives the Core ML model DIRECTLY with synthetic inputs shaped from the
model's own `expected_inputs`. This isolates the ANE/Core ML UNet latency —
the exact quantity that quantization (Phase 6) and the coremltools upgrade
(Phase 5) are expected to move — WITHOUT needing a live ComfyUI server, CLIP,
VAE, or a checkpoint. It is therefore the cleanest, most reproducible signal.
WHAT THIS DOES NOT MEASURE
--------------------------
* Image quality. PSNR is a separate, decoupled concern: pass --reference-images
and --candidate-images (e.g. the E2E-1.5-MPS / E2E-1.5-CoreML PNGs your
existing integration workflow already produces) and PSNR is computed per
matching filename. Without them, quality_psnr is null (and that's honest).
* ANE / wired GPU memory. peak_rss_mb is *process* RSS only; treat as a coarse
floor, not the true device footprint.
REQUIREMENTS
------------
* Run on Apple Silicon (macOS) with the suite's deps installed
(coremltools + apple/ml-stable-diffusion providing `python_coreml_stable_diffusion`).
* numpy is required. psutil and Pillow are optional (graceful fallback).
USAGE
-----
# Auto-discover every .mlmodelc/.mlpackage under a dir, sweep compute units:
python bench/run.py --models-dir /path/to/ComfyUI/models/unet
# Explicit matrix (recommended for a stable, committed baseline):
python bench/run.py --matrix bench/matrix.json --models-dir /path/to/models/unet
# With quality from already-produced images:
python bench/run.py --models-dir ... \
--reference-images /path/to/mps_pngs --candidate-images /path/to/coreml_pngs
matrix.json format (a list of configs):
[
{"label": "sd15-se", "model": "dreamshaper_8_1x512x512_se.mlmodelc", "compute_unit": "CPU_AND_NE"},
{"label": "sdxl-orig","model": "sdxl_base_1x1024x1024_orig.mlmodelc", "compute_unit": "CPU_AND_GPU"}
]
Each "model" is resolved against --models-dir unless it is an absolute path.
Output: bench/results/<gitsha>.json and bench/results/<gitsha>.md
"""
import argparse
import dataclasses
import datetime as _dt
import glob
import json
import os
import platform
import statistics
import subprocess
import sys
import time
from pathlib import Path
from typing import Any, Optional
import numpy as np
# ----- optional deps (graceful) ---------------------------------------------
try:
import psutil # type: ignore
_PROC = psutil.Process(os.getpid())
except Exception: # pragma: no cover - optional
psutil = None
_PROC = None
DEFAULT_COMPUTE_UNITS = ["CPU_AND_NE", "CPU_AND_GPU"]
COREML_EXTS = (".mlmodelc", ".mlpackage")
# ----- small helpers --------------------------------------------------------
def git_sha() -> str:
try:
out = subprocess.check_output(
["git", "rev-parse", "--short", "HEAD"], stderr=subprocess.DEVNULL
)
return out.decode().strip() or "nogit"
except Exception:
return "nogit"
def coremltools_version() -> str:
try:
import coremltools as ct # noqa: WPS433 (local import is intentional)
return getattr(ct, "__version__", "unknown")
except Exception as exc: # pragma: no cover
return f"import-failed: {exc}"
def dir_size_bytes(path: Path) -> int:
"""Core ML models are directories (.mlmodelc / .mlpackage)."""
if path.is_file():
return path.stat().st_size
total = 0
for root, _dirs, files in os.walk(path):
for f in files:
try:
total += (Path(root) / f).stat().st_size
except OSError:
pass
return total
def sample_rss_mb() -> Optional[float]:
if _PROC is None:
return None
try:
return _PROC.memory_info().rss / (1024 * 1024)
except Exception:
return None
def detect_kind(expected_inputs: dict) -> str:
"""Mirror the suite's detection WITHOUT importing comfy (keeps bench light)."""
if "time_ids" in expected_inputs and "text_embeds" in expected_inputs:
try:
n = expected_inputs["time_ids"]["shape"][1]
except Exception:
n = -1
return "sdxl_base" if n == 6 else "sdxl_refiner" if n == 5 else "sdxl_unknown"
if "timestep_cond" in expected_inputs:
return "lcm"
return "sd15"
def make_inputs(expected_inputs: dict, rng: np.random.Generator) -> dict[str, np.ndarray]:
"""Build synthetic fp16 inputs matching the model's expected shapes.
Values are random; they do not affect latency in any meaningful way, and
we feed fp16 to match how CoreMLModelWrapper feeds the model at runtime.
"""
inputs: dict[str, np.ndarray] = {}
for name, spec in expected_inputs.items():
shape = tuple(int(d) for d in spec["shape"])
inputs[name] = rng.standard_normal(shape).astype(np.float16)
return inputs
def psnr(img_a: "np.ndarray", img_b: "np.ndarray") -> float:
mse = float(np.mean((img_a.astype(np.float64) - img_b.astype(np.float64)) ** 2))
if mse == 0:
return 100.0
return 20.0 * float(np.log10(255.0 / np.sqrt(mse)))
# ----- core measurement -----------------------------------------------------
@dataclasses.dataclass
class Config:
label: str
model_path: str
compute_unit: str
def resolve_matrix(args) -> list[Config]:
configs: list[Config] = []
if args.matrix:
entries = json.loads(Path(args.matrix).read_text())
for e in entries:
model = e["model"]
if not os.path.isabs(model):
if not args.models_dir:
raise SystemExit("matrix uses relative model names but --models-dir not given")
model = os.path.join(args.models_dir, model)
configs.append(Config(e.get("label", Path(model).stem), model, e.get("compute_unit", "CPU_AND_NE")))
return configs
if args.model:
for m in args.model:
for cu in args.compute_units:
configs.append(Config(f"{Path(m).stem}-{cu}", m, cu))
return configs
if args.models_dir:
found: list[str] = []
for ext in COREML_EXTS:
found += glob.glob(os.path.join(args.models_dir, f"*{ext}"))
found = sorted(set(found))
if not found:
raise SystemExit(f"No {COREML_EXTS} models found under {args.models_dir}")
for m in found:
for cu in args.compute_units:
configs.append(Config(f"{Path(m).stem}-{cu}", m, cu))
return configs
raise SystemExit("Provide one of: --matrix, --model, or --models-dir")
def measure(cfg: Config, args, rng: np.random.Generator) -> dict[str, Any]:
from python_coreml_stable_diffusion.coreml_model import CoreMLModel # local import: M2 only
result: dict[str, Any] = {
"label": cfg.label,
"model_path": cfg.model_path,
"compute_unit": cfg.compute_unit,
"error": None,
}
path = Path(cfg.model_path)
if not path.exists():
result["error"] = "model path does not exist"
return result
sources = "compiled" if cfg.model_path.endswith(".mlmodelc") else "packages"
peak_rss = sample_rss_mb()
try:
result["model_size_bytes"] = dir_size_bytes(path)
t0 = time.perf_counter()
model = CoreMLModel(cfg.model_path, cfg.compute_unit, sources)
result["load_time_s"] = round(time.perf_counter() - t0, 4)
peak_rss = max(filter(None, [peak_rss, sample_rss_mb()]), default=None)
expected = dict(model.expected_inputs)
result["kind"] = detect_kind(expected)
result["expected_inputs"] = {k: list(v["shape"]) for k, v in expected.items()}
# Pre-generate all input sets so timing excludes array allocation.
warm_in = make_inputs(expected, rng)
step_inputs = [make_inputs(expected, rng) for _ in range(args.repeats)]
# Warmup (first forward — ANE prepares its graph here).
t0 = time.perf_counter()
out = model(**warm_in)
result["warmup_ms"] = round((time.perf_counter() - t0) * 1000.0, 3)
if not (isinstance(out, dict) and "noise_pred" in out):
result["error"] = "unexpected output (no 'noise_pred')"
peak_rss = max(filter(None, [peak_rss, sample_rss_mb()]), default=None)
# Steady-state forward latency.
times_ms: list[float] = []
for inp in step_inputs:
t0 = time.perf_counter()
model(**inp)
times_ms.append((time.perf_counter() - t0) * 1000.0)
rss = sample_rss_mb()
if rss is not None:
peak_rss = rss if peak_rss is None else max(peak_rss, rss)
result["forward_ms"] = {
"mean": round(statistics.fmean(times_ms), 3),
"std": round(statistics.pstdev(times_ms), 3) if len(times_ms) > 1 else 0.0,
"min": round(min(times_ms), 3),
"median": round(statistics.median(times_ms), 3),
"n": len(times_ms),
}
result["est_e2e_ms"] = round(result["forward_ms"]["median"] * args.assumed_steps, 1)
result["peak_rss_mb"] = round(peak_rss, 1) if peak_rss is not None else None
except Exception as exc: # keep one bad model from killing the whole run
result["error"] = f"{type(exc).__name__}: {exc}"
return result
def compute_quality(reference_dir: str, candidate_dir: str) -> dict[str, Any]:
try:
from PIL import Image # type: ignore
except Exception as exc: # pragma: no cover
return {"error": f"Pillow not available: {exc}"}
ref = {p.name: p for p in Path(reference_dir).glob("*.png")}
cand = {p.name: p for p in Path(candidate_dir).glob("*.png")}
common = sorted(set(ref) & set(cand))
pairs: dict[str, Any] = {}
for name in common:
try:
a = np.array(Image.open(ref[name]).convert("RGB"))
b = np.array(Image.open(cand[name]).convert("RGB"))
if a.shape != b.shape:
pairs[name] = {"error": f"shape mismatch {a.shape} vs {b.shape}"}
continue
pairs[name] = {"psnr_db": round(psnr(a, b), 2)}
except Exception as exc:
pairs[name] = {"error": str(exc)}
return {
"reference_dir": reference_dir,
"candidate_dir": candidate_dir,
"matched": len(common),
"pairs": pairs,
}
# ----- reporting ------------------------------------------------------------
def to_markdown(report: dict[str, Any]) -> str:
lines = [
f"# CoreMLSuite baseline — `{report['git_sha']}`",
"",
f"- timestamp: {report['timestamp']}",
f"- host: {report['host']['machine']} / {report['host']['system']} {report['host']['release']}",
f"- python: {report['host']['python']}",
f"- coremltools: {report['tool_versions']['coremltools']}, numpy: {report['tool_versions']['numpy']}",
f"- settings: repeats={report['settings']['repeats']}, "
f"input_seed={report['settings']['input_seed']}, "
f"assumed_steps={report['settings']['assumed_steps']}",
"",
"| label | kind | compute | size (MB) | load (s) | warmup (ms) | fwd median (ms) | fwd min (ms) | est e2e (ms) | peak RSS (MB) | error |",
"|---|---|---|---|---|---|---|---|---|---|---|",
]
for r in report["results"]:
if r.get("error") and "forward_ms" not in r:
lines.append(
f"| {r['label']} | - | {r['compute_unit']} | - | - | - | - | - | - | - | {r['error']} |"
)
continue
size_mb = round(r.get("model_size_bytes", 0) / (1024 * 1024), 1)
fwd = r.get("forward_ms", {})
lines.append(
f"| {r['label']} | {r.get('kind','?')} | {r['compute_unit']} | {size_mb} | "
f"{r.get('load_time_s','-')} | {r.get('warmup_ms','-')} | {fwd.get('median','-')} | "
f"{fwd.get('min','-')} | {r.get('est_e2e_ms','-')} | {r.get('peak_rss_mb','-')} | "
f"{r.get('error') or ''} |"
)
q = report.get("quality")
if q and q.get("pairs"):
lines += ["", "## Quality (PSNR vs reference)", "", "| image | PSNR (dB) |", "|---|---|"]
for name, val in q["pairs"].items():
lines.append(f"| {name} | {val.get('psnr_db', val.get('error','?'))} |")
if report.get("notes"):
lines += ["", "## Notes", ""] + [f"- {n}" for n in report["notes"]]
return "\n".join(lines) + "\n"
def main(argv: Optional[list[str]] = None) -> int:
ap = argparse.ArgumentParser(description="CoreMLSuite Phase-1 baseline benchmark")
ap.add_argument("--matrix", help="path to matrix.json (list of {label, model, compute_unit})")
ap.add_argument("--models-dir", help="dir holding .mlmodelc/.mlpackage models")
ap.add_argument("--model", action="append", help="explicit model path (repeatable)")
ap.add_argument("--compute-units", nargs="+", default=DEFAULT_COMPUTE_UNITS,
help="compute units to sweep in auto/--model mode")
ap.add_argument("--repeats", type=int, default=30, help="steady-state forward passes")
ap.add_argument("--assumed-steps", type=int, default=20,
help="sampler steps used ONLY for the est_e2e_ms estimate")
ap.add_argument("--input-seed", type=int, default=0, help="seed for synthetic input generation")
ap.add_argument("--reference-images", help="dir of reference PNGs (e.g. MPS) for optional PSNR")
ap.add_argument("--candidate-images", help="dir of candidate PNGs (e.g. CoreML) for optional PSNR")
ap.add_argument("--out-dir", default="bench/results", help="where to write <gitsha>.json/.md")
args = ap.parse_args(argv)
sha = git_sha()
rng = np.random.default_rng(args.input_seed)
configs = resolve_matrix(args)
notes = [
"forward_ms is the latency of a single Core ML UNet forward pass with synthetic "
"inputs; with classifier-free guidance one sampler step typically maps to one "
"batched forward (cond+uncond) — interpret est_e2e_ms accordingly.",
"peak_rss_mb is process RSS only and excludes ANE/wired GPU memory; treat as a floor.",
"quality_psnr is null unless --reference-images and --candidate-images are provided.",
"Latency varies run-to-run; 'min' is the most reproducible figure for comparisons.",
]
if psutil is None:
notes.append("psutil not installed -> peak_rss_mb is null. `pip install psutil` to capture it.")
print(f"[bench] git={sha} coremltools={coremltools_version()} configs={len(configs)}", file=sys.stderr)
results = []
for cfg in configs:
print(f"[bench] measuring {cfg.label} ({cfg.compute_unit}) ...", file=sys.stderr)
results.append(measure(cfg, args, rng))
report: dict[str, Any] = {
"git_sha": sha,
"timestamp": _dt.datetime.now(_dt.timezone.utc).isoformat(timespec="seconds"),
"host": {
"machine": platform.machine(),
"system": platform.system(),
"release": platform.release(),
"python": platform.python_version(),
},
"tool_versions": {"coremltools": coremltools_version(), "numpy": np.__version__},
"settings": {
"repeats": args.repeats,
"assumed_steps": args.assumed_steps,
"input_seed": args.input_seed,
"compute_units": args.compute_units,
},
"results": results,
"notes": notes,
}
if args.reference_images and args.candidate_images:
report["quality"] = compute_quality(args.reference_images, args.candidate_images)
out_dir = Path(args.out_dir)
out_dir.mkdir(parents=True, exist_ok=True)
json_path = out_dir / f"{sha}.json"
md_path = out_dir / f"{sha}.md"
json_path.write_text(json.dumps(report, indent=2))
md_path.write_text(to_markdown(report))
print(f"[bench] wrote {json_path}", file=sys.stderr)
print(f"[bench] wrote {md_path}", file=sys.stderr)
return 0
if __name__ == "__main__":
raise SystemExit(main())
+87
View File
@@ -0,0 +1,87 @@
#!/usr/bin/env python3
"""bench/scripts/convert_sd15.py — Phase 1 baseline UNet conversion.
Converts a SD1.5 checkpoint to a Core ML UNet (.mlmodelc) for the bench harness.
Bypasses the ComfyUI node graph and calls coreml_suite.converter directly so
the conversion is reproducible from a single command.
Run from the repo root with the project venv:
.venv/bin/python bench/scripts/convert_sd15.py
Environment overrides:
COMFY_DIR=... (default: ../..)
CKPT_NAME=v1-5-pruned-emaonly.safetensors
ATTN=SPLIT_EINSUM (SPLIT_EINSUM | SPLIT_EINSUM_V2 | ORIGINAL)
HEIGHT=512 WIDTH=512 BATCH_SIZE=1
CONTROLNET=0
"""
import logging
import os
import sys
from pathlib import Path
logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(message)s")
log = logging.getLogger("convert_sd15")
REPO_ROOT = Path(__file__).resolve().parents[2]
COMFY_DIR = Path(os.environ.get("COMFY_DIR", REPO_ROOT.parents[1])).resolve()
if str(COMFY_DIR) not in sys.path:
sys.path.insert(0, str(COMFY_DIR))
if str(REPO_ROOT) not in sys.path:
sys.path.insert(0, str(REPO_ROOT))
import folder_paths # noqa: E402 (ComfyUI module — needs sys.path above)
# Ensure ComfyUI's folder_paths is pointed at the real install for checkpoints.
folder_paths.base_path = str(COMFY_DIR)
folder_paths.add_model_folder_path("checkpoints", str(COMFY_DIR / "models" / "checkpoints"))
folder_paths.add_model_folder_path("unet", str(COMFY_DIR / "models" / "unet"))
from coreml_suite import converter # noqa: E402
from coreml_suite.config import ModelVersion # noqa: E402
def main() -> int:
ckpt_name = os.environ.get("CKPT_NAME", "v1-5-pruned-emaonly.safetensors")
attn = os.environ.get("ATTN", "SPLIT_EINSUM")
height = int(os.environ.get("HEIGHT", "512"))
width = int(os.environ.get("WIDTH", "512"))
batch_size = int(os.environ.get("BATCH_SIZE", "1"))
controlnet = os.environ.get("CONTROLNET", "0") not in ("0", "false", "False", "")
ckpt_path = folder_paths.get_full_path("checkpoints", ckpt_name)
if not ckpt_path:
log.error("checkpoint not found: %s under %s", ckpt_name, COMFY_DIR / "models" / "checkpoints")
return 2
attn_suffix = {"SPLIT_EINSUM": "se", "SPLIT_EINSUM_V2": "se2", "ORIGINAL": "orig"}[attn]
cn_suffix = "_cn" if controlnet else ""
stem = ckpt_name.split(".")[0]
out_name = f"{stem}_{batch_size}x{width}x{height}{cn_suffix}_{attn_suffix}"
unet_out_path = converter.get_out_path("unet", out_name)
log.info("repo_root=%s comfy_dir=%s", REPO_ROOT, COMFY_DIR)
log.info("ckpt=%s out_name=%s", ckpt_path, out_name)
log.info("attn=%s size=%dx%d batch=%d controlnet=%s", attn, width, height, batch_size, controlnet)
converter.convert(
ckpt_path=ckpt_path,
model_version=ModelVersion.SD15,
unet_out_path=unet_out_path,
batch_size=batch_size,
sample_size=(height // 8, width // 8),
controlnet_support=controlnet,
lora_weights=[],
attn_impl=attn,
config_path=None,
)
target_path = converter.compile_model(out_path=unet_out_path, out_name=out_name, submodule_name="unet")
log.info("compiled: %s", target_path)
print(target_path)
return 0
if __name__ == "__main__":
raise SystemExit(main())
+163
View File
@@ -0,0 +1,163 @@
#!/usr/bin/env python3
"""bench/scripts/smoke_image.py — Phase 1 known-good-workflow image smoke test.
Posts a Stable Diffusion 1.5 workflow (CLIP → Core ML UNet → VAE → PNG) to a
locally running ComfyUI server, waits for the queue to drain, then copies the
generated CoreML + MPS reference images into bench/results/smoke/<gitsha>/
and writes a small report.json with the PSNR between them. This is the Phase 1
"known-good workflow still produces an image [M2-ANE]" artifact.
Prereqs:
- ComfyUI server running locally on $COMFY_HOST:$COMFY_PORT (default
http://localhost:8188), started against the project .venv so it picks up
the pinned ml-stable-diffusion + Core ML Suite nodes.
- The Core ML UNet has already been converted (see convert_sd15.py); the
workflow's Core ML Converter node will skip if the .mlmodelc exists.
Run from the repo root:
.venv/bin/python bench/scripts/smoke_image.py
"""
import json
import os
import shutil
import subprocess
import sys
import time
import urllib.error
import urllib.request
from pathlib import Path
import numpy as np
from PIL import Image
REPO_ROOT = Path(__file__).resolve().parents[2]
COMFY_DIR = Path(os.environ.get("COMFY_DIR", REPO_ROOT.parents[1])).resolve()
COMFY_HOST = os.environ.get("COMFY_HOST", "localhost")
COMFY_PORT = int(os.environ.get("COMFY_PORT", "8188"))
COMFY_URL = f"http://{COMFY_HOST}:{COMFY_PORT}"
CKPT_NAME = os.environ.get("CKPT_NAME", "v1-5-pruned-emaonly.safetensors")
WORKFLOW_PATH = Path(os.environ.get(
"WORKFLOW_PATH",
REPO_ROOT / "tests" / "integration" / "workflows" / "e2e-1.5-basic-conversion.json",
))
TIMEOUT_S = int(os.environ.get("TIMEOUT_S", "1800")) # 30min cap
def git_sha() -> str:
try:
return subprocess.check_output(
["git", "-C", str(REPO_ROOT), "rev-parse", "--short", "HEAD"],
stderr=subprocess.DEVNULL,
).decode().strip()
except Exception:
return "nogit"
def http_get_json(path: str) -> dict:
with urllib.request.urlopen(f"{COMFY_URL}{path}", timeout=60) as r:
return json.loads(r.read().decode())
def http_post_json(path: str, payload: dict) -> dict:
data = json.dumps(payload).encode("utf-8")
req = urllib.request.Request(
f"{COMFY_URL}{path}", data=data,
headers={"Content-Type": "application/json"}, method="POST",
)
# Long timeout: first POST blocks while the server warms model loaders +
# the Core ML compile-on-load pass (~60-90s on a cold cache).
with urllib.request.urlopen(req, timeout=300) as r:
return json.loads(r.read().decode())
def psnr(a: np.ndarray, b: np.ndarray) -> float:
mse = float(np.mean((a.astype(np.float64) - b.astype(np.float64)) ** 2))
if mse == 0:
return 100.0
return 20.0 * float(np.log10(255.0 / np.sqrt(mse)))
def find_image(out_dir: Path, prefix: str) -> Path | None:
matches = sorted(out_dir.glob(f"{prefix}_*.png"), reverse=True)
return matches[0] if matches else None
def main() -> int:
sha = git_sha()
out_root = REPO_ROOT / "bench" / "results" / "smoke" / sha
out_root.mkdir(parents=True, exist_ok=True)
workflow = json.loads(WORKFLOW_PATH.read_text())
# Point both checkpoint nodes at the maintainer's available SD1.5 ckpt.
for nid in ("4", "10"):
if nid in workflow:
workflow[nid]["inputs"]["ckpt_name"] = CKPT_NAME
# Fixed seed for reproducibility (Phase 1 baseline).
seed = int(os.environ.get("SEED", "42"))
for nid in ("3", "11"):
if nid in workflow and "seed" in workflow[nid].get("inputs", {}):
workflow[nid]["inputs"]["seed"] = seed
# Phase 1 only needs a Core ML image. The reference MPS branch (nodes
# 3/8/9) trips a known MPS f16/f32 mismatch on torch 2.0.1 + macOS 26+,
# which would crash the whole server. Strip those nodes so the queue
# only runs the Core ML path; SKIP_MPS=0 to opt back in.
if os.environ.get("SKIP_MPS", "1") not in ("0", "false", "False"):
for nid in ("3", "8", "9"):
workflow.pop(nid, None)
print(f"[smoke] git={sha} server={COMFY_URL} ckpt={CKPT_NAME} seed={seed}", file=sys.stderr)
resp = http_post_json("/prompt", {"prompt": workflow})
print(f"[smoke] queued: {resp}", file=sys.stderr)
t0 = time.time()
while True:
if time.time() - t0 > TIMEOUT_S:
print(f"[smoke] TIMEOUT after {TIMEOUT_S}s waiting for queue drain", file=sys.stderr)
return 2
try:
q = http_get_json("/prompt")
remaining = q.get("exec_info", {}).get("queue_remaining", -1)
if remaining == 0:
break
except (urllib.error.URLError, json.JSONDecodeError) as exc:
print(f"[smoke] poll error: {exc}", file=sys.stderr)
time.sleep(2)
comfy_out = COMFY_DIR / "output"
coreml_png = find_image(comfy_out, "E2E-1.5-CoreML")
if coreml_png is None:
print(f"[smoke] missing Core ML image under {comfy_out}", file=sys.stderr)
return 3
coreml_dst = out_root / coreml_png.name
shutil.copy2(coreml_png, coreml_dst)
mps_png = find_image(comfy_out, "E2E-1.5-MPS")
mps_dst = None
psnr_db = None
if mps_png is not None:
mps_dst = out_root / mps_png.name
shutil.copy2(mps_png, mps_dst)
a = np.array(Image.open(coreml_dst).convert("RGB"))
b = np.array(Image.open(mps_dst).convert("RGB"))
if a.shape == b.shape:
psnr_db = round(psnr(a, b), 2)
report = {
"git_sha": sha,
"seed": seed,
"ckpt_name": CKPT_NAME,
"workflow": str(WORKFLOW_PATH.relative_to(REPO_ROOT)),
"coreml_image": str(coreml_dst.relative_to(REPO_ROOT)),
"mps_image": str(mps_dst.relative_to(REPO_ROOT)) if mps_dst else None,
"psnr_db": psnr_db,
"wall_seconds": round(time.time() - t0, 1),
}
(out_root / "report.json").write_text(json.dumps(report, indent=2))
print(json.dumps(report, indent=2))
return 0
if __name__ == "__main__":
raise SystemExit(main())
+26 -2
View File
@@ -2,8 +2,23 @@
name = "comfyui-coremlsuite" name = "comfyui-coremlsuite"
description = "This extension contains a set of custom nodes for ComfyUI that allow you to use Core ML models in your ComfyUI workflows." description = "This extension contains a set of custom nodes for ComfyUI that allow you to use Core ML models in your ComfyUI workflows."
version = "1.0.1" version = "1.0.1"
license = { file = "LICENSE" } license = "MIT"
dependencies = ["git+https://github.com/apple/ml-stable-diffusion.git", "coremltools>=7.1", "overrides", "diffusers>=0.22", "peft>=0.6.2", "omegaconf>=2.3"] requires-python = ">=3.11,<3.12"
packages = [{ include = "coreml_suite" }]
dependencies = [
# Phase 1 baseline pins: matches the apple_env that last produced working
# conversions. Comfy's checkpoint-safe-loading branch (utils.py:33) is gated
# on torch>=2.4, so newer torch + numpy 1.23 (ml-sd's pin) breaks at import.
# Bump the whole set together in Phase 5; do not bump individually.
"python-coreml-stable-diffusion @ git+https://github.com/apple/ml-stable-diffusion.git@e5d960c41a6a4ab200b8db379194127607b1c590",
"torch==2.0.1",
"coremltools==8.2",
"numpy<1.25",
"overrides",
"diffusers>=0.22",
"peft>=0.6.2",
"omegaconf>=2.3",
]
[project.urls] [project.urls]
Repository = "https://github.com/aszc-dev/ComfyUI-CoreMLSuite" Repository = "https://github.com/aszc-dev/ComfyUI-CoreMLSuite"
@@ -13,3 +28,12 @@ Repository = "https://github.com/aszc-dev/ComfyUI-CoreMLSuite"
PublisherId = "aszc-dev" PublisherId = "aszc-dev"
DisplayName = "ComfyUI-CoreMLSuite" DisplayName = "ComfyUI-CoreMLSuite"
Icon = "" Icon = ""
# Pinned to the ComfyUI commit this Phase 1 baseline was validated against.
# Bump together with the toolchain upgrade in Phase 5.
requires-comfyui = "==ab5413351eee61f3d7f10c74e75286df0058bb18"
[dependency-groups]
dev = [
"pillow>=12.2.0",
"psutil>=7.2.2",
]
+4 -2
View File
@@ -1,5 +1,7 @@
git+https://github.com/apple/ml-stable-diffusion.git git+https://github.com/apple/ml-stable-diffusion.git@e5d960c41a6a4ab200b8db379194127607b1c590
coremltools>=7.1 torch==2.0.1
coremltools==8.2
numpy<1.25
overrides overrides
diffusers>=0.22 diffusers>=0.22
peft>=0.6.2 peft>=0.6.2
+2 -2
View File
@@ -8,7 +8,7 @@ from coreml_suite.controlnet import chunk_control
from coreml_suite.models import ( from coreml_suite.models import (
CoreMLInputs, CoreMLInputs,
) )
from coreml_suite.config import get_model_config from coreml_suite.config import ModelVersion, get_model_config
@pytest.fixture @pytest.fixture
@@ -26,7 +26,7 @@ def expected_inputs():
@pytest.fixture @pytest.fixture
def model_config(): def model_config():
return get_model_config() return get_model_config(ModelVersion.SD15)
@pytest.mark.parametrize("batch_size", [1, 2, 4, 5, 9]) @pytest.mark.parametrize("batch_size", [1, 2, 4, 5, 9])
Generated
+1404
View File
File diff suppressed because it is too large Load Diff