chore(phase1): pin baseline toolchain and add bench harness scaffold
Phase 1 of the modernization plan: freeze the currently-working environment so later refactors have a measured reference point. - Pin python-coreml-stable-diffusion to commit e5d960c4 (the one already installed in the maintainer's apple_env), plus torch==2.0.1, coremltools==8.2 and numpy<1.25 to match the only env that loads ComfyUI successfully (Comfy's checkpoint-safe-loading branch in utils.py is gated on torch>=2.4, so newer torch + numpy 1.23 breaks at import). - Mirror the same pins in requirements.txt and commit uv.lock for reproducible installs. - Add requires-comfyui pinning ComfyUI to ab541335 (the validated commit). - Fix tests/unit/test_chunks.py fixture: get_model_config() now takes a ModelVersion argument; pass ModelVersion.SD15 (the previously-broken test was the only Phase 1 production-code change required). - Add the Phase 1 baseline harness: bench/run.py (direct Core ML UNet latency, deterministic), bench/scripts/convert_sd15.py (one-command conversion bypassing the node graph), bench/scripts/smoke_image.py (POSTs the existing e2e workflow to a local ComfyUI server and saves the Core ML image), bench/env/capture.sh (env snapshot), bench/prompts.json (fixed prompt set). - Ignore apple_env/, comfy_env/, and bench/scripts/*.log. Tests: 20/20 unit pass (test_chunks + test_controlnet).
This commit is contained in:
@@ -1,3 +1,11 @@
|
||||
playground/
|
||||
experiments/
|
||||
__pycache__/
|
||||
models/
|
||||
.venv/
|
||||
apple_env/
|
||||
comfy_env/
|
||||
coremlsuite-venv/
|
||||
experiment_results/
|
||||
test_results/
|
||||
bench/scripts/*.log
|
||||
|
||||
+100
@@ -0,0 +1,100 @@
|
||||
#!/usr/bin/env bash
|
||||
# Phase 1 baseline environment capture.
|
||||
#
|
||||
# Default mode: query the project's .venv via `uv pip` (matches how uv builds
|
||||
# the venv — without a bundled pip executable). For a non-uv setup, point
|
||||
# PYTHON_BIN at the right interpreter; pip freeze then falls back to
|
||||
# `python -m pip` if available, otherwise `uv pip`.
|
||||
#
|
||||
# Usage:
|
||||
# bash bench/env/capture.sh # uses .venv (uv-managed)
|
||||
# PYTHON_BIN=apple_env/bin/python bash bench/env/capture.sh
|
||||
#
|
||||
# Writes bench/env/baseline-<gitsha>.txt with: this repo's git sha,
|
||||
# ComfyUI's git sha (override path with COMFY_DIR=...), python + macOS
|
||||
# versions, the resolved git commit of python-coreml-stable-diffusion as
|
||||
# installed, and the full freeze of the active venv (for fresh-venv
|
||||
# reproducibility).
|
||||
set -euo pipefail
|
||||
|
||||
REPO_ROOT="$(git rev-parse --show-toplevel)"
|
||||
cd "$REPO_ROOT"
|
||||
|
||||
SHA_SHORT="$(git rev-parse --short HEAD)"
|
||||
OUT_DIR="bench/env"
|
||||
OUT="${OUT_DIR}/baseline-${SHA_SHORT}.txt"
|
||||
mkdir -p "$OUT_DIR"
|
||||
|
||||
COMFY_DIR="${COMFY_DIR:-$(cd ../.. && pwd)}"
|
||||
PYTHON_BIN="${PYTHON_BIN:-$REPO_ROOT/.venv/bin/python}"
|
||||
|
||||
if [ ! -x "$PYTHON_BIN" ]; then
|
||||
echo "PYTHON_BIN not executable: $PYTHON_BIN" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Prefer `python -m pip` (works in classic venvs). Fall back to `uv pip`,
|
||||
# which queries any interpreter without needing pip installed in it.
|
||||
run_freeze() {
|
||||
if "$PYTHON_BIN" -m pip --version >/dev/null 2>&1; then
|
||||
"$PYTHON_BIN" -m pip freeze
|
||||
elif command -v uv >/dev/null 2>&1; then
|
||||
uv pip freeze --python "$PYTHON_BIN"
|
||||
else
|
||||
echo "neither pip nor uv available for freeze"
|
||||
fi
|
||||
}
|
||||
|
||||
run_show_msd() {
|
||||
if "$PYTHON_BIN" -m pip --version >/dev/null 2>&1; then
|
||||
"$PYTHON_BIN" -m pip show python-coreml-stable-diffusion 2>/dev/null || echo "not-installed"
|
||||
elif command -v uv >/dev/null 2>&1; then
|
||||
uv pip show --python "$PYTHON_BIN" python-coreml-stable-diffusion 2>/dev/null || echo "not-installed"
|
||||
else
|
||||
echo "neither pip nor uv available for show"
|
||||
fi
|
||||
}
|
||||
|
||||
{
|
||||
echo "# Phase 1 baseline environment capture"
|
||||
echo "timestamp_utc: $(date -u +%Y-%m-%dT%H:%M:%SZ)"
|
||||
echo "repo_sha: $(git rev-parse HEAD)"
|
||||
echo "repo_branch: $(git rev-parse --abbrev-ref HEAD)"
|
||||
echo "comfyui_dir: ${COMFY_DIR}"
|
||||
echo "comfyui_sha: $(git -C "$COMFY_DIR" rev-parse HEAD 2>/dev/null || echo unknown)"
|
||||
echo "python_bin: ${PYTHON_BIN}"
|
||||
echo "python_version: $("$PYTHON_BIN" --version 2>&1)"
|
||||
echo "platform_machine: $(uname -m)"
|
||||
echo "platform_uname: $(uname -a)"
|
||||
if command -v sw_vers >/dev/null 2>&1; then
|
||||
echo "macos_product: $(sw_vers -productName)"
|
||||
echo "macos_version: $(sw_vers -productVersion)"
|
||||
echo "macos_build: $(sw_vers -buildVersion)"
|
||||
fi
|
||||
echo
|
||||
echo "## python-coreml-stable-diffusion (resolved)"
|
||||
run_show_msd
|
||||
echo
|
||||
echo "## python-coreml-stable-diffusion installed git metadata"
|
||||
DIST_INFO="$("$PYTHON_BIN" -c "import importlib.metadata as m; d=m.distribution('python-coreml-stable-diffusion'); print(d._path)" 2>/dev/null || true)"
|
||||
if [ -n "${DIST_INFO}" ] && [ -f "${DIST_INFO}/direct_url.json" ]; then
|
||||
cat "${DIST_INFO}/direct_url.json"
|
||||
echo
|
||||
else
|
||||
echo "direct_url.json not found"
|
||||
fi
|
||||
echo
|
||||
echo "## coremltools version"
|
||||
"$PYTHON_BIN" -c "import coremltools; print('coremltools', coremltools.__version__)" 2>&1 || true
|
||||
echo
|
||||
echo "## torch version"
|
||||
"$PYTHON_BIN" -c "import torch; print('torch', torch.__version__)" 2>&1 || true
|
||||
echo
|
||||
echo "## numpy version"
|
||||
"$PYTHON_BIN" -c "import numpy; print('numpy', numpy.__version__)" 2>&1 || true
|
||||
echo
|
||||
echo "## freeze"
|
||||
run_freeze
|
||||
} > "$OUT"
|
||||
|
||||
echo "wrote $OUT"
|
||||
@@ -0,0 +1,19 @@
|
||||
{
|
||||
"_comment": "Fixed prompt set for Phase 1 baseline reproducibility. Do not edit without bumping the bench version: changing prompts invalidates cross-baseline PSNR comparisons.",
|
||||
"seed": 0,
|
||||
"negative_prompt": "blurry, low quality, text, watermark",
|
||||
"prompts": [
|
||||
{
|
||||
"id": "p01_portrait",
|
||||
"text": "a portrait of an astronaut in a sunflower field, soft natural light, 35mm film"
|
||||
},
|
||||
{
|
||||
"id": "p02_cityscape",
|
||||
"text": "a futuristic city skyline at sunset, glass towers, volumetric lighting, cinematic"
|
||||
},
|
||||
{
|
||||
"id": "p03_still_life",
|
||||
"text": "a still life of fruit on a wooden table, oil painting, dramatic chiaroscuro"
|
||||
}
|
||||
]
|
||||
}
|
||||
+417
@@ -0,0 +1,417 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
bench/run.py — Phase 1 baseline benchmark harness for ComfyUI-CoreMLSuite.
|
||||
|
||||
WHAT THIS MEASURES (the deterministic, easy-to-trust numbers)
|
||||
-------------------------------------------------------------
|
||||
For each converted Core ML UNet model in the matrix, in-process:
|
||||
* model load time (includes mlpackage compile; mlmodelc is precompiled)
|
||||
* model file size on disk
|
||||
* warmup (first forward) time
|
||||
* steady-state UNet forward latency (mean/std/min/median over N repeats)
|
||||
* an end-to-end estimate (median forward * assumed sampler steps)
|
||||
* peak process RSS during the run
|
||||
|
||||
It drives the Core ML model DIRECTLY with synthetic inputs shaped from the
|
||||
model's own `expected_inputs`. This isolates the ANE/Core ML UNet latency —
|
||||
the exact quantity that quantization (Phase 6) and the coremltools upgrade
|
||||
(Phase 5) are expected to move — WITHOUT needing a live ComfyUI server, CLIP,
|
||||
VAE, or a checkpoint. It is therefore the cleanest, most reproducible signal.
|
||||
|
||||
WHAT THIS DOES NOT MEASURE
|
||||
--------------------------
|
||||
* Image quality. PSNR is a separate, decoupled concern: pass --reference-images
|
||||
and --candidate-images (e.g. the E2E-1.5-MPS / E2E-1.5-CoreML PNGs your
|
||||
existing integration workflow already produces) and PSNR is computed per
|
||||
matching filename. Without them, quality_psnr is null (and that's honest).
|
||||
* ANE / wired GPU memory. peak_rss_mb is *process* RSS only; treat as a coarse
|
||||
floor, not the true device footprint.
|
||||
|
||||
REQUIREMENTS
|
||||
------------
|
||||
* Run on Apple Silicon (macOS) with the suite's deps installed
|
||||
(coremltools + apple/ml-stable-diffusion providing `python_coreml_stable_diffusion`).
|
||||
* numpy is required. psutil and Pillow are optional (graceful fallback).
|
||||
|
||||
USAGE
|
||||
-----
|
||||
# Auto-discover every .mlmodelc/.mlpackage under a dir, sweep compute units:
|
||||
python bench/run.py --models-dir /path/to/ComfyUI/models/unet
|
||||
|
||||
# Explicit matrix (recommended for a stable, committed baseline):
|
||||
python bench/run.py --matrix bench/matrix.json --models-dir /path/to/models/unet
|
||||
|
||||
# With quality from already-produced images:
|
||||
python bench/run.py --models-dir ... \
|
||||
--reference-images /path/to/mps_pngs --candidate-images /path/to/coreml_pngs
|
||||
|
||||
matrix.json format (a list of configs):
|
||||
[
|
||||
{"label": "sd15-se", "model": "dreamshaper_8_1x512x512_se.mlmodelc", "compute_unit": "CPU_AND_NE"},
|
||||
{"label": "sdxl-orig","model": "sdxl_base_1x1024x1024_orig.mlmodelc", "compute_unit": "CPU_AND_GPU"}
|
||||
]
|
||||
Each "model" is resolved against --models-dir unless it is an absolute path.
|
||||
|
||||
Output: bench/results/<gitsha>.json and bench/results/<gitsha>.md
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import dataclasses
|
||||
import datetime as _dt
|
||||
import glob
|
||||
import json
|
||||
import os
|
||||
import platform
|
||||
import statistics
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import Any, Optional
|
||||
|
||||
import numpy as np
|
||||
|
||||
# ----- optional deps (graceful) ---------------------------------------------
|
||||
try:
|
||||
import psutil # type: ignore
|
||||
|
||||
_PROC = psutil.Process(os.getpid())
|
||||
except Exception: # pragma: no cover - optional
|
||||
psutil = None
|
||||
_PROC = None
|
||||
|
||||
DEFAULT_COMPUTE_UNITS = ["CPU_AND_NE", "CPU_AND_GPU"]
|
||||
COREML_EXTS = (".mlmodelc", ".mlpackage")
|
||||
|
||||
|
||||
# ----- small helpers --------------------------------------------------------
|
||||
def git_sha() -> str:
|
||||
try:
|
||||
out = subprocess.check_output(
|
||||
["git", "rev-parse", "--short", "HEAD"], stderr=subprocess.DEVNULL
|
||||
)
|
||||
return out.decode().strip() or "nogit"
|
||||
except Exception:
|
||||
return "nogit"
|
||||
|
||||
|
||||
def coremltools_version() -> str:
|
||||
try:
|
||||
import coremltools as ct # noqa: WPS433 (local import is intentional)
|
||||
|
||||
return getattr(ct, "__version__", "unknown")
|
||||
except Exception as exc: # pragma: no cover
|
||||
return f"import-failed: {exc}"
|
||||
|
||||
|
||||
def dir_size_bytes(path: Path) -> int:
|
||||
"""Core ML models are directories (.mlmodelc / .mlpackage)."""
|
||||
if path.is_file():
|
||||
return path.stat().st_size
|
||||
total = 0
|
||||
for root, _dirs, files in os.walk(path):
|
||||
for f in files:
|
||||
try:
|
||||
total += (Path(root) / f).stat().st_size
|
||||
except OSError:
|
||||
pass
|
||||
return total
|
||||
|
||||
|
||||
def sample_rss_mb() -> Optional[float]:
|
||||
if _PROC is None:
|
||||
return None
|
||||
try:
|
||||
return _PROC.memory_info().rss / (1024 * 1024)
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
def detect_kind(expected_inputs: dict) -> str:
|
||||
"""Mirror the suite's detection WITHOUT importing comfy (keeps bench light)."""
|
||||
if "time_ids" in expected_inputs and "text_embeds" in expected_inputs:
|
||||
try:
|
||||
n = expected_inputs["time_ids"]["shape"][1]
|
||||
except Exception:
|
||||
n = -1
|
||||
return "sdxl_base" if n == 6 else "sdxl_refiner" if n == 5 else "sdxl_unknown"
|
||||
if "timestep_cond" in expected_inputs:
|
||||
return "lcm"
|
||||
return "sd15"
|
||||
|
||||
|
||||
def make_inputs(expected_inputs: dict, rng: np.random.Generator) -> dict[str, np.ndarray]:
|
||||
"""Build synthetic fp16 inputs matching the model's expected shapes.
|
||||
|
||||
Values are random; they do not affect latency in any meaningful way, and
|
||||
we feed fp16 to match how CoreMLModelWrapper feeds the model at runtime.
|
||||
"""
|
||||
inputs: dict[str, np.ndarray] = {}
|
||||
for name, spec in expected_inputs.items():
|
||||
shape = tuple(int(d) for d in spec["shape"])
|
||||
inputs[name] = rng.standard_normal(shape).astype(np.float16)
|
||||
return inputs
|
||||
|
||||
|
||||
def psnr(img_a: "np.ndarray", img_b: "np.ndarray") -> float:
|
||||
mse = float(np.mean((img_a.astype(np.float64) - img_b.astype(np.float64)) ** 2))
|
||||
if mse == 0:
|
||||
return 100.0
|
||||
return 20.0 * float(np.log10(255.0 / np.sqrt(mse)))
|
||||
|
||||
|
||||
# ----- core measurement -----------------------------------------------------
|
||||
@dataclasses.dataclass
|
||||
class Config:
|
||||
label: str
|
||||
model_path: str
|
||||
compute_unit: str
|
||||
|
||||
|
||||
def resolve_matrix(args) -> list[Config]:
|
||||
configs: list[Config] = []
|
||||
|
||||
if args.matrix:
|
||||
entries = json.loads(Path(args.matrix).read_text())
|
||||
for e in entries:
|
||||
model = e["model"]
|
||||
if not os.path.isabs(model):
|
||||
if not args.models_dir:
|
||||
raise SystemExit("matrix uses relative model names but --models-dir not given")
|
||||
model = os.path.join(args.models_dir, model)
|
||||
configs.append(Config(e.get("label", Path(model).stem), model, e.get("compute_unit", "CPU_AND_NE")))
|
||||
return configs
|
||||
|
||||
if args.model:
|
||||
for m in args.model:
|
||||
for cu in args.compute_units:
|
||||
configs.append(Config(f"{Path(m).stem}-{cu}", m, cu))
|
||||
return configs
|
||||
|
||||
if args.models_dir:
|
||||
found: list[str] = []
|
||||
for ext in COREML_EXTS:
|
||||
found += glob.glob(os.path.join(args.models_dir, f"*{ext}"))
|
||||
found = sorted(set(found))
|
||||
if not found:
|
||||
raise SystemExit(f"No {COREML_EXTS} models found under {args.models_dir}")
|
||||
for m in found:
|
||||
for cu in args.compute_units:
|
||||
configs.append(Config(f"{Path(m).stem}-{cu}", m, cu))
|
||||
return configs
|
||||
|
||||
raise SystemExit("Provide one of: --matrix, --model, or --models-dir")
|
||||
|
||||
|
||||
def measure(cfg: Config, args, rng: np.random.Generator) -> dict[str, Any]:
|
||||
from python_coreml_stable_diffusion.coreml_model import CoreMLModel # local import: M2 only
|
||||
|
||||
result: dict[str, Any] = {
|
||||
"label": cfg.label,
|
||||
"model_path": cfg.model_path,
|
||||
"compute_unit": cfg.compute_unit,
|
||||
"error": None,
|
||||
}
|
||||
path = Path(cfg.model_path)
|
||||
if not path.exists():
|
||||
result["error"] = "model path does not exist"
|
||||
return result
|
||||
|
||||
sources = "compiled" if cfg.model_path.endswith(".mlmodelc") else "packages"
|
||||
peak_rss = sample_rss_mb()
|
||||
|
||||
try:
|
||||
result["model_size_bytes"] = dir_size_bytes(path)
|
||||
|
||||
t0 = time.perf_counter()
|
||||
model = CoreMLModel(cfg.model_path, cfg.compute_unit, sources)
|
||||
result["load_time_s"] = round(time.perf_counter() - t0, 4)
|
||||
peak_rss = max(filter(None, [peak_rss, sample_rss_mb()]), default=None)
|
||||
|
||||
expected = dict(model.expected_inputs)
|
||||
result["kind"] = detect_kind(expected)
|
||||
result["expected_inputs"] = {k: list(v["shape"]) for k, v in expected.items()}
|
||||
|
||||
# Pre-generate all input sets so timing excludes array allocation.
|
||||
warm_in = make_inputs(expected, rng)
|
||||
step_inputs = [make_inputs(expected, rng) for _ in range(args.repeats)]
|
||||
|
||||
# Warmup (first forward — ANE prepares its graph here).
|
||||
t0 = time.perf_counter()
|
||||
out = model(**warm_in)
|
||||
result["warmup_ms"] = round((time.perf_counter() - t0) * 1000.0, 3)
|
||||
if not (isinstance(out, dict) and "noise_pred" in out):
|
||||
result["error"] = "unexpected output (no 'noise_pred')"
|
||||
peak_rss = max(filter(None, [peak_rss, sample_rss_mb()]), default=None)
|
||||
|
||||
# Steady-state forward latency.
|
||||
times_ms: list[float] = []
|
||||
for inp in step_inputs:
|
||||
t0 = time.perf_counter()
|
||||
model(**inp)
|
||||
times_ms.append((time.perf_counter() - t0) * 1000.0)
|
||||
rss = sample_rss_mb()
|
||||
if rss is not None:
|
||||
peak_rss = rss if peak_rss is None else max(peak_rss, rss)
|
||||
|
||||
result["forward_ms"] = {
|
||||
"mean": round(statistics.fmean(times_ms), 3),
|
||||
"std": round(statistics.pstdev(times_ms), 3) if len(times_ms) > 1 else 0.0,
|
||||
"min": round(min(times_ms), 3),
|
||||
"median": round(statistics.median(times_ms), 3),
|
||||
"n": len(times_ms),
|
||||
}
|
||||
result["est_e2e_ms"] = round(result["forward_ms"]["median"] * args.assumed_steps, 1)
|
||||
result["peak_rss_mb"] = round(peak_rss, 1) if peak_rss is not None else None
|
||||
|
||||
except Exception as exc: # keep one bad model from killing the whole run
|
||||
result["error"] = f"{type(exc).__name__}: {exc}"
|
||||
|
||||
return result
|
||||
|
||||
|
||||
def compute_quality(reference_dir: str, candidate_dir: str) -> dict[str, Any]:
|
||||
try:
|
||||
from PIL import Image # type: ignore
|
||||
except Exception as exc: # pragma: no cover
|
||||
return {"error": f"Pillow not available: {exc}"}
|
||||
|
||||
ref = {p.name: p for p in Path(reference_dir).glob("*.png")}
|
||||
cand = {p.name: p for p in Path(candidate_dir).glob("*.png")}
|
||||
common = sorted(set(ref) & set(cand))
|
||||
pairs: dict[str, Any] = {}
|
||||
for name in common:
|
||||
try:
|
||||
a = np.array(Image.open(ref[name]).convert("RGB"))
|
||||
b = np.array(Image.open(cand[name]).convert("RGB"))
|
||||
if a.shape != b.shape:
|
||||
pairs[name] = {"error": f"shape mismatch {a.shape} vs {b.shape}"}
|
||||
continue
|
||||
pairs[name] = {"psnr_db": round(psnr(a, b), 2)}
|
||||
except Exception as exc:
|
||||
pairs[name] = {"error": str(exc)}
|
||||
return {
|
||||
"reference_dir": reference_dir,
|
||||
"candidate_dir": candidate_dir,
|
||||
"matched": len(common),
|
||||
"pairs": pairs,
|
||||
}
|
||||
|
||||
|
||||
# ----- reporting ------------------------------------------------------------
|
||||
def to_markdown(report: dict[str, Any]) -> str:
|
||||
lines = [
|
||||
f"# CoreMLSuite baseline — `{report['git_sha']}`",
|
||||
"",
|
||||
f"- timestamp: {report['timestamp']}",
|
||||
f"- host: {report['host']['machine']} / {report['host']['system']} {report['host']['release']}",
|
||||
f"- python: {report['host']['python']}",
|
||||
f"- coremltools: {report['tool_versions']['coremltools']}, numpy: {report['tool_versions']['numpy']}",
|
||||
f"- settings: repeats={report['settings']['repeats']}, "
|
||||
f"input_seed={report['settings']['input_seed']}, "
|
||||
f"assumed_steps={report['settings']['assumed_steps']}",
|
||||
"",
|
||||
"| label | kind | compute | size (MB) | load (s) | warmup (ms) | fwd median (ms) | fwd min (ms) | est e2e (ms) | peak RSS (MB) | error |",
|
||||
"|---|---|---|---|---|---|---|---|---|---|---|",
|
||||
]
|
||||
for r in report["results"]:
|
||||
if r.get("error") and "forward_ms" not in r:
|
||||
lines.append(
|
||||
f"| {r['label']} | - | {r['compute_unit']} | - | - | - | - | - | - | - | {r['error']} |"
|
||||
)
|
||||
continue
|
||||
size_mb = round(r.get("model_size_bytes", 0) / (1024 * 1024), 1)
|
||||
fwd = r.get("forward_ms", {})
|
||||
lines.append(
|
||||
f"| {r['label']} | {r.get('kind','?')} | {r['compute_unit']} | {size_mb} | "
|
||||
f"{r.get('load_time_s','-')} | {r.get('warmup_ms','-')} | {fwd.get('median','-')} | "
|
||||
f"{fwd.get('min','-')} | {r.get('est_e2e_ms','-')} | {r.get('peak_rss_mb','-')} | "
|
||||
f"{r.get('error') or ''} |"
|
||||
)
|
||||
|
||||
q = report.get("quality")
|
||||
if q and q.get("pairs"):
|
||||
lines += ["", "## Quality (PSNR vs reference)", "", "| image | PSNR (dB) |", "|---|---|"]
|
||||
for name, val in q["pairs"].items():
|
||||
lines.append(f"| {name} | {val.get('psnr_db', val.get('error','?'))} |")
|
||||
|
||||
if report.get("notes"):
|
||||
lines += ["", "## Notes", ""] + [f"- {n}" for n in report["notes"]]
|
||||
return "\n".join(lines) + "\n"
|
||||
|
||||
|
||||
def main(argv: Optional[list[str]] = None) -> int:
|
||||
ap = argparse.ArgumentParser(description="CoreMLSuite Phase-1 baseline benchmark")
|
||||
ap.add_argument("--matrix", help="path to matrix.json (list of {label, model, compute_unit})")
|
||||
ap.add_argument("--models-dir", help="dir holding .mlmodelc/.mlpackage models")
|
||||
ap.add_argument("--model", action="append", help="explicit model path (repeatable)")
|
||||
ap.add_argument("--compute-units", nargs="+", default=DEFAULT_COMPUTE_UNITS,
|
||||
help="compute units to sweep in auto/--model mode")
|
||||
ap.add_argument("--repeats", type=int, default=30, help="steady-state forward passes")
|
||||
ap.add_argument("--assumed-steps", type=int, default=20,
|
||||
help="sampler steps used ONLY for the est_e2e_ms estimate")
|
||||
ap.add_argument("--input-seed", type=int, default=0, help="seed for synthetic input generation")
|
||||
ap.add_argument("--reference-images", help="dir of reference PNGs (e.g. MPS) for optional PSNR")
|
||||
ap.add_argument("--candidate-images", help="dir of candidate PNGs (e.g. CoreML) for optional PSNR")
|
||||
ap.add_argument("--out-dir", default="bench/results", help="where to write <gitsha>.json/.md")
|
||||
args = ap.parse_args(argv)
|
||||
|
||||
sha = git_sha()
|
||||
rng = np.random.default_rng(args.input_seed)
|
||||
configs = resolve_matrix(args)
|
||||
|
||||
notes = [
|
||||
"forward_ms is the latency of a single Core ML UNet forward pass with synthetic "
|
||||
"inputs; with classifier-free guidance one sampler step typically maps to one "
|
||||
"batched forward (cond+uncond) — interpret est_e2e_ms accordingly.",
|
||||
"peak_rss_mb is process RSS only and excludes ANE/wired GPU memory; treat as a floor.",
|
||||
"quality_psnr is null unless --reference-images and --candidate-images are provided.",
|
||||
"Latency varies run-to-run; 'min' is the most reproducible figure for comparisons.",
|
||||
]
|
||||
if psutil is None:
|
||||
notes.append("psutil not installed -> peak_rss_mb is null. `pip install psutil` to capture it.")
|
||||
|
||||
print(f"[bench] git={sha} coremltools={coremltools_version()} configs={len(configs)}", file=sys.stderr)
|
||||
|
||||
results = []
|
||||
for cfg in configs:
|
||||
print(f"[bench] measuring {cfg.label} ({cfg.compute_unit}) ...", file=sys.stderr)
|
||||
results.append(measure(cfg, args, rng))
|
||||
|
||||
report: dict[str, Any] = {
|
||||
"git_sha": sha,
|
||||
"timestamp": _dt.datetime.now(_dt.timezone.utc).isoformat(timespec="seconds"),
|
||||
"host": {
|
||||
"machine": platform.machine(),
|
||||
"system": platform.system(),
|
||||
"release": platform.release(),
|
||||
"python": platform.python_version(),
|
||||
},
|
||||
"tool_versions": {"coremltools": coremltools_version(), "numpy": np.__version__},
|
||||
"settings": {
|
||||
"repeats": args.repeats,
|
||||
"assumed_steps": args.assumed_steps,
|
||||
"input_seed": args.input_seed,
|
||||
"compute_units": args.compute_units,
|
||||
},
|
||||
"results": results,
|
||||
"notes": notes,
|
||||
}
|
||||
|
||||
if args.reference_images and args.candidate_images:
|
||||
report["quality"] = compute_quality(args.reference_images, args.candidate_images)
|
||||
|
||||
out_dir = Path(args.out_dir)
|
||||
out_dir.mkdir(parents=True, exist_ok=True)
|
||||
json_path = out_dir / f"{sha}.json"
|
||||
md_path = out_dir / f"{sha}.md"
|
||||
json_path.write_text(json.dumps(report, indent=2))
|
||||
md_path.write_text(to_markdown(report))
|
||||
|
||||
print(f"[bench] wrote {json_path}", file=sys.stderr)
|
||||
print(f"[bench] wrote {md_path}", file=sys.stderr)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,87 @@
|
||||
#!/usr/bin/env python3
|
||||
"""bench/scripts/convert_sd15.py — Phase 1 baseline UNet conversion.
|
||||
|
||||
Converts a SD1.5 checkpoint to a Core ML UNet (.mlmodelc) for the bench harness.
|
||||
Bypasses the ComfyUI node graph and calls coreml_suite.converter directly so
|
||||
the conversion is reproducible from a single command.
|
||||
|
||||
Run from the repo root with the project venv:
|
||||
.venv/bin/python bench/scripts/convert_sd15.py
|
||||
|
||||
Environment overrides:
|
||||
COMFY_DIR=... (default: ../..)
|
||||
CKPT_NAME=v1-5-pruned-emaonly.safetensors
|
||||
ATTN=SPLIT_EINSUM (SPLIT_EINSUM | SPLIT_EINSUM_V2 | ORIGINAL)
|
||||
HEIGHT=512 WIDTH=512 BATCH_SIZE=1
|
||||
CONTROLNET=0
|
||||
"""
|
||||
import logging
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(message)s")
|
||||
log = logging.getLogger("convert_sd15")
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parents[2]
|
||||
COMFY_DIR = Path(os.environ.get("COMFY_DIR", REPO_ROOT.parents[1])).resolve()
|
||||
|
||||
if str(COMFY_DIR) not in sys.path:
|
||||
sys.path.insert(0, str(COMFY_DIR))
|
||||
if str(REPO_ROOT) not in sys.path:
|
||||
sys.path.insert(0, str(REPO_ROOT))
|
||||
|
||||
import folder_paths # noqa: E402 (ComfyUI module — needs sys.path above)
|
||||
|
||||
# Ensure ComfyUI's folder_paths is pointed at the real install for checkpoints.
|
||||
folder_paths.base_path = str(COMFY_DIR)
|
||||
folder_paths.add_model_folder_path("checkpoints", str(COMFY_DIR / "models" / "checkpoints"))
|
||||
folder_paths.add_model_folder_path("unet", str(COMFY_DIR / "models" / "unet"))
|
||||
|
||||
from coreml_suite import converter # noqa: E402
|
||||
from coreml_suite.config import ModelVersion # noqa: E402
|
||||
|
||||
|
||||
def main() -> int:
|
||||
ckpt_name = os.environ.get("CKPT_NAME", "v1-5-pruned-emaonly.safetensors")
|
||||
attn = os.environ.get("ATTN", "SPLIT_EINSUM")
|
||||
height = int(os.environ.get("HEIGHT", "512"))
|
||||
width = int(os.environ.get("WIDTH", "512"))
|
||||
batch_size = int(os.environ.get("BATCH_SIZE", "1"))
|
||||
controlnet = os.environ.get("CONTROLNET", "0") not in ("0", "false", "False", "")
|
||||
|
||||
ckpt_path = folder_paths.get_full_path("checkpoints", ckpt_name)
|
||||
if not ckpt_path:
|
||||
log.error("checkpoint not found: %s under %s", ckpt_name, COMFY_DIR / "models" / "checkpoints")
|
||||
return 2
|
||||
|
||||
attn_suffix = {"SPLIT_EINSUM": "se", "SPLIT_EINSUM_V2": "se2", "ORIGINAL": "orig"}[attn]
|
||||
cn_suffix = "_cn" if controlnet else ""
|
||||
stem = ckpt_name.split(".")[0]
|
||||
out_name = f"{stem}_{batch_size}x{width}x{height}{cn_suffix}_{attn_suffix}"
|
||||
unet_out_path = converter.get_out_path("unet", out_name)
|
||||
|
||||
log.info("repo_root=%s comfy_dir=%s", REPO_ROOT, COMFY_DIR)
|
||||
log.info("ckpt=%s out_name=%s", ckpt_path, out_name)
|
||||
log.info("attn=%s size=%dx%d batch=%d controlnet=%s", attn, width, height, batch_size, controlnet)
|
||||
|
||||
converter.convert(
|
||||
ckpt_path=ckpt_path,
|
||||
model_version=ModelVersion.SD15,
|
||||
unet_out_path=unet_out_path,
|
||||
batch_size=batch_size,
|
||||
sample_size=(height // 8, width // 8),
|
||||
controlnet_support=controlnet,
|
||||
lora_weights=[],
|
||||
attn_impl=attn,
|
||||
config_path=None,
|
||||
)
|
||||
|
||||
target_path = converter.compile_model(out_path=unet_out_path, out_name=out_name, submodule_name="unet")
|
||||
log.info("compiled: %s", target_path)
|
||||
print(target_path)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,163 @@
|
||||
#!/usr/bin/env python3
|
||||
"""bench/scripts/smoke_image.py — Phase 1 known-good-workflow image smoke test.
|
||||
|
||||
Posts a Stable Diffusion 1.5 workflow (CLIP → Core ML UNet → VAE → PNG) to a
|
||||
locally running ComfyUI server, waits for the queue to drain, then copies the
|
||||
generated CoreML + MPS reference images into bench/results/smoke/<gitsha>/
|
||||
and writes a small report.json with the PSNR between them. This is the Phase 1
|
||||
"known-good workflow still produces an image [M2-ANE]" artifact.
|
||||
|
||||
Prereqs:
|
||||
- ComfyUI server running locally on $COMFY_HOST:$COMFY_PORT (default
|
||||
http://localhost:8188), started against the project .venv so it picks up
|
||||
the pinned ml-stable-diffusion + Core ML Suite nodes.
|
||||
- The Core ML UNet has already been converted (see convert_sd15.py); the
|
||||
workflow's Core ML Converter node will skip if the .mlmodelc exists.
|
||||
|
||||
Run from the repo root:
|
||||
.venv/bin/python bench/scripts/smoke_image.py
|
||||
"""
|
||||
import json
|
||||
import os
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
from PIL import Image
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parents[2]
|
||||
COMFY_DIR = Path(os.environ.get("COMFY_DIR", REPO_ROOT.parents[1])).resolve()
|
||||
COMFY_HOST = os.environ.get("COMFY_HOST", "localhost")
|
||||
COMFY_PORT = int(os.environ.get("COMFY_PORT", "8188"))
|
||||
COMFY_URL = f"http://{COMFY_HOST}:{COMFY_PORT}"
|
||||
|
||||
CKPT_NAME = os.environ.get("CKPT_NAME", "v1-5-pruned-emaonly.safetensors")
|
||||
WORKFLOW_PATH = Path(os.environ.get(
|
||||
"WORKFLOW_PATH",
|
||||
REPO_ROOT / "tests" / "integration" / "workflows" / "e2e-1.5-basic-conversion.json",
|
||||
))
|
||||
TIMEOUT_S = int(os.environ.get("TIMEOUT_S", "1800")) # 30min cap
|
||||
|
||||
|
||||
def git_sha() -> str:
|
||||
try:
|
||||
return subprocess.check_output(
|
||||
["git", "-C", str(REPO_ROOT), "rev-parse", "--short", "HEAD"],
|
||||
stderr=subprocess.DEVNULL,
|
||||
).decode().strip()
|
||||
except Exception:
|
||||
return "nogit"
|
||||
|
||||
|
||||
def http_get_json(path: str) -> dict:
|
||||
with urllib.request.urlopen(f"{COMFY_URL}{path}", timeout=60) as r:
|
||||
return json.loads(r.read().decode())
|
||||
|
||||
|
||||
def http_post_json(path: str, payload: dict) -> dict:
|
||||
data = json.dumps(payload).encode("utf-8")
|
||||
req = urllib.request.Request(
|
||||
f"{COMFY_URL}{path}", data=data,
|
||||
headers={"Content-Type": "application/json"}, method="POST",
|
||||
)
|
||||
# Long timeout: first POST blocks while the server warms model loaders +
|
||||
# the Core ML compile-on-load pass (~60-90s on a cold cache).
|
||||
with urllib.request.urlopen(req, timeout=300) as r:
|
||||
return json.loads(r.read().decode())
|
||||
|
||||
|
||||
def psnr(a: np.ndarray, b: np.ndarray) -> float:
|
||||
mse = float(np.mean((a.astype(np.float64) - b.astype(np.float64)) ** 2))
|
||||
if mse == 0:
|
||||
return 100.0
|
||||
return 20.0 * float(np.log10(255.0 / np.sqrt(mse)))
|
||||
|
||||
|
||||
def find_image(out_dir: Path, prefix: str) -> Path | None:
|
||||
matches = sorted(out_dir.glob(f"{prefix}_*.png"), reverse=True)
|
||||
return matches[0] if matches else None
|
||||
|
||||
|
||||
def main() -> int:
|
||||
sha = git_sha()
|
||||
out_root = REPO_ROOT / "bench" / "results" / "smoke" / sha
|
||||
out_root.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
workflow = json.loads(WORKFLOW_PATH.read_text())
|
||||
# Point both checkpoint nodes at the maintainer's available SD1.5 ckpt.
|
||||
for nid in ("4", "10"):
|
||||
if nid in workflow:
|
||||
workflow[nid]["inputs"]["ckpt_name"] = CKPT_NAME
|
||||
# Fixed seed for reproducibility (Phase 1 baseline).
|
||||
seed = int(os.environ.get("SEED", "42"))
|
||||
for nid in ("3", "11"):
|
||||
if nid in workflow and "seed" in workflow[nid].get("inputs", {}):
|
||||
workflow[nid]["inputs"]["seed"] = seed
|
||||
# Phase 1 only needs a Core ML image. The reference MPS branch (nodes
|
||||
# 3/8/9) trips a known MPS f16/f32 mismatch on torch 2.0.1 + macOS 26+,
|
||||
# which would crash the whole server. Strip those nodes so the queue
|
||||
# only runs the Core ML path; SKIP_MPS=0 to opt back in.
|
||||
if os.environ.get("SKIP_MPS", "1") not in ("0", "false", "False"):
|
||||
for nid in ("3", "8", "9"):
|
||||
workflow.pop(nid, None)
|
||||
|
||||
print(f"[smoke] git={sha} server={COMFY_URL} ckpt={CKPT_NAME} seed={seed}", file=sys.stderr)
|
||||
|
||||
resp = http_post_json("/prompt", {"prompt": workflow})
|
||||
print(f"[smoke] queued: {resp}", file=sys.stderr)
|
||||
|
||||
t0 = time.time()
|
||||
while True:
|
||||
if time.time() - t0 > TIMEOUT_S:
|
||||
print(f"[smoke] TIMEOUT after {TIMEOUT_S}s waiting for queue drain", file=sys.stderr)
|
||||
return 2
|
||||
try:
|
||||
q = http_get_json("/prompt")
|
||||
remaining = q.get("exec_info", {}).get("queue_remaining", -1)
|
||||
if remaining == 0:
|
||||
break
|
||||
except (urllib.error.URLError, json.JSONDecodeError) as exc:
|
||||
print(f"[smoke] poll error: {exc}", file=sys.stderr)
|
||||
time.sleep(2)
|
||||
|
||||
comfy_out = COMFY_DIR / "output"
|
||||
coreml_png = find_image(comfy_out, "E2E-1.5-CoreML")
|
||||
if coreml_png is None:
|
||||
print(f"[smoke] missing Core ML image under {comfy_out}", file=sys.stderr)
|
||||
return 3
|
||||
coreml_dst = out_root / coreml_png.name
|
||||
shutil.copy2(coreml_png, coreml_dst)
|
||||
|
||||
mps_png = find_image(comfy_out, "E2E-1.5-MPS")
|
||||
mps_dst = None
|
||||
psnr_db = None
|
||||
if mps_png is not None:
|
||||
mps_dst = out_root / mps_png.name
|
||||
shutil.copy2(mps_png, mps_dst)
|
||||
a = np.array(Image.open(coreml_dst).convert("RGB"))
|
||||
b = np.array(Image.open(mps_dst).convert("RGB"))
|
||||
if a.shape == b.shape:
|
||||
psnr_db = round(psnr(a, b), 2)
|
||||
|
||||
report = {
|
||||
"git_sha": sha,
|
||||
"seed": seed,
|
||||
"ckpt_name": CKPT_NAME,
|
||||
"workflow": str(WORKFLOW_PATH.relative_to(REPO_ROOT)),
|
||||
"coreml_image": str(coreml_dst.relative_to(REPO_ROOT)),
|
||||
"mps_image": str(mps_dst.relative_to(REPO_ROOT)) if mps_dst else None,
|
||||
"psnr_db": psnr_db,
|
||||
"wall_seconds": round(time.time() - t0, 1),
|
||||
}
|
||||
(out_root / "report.json").write_text(json.dumps(report, indent=2))
|
||||
print(json.dumps(report, indent=2))
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
+26
-2
@@ -2,8 +2,23 @@
|
||||
name = "comfyui-coremlsuite"
|
||||
description = "This extension contains a set of custom nodes for ComfyUI that allow you to use Core ML models in your ComfyUI workflows."
|
||||
version = "1.0.1"
|
||||
license = { file = "LICENSE" }
|
||||
dependencies = ["git+https://github.com/apple/ml-stable-diffusion.git", "coremltools>=7.1", "overrides", "diffusers>=0.22", "peft>=0.6.2", "omegaconf>=2.3"]
|
||||
license = "MIT"
|
||||
requires-python = ">=3.11,<3.12"
|
||||
packages = [{ include = "coreml_suite" }]
|
||||
dependencies = [
|
||||
# Phase 1 baseline pins: matches the apple_env that last produced working
|
||||
# conversions. Comfy's checkpoint-safe-loading branch (utils.py:33) is gated
|
||||
# on torch>=2.4, so newer torch + numpy 1.23 (ml-sd's pin) breaks at import.
|
||||
# Bump the whole set together in Phase 5; do not bump individually.
|
||||
"python-coreml-stable-diffusion @ git+https://github.com/apple/ml-stable-diffusion.git@e5d960c41a6a4ab200b8db379194127607b1c590",
|
||||
"torch==2.0.1",
|
||||
"coremltools==8.2",
|
||||
"numpy<1.25",
|
||||
"overrides",
|
||||
"diffusers>=0.22",
|
||||
"peft>=0.6.2",
|
||||
"omegaconf>=2.3",
|
||||
]
|
||||
|
||||
[project.urls]
|
||||
Repository = "https://github.com/aszc-dev/ComfyUI-CoreMLSuite"
|
||||
@@ -13,3 +28,12 @@ Repository = "https://github.com/aszc-dev/ComfyUI-CoreMLSuite"
|
||||
PublisherId = "aszc-dev"
|
||||
DisplayName = "ComfyUI-CoreMLSuite"
|
||||
Icon = ""
|
||||
# Pinned to the ComfyUI commit this Phase 1 baseline was validated against.
|
||||
# Bump together with the toolchain upgrade in Phase 5.
|
||||
requires-comfyui = "==ab5413351eee61f3d7f10c74e75286df0058bb18"
|
||||
|
||||
[dependency-groups]
|
||||
dev = [
|
||||
"pillow>=12.2.0",
|
||||
"psutil>=7.2.2",
|
||||
]
|
||||
|
||||
+4
-2
@@ -1,5 +1,7 @@
|
||||
git+https://github.com/apple/ml-stable-diffusion.git
|
||||
coremltools>=7.1
|
||||
git+https://github.com/apple/ml-stable-diffusion.git@e5d960c41a6a4ab200b8db379194127607b1c590
|
||||
torch==2.0.1
|
||||
coremltools==8.2
|
||||
numpy<1.25
|
||||
overrides
|
||||
diffusers>=0.22
|
||||
peft>=0.6.2
|
||||
|
||||
@@ -8,7 +8,7 @@ from coreml_suite.controlnet import chunk_control
|
||||
from coreml_suite.models import (
|
||||
CoreMLInputs,
|
||||
)
|
||||
from coreml_suite.config import get_model_config
|
||||
from coreml_suite.config import ModelVersion, get_model_config
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
@@ -26,7 +26,7 @@ def expected_inputs():
|
||||
|
||||
@pytest.fixture
|
||||
def model_config():
|
||||
return get_model_config()
|
||||
return get_model_config(ModelVersion.SD15)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("batch_size", [1, 2, 4, 5, 9])
|
||||
|
||||
Reference in New Issue
Block a user