diff --git a/specs/009-vlm-image-input/tasks.md b/specs/009-vlm-image-input/tasks.md index a7f5ea0..6d2286c 100644 --- a/specs/009-vlm-image-input/tasks.md +++ b/specs/009-vlm-image-input/tasks.md @@ -33,8 +33,8 @@ handling lives only in the `comfy`-guarded `ollama.py`. **Purpose**: the image carrier every path depends on. **⚠️ No user story work can begin until this is complete.** -- [ ] T002-T Write FAILING test: `Message.images` defaults to `None`, round-trips a base64 list, and a text-only message's transport dump **omits** the `images` key (byte-identical to today), in `tests/test_llm_provider.py` (contract T1) -- [ ] T002-I Add `images: list[str] | None = None` to `Message` in `src/comfydv/_llm/provider.py` — makes T002-T pass +- [x] T002-T Write FAILING test: `Message.images` defaults to `None`, round-trips a base64 list, and a text-only message's transport dump **omits** the `images` key (byte-identical to today), in `tests/test_llm_provider.py` (contract T1) +- [x] T002-I Add `images: list[str] | None = None` to `Message` in `src/comfydv/_llm/provider.py` — makes T002-T pass **Checkpoint**: carrier ready — user stories can begin. diff --git a/src/comfydv/_llm/provider.py b/src/comfydv/_llm/provider.py index 353e594..25f65e5 100644 --- a/src/comfydv/_llm/provider.py +++ b/src/comfydv/_llm/provider.py @@ -39,10 +39,21 @@ class ModelInfo(BaseModel): class Message(BaseModel): - """One turn in a chat request.""" + """One turn in a chat request. + + ``images`` carries optional base64-encoded image payloads (no ``data:`` + prefix) associated with this turn, for vision-capable models. ``None`` + (the default) means a text-only turn that serializes byte-for-byte as + before — providers dump with ``exclude_none=True`` so no ``images`` key + reaches the wire for image-less turns. Each provider translates this + neutral carrier into its own native shape (ADR-008): Ollama's flat + per-message ``images`` array, llama.cpp's OpenAI ``image_url`` content + parts, and pydantic-ai ``BinaryContent`` on the structured path. + """ role: Literal["system", "user", "assistant"] content: str + images: list[str] | None = None class LLMProvider(Protocol):