diff --git a/.github/workflows/docs.yml b/.github/workflows/docs.yml new file mode 100644 index 0000000..3abdfa5 --- /dev/null +++ b/.github/workflows/docs.yml @@ -0,0 +1,33 @@ +name: Deploy docs + +on: + push: + branches: [main] + workflow_dispatch: + +permissions: + contents: write + +jobs: + deploy: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + with: + fetch-depth: 0 + + - name: Install uv + uses: astral-sh/setup-uv@v5 + + - name: Install dependencies + run: uv sync --extra docs + + - name: Configure git for gh-pages push + run: | + git config user.name "github-actions[bot]" + git config user.email "github-actions[bot]@users.noreply.github.com" + + - name: Deploy docs with mike + run: | + VERSION=$(uv run python -c "import tomllib; d=tomllib.load(open('pyproject.toml','rb')); print(d['project']['version'])") + uv run mike deploy --push --update-aliases "$VERSION" stable diff --git a/README.md b/README.md index d77308f..a5e43d8 100644 --- a/README.md +++ b/README.md @@ -15,7 +15,7 @@ A collection of workflow efficiency and quality-of-life nodes built out of neces | **Ollama Model Selector** | Fetches the live model list from Ollama and presents it as a dropdown. Outputs the selected model name. | | **Ollama Load Model** | Loads a model into Ollama's memory using `/api/generate` with `keep_alive=-1`. | | **Ollama Unload Model** | Evicts a model from Ollama's memory using `/api/generate` with `keep_alive=0`. | -| **Ollama Chat Completion** | Sends a prompt (and optional conversation history) to Ollama `/api/chat` and returns the response text plus the updated history. | +| **Ollama Chat Completion** | Sends a prompt (and optional conversation history) to Ollama `/api/chat`. Response and history are shown inline in the node body and available as output sockets. | | **Ollama Option — \*** | Seven composable option nodes (Temperature, Seed, Max Tokens, Top P, Top K, Repeat Penalty, Extra Body) that merge into an `OLLAMA_OPTIONS` dict wired into Chat Completion. | | **Ollama Debug History** | Serialises an `OLLAMA_HISTORY` list to a pretty-printed JSON string for inspection. | | **Ollama History Length** | Returns the number of messages in an `OLLAMA_HISTORY` list as an integer. | @@ -117,15 +117,15 @@ On memory-constrained machines and single-GPU setups, explicitly loading and unl The correct chain is **Load → Chat → Unload**, enforced through data dependencies: -1. Wire `OllamaLoadModel.model_name` → `OllamaChatCompletion.model_name` (optional input). This guarantees Load runs before Chat and overrides the Chat dropdown with the same model. -2. Wire `OllamaChatCompletion.model_name` → `OllamaUnloadModel.model`. This guarantees Unload runs after Chat. +1. Wire `OllamaLoadModel.model_name` → `OllamaChatCompletion.model`. This creates the data dependency that guarantees Load runs before Chat and passes the model name into the Chat node's plain-string `model` input. +2. Wire `OllamaChatCompletion.model_name` → `OllamaUnloadModel.model`. This guarantees Unload runs after Chat completes. 3. Optionally wire `OllamaChatCompletion.response` → `OllamaUnloadModel.passthrough` — Unload returns the response unchanged so the rest of your workflow can still consume it. ### Minimal chat workflow 1. **Ollama Client** → set host (default `http://localhost:11434`) -2. **Ollama Model Selector** → pick a model from the live dropdown -3. **Ollama Chat Completion** → wire client + model + prompt → response string +2. **Ollama Model Selector** → pick a model from the live dropdown (or type/wire a model name directly into Chat Completion's `model` input) +3. **Ollama Chat Completion** → wire client + model + prompt; the response appears inline in the node body and is also available as an output socket ![Ollama Chat Completion](docs/assets/ollama_chat.png) diff --git a/docker-compose.dev.yml b/docker-compose.dev.yml index 692f80c..c9098bd 100644 --- a/docker-compose.dev.yml +++ b/docker-compose.dev.yml @@ -9,6 +9,10 @@ services: # On macOS/Windows Docker Desktop: host.docker.internal is built-in. extra_hosts: - "host.docker.internal:host-gateway" + environment: + # Tells comfydv where to reach Ollama on the host so model dropdowns are + # pre-populated at server start without needing a manual Refresh click. + - OLLAMA_HOST=http://host.docker.internal:11434 command: - sh - -c diff --git a/docs/assets/ollama_chat.png b/docs/assets/ollama_chat.png index 72ae97c..de55783 100644 Binary files a/docs/assets/ollama_chat.png and b/docs/assets/ollama_chat.png differ diff --git a/docs/assets/ollama_lifecycle.png b/docs/assets/ollama_lifecycle.png index 9579107..a8d8c97 100644 Binary files a/docs/assets/ollama_lifecycle.png and b/docs/assets/ollama_lifecycle.png differ diff --git a/docs/assets/ollama_workflow.png b/docs/assets/ollama_workflow.png index de17548..298f828 100644 Binary files a/docs/assets/ollama_workflow.png and b/docs/assets/ollama_workflow.png differ diff --git a/docs/index.md b/docs/index.md index 18f5da5..ee5c3c8 100644 --- a/docs/index.md +++ b/docs/index.md @@ -11,7 +11,7 @@ A collection of workflow efficiency and quality-of-life nodes built out of neces | **Ollama Model Selector** | Fetches the live model list from Ollama and presents it as a dropdown. Outputs the selected model name. | | **Ollama Load Model** | Loads a model into Ollama's memory using `/api/generate` with `keep_alive=-1`. | | **Ollama Unload Model** | Evicts a model from Ollama's memory using `/api/generate` with `keep_alive=0`. | -| **Ollama Chat Completion** | Sends a prompt (and optional conversation history) to Ollama `/api/chat` and returns the response text plus the updated history. | +| **Ollama Chat Completion** | Sends a prompt (and optional conversation history) to Ollama `/api/chat`. Response and history are shown inline in the node body and available as output sockets. | | **Ollama Option — \*** | Seven composable option nodes (Temperature, Seed, Max Tokens, Top P, Top K, Repeat Penalty, Extra Body) that merge into an `OLLAMA_OPTIONS` dict wired into Chat Completion. | | **Ollama Debug History** | Serialises an `OLLAMA_HISTORY` list to a pretty-printed JSON string for inspection. | | **Ollama History Length** | Returns the number of messages in an `OLLAMA_HISTORY` list as an integer. | @@ -101,15 +101,15 @@ On memory-constrained machines and single-GPU setups, explicitly loading and unl The correct chain is **Load → Chat → Unload**, enforced through data dependencies: -1. Wire `OllamaLoadModel.model_name` → `OllamaChatCompletion.model_name` (optional input). This guarantees Load runs before Chat and overrides the Chat dropdown with the same model. -2. Wire `OllamaChatCompletion.model_name` → `OllamaUnloadModel.model`. This guarantees Unload runs after Chat. +1. Wire `OllamaLoadModel.model_name` → `OllamaChatCompletion.model`. This creates the data dependency that guarantees Load runs before Chat and passes the model name into the Chat node's plain-string `model` input. +2. Wire `OllamaChatCompletion.model_name` → `OllamaUnloadModel.model`. This guarantees Unload runs after Chat completes. 3. Optionally wire `OllamaChatCompletion.response` → `OllamaUnloadModel.passthrough` — Unload returns the response unchanged so the rest of your workflow can still consume it. ### Minimal chat workflow 1. **Ollama Client** → set host (default `http://localhost:11434`) -2. **Ollama Model Selector** → pick a model from the live dropdown -3. **Ollama Chat Completion** → wire client + model + prompt → response string +2. **Ollama Model Selector** → pick a model from the live dropdown (or type/wire a model name directly into Chat Completion's `model` input) +3. **Ollama Chat Completion** → wire client + model + prompt; the response appears inline in the node body and is also available as an output socket ![Ollama Chat Completion](assets/ollama_chat.png) diff --git a/src/comfydv/ollama.py b/src/comfydv/ollama.py index d9cdb4e..a8f97a6 100644 --- a/src/comfydv/ollama.py +++ b/src/comfydv/ollama.py @@ -11,6 +11,7 @@ ADR-005: OllamaClient node is the single source of the host URL. import asyncio import json import logging +import os import sys logger = logging.getLogger(__name__) @@ -82,10 +83,36 @@ async def _post_json(url: str, payload: dict, *, timeout: float = 120.0) -> dict raise RuntimeError(f"Cannot reach Ollama at {url}: {exc}") from exc -# Static placeholder — avoids a network call at import time (which always fails -# in CI and slows cold-start). The JS refresh button calls /dv/ollama/models -# at runtime to populate the live list. -_DEFAULT_MODELS: list[str] = ["(⟳ click Refresh models)"] +def _load_default_models() -> list[str]: + """Fetch the installed model list at server start-up. + + Tries OLLAMA_HOST env var first (set in docker-compose for host.docker.internal), + then falls back to localhost. Returns a one-element placeholder list only when + Ollama is genuinely unreachable so that COMBO validation doesn't reject saved + workflow values. + """ + candidates = [] + env_host = os.environ.get("OLLAMA_HOST", "").strip() + if env_host: + candidates.append(env_host) + candidates.append("http://host.docker.internal:11434") + candidates.append("http://localhost:11434") + + for host in candidates: + models = _run_async(_fetch_models(host)) + if models: + logger.info("Ollama models loaded from %s: %s", host, models) + return models + + logger.warning( + "Could not reach Ollama at any candidate host %s — " + "model dropdowns will be empty until a Refresh is clicked.", + candidates, + ) + return ["(start Ollama — click ⟳ Refresh)"] + + +_DEFAULT_MODELS: list[str] = _load_default_models() # --------------------------------------------------------------------------- @@ -235,22 +262,35 @@ class OllamaUnloadModel: # --------------------------------------------------------------------------- +def _history_preview(messages: list[dict]) -> str: + """Compact multi-turn summary shown below the response in the node body.""" + lines = [] + for m in messages[-6:]: + prefix = "▶" if m["role"] == "user" else "·" + snippet = m["content"][:100] + if len(m["content"]) > 100: + snippet += "…" + lines.append(f"{prefix} {snippet}") + return "\n".join(lines) + + class OllamaChatCompletion: + OUTPUT_NODE = True + @classmethod def INPUT_TYPES(s): return { "required": { "client": ("OLLAMA_CLIENT",), - "model": (_DEFAULT_MODELS, {}), + # Plain STRING so it can receive a wired value from OllamaLoadModel + # (or OllamaModelSelector) without needing a separate model_name socket. + "model": ("STRING", {"default": ""}), "prompt": ("STRING", {"multiline": True, "default": ""}), }, "optional": { "system": ("STRING", {"multiline": True, "default": ""}), "history": ("OLLAMA_HISTORY",), "options": ("OLLAMA_OPTIONS",), - # Wire OllamaLoadModel.model_name here to guarantee load runs - # before chat and to override the dropdown with the wired value. - "model_name": ("STRING", {"forceInput": True}), "timeout_secs": ("INT", {"default": 300, "min": 30, "max": 3600}), }, } @@ -268,12 +308,11 @@ class OllamaChatCompletion: system="", history=None, options=None, - model_name=None, timeout_secs=300, ): - effective_model = ( - model_name.strip() if model_name and model_name.strip() else model - ) + effective_model = model.strip() + if not effective_model: + raise ValueError("model cannot be empty — type a model name or wire one in") if history is None: history = [] messages = list(history) @@ -294,7 +333,16 @@ class OllamaChatCompletion: updated = list(history) updated.append({"role": "user", "content": prompt}) updated.append({"role": "assistant", "content": response_text}) - return (response_text, updated, effective_model) + n = len(updated) + ui_text = ( + f"{response_text}\n\n── History: {n} message(s) ──\n{_history_preview(updated)}" + if n > 2 + else response_text + ) + return { + "ui": {"text": [ui_text]}, + "result": (response_text, updated, effective_model), + } # --------------------------------------------------------------------------- diff --git a/src/js/ollama.js b/src/js/ollama.js index c840535..bfe1248 100644 --- a/src/js/ollama.js +++ b/src/js/ollama.js @@ -1,24 +1,33 @@ /** * ollama.js — ComfyUI frontend extension for comfydv Ollama nodes. * - * Populates model COMBO widgets on OllamaModelSelector, OllamaLoadModel, and - * OllamaChatCompletion from a live call to GET /dv/ollama/models?host=. + * Populates the model widget on Ollama nodes from a live call to + * GET /dv/ollama/models?host=. + * + * OllamaModelSelector and OllamaLoadModel use a COMBO widget (dropdown). + * OllamaChatCompletion uses a plain STRING widget (accepts wired values). + * The Refresh button works the same way for both: it fetches the live list + * and sets the widget value / updates COMBO options as appropriate. */ import { app } from "../../scripts/app.js"; -const OLLAMA_COMBO_NODES = new Set([ - "OllamaModelSelector", - "OllamaLoadModel", - "OllamaChatCompletion", -]); +/** Nodes whose model widget is a COMBO dropdown. */ +const OLLAMA_COMBO_NODES = new Set(["OllamaModelSelector", "OllamaLoadModel"]); + +/** Nodes whose model widget is a plain STRING (accepts wired input). */ +const OLLAMA_STRING_MODEL_NODES = new Set(["OllamaChatCompletion"]); + +const OLLAMA_ALL_NODES = new Set([...OLLAMA_COMBO_NODES, ...OLLAMA_STRING_MODEL_NODES]); /** - * Fetch model list from the backend and repopulate the COMBO widget. - * @param {LGraphNode} node - * @param {string} host - Ollama host URL, e.g. "http://localhost:11434" + * Fetch model list and update the node's model widget. + * Works for both COMBO and STRING widgets: + * - COMBO: updates options.values + preserves selection if model still exists. + * - STRING: sets the value to the first model; keeps existing value if it + * still appears in the live list (user may have typed a valid name). */ -async function refreshModelDropdown(node, host) { +async function refreshModelWidget(node, host) { try { const resp = await fetch(`/dv/ollama/models?host=${encodeURIComponent(host)}`); if (!resp.ok) return; @@ -29,12 +38,22 @@ async function refreshModelDropdown(node, host) { const modelWidget = node.widgets?.find(w => w.name === "model"); if (!modelWidget) return; - const current = modelWidget.value; - modelWidget.options.values = models; - modelWidget.value = models.includes(current) ? current : models[0]; + const current = (modelWidget.value ?? "").trim(); + + if (Array.isArray(modelWidget.options?.values)) { + // COMBO widget + modelWidget.options.values = models; + modelWidget.value = models.includes(current) ? current : models[0]; + } else { + // STRING widget — keep current value if it is a known model + if (!current || !models.includes(current)) { + modelWidget.value = models[0]; + } + } + node.setDirtyCanvas(true, false); } catch (_) { - // Ollama unreachable — leave COMBO with server-side defaults + // Ollama unreachable — leave widget unchanged } } @@ -43,8 +62,7 @@ async function refreshModelDropdown(node, host) { * * First checks the node's own widgets (OllamaClient has a "host" widget). * Otherwise traverses graph links to find a connected OllamaClient node and - * reads its "host" widget — this is the common case for downstream nodes - * (ModelSelector, LoadModel, ChatCompletion) that receive the client socket. + * reads its "host" widget — this is the common case for downstream nodes. */ function getHostFromNode(node) { const ownHostWidget = node.widgets?.find(w => w.name === "host"); @@ -68,21 +86,21 @@ app.registerExtension({ name: "comfydv.ollama", async beforeRegisterNodeDef(nodeType, nodeData) { - if (!OLLAMA_COMBO_NODES.has(nodeData.name)) return; + if (!OLLAMA_ALL_NODES.has(nodeData.name)) return; const onNodeCreated = nodeType.prototype.onNodeCreated; nodeType.prototype.onNodeCreated = function () { const result = onNodeCreated?.apply(this, arguments); - // Add a refresh button below the model widget + // Add a Refresh button below the model widget this.addWidget("button", "⟳ Refresh models", null, () => { const host = getHostFromNode(this); - refreshModelDropdown(this, host); + refreshModelWidget(this, host); }); - // Initial population + // Initial population on node creation const host = getHostFromNode(this); - refreshModelDropdown(this, host); + refreshModelWidget(this, host); return result; }; diff --git a/tests/conftest.py b/tests/conftest.py index 117949b..0b57861 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -89,6 +89,25 @@ def skip_if_no_ollama(ollama_available): ) +@pytest.fixture(scope="session") +def first_generative_model(ollama_host, ollama_available): + """Return the first model available from Ollama, skipping embedding-only models. + + Used by lifecycle tests that call /api/generate — embedding models like + embeddinggemma reject that endpoint with HTTP 400. + """ + if not ollama_available: + pytest.skip("Ollama not reachable at localhost:11434") + import asyncio + + from comfydv.ollama import _fetch_models + + models = asyncio.run(_fetch_models(ollama_host)) + if not models: + pytest.skip("No models installed in Ollama") + return models[0] + + # --------------------------------------------------------------------------- # Existing ComfyUI node fixtures # --------------------------------------------------------------------------- diff --git a/tests/test_ollama.py b/tests/test_ollama.py index 50a8bc1..2372692 100644 --- a/tests/test_ollama.py +++ b/tests/test_ollama.py @@ -374,12 +374,14 @@ class TestUS3ModelLifecycle: ) @pytest.mark.integration - def test_load_model_returns_name(self, ollama_host, skip_if_no_ollama): + def test_load_model_returns_name( + self, ollama_host, skip_if_no_ollama, first_generative_model + ): """Scenario: Load Model loads model into Ollama memory.""" (result,) = OllamaLoadModel().load_model( - client=ollama_host, model="embeddinggemma:latest" + client=ollama_host, model=first_generative_model ) - assert result == "embeddinggemma:latest" + assert result == first_generative_model def test_unload_returns_two_values(self): """OllamaUnloadModel returns (model_name, passthrough) tuple.""" @@ -396,15 +398,15 @@ class TestUS3ModelLifecycle: @pytest.mark.integration def test_unload_model_returns_name_and_passthrough( - self, ollama_host, skip_if_no_ollama + self, ollama_host, skip_if_no_ollama, first_generative_model ): """Scenario: Unload evicts model; passthrough flows through unchanged.""" model_name, passthrough = OllamaUnloadModel().unload_model( client=ollama_host, - model="embeddinggemma:latest", + model=first_generative_model, passthrough="sentinel", ) - assert model_name == "embeddinggemma:latest" + assert model_name == first_generative_model assert passthrough == "sentinel" @@ -416,22 +418,48 @@ _CHAT_MODEL = "lukey03/qwen3.5-9b-abliterated-vision:latest" class TestUS4ChatCompletion: - def test_chat_completion_input_types_uses_combo(self): - """Scenario: Chat Completion shows live dropdown (fixes Issue #1).""" + def test_chat_completion_model_is_plain_string(self): + """model input must be STRING (not COMBO) so it can receive wired values. + + COMBO inputs cannot accept wired connections. Using STRING lets users + wire OllamaLoadModel.model_name → OllamaChatCompletion.model directly, + removing the need for a separate model_name socket. + """ input_types = OllamaChatCompletion.INPUT_TYPES() model_input = input_types["required"]["model"] - assert isinstance(model_input[0], list), ( - "OllamaChatCompletion model input must be a COMBO (list) — Issue #1 fix" + assert model_input[0] == "STRING", ( + f"OllamaChatCompletion model input must be STRING so it can be wired, " + f"got {model_input[0]!r}" ) - def test_chat_completion_accepts_wired_model_name(self): - """model_name optional input lets OllamaLoadModel.model_name wire in for ordering.""" + def test_chat_has_no_model_name_input(self): + """model_name optional input must be removed — model STRING accepts wired values.""" inputs = OllamaChatCompletion.INPUT_TYPES() - assert "model_name" in inputs.get("optional", {}), ( - "model_name must be optional so LoadModel can wire its output here " - "to guarantee Load → Chat execution order" + assert "model_name" not in inputs.get("optional", {}), ( + "model_name optional input is redundant now that model is a plain STRING " + "that accepts wired connections" ) + def test_chat_model_receives_wired_string(self, monkeypatch): + """Wiring OllamaLoadModel.model_name → OllamaChatCompletion.model works.""" + captured = {} + + async def fake_post(url, payload, *, timeout=120.0): + captured["model"] = payload.get("model") + return {"message": {"content": "ok"}} + + import comfydv.ollama as ollama_mod + + monkeypatch.setattr(ollama_mod, "_post_json", fake_post) + + _, _, used_model = OllamaChatCompletion().chat( + client="http://localhost:11434", + model="llama3:latest", + prompt="hi", + )["result"] + assert used_model == "llama3:latest" + assert captured.get("model") == "llama3:latest" + def test_chat_completion_returns_model_name(self): """Third return value carries the effective model name for downstream unload.""" assert OllamaChatCompletion.RETURN_TYPES == ( @@ -445,26 +473,55 @@ class TestUS4ChatCompletion: "model_name", ) - def test_wired_model_name_overrides_combo(self, monkeypatch): - """model_name kwarg takes precedence over the COMBO widget value.""" - captured = {} + def test_chat_is_output_node(self): + """OllamaChatCompletion must have OUTPUT_NODE=True for inline display.""" + assert getattr(OllamaChatCompletion, "OUTPUT_NODE", False) is True + + def test_chat_returns_ui_result_dict(self, monkeypatch): + """chat() must return {'ui': ..., 'result': ...} not a bare tuple.""" async def fake_post(url, payload, *, timeout=120.0): - captured["model"] = payload.get("model") - return {"message": {"content": "ok"}} + return {"message": {"content": "hello"}} import comfydv.ollama as ollama_mod monkeypatch.setattr(ollama_mod, "_post_json", fake_post) - response, _, effective = OllamaChatCompletion().chat( - client="http://localhost:11434", - model="dropdown-value", - prompt="hi", - model_name="wired-value", - ) - assert effective == "wired-value" - assert captured.get("model") == "wired-value" + ret = OllamaChatCompletion().chat(client="http://x", model="m", prompt="hi") + assert isinstance(ret, dict), f"Expected dict, got {type(ret)}" + assert "ui" in ret, "Missing 'ui' key" + assert "result" in ret, "Missing 'result' key" + + def test_chat_ui_contains_response_text(self, monkeypatch): + """Response text must appear in ui['text'] for the inline display.""" + + async def fake_post(url, payload, *, timeout=120.0): + return {"message": {"content": "hello world"}} + + import comfydv.ollama as ollama_mod + + monkeypatch.setattr(ollama_mod, "_post_json", fake_post) + + ret = OllamaChatCompletion().chat(client="http://x", model="m", prompt="hi") + assert "hello world" in ret["ui"]["text"][0] + + def test_chat_result_is_3_tuple(self, monkeypatch): + """result key must be the 3-tuple (response, history, model_name).""" + + async def fake_post(url, payload, *, timeout=120.0): + return {"message": {"content": "hello"}} + + import comfydv.ollama as ollama_mod + + monkeypatch.setattr(ollama_mod, "_post_json", fake_post) + + ret = OllamaChatCompletion().chat(client="http://x", model="m", prompt="hi") + assert isinstance(ret["result"], tuple) + assert len(ret["result"]) == 3 + response, history, model_name = ret["result"] + assert response == "hello" + assert isinstance(history, list) + assert model_name == "m" # ---- Issue 6: Chat timeout widget ----------------------------------------- @@ -520,7 +577,7 @@ class TestUS4ChatCompletion: model=_CHAT_MODEL, prompt="Say exactly the word: pong", history=[], - ) + )["result"] assert isinstance(response, str) assert len(response) > 0 assert len(updated_history) == 2 @@ -530,19 +587,26 @@ class TestUS4ChatCompletion: @pytest.mark.integration def test_multi_turn_receives_context(self, ollama_host, skip_if_no_ollama): - """Scenario: Multi-turn completion receives full conversation context.""" + """Scenario: Multi-turn completion receives full conversation context. + + Passes think=False to prevent Qwen3-family models from returning all + output as thinking tokens with an empty content field. + """ + no_think = {"think": False} _, history, _ = OllamaChatCompletion().chat( client=ollama_host, model=_CHAT_MODEL, prompt="My name is Alice. Remember it.", history=[], - ) + options=no_think, + )["result"] response, updated, _ = OllamaChatCompletion().chat( client=ollama_host, model=_CHAT_MODEL, prompt="What is my name?", history=history, - ) + options=no_think, + )["result"] assert "Alice" in response assert len(updated) == 4 @@ -551,11 +615,11 @@ class TestUS4ChatCompletion: """History list grows by 2 entries per turn.""" _, h1, _ = OllamaChatCompletion().chat( client=ollama_host, model=_CHAT_MODEL, prompt="Turn 1", history=[] - ) + )["result"] assert len(h1) == 2 _, h2, _ = OllamaChatCompletion().chat( client=ollama_host, model=_CHAT_MODEL, prompt="Turn 2", history=h1 - ) + )["result"] assert len(h2) == 4 @@ -630,8 +694,8 @@ class TestUS5ComposableOptions: history=[], options=opts2, ) - r1, _, _model = OllamaChatCompletion().chat(**kwargs) - r2, _, _model = OllamaChatCompletion().chat(**kwargs) + r1, _, _model = OllamaChatCompletion().chat(**kwargs)["result"] + r2, _, _model = OllamaChatCompletion().chat(**kwargs)["result"] assert r1 == r2