From 6d5391ecc0e772d0f3e1fc29153ad4b66058b1a6 Mon Sep 17 00:00:00 2001 From: qnsh Date: Sun, 30 Aug 2026 14:08:15 +0800 Subject: [PATCH] update final answer --- README.md | 4 +- README_zh_CN.md | 4 +- openai/openai_text_node.py | 5 ++- utils/skill_utils.py | 71 ++++++++++++++++++++++++++++++++++-- web/js/preview_api_result.js | 3 ++ 5 files changed, 79 insertions(+), 8 deletions(-) diff --git a/README.md b/README.md index 0758d40..c3e04b0 100644 --- a/README.md +++ b/README.md @@ -33,7 +33,7 @@ The ModelScope image generation interface only requires you to fill in the corre The node supports text, image, and video inputs. Video is sent directly using the current protocol's content format; if the target API or protocol does not support video, its error is surfaced as an explicit video-unsupported message. File input is not implemented in this version. The `OpenAI Text Advanced Options` node accepts a JSON object for protocol/API-specific parameters, for example `{"temperature":0.7,"max_output_tokens":4096}`. JSON options cannot override request fields such as `model`, `messages`, `input`, `instructions`, `stream`, `api_key`, `base_url`, or `timeout`. -When `stream` is enabled, the model response is streamed to the client as it is generated using server-sent events (SSE). Streaming and `skill_options` cannot currently be enabled together. +When `stream` is enabled, the model response is streamed to the client as it is generated using server-sent events (SSE). This also works with `skill_options`. In Skill mode, text from rounds that request a tool is treated as a candidate and is discarded after the tool call; only the first round that completes without tool calls is promoted to the final answer. The final `end` event is authoritative and reconciles any difference between streamed deltas and the returned node text. Terminal states distinguish normal completion, truncation, and errors. ### skills @@ -41,6 +41,8 @@ The `OpenAI Text Skill Options` node discovers local skills organized around `SK When a Skill is selected, the model follows its instructions and reads bundled text resources as needed. Script files can be read as text but are never executed. Skill mode cannot run commands, edit or write files, or independently access the network. The target model service must support tool calls. +Skill streaming is deliberately conservative: candidate text is buffered per model round, tool-call rounds are cleared, and final text is displayed only after a no-tool round completes. This avoids showing an intermediate answer that the model later revises after reading a Skill resource. + ## Advanced usage instructions API nodes support `Config Options` and `Proxy Options`. Both can be used to override configuration file parameters by configuring parameters through the front-end node. diff --git a/README_zh_CN.md b/README_zh_CN.md index 6c202a2..f73052e 100644 --- a/README_zh_CN.md +++ b/README_zh_CN.md @@ -33,7 +33,7 @@ git clone https://github.com/ycyy/ComfyUI-YCYY-API.git 节点支持文本、图像和视频输入。视频会按当前协议的格式直接发送;如果目标 API 或协议不支持视频,接口错误会转换为明确的视频不支持提示。`files` 输入当前版本暂不支持。`OpenAI 文本高级选项(JSON)` 节点接受协议或 API 特有参数,例如 `{"temperature":0.7,"max_output_tokens":4096}`。JSON 参数不能覆盖 `model`、`messages`、`input`、`instructions`、`stream`、`api_key`、`base_url` 或 `timeout` 等请求字段。 -启用 `stream` 后,模型响应会在生成过程中通过服务器发送事件(SSE)流式传输到客户端。当前流式模式不能与 `skill_options` 同时启用。 +启用 `stream` 后,模型响应会通过服务器发送事件(SSE)在生成过程中流式传输到客户端,也支持与 `skill_options` 同时使用。在 Skill 模式中,包含工具调用的轮次产生的文本只作为候选内容;工具调用完成后会丢弃该轮候选文本,只有“不再调用工具”的轮次才会提升为最终答案。最终的 `end` 事件具有权威性,并会校准流式增量与节点最终输出之间的差异。终止状态会区分正常完成、输出截断和错误。 ### skills @@ -41,6 +41,8 @@ git clone https://github.com/ycyy/ComfyUI-YCYY-API.git 运行时,模型会遵循所选 Skill 的说明,并按需读取其中的文本资源。脚本文件只能作为文本读取,不会被执行。Skill 模式不能运行命令、编辑或写入文件,也不能自行访问网络。目标模型服务需要支持工具调用。 +Skill 流式输出采用稳妥策略:按模型轮次缓冲候选文本;发生工具调用的轮次会清空候选内容;只有无工具调用且正常结束的轮次才会显示为最终文本。这样可以避免模型读取 Skill 资源后修改答案时,前端先展示错误的中间答案。 + ### proxy `proxy` 支持配置http代理,适用于特殊网络环境 diff --git a/openai/openai_text_node.py b/openai/openai_text_node.py index 976f0c2..477516e 100644 --- a/openai/openai_text_node.py +++ b/openai/openai_text_node.py @@ -110,13 +110,14 @@ class _TextStreamSink: "round_end", round=int(round_index), has_tool_calls=bool(has_tool_calls), + text_status="candidate" if has_tool_calls else "final", ) def end(self, value): - self._send("end", text=value) + self._send("end", text=value, stop_reason="stop", text_status="final") def error(self, exc): - self._send("error", message=_safe_stream_error(exc)) + self._send("error", message=_safe_stream_error(exc), stop_reason="error", text_status="error") @PromptServer.instance.routes.get("/ycyy/openai/apis/all") diff --git a/utils/skill_utils.py b/utils/skill_utils.py index bf02cf5..5cc9c92 100644 --- a/utils/skill_utils.py +++ b/utils/skill_utils.py @@ -780,6 +780,11 @@ class NormalizedResponse: native_items: list[dict] tool_calls: list[dict] final_text: str | None + # Provider-neutral terminal state. ``tool_use`` is inferred when a + # response contains tool calls; ``stop`` means a normal final answer. + stop_reason: str | None = None + terminal: bool = True + incomplete_reason: str | None = None class ProviderAdapter: @@ -876,10 +881,10 @@ class ResponsesProviderAdapter(ProviderAdapter): def consume(event): nonlocal completed_response, completed event_type = event.get("type") - if event_type in {"error", "response.failed", "response.incomplete"}: + if event_type in {"error", "response.failed"}: detail = event.get("error") or event.get("response") or event raise ValueError(f"Streaming API failed: {detail}") - if event_type == "response.completed": + if event_type in {"response.completed", "response.incomplete"}: completed_response = event.get("response") completed = True return @@ -988,8 +993,25 @@ class ResponsesProviderAdapter(ProviderAdapter): and block.get("text") ): chunks.append(block["text"]) + status = data.get("status") + incomplete_reason = None + if isinstance(data.get("incomplete_details"), dict): + incomplete_reason = data["incomplete_details"].get("reason") + if status == "incomplete": + stop_reason = "length" if incomplete_reason == "max_output_tokens" else "incomplete" + elif status in {"failed", "cancelled"}: + stop_reason = "error" + elif calls: + stop_reason = "tool_use" + elif status == "completed": + stop_reason = "stop" + else: + stop_reason = status return NormalizedResponse( - output, calls, "\n".join(chunks) if chunks else None + output, calls, "\n".join(chunks) if chunks else None, + stop_reason=stop_reason, + terminal=status in {"completed", "incomplete", "failed", "cancelled"}, + incomplete_reason=incomplete_reason, ) def append_native_items(self, history, normalized): @@ -1135,7 +1157,22 @@ class CompletionsProviderAdapter(ProviderAdapter): }) content = message.get("content") final_text = content if isinstance(content, str) and content.strip() else None - return NormalizedResponse([deepcopy(message)], calls, final_text) + finish_reason = choices[0].get("finish_reason") if isinstance(choices[0], dict) else None + if finish_reason == "length": + stop_reason = "length" + elif finish_reason in {"content_filter", "error"}: + stop_reason = "error" + elif calls or finish_reason in {"tool_calls", "function_call"}: + stop_reason = "tool_use" + elif finish_reason in {None, "stop"}: + stop_reason = "stop" + else: + stop_reason = str(finish_reason) + return NormalizedResponse( + [deepcopy(message)], calls, final_text, + stop_reason=stop_reason, + terminal=finish_reason is not None or bool(final_text or calls), + ) def append_native_items(self, history, normalized): history.extend(deepcopy(normalized.native_items)) @@ -1237,12 +1274,38 @@ class PiSkillAgentLoop: final_text_present=normalized.final_text is not None, ) + # Never execute a tool call whose arguments/response were cut off + # or failed at the provider level. This mirrors pi's safeguard + # against running truncated tool arguments. + if normalized.stop_reason == "length" and normalized.tool_calls: + raise SkillExecutionError( + "pi_skill_response_truncated", + "Provider response was truncated before Skill tool arguments completed", + ) + if normalized.stop_reason in {"error", "incomplete"} and normalized.tool_calls: + detail = normalized.incomplete_reason or normalized.stop_reason + raise SkillExecutionError( + "pi_skill_response_incomplete", + f"Provider response did not complete: {detail}", + ) + if not normalized.tool_calls: if not loaded: raise SkillExecutionError( "skill_not_loaded", "The model returned final output before calling load_skill", ) + if normalized.stop_reason == "length": + raise SkillExecutionError( + "pi_skill_response_truncated", + "Provider response was truncated before a complete final answer", + ) + if normalized.stop_reason in {"error", "incomplete"}: + detail = normalized.incomplete_reason or normalized.stop_reason + raise SkillExecutionError( + "pi_skill_response_incomplete", + f"Provider response did not complete: {detail}", + ) if normalized.final_text is None: raise SkillExecutionError( "pi_skill_response_invalid", diff --git a/web/js/preview_api_result.js b/web/js/preview_api_result.js index 9ab6430..8f983a2 100644 --- a/web/js/preview_api_result.js +++ b/web/js/preview_api_result.js @@ -759,6 +759,9 @@ function receiveStreamEvent(event) { setStatus(state, "displaying"); } enqueueFinalText(state, data.text ?? state.receivedText); + if (data.stop_reason && data.stop_reason !== "stop") { + state.terminalError = `stop_reason: ${data.stop_reason}`; + } state.candidateText = ""; state.currentActivity = null; renderActivity(state);