Files
darth-veitcher-comfydv/src/comfydv/__init__.py
T
James VeitchandClaude Sonnet 5 1f138e5991 feat(llm): detect refusal/deflection responses and retry with a new seed (#36)
Some models (observed with an abliterated Qwen variant) answer a request
they judge "politically sensitive" with hedging refusal language instead of
erroring or returning blank — neither of which the existing blank-response
or schema-validation retry triggers catch, so the refusal just passed
through as a normal result.

Adds a hybrid detector to _llm/retry.py: a free lexical/regex pass catches
blatant refusals ("I cannot generate...") without any network call; a
response that's short and/or hedge-y enough to be ambiguous additionally
gets an embedding-similarity check against canonical refusal exemplars, via
a new optional LLMProvider.embed() (Ollama's /api/embed, llama-server's
/v1/embeddings) — deliberately model-agnostic to the backend, since this is
a model-behavior concern, not a backend one. A detected refusal is treated
exactly like a blank response or validation failure: retried with a bumped
seed (existing next_seed()/RETRY_BACKOFF_SECS machinery), never returned to
the caller.

Wired into all four retry loops: OllamaProvider.chat()/chat_structured(),
LlamaCppProvider.chat(), and the shared _llm/chat.py chat_structured() used
by LlamaCppProvider.chat_structured(). New OllamaOptionRefusalRetry node
(same composable OLLAMA_OPTIONS chain as OllamaOptionDisableThinking)
configures it: enabled toggle, optional embedding_model (blank = lexical-
only), similarity threshold.


Claude-Session: https://claude.ai/code/session_01YArD9ZjBWKsAvazmS48amA

Co-authored-by: Claude Sonnet 5 <noreply@anthropic.com>
2026-07-29 17:11:08 +01:00

91 lines
3.4 KiB
Python

import logging
from .circuit_breaker import CircuitBreaker
from .format_string import FormatString
from .llamacpp import LlamaCppClient
from .ollama import (
ChatCompletion,
LLMLoadModel,
LLMModelSelector,
LLMUnloadModel,
OllamaClient,
OllamaDebugHistory,
OllamaHeaderBasicAuth,
OllamaHeaderBearerToken,
OllamaHeaderCustom,
OllamaHistoryLength,
OllamaOptionDisableThinking,
OllamaOptionExtraBody,
OllamaOptionMaxTokens,
OllamaOptionRefusalRetry,
OllamaOptionRepeatPenalty,
OllamaOptionSeed,
OllamaOptionTemperature,
OllamaOptionTopK,
OllamaOptionTopP,
)
from .random_choice import RandomChoice
logging.getLogger(__name__).addHandler(logging.NullHandler())
# A dictionary that contains all nodes you want to export with their names
# NOTE: names should be globally unique
NODE_CLASS_MAPPINGS = {
"RandomChoice": RandomChoice,
"CircuitBreaker": CircuitBreaker,
"FormatString": FormatString,
# LLM nodes (generic, ADR-007) — see comfydv.ollama.MIGRATION_MAP for
# the pre-cutover Ollama-specific names these replace
"OllamaClient": OllamaClient,
"LlamaCppClient": LlamaCppClient,
"LLMModelSelector": LLMModelSelector,
"LLMLoadModel": LLMLoadModel,
"LLMUnloadModel": LLMUnloadModel,
"ChatCompletion": ChatCompletion,
"OllamaOptionTemperature": OllamaOptionTemperature,
"OllamaOptionSeed": OllamaOptionSeed,
"OllamaOptionMaxTokens": OllamaOptionMaxTokens,
"OllamaOptionTopP": OllamaOptionTopP,
"OllamaOptionTopK": OllamaOptionTopK,
"OllamaOptionRepeatPenalty": OllamaOptionRepeatPenalty,
"OllamaOptionDisableThinking": OllamaOptionDisableThinking,
"OllamaOptionRefusalRetry": OllamaOptionRefusalRetry,
"OllamaOptionExtraBody": OllamaOptionExtraBody,
"OllamaDebugHistory": OllamaDebugHistory,
"OllamaHistoryLength": OllamaHistoryLength,
"OllamaHeaderBasicAuth": OllamaHeaderBasicAuth,
"OllamaHeaderBearerToken": OllamaHeaderBearerToken,
"OllamaHeaderCustom": OllamaHeaderCustom,
}
# A dictionary that contains the friendly/humanly readable titles for the nodes
NODE_DISPLAY_NAME_MAPPINGS = {
"RandomChoice": "Random Choice",
"CircuitBreaker": "Circuit Breaker",
"FormatString": "Format String (Python f-strings)",
# LLM nodes (generic, ADR-007)
"OllamaClient": "Ollama Client",
"LlamaCppClient": "LlamaCpp Client",
"LLMModelSelector": "LLM Model Selector",
"LLMLoadModel": "LLM Load Model",
"LLMUnloadModel": "LLM Unload Model",
"ChatCompletion": "Chat Completion",
"OllamaOptionTemperature": "Ollama Option — Temperature",
"OllamaOptionSeed": "Ollama Option — Seed",
"OllamaOptionMaxTokens": "Ollama Option — Max Tokens",
"OllamaOptionTopP": "Ollama Option — Top P",
"OllamaOptionTopK": "Ollama Option — Top K",
"OllamaOptionRepeatPenalty": "Ollama Option — Repeat Penalty",
"OllamaOptionDisableThinking": "Ollama Option — Disable Thinking",
"OllamaOptionRefusalRetry": "Ollama Option — Refusal Retry",
"OllamaOptionExtraBody": "Ollama Option — Extra Body",
"OllamaDebugHistory": "Ollama Debug History",
"OllamaHistoryLength": "Ollama History Length",
"OllamaHeaderBasicAuth": "Ollama Header — Basic Auth",
"OllamaHeaderBearerToken": "Ollama Header — Bearer Token",
"OllamaHeaderCustom": "Ollama Header — Custom",
}
WEB_DIRECTORY = "../js"
__all__ = ["NODE_CLASS_MAPPINGS", "NODE_DISPLAY_NAME_MAPPINGS", "WEB_DIRECTORY"]