Some models (observed with an abliterated Qwen variant) answer a request
they judge "politically sensitive" with hedging refusal language instead of
erroring or returning blank — neither of which the existing blank-response
or schema-validation retry triggers catch, so the refusal just passed
through as a normal result.
Adds a hybrid detector to _llm/retry.py: a free lexical/regex pass catches
blatant refusals ("I cannot generate...") without any network call; a
response that's short and/or hedge-y enough to be ambiguous additionally
gets an embedding-similarity check against canonical refusal exemplars, via
a new optional LLMProvider.embed() (Ollama's /api/embed, llama-server's
/v1/embeddings) — deliberately model-agnostic to the backend, since this is
a model-behavior concern, not a backend one. A detected refusal is treated
exactly like a blank response or validation failure: retried with a bumped
seed (existing next_seed()/RETRY_BACKOFF_SECS machinery), never returned to
the caller.
Wired into all four retry loops: OllamaProvider.chat()/chat_structured(),
LlamaCppProvider.chat(), and the shared _llm/chat.py chat_structured() used
by LlamaCppProvider.chat_structured(). New OllamaOptionRefusalRetry node
(same composable OLLAMA_OPTIONS chain as OllamaOptionDisableThinking)
configures it: enabled toggle, optional embedding_model (blank = lexical-
only), similarity threshold.
Claude-Session: https://claude.ai/code/session_01YArD9ZjBWKsAvazmS48amA
Co-authored-by: Claude Sonnet 5 <noreply@anthropic.com>
91 lines
3.4 KiB
Python
91 lines
3.4 KiB
Python
import logging
|
|
|
|
from .circuit_breaker import CircuitBreaker
|
|
from .format_string import FormatString
|
|
from .llamacpp import LlamaCppClient
|
|
from .ollama import (
|
|
ChatCompletion,
|
|
LLMLoadModel,
|
|
LLMModelSelector,
|
|
LLMUnloadModel,
|
|
OllamaClient,
|
|
OllamaDebugHistory,
|
|
OllamaHeaderBasicAuth,
|
|
OllamaHeaderBearerToken,
|
|
OllamaHeaderCustom,
|
|
OllamaHistoryLength,
|
|
OllamaOptionDisableThinking,
|
|
OllamaOptionExtraBody,
|
|
OllamaOptionMaxTokens,
|
|
OllamaOptionRefusalRetry,
|
|
OllamaOptionRepeatPenalty,
|
|
OllamaOptionSeed,
|
|
OllamaOptionTemperature,
|
|
OllamaOptionTopK,
|
|
OllamaOptionTopP,
|
|
)
|
|
from .random_choice import RandomChoice
|
|
|
|
logging.getLogger(__name__).addHandler(logging.NullHandler())
|
|
|
|
# A dictionary that contains all nodes you want to export with their names
|
|
# NOTE: names should be globally unique
|
|
NODE_CLASS_MAPPINGS = {
|
|
"RandomChoice": RandomChoice,
|
|
"CircuitBreaker": CircuitBreaker,
|
|
"FormatString": FormatString,
|
|
# LLM nodes (generic, ADR-007) — see comfydv.ollama.MIGRATION_MAP for
|
|
# the pre-cutover Ollama-specific names these replace
|
|
"OllamaClient": OllamaClient,
|
|
"LlamaCppClient": LlamaCppClient,
|
|
"LLMModelSelector": LLMModelSelector,
|
|
"LLMLoadModel": LLMLoadModel,
|
|
"LLMUnloadModel": LLMUnloadModel,
|
|
"ChatCompletion": ChatCompletion,
|
|
"OllamaOptionTemperature": OllamaOptionTemperature,
|
|
"OllamaOptionSeed": OllamaOptionSeed,
|
|
"OllamaOptionMaxTokens": OllamaOptionMaxTokens,
|
|
"OllamaOptionTopP": OllamaOptionTopP,
|
|
"OllamaOptionTopK": OllamaOptionTopK,
|
|
"OllamaOptionRepeatPenalty": OllamaOptionRepeatPenalty,
|
|
"OllamaOptionDisableThinking": OllamaOptionDisableThinking,
|
|
"OllamaOptionRefusalRetry": OllamaOptionRefusalRetry,
|
|
"OllamaOptionExtraBody": OllamaOptionExtraBody,
|
|
"OllamaDebugHistory": OllamaDebugHistory,
|
|
"OllamaHistoryLength": OllamaHistoryLength,
|
|
"OllamaHeaderBasicAuth": OllamaHeaderBasicAuth,
|
|
"OllamaHeaderBearerToken": OllamaHeaderBearerToken,
|
|
"OllamaHeaderCustom": OllamaHeaderCustom,
|
|
}
|
|
|
|
# A dictionary that contains the friendly/humanly readable titles for the nodes
|
|
NODE_DISPLAY_NAME_MAPPINGS = {
|
|
"RandomChoice": "Random Choice",
|
|
"CircuitBreaker": "Circuit Breaker",
|
|
"FormatString": "Format String (Python f-strings)",
|
|
# LLM nodes (generic, ADR-007)
|
|
"OllamaClient": "Ollama Client",
|
|
"LlamaCppClient": "LlamaCpp Client",
|
|
"LLMModelSelector": "LLM Model Selector",
|
|
"LLMLoadModel": "LLM Load Model",
|
|
"LLMUnloadModel": "LLM Unload Model",
|
|
"ChatCompletion": "Chat Completion",
|
|
"OllamaOptionTemperature": "Ollama Option — Temperature",
|
|
"OllamaOptionSeed": "Ollama Option — Seed",
|
|
"OllamaOptionMaxTokens": "Ollama Option — Max Tokens",
|
|
"OllamaOptionTopP": "Ollama Option — Top P",
|
|
"OllamaOptionTopK": "Ollama Option — Top K",
|
|
"OllamaOptionRepeatPenalty": "Ollama Option — Repeat Penalty",
|
|
"OllamaOptionDisableThinking": "Ollama Option — Disable Thinking",
|
|
"OllamaOptionRefusalRetry": "Ollama Option — Refusal Retry",
|
|
"OllamaOptionExtraBody": "Ollama Option — Extra Body",
|
|
"OllamaDebugHistory": "Ollama Debug History",
|
|
"OllamaHistoryLength": "Ollama History Length",
|
|
"OllamaHeaderBasicAuth": "Ollama Header — Basic Auth",
|
|
"OllamaHeaderBearerToken": "Ollama Header — Bearer Token",
|
|
"OllamaHeaderCustom": "Ollama Header — Custom",
|
|
}
|
|
|
|
WEB_DIRECTORY = "../js"
|
|
__all__ = ["NODE_CLASS_MAPPINGS", "NODE_DISPLAY_NAME_MAPPINGS", "WEB_DIRECTORY"]
|