Increase default context size from 8192 to 16384 tokens
Fixes 400 error when image requests exceed 8192 token context window. Aligns shell script and Server Control node defaults with GGUF node auto-start. Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 4.6
parent
b77bb3a581
commit
c03a2193c1
@@ -844,7 +844,7 @@ class ArchAi3D_QwenVL_Server_Control:
|
||||
"tooltip": "GPU layers (99=all on GPU, lower values use more CPU to save VRAM)"
|
||||
}),
|
||||
"context_size": ("INT", {
|
||||
"default": 8192,
|
||||
"default": 16384,
|
||||
"min": 1024,
|
||||
"max": 32768,
|
||||
"step": 1024,
|
||||
@@ -900,7 +900,7 @@ class ArchAi3D_QwenVL_Server_Control:
|
||||
return False, str(e)
|
||||
|
||||
def control(self, action, model_size, quantization="Q4_K_M (Smaller, ~5GB)",
|
||||
gpu_layers=99, context_size=8192,
|
||||
gpu_layers=99, context_size=16384,
|
||||
flash_attention="auto (Recommended)", kv_cache_type="f16 (Best Quality)",
|
||||
parallel_slots=2, trigger=None, cache_to_local_ssd=True):
|
||||
"""Control the llama-server with quality and speed optimizations."""
|
||||
|
||||
@@ -24,7 +24,7 @@
|
||||
# ============================================================================
|
||||
|
||||
# Configuration
|
||||
CTX="${CTX:-8192}"
|
||||
CTX="${CTX:-16384}"
|
||||
GPU_LAYERS="${GPU_LAYERS:-99}" # 99 for all layers on GPU
|
||||
|
||||
# ============================================================================
|
||||
|
||||
Reference in New Issue
Block a user