From 01bd5568d09fbb845f1066a667262d3eaac404af Mon Sep 17 00:00:00 2001 From: asagi4 <130366179+asagi4@users.noreply.github.com> Date: Tue, 20 May 2025 20:02:55 +0300 Subject: [PATCH] New function: TE Fixes #106 --- README.md | 1 + doc/syntax.md | 24 ++++++++++++++------ prompt_control/prompts.py | 47 ++++++++++++++++++++++++++++++--------- 3 files changed, 54 insertions(+), 18 deletions(-) diff --git a/README.md b/README.md index 4ad0869..086c6e0 100644 --- a/README.md +++ b/README.md @@ -28,6 +28,7 @@ See [features](#features) below. Things you can control via the prompt: - Prompt editing and filtering without noodle soup - LoRA loading and scheduling via ComfyUI's hook system - Masking, composition and area control (regional prompting) +- Per-encoder prompts for models with multiple text encoders, such as SDXL and Flux - Prompt operations like `BREAK` and `AND` - Weight interpretation types (comfy, A1111, etc.) - Prompt masking with [cutoff](#cutoff) diff --git a/doc/syntax.md b/doc/syntax.md index 932f059..2a0840c 100644 --- a/doc/syntax.md +++ b/doc/syntax.md @@ -120,15 +120,23 @@ The nodes do not treat SDXL models specially, but there are some utilities that You can use the function `SDXL(width height, target_width target_height, crop_w crop_h)` to set SDXL prompt parameters. `SDXL()` is equivalent to `SDXL(1024 1024, 1024 1024, 0 0)` unless the default values have been overridden by `PCScheduleSettings`. -To set the `clip_l` prompt, as with `CLIPTextEncodeSDXL`, use the function `CLIP_L(prompt text goes here)`. +### Multiple text encoders: TE + +You can specify per-encoder prompts using the `TE` function. The syntax is as follows: +`TE(encoder_name=prompt)`. Whitespace surrounding the prompt and encoder name are ignored. + +For example: +``` +TE(l=cat) TE(g = (dog:1.1)) TE(t5xxl=tiger) +``` +The keys to use depend on what key ComfyUI uses for the encoder; for example `l` for CLIP L, `g` for CLIP G, and `t5xxl` for T5 XXL (Flux text encoder). + +Use `TE(help)` to print a help text listing available keys. Things to note: -- Multiple instances of `CLIP_L` are joined with a space. That is, `CLIP_L(foo)CLIP_L(bar)` is the same as `CLIP_L(foo bar)` -- Using `BREAK` isn't supported in it; it'll just parse as the plain word BREAK. -- similarly, `AND` inside `CLIP_L` does not do anything sensible; `CLIP_L(foo AND bar)` will parse as two prompts `CLIP_L(foo` and `bar)` -- `CLIP_L` and `SDXL` have no effect on SD 1.5. -- The rest of the prompt becomes the `clip_g` prompt. -- If there is no `CLIP_L` or `SDXL`, the prompts will work as with `CLIPTextEncode`. +- If you set a prompt with `TE`, it will override the prompt outside the function for the specified text encoder. +- Multiple instances of `TE` are joined with a space. That is, `TE(l=foo)TE(l=bar)` is the same as `TE(l=foo bar)` +- `AND` inside `TE` does not do anything sensible; `TE(l=foo AND bar)` will parse as two prompts `TE(foo` and `bar)`. `BREAK`, `SHIFT` and `SHUFFLE` do work, however ### SHUFFLE and SHIFT @@ -252,3 +260,5 @@ Use `ATTN()` in combination with `MASK()` or `IMASK()` to enable attention maski ## TE_WEIGHT For models using multiple text encoders, you can set weights per TE using the syntax `TE_WEIGHT(clipname=weight, clipname2=weight2, ...)` where `clipname` is one of `g`, `l`, or `t5xxl`. For example with SDXL, try `TE_WEIGHT(g=0.25, l=0.75`). The weights are applied as a multiplier to the TE output. + +To set a default value for all encoders, use `TE_WEIGHT(all=weight)` diff --git a/prompt_control/prompts.py b/prompt_control/prompts.py index fd11520..bdc29f4 100644 --- a/prompt_control/prompts.py +++ b/prompt_control/prompts.py @@ -179,21 +179,46 @@ def encode_prompt_segment( # defaults=None means there is no argument parsing at all text, l_prompts = get_function(text, "CLIP_L", defaults=None) + text, te_prompts = get_function(text, "TE", defaults=None) need_word_ids = True tokens = tokenize_chunks(clip, text, need_word_ids) - # Non-SDXL has only "l" - if "g" in tokens and l_prompts: - text_l = " ".join(l_prompts) - log.info("Encoded SDXL CLIP_L prompt: %s", text_l) - tokens["l"] = clip.tokenize(text_l, return_word_ids=need_word_ids)["l"] + per_te_prompts = {} + if l_prompts: + log.warning("Note: CLIP_L is deprecated. Use TE(l=prompt) instead") + per_te_prompts["l"] = l_prompts - if "g" in tokens and "l" in tokens and len(tokens["l"]) != len(tokens["g"]): - empty = clip.tokenize("", return_word_ids=need_word_ids) - while len(tokens["l"]) < len(tokens["g"]): - tokens["l"] += empty["l"] - while len(tokens["l"]) > len(tokens["g"]): - tokens["g"] += empty["g"] + for prompt in te_prompts: + if prompt.strip() == "help": + log.info("Encoders available for TE: %s", ", ".join(tokens.keys())) + continue + params = prompt.split("=", 1) + if len(params) != 2: + log.warning("Invalid TE call, ignoring: %s", prompt) + continue + te = params[0].strip() + prompt = params[1].strip() + if te not in tokens: + log.warning("Invalid TE call, no TE with key '%s', ignoring: %s", te) + log.info("Encoders available for TE: %s", ", ".join(tokens.keys())) + continue + l = per_te_prompts.get(te, []) + l.append(prompt) + per_te_prompts[te] = l + + if per_te_prompts: + for key in per_te_prompts: + prompt = " ".join(per_te_prompts[key]) + tokens[key] = tokenize_chunks(clip, prompt, need_word_ids)[key] + log.info("Encoded prompt with TE '%s': %s", key, prompt) + + maxlen = max(len(tokens[k]) for k in tokens) + empty = None + for k in tokens: + while len(tokens[k]) < maxlen: + if empty is None: + empty = clip.tokenize("", return_word_ids=need_word_ids) + tokens[k] += empty[k] tokens = fix_word_ids(tokens)