39 Commits
Author SHA1 Message Date
Jose Luis Cordova 33c21baff8 Refresh CaptionForge 1.0.2 workflow assets 2026-09-19 19:47:44 -04:00
Jose Luis Cordova b116b08033 Sync CaptionForge 1.0.2 Ollama witness default 2026-09-19 19:16:38 -04:00
Jose Luis Cordova 1bc70ccb2a Polish CaptionForge 1.0.2 release workflows 2026-09-19 17:34:15 -04:00
Jose Luis Cordova 9d5b131b08 Release CaptionForge 1.0.2 2026-09-19 13:08:14 -04:00
Jose Luis Cordova fd5f519aae Harden end-to-end caption cleanup 2026-09-19 13:00:22 -04:00
José Luis Córdova 90753aa7c6 Align replacement helper with boundary-safe default 2026-09-19 09:42:15 -04:00
José Luis Córdova a84724a38a Align replacement helper with boundary-safe default 2026-09-19 09:42:13 -04:00
José Luis Córdova 7fa2871810 Add replace_pairs boundary regression coverage 2026-09-19 09:41:59 -04:00
José Luis Córdova 49080cbae2 Make Ollama replace_pairs boundary-safe 2026-09-19 09:41:41 -04:00
José Luis Córdova 9f76aae7e7 Make Qwen replace_pairs boundary-safe by default 2026-09-19 09:41:35 -04:00
José Luis Córdova 49290e8812 Make Joy replace_pairs boundary-safe by default 2026-09-19 09:41:33 -04:00
José Luis Córdova f7f770f953 Add boundary-safe shared replacement helper 2026-09-19 09:41:13 -04:00
José Luis Córdova a1e613a6ac Add cleanup boundary regression coverage 2026-09-19 09:23:59 -04:00
José Luis Córdova 25425d41b5 Fix Ollama forbidden matching and audit cleanup settings 2026-09-19 09:23:36 -04:00
José Luis Córdova b438fddbc7 Use boundary-safe forbidden matching in Qwen 2026-09-19 09:23:26 -04:00
José Luis Córdova 32a8d4ecc7 Use boundary-safe forbidden matching in Joy 2026-09-19 09:23:24 -04:00
José Luis Córdova af785e3d36 Add boundary-safe shared cleanup matcher 2026-09-19 09:23:06 -04:00
Jose Luis Cordova a373a7e8d9 Restore scoped Joy/Qwen bitsandbytes diagnostics and install dependencies 2026-09-18 14:49:19 -04:00
José Luis Córdova b234e8020c Test explicit reuse of generated dataset inputs 2026-09-18 11:49:20 -04:00
José Luis Córdova 6c111930d5 Allow explicit reuse of generated dataset inputs 2026-09-18 11:49:07 -04:00
José Luis Córdova 31a10619d9 Allow explicit reuse of generated dataset inputs 2026-09-18 11:49:05 -04:00
José Luis Córdova 5ed231eaad Allow explicit reuse of generated dataset inputs 2026-09-18 11:49:02 -04:00
Jose Luis Cordova 435ad73ad6 CaptionForge: add paired dataset export 2026-09-17 20:19:55 -04:00
Jose Luis Cordova 3cf6b3536d CaptionForge: release 1.0.1 2026-09-13 19:03:50 -04:00
Jose Luis Cordova cb5741b604 CaptionForge: harden orchestrator resume and recovery 2026-09-13 17:11:16 -04:00
Jose Luis Cordova 59c4eb31e9 CaptionForge: cap validator image size from planner 2026-09-13 13:09:46 -04:00
Jose Luis Cordova 2a5d62ef8a CaptionForge: improve 1.0.0 project landing page 2026-09-12 14:05:16 -04:00
Jose Luis Cordova 5b1c395a59 CaptionForge: refresh canonical workflows for 1.0.0 2026-09-12 13:24:45 -04:00
Jose Luis Cordova a6a13d389c CaptionForge: relax SHORT length constraint 2026-09-12 13:16:10 -04:00
Jose Luis Cordova e69d9a0dd8 CaptionForge: tighten public Ollama model choices 2026-09-12 10:31:34 -04:00
Jose Luis Cordova 416555fdfb CaptionForge: reconcile remaining release polish 2026-09-12 10:15:26 -04:00
Jose Luis Cordova c26c8a613f CaptionForge: refine Orchestrator output contract 2026-09-12 09:29:17 -04:00
Jose Luis Cordova 093b6908f5 CaptionForge: harden Pass-A source identity and artifacts 2026-09-12 09:00:03 -04:00
Jose Luis Cordova 9b210ad811 CaptionForge: polish documentation and release metadata for 1.0.0 2026-09-11 15:18:52 -04:00
Jose Luis Cordova cd2d0a675f CaptionForge: freeze downstream ownership and production defaults 2026-09-11 00:41:05 -04:00
Jose Luis Cordova 4d2124a8f4 CaptionForge: refine short captions from validated long 2026-09-09 08:10:41 -04:00
Jose Luis Cordova 5568f74dbb CaptionForge: harden seed contract and planner ownership 2026-09-08 22:58:07 -04:00
Jose Luis Cordova dc2d05899b CaptionForge: reconcile planner and capstone contracts 2026-09-07 19:07:26 -04:00
Jose Luis Cordova 3dd61f527c CaptionForge: harden release workflow and output handling 2026-09-07 11:57:05 -04:00
50 changed files with 8802 additions and 2876 deletions
+5 -1
View File
@@ -69,4 +69,8 @@ config/.bak/
# CaptionForge local experimental/quarantined code
/nodes/experimental/
/engines/experimental/
node.zip
node.zip
# CaptionForge local/private model configuration references
config/captionforge_ollama_models_new.json
config/captionforge_ollama_models_unrestricted.json
config/Ollama_Models_Catalog.md
+34 -21
View File
@@ -1,31 +1,44 @@
Project Folder/File Structure:
# Repository structure
CaptionForge keeps ComfyUI-facing nodes separate from reusable backend code.
Only the explicit production packages and package data listed in
`pyproject.toml` are included in a distribution.
```text
CaptionForge/
├─ __init__.py
├─ __init__.py ComfyUI registration entry point
├─ captionforge_version.py authoritative release version
├─ README.md
├─ pyproject.toml
│
├─ config/
│ └─ captionforge_ollama_models.json user-editable Ollama model tags
├─ nodes/
│ ├─ __init__.py
│ ├─ captionforge_extra_options_CUI_node.py
│ ├─ jlc_captionforge_node.py
│ ├─ jlc_captionforge_pipeline_planner_node.py
| ├─ caption_nodes
│ │ ├─ jlc_joy_caption_node.py
│ │ ├─ jlc_qwen_caption_node.py
│ │ ├─ jlc_ollama_caption_node.py
│
│ ├─ jlc_captionforge_node.py production B/C/D capstone
│ ├─ jlc_captionforge_template_options.py
│ ├─ captionforge_ollama_model_dropdowns.py
│ └─ caption_nodes/
│ ├─ jlc_captionforge_joy_caption_node.py
│ ├─ jlc_captionforge_qwen_caption_node.py
│ └─ jlc_captionforge_ollama_caption_node.py
├─ engines/
│ ├─ __init__.py
│ ├─ captionforge_claim_engine.py
│ ├─ captionforge_model_cache.py
│ ├─ captionforge_pipeline_planner_engine.py
│ ├─ captionforge_prompt_defaults.py
│ ├─ captionforge_caption_prompt_kit.py
│ ├─ captionforge_joy_space_prompt_kit.py
│ ├─ captionforge_model_cache.py
│ ├─ jlc_joy_caption_engine.py
│ ├─ jlc_qwen_caption_engine.py
│
├─ config/
│ ├─ captionforge_ollama_models.json
│
└─ web/
├─ jlc_captionforge_icons.js
└─ ...
│ ├─ captionforge_distiller_engine.py standalone reference/CLI engine
│ └─ captionforge_vlm_validator_engine.py standalone reference/CLI engine
├─ assets/
│ ├─ icons/
│ └─ workflows/ canonical JSON/API JSON/PNG workflows
├─ docs/ release-facing studies
├─ tests/ CPU contract tests
└─ web/ ComfyUI frontend/icon support
```
Local `.backups/`, `.tests/`, `.vscode/`, `.bak/`, `.deprecated/`, and
experimental trees are archival or development-only. They are neither
registered by CaptionForge nor included in the production package.
+400 -438
View File
@@ -1,257 +1,382 @@
# CaptionForge
<p align="center">
<img src="assets/icons/jlc-comfyui-nodes_Logo-0512.png" width="150" alt="CaptionForge logo">
</p>
**Accurate, auditable image captions for LoRA dataset preparation in ComfyUI.**
<h1 align="center">CaptionForge for ComfyUI</h1>
CaptionForge is built around a simple idea: one captioner can be useful, but one captioner is also easy to fool. Instead of asking a single model to describe an image and hoping it gets everything right, CaptionForge can ask multiple independent captioning engines to produce separate “witness accounts” of the same image. Those accounts are then merged by a text-LLM distillation pass that looks for agreement, preserves useful details, and separates likely contradictions or unsupported claims. The resulting draft is checked against the image by a final vision-language model, which acts as the image-aware judge before the final captions are exported.
<p align="center">
<strong>Accurate, auditable multi-model image captions for LoRA dataset preparation.</strong>
</p>
The goal is not magic, and it is not perfection. CaptionForge is meant for automated captioning of large image archives and LoRA training sets where hand-captioning would be too slow, but where the usual hallucinations, omissions, and inconsistencies from a single captioning model are still a problem. The pipeline is intentionally heavier than a normal caption node, so it is best used when caption quality, auditability, and consistency matter enough to justify the extra computation.
The current v0.1.x workflow is tuned primarily for character, fashion, portrait, doll/render, cosplay, pageant, glamour, and style-LoRA datasets, where visible details such as face, hair, eyes, expression, pose, body shape, clothing construction, accessories, colors, materials, lighting, background, framing, and visual style matter.
<p align="center">
<img alt="ComfyUI" src="https://img.shields.io/badge/ComfyUI-Custom%20Nodes-blue">
<img alt="License" src="https://img.shields.io/badge/license-MIT-green">
<img alt="Version" src="https://img.shields.io/badge/version-1.0.2-blue">
</p>
---
<p align="center">
<img src="assets/icons/jlc-comfyui-nodes_Logo-0512.png" width="120">
&nbsp;&nbsp;&nbsp;
<img src="assets/icons/jlc-comfyui-nodes_Logo-Dark-0512.png" width="120">
</p>
## Overview
[![ComfyUI](https://img.shields.io/badge/ComfyUI-Custom%20Nodes-blue)]()
[![License](https://img.shields.io/badge/license-MIT-green)]()
![Status](https://img.shields.io/badge/status-v0.1.x%20preview-orange)
![Version](https://img.shields.io/badge/version-0.1.0-orange)
CaptionForge is a local, model-agnostic captioning framework for building richer, more auditable captions for LoRA dataset preparation inside ComfyUI.
## Starter workflow
The core idea is simple: a single image captioner can be useful, but it should not be treated as authoritative. CaptionForge can collect several independent **Pass A witness captions**, synthesize them with a text LLM, validate the resulting draft against the original image with a vision-language model, and then export three useful caption forms from the same validated semantic result:
A full workflow sample is included as a PNG with embedded ComfyUI workflow metadata:
- `*_long.txt` — authoritative image-validated natural-language caption
- `*_short.txt` — concise AI-compressed natural-language caption intended for modern LoRA training workflows such as FLUX-family training
- `*_taggy.txt` — compact comma-separated caption suited to tag-oriented SD-style training workflows
```text
assets/workflows/CaptionForge_FullWorkflow.png
```
CaptionForge also writes structured JSONL audit records so intermediate evidence, prompts, model settings, and final outputs can be inspected instead of treated as a black box.
> **Current release: CaptionForge 1.0.2.**
> The A/B/C/D semantic pipeline, Planner/Orchestrator authority model, seed contract, and production defaults are frozen for this release.
---
## Sample Workflow
For installation dependencies and Joy Balanced (8-bit) warning handling, see
[Joy 8-bit and clean installation](docs/joy-8bit-installation.md).
<p align="center">
<a href="assets/workflows/CaptionForge_FullWorkflow.png">Download workflow PNG</a>
<a href="assets/workflows/CaptionForge_FullWorkflow_Rel_v1.0.2.png">
<img src="assets/workflows/CaptionForge_FullWorkflow_Rel_v1.0.2.png"
alt="CaptionForge Full Workflow"
width="100%">
</a>
</p>
<p align="center">
<img src="assets/workflows/CaptionForge_FullWorkflow.png" alt="CaptionForge full starter workflow" width="900">
</p>
Canonical workflow files:
A separate JSON export of the same workflow is also included:
- [`CaptionForge_FullWorkflow_Rel_v1.0.2.json`](assets/workflows/CaptionForge_FullWorkflow_Rel_v1.0.2.json) — editable ComfyUI workflow
- [`CaptionForge_FullWorkflow_API_Rel_v1.0.2.json`](assets/workflows/CaptionForge_FullWorkflow_API_Rel_v1.0.2.json) — API-format workflow
- [`CaptionForge_FullWorkflow_Rel_v1.0.2.png`](assets/workflows/CaptionForge_FullWorkflow_Rel_v1.0.2.png) — workflow PNG with embedded metadata
Click the workflow image above to view it at full size.
---
## Why CaptionForge exists
A strong standalone captioner may be enough for many datasets. CaptionForge is for cases where one captioner is not accurate, complete, consistent, or auditable enough.
Different captioning models often notice different details. One may capture face and hair accurately but miss clothing construction. Another may catch materials or accessories but misread pose. A third may notice background or style cues that the others omit.
CaptionForge treats those captions as **witness statements**, not final truth.
The pipeline then:
1. gathers independent witness captions;
2. combines useful evidence into a richer draft;
3. checks that draft against the original image;
4. produces validated LONG, SHORT, and TAGGY forms;
5. records the process in auditable JSONL artifacts.
The goal is not perfect captions. The goal is a more defensible automated captioning process for large datasets where hand-captioning would be impractical.
---
## Production pipeline
```text
assets/workflows/CaptionForge_FullWorkflow.json
```
<p align="center">
<a href="assets/workflows/CaptionForge_FullWorkflow.json">Download workflow JSON</a>
</p>
In ComfyUI, load the workflow by dragging either `CaptionForge_FullWorkflow.png` or `CaptionForge_FullWorkflow.json` onto the canvas.
## Install
Clone CaptionForge into your ComfyUI custom nodes folder:
```bash
git clone https://github.com/Damkohler/CaptionForge.git ComfyUI/custom_nodes/CaptionForge
```
Or copy the repository manually so the folder layout is:
```text
ComfyUI/custom_nodes/CaptionForge/
```
Then restart ComfyUI.
If your ComfyUI environment does not already include the needed Python packages, install CaptionForge dependencies from inside your ComfyUI Python environment. The exact command depends on how your ComfyUI install is managed, but typical options are:
```bash
cd ComfyUI/custom_nodes/CaptionForge
pip install -e .
```
or, if you maintain dependencies manually:
```bash
pip install torch transformers accelerate huggingface-hub pillow numpy safetensors qwen-vl-utils
```
Optional 8-bit loading may require:
```bash
pip install bitsandbytes
```
Ollama-backed stages require a working local Ollama installation and installed Ollama model tags.
Example:
```bash
ollama pull mistral-small:24b
ollama pull gemma4:26b
```
CaptionForge does **not** ship model weights. Joy, Qwen, and Ollama model downloads remain user-controlled.
## What the workflow does
CaptionForge's main pipeline is:
```text
Pass A — raw witness captions
Pass A — witness captions
Joy Caption xN
Qwen Caption xN
optional Ollama VLM Caption xN
Ollama VLM Caption xN
Pass B — text-LLM distillation
combine witness captions
preserve repeated and useful details
separate contradictions and weak claims
build a rich draft caption
Pass B — Distiller
text LLM combines witness evidence
preserves useful repeated and plausible details
builds a rich draft caption
Pass C — image-aware VLM validation
inspect the actual image
keep image-supported details
remove unsupported hallucinations
correct visible errors
produce the authoritative long caption
Pass C — Validator
image-aware VLM inspects the source image
checks the draft against visible evidence
removes unsupported claims where possible
corrects visible errors where possible
produces the authoritative LONG caption
Pass D — deterministic export formatting
write the validated long caption
derive a shorter LoRA-length caption
derive a compact taggy caption
write TXT and JSONL audit records
Pass D — Formatter
text LLM transforms validated LONG into:
SHORT
TAGGY
```
The important distinction is that the expensive semantic work should mostly end at the VLM-validated long caption. The short and taggy outputs are intentionally lighter recipe-style formatting steps derived from that validated caption, not new attempts to reinterpret the image.
The final semantic authority is the image-aware validation pass. Pass D formats that validated result; it is not intended to reinterpret the image from scratch.
## Current status
---
CaptionForge v0.1.0 is a working experimental preview for ComfyUI users and node developers who want to test a multi-pass captioning pipeline.
## Recommended production defaults
It is not presented as a universal replacement for a strong standalone captioner. If JoyCaption, Qwen, Florence, BLIP, WD14, or another captioning tool already gives you exactly what your dataset needs, you may not need CaptionForge. This project is aimed at cases where a single captioner is not accurate, complete, consistent, or auditable enough.
### Pass A — witnesses
Expected v0.1.x realities:
The current recommended witness profile is:
- the workflow is computationally heavy
- large models may be slow
- model choices matter a lot
- output schemas may still evolve
- prompts and defaults may continue to be refined
- not every dataset will benefit equally
- comparison feedback is welcome
- **Joy:** 2 runs per image
- **Qwen:** 1 run per image
- **Ollama:** 1 run per connected Ollama witness node
This is a heavy tool. Use it when the extra caption quality and audit trail of large automated jobs are worth the runtime cost.
Multiple Ollama witness nodes may be connected when additional independent VLM caption voices are desired.
## Why use this instead of a standalone captioner?
### Pass B — Distiller
You may want CaptionForge when:
- one captioner notices the face but misses clothing details
- another captioner notices clothing but misreads the pose
- a third captioner catches style or material details the others miss
- you want an LLM to consolidate agreement instead of merely accepting one model's wording
- you want a final VLM to check the draft against the actual image
- you want intermediate JSONL records for debugging and audit
- you want final captions written as sidecars beside the source images
- you need both long natural captions and compact LoRA-style derivatives
The project question is practical:
> Can independent caption witnesses plus text distillation plus image-aware validation produce better dataset captions than a single captioning model alone?
For some datasets, the answer may be yes. For others, a simpler captioner may be enough. CaptionForge is designed to make that comparison visible.
## What CaptionForge tries to optimize
CaptionForge currently favors captions that are:
- rich enough for LoRA training
- visually grounded
- less hallucinated than unvalidated text-only synthesis
- explicit about visible, trainable details
- auditable through JSONL records
- locally runnable
- model-agnostic enough to improve as better captioners, distillers, and validators become available
Useful caption details often include:
- subject type and visible style
- face shape and facial traits
- hair color and hairstyle
- eye color and makeup as separate details
- expression and pose
- hands and body position
- body shape and visible proportions when relevant
- clothing construction, layers, fit, and materials
- accessories, jewelry, nails, props, and distinctive details
- colors, textures, lighting, background, framing, and crop
Visible glamour, swimwear, lingerie, revealing clothing, cleavage, side openings, exposed midriff, or similar styling may be described neutrally when it is actually visible and relevant to the dataset. CaptionForge prompts should not invent hidden anatomy, unseen clothing, explicit acts, or contradicted details.
## Active node families
Node categories are being normalized under:
Default model:
```text
Captioning/CaptionForge
mistral-small:24b
```
with active caption nodes under:
Default production settings:
```text
Captioning/CaptionForge/Caption Nodes
source cap = 1536
num_predict = 3096
temperature = 0.24
top_p = 0.90
top_k = 60
seed = -1
```
### JLC CaptionForge Pipeline Planner
### Pass C — Validator
The central planning node for normal runs.
Default model:
```text
gemma4:26b
```
Default production settings:
```text
num_predict = 2112
temperature = 0.00
top_p = 0.92
top_k = 80
seed = -1
```
### Pass D — Formatter
Default model:
```text
mistral-small:24b
```
Default production settings:
```text
num_predict = 3200
temperature = 0.12
top_p = 0.88
top_k = 50
seed = -1
```
`seed = -1` means CaptionForge does **not** send an explicit seed to Ollama. This is intentionally unseeded behavior and is not guaranteed to be repeatable.
---
## LONG, SHORT, and TAGGY
### LONG
The authoritative image-validated natural-language caption.
Use LONG when maximum descriptive detail is useful for:
- review
- auditing
- experimentation
- highly descriptive datasets
### SHORT
A concise semantic compression of validated LONG.
SHORT is intended as the best starting point for many FLUX-family LoRA workflows.
The Formatter aims for **roughly 100 words, give or take**. CaptionForge does not hard-cut a valid SHORT at a word or character boundary; the caption is allowed to finish naturally.
### TAGGY
A compact comma-separated caption derived from validated LONG.
TAGGY is intended primarily for SD-style or other tag-oriented training workflows.
CaptionForge writes all three forms so a dataset can later be trained or compared using different caption styles without recaptioning the source images.
---
## Main nodes
### CaptionForge Pipeline Planner
The Planner is the control center for the full workflow.
It coordinates:
- input image path or direct image passthrough
- recursive folder traversal
- image or folder input
- recursive traversal
- filename glob filtering
- output directory
- output location
- run name
- overwrite behavior
- Pass A witness run counts
- seed schedules
- sampling schedules
- max image size
- max token budget
- LoRA trigger word
- user caption anchor
- distiller settings
- validator settings
- final export settings
- derived JSONL/TXT/config paths
- witness run counts
- Pass A seed scheduling
- shared Ollama configuration
- Distiller settings
- Validator settings
- Formatter settings
- audit options
### JLC CaptionForge
In the full workflow, **the Planner is authoritative**.
The main capstone/orchestration node.
The Planner now owns `forbidden_phrases` and `replace_pairs` for the full workflow. Joy, Qwen, Ollama Caption, and the Orchestrator retain their corresponding standalone inputs; when a Planner is connected, its values take precedence.
It consumes Pass A raw caption records, runs the distillation and validation stages, and exports final captions. The VLM-validated natural paragraph is the authoritative long caption. Formatting stages should not blindly rewrite that natural caption.
### CaptionForge Joy Caption
### JLC CaptionForge Joy Caption
Python/Hugging Face JoyCaption-family Pass A witness.
Python/Hugging Face JoyCaption/LLaVA-family Pass A witness.
Joy remains a strong first-class caption source and is commonly run twice per image to gain useful diversity.
Joy is treated as a first-class CaptionForge caption source and is often one of the strongest raw caption witnesses.
### JLC CaptionForge Qwen Caption
### CaptionForge Qwen Caption
Python/Hugging Face Qwen-family Pass A witness.
Qwen is useful as a second independent captioning voice, especially when its behavior complements Joy. Optional 8-bit loading may be available where supported.
Qwen provides a complementary caption voice and is commonly run once per image in the production profile.
### JLC CaptionForge Ollama Caption
### CaptionForge Ollama Caption
Ollama-backed VLM Pass A witness.
This node delegates image-caption generation to a local Ollama server rather than loading Hugging Face/PyTorch weights inside ComfyUI. It can use configured Ollama VLM tags such as:
Multiple Ollama witness nodes may be chained to contribute captions from different local VLMs.
### CaptionForge Template Options
Provides shared captioning guidance for witness nodes so they can emphasize LoRA-relevant visual detail without forcing every backend into an identical prompt implementation.
### CaptionForge Orchestrator
The Orchestrator performs the downstream B/C/D work.
It:
- consumes Pass A witness records;
- runs Distiller synthesis;
- runs image-aware validation;
- produces LONG, SHORT, and TAGGY;
- writes final sidecars and audit artifacts.
The Orchestrator exposes exactly five public outputs:
```text
long_captions
short_captions
taggy_captions
final_records
status
```
In standalone mode, Orchestrator-local B/C/D settings are authoritative.
When connected to the Pipeline Planner, Planner settings take precedence.
Cleanup uses boundary-safe whole-word and phrase matching, so a rule such as `old` does not alter `bold`, `holding`, or `gold`. The effective rules are enforced through final LONG, SHORT, and TAGGY generation so downstream models cannot silently reintroduce forbidden or superseded wording.
---
## Output files
For each successfully processed image, CaptionForge writes:
```text
*_long.txt
*_short.txt
*_taggy.txt
```
The run also writes structured JSON/JSONL records for intermediate and final pipeline stages.
`A_RAW_CAPTIONS.jsonl` is the authoritative Pass A witness ledger. Planned CaptionForge execution does **not** create redundant individual Joy/Qwen/Ollama witness `.txt` files.
Optional prompt and raw-response auditing can be enabled when deeper inspection is useful.
---
## Source-image identity
CaptionForge keeps human-readable names separate from collision-safe canonical identity.
For normal folder input:
```text
image = exact source filename stem
image_key = relative dataset path including extension
```
Example:
```text
People/Session 1/_DSC2094-Edit-2.jpg
image = _DSC2094-Edit-2
image_key = People/Session 1/_DSC2094-Edit-2.jpg
```
This preserves leading underscores, spaces, punctuation, capitalization, relative directory hierarchy, and the extension in the canonical key.
It also prevents collisions between files such as:
```text
Set A/photo.jpg
Set A/photo.png
Set B/photo.jpg
```
Canonical JSON/JSONL identity uses portable `/` separators.
---
## Optional IMAGE input
Direct ComfyUI IMAGE input can be used independently or together with Planner folder input.
Filename-less optional tensors receive deterministic names such as:
```text
comfy_image_0000
```
and collision-safe canonical keys such as:
```text
captionforge-optional-image://comfy_image_0000.png
```
The Orchestrator materializes these images into the run's `opt_images` directory so the generated sidecars can be matched back to their corresponding images.
---
## Ollama model choices
The public dropdown configuration lives at:
```text
config/captionforge_ollama_models.json
```
Current public choices include:
### Distiller / Formatter
```text
mistral-small:24b
gpt-oss:20b
VladimirGav/gemma4-26b-16GB-VRAM-Uncensored
```
### Validator / Ollama witness
```text
gemma4:26b
@@ -259,274 +384,111 @@ qwen3.6:35B-A3B
huihui_ai/gemma-4-abliterated:26b
```
Its purpose is to provide access to other raw-caption witness alternatives. It's function is parallel to the Joy Caption and Qwen Caption nodes, and should not be confused with the later VLM validator/capstone role.
### JLC CaptionForge Template Options
Shared prompt-option sidecar for caption nodes.
Template Options let one sidecar node feed consistent LoRA-relevant prompt modifiers into Joy, Qwen, Ollama, and later caption witnesses without duplicating the same option widgets on every caption node.
## Model and memory behavior
CaptionForge uses two model ecosystems:
1. **Python / Hugging Face model folders** for Joy and Qwen witness engines.
2. **Ollama models** for text-LLM distillation, image-aware VLM validation, optional formatting, and Ollama-backed caption witnesses.
Joy and Qwen use Python/Hugging Face engines that integrate with the CaptionForge process-local model cache. Those engines manage Python model residency, reuse, and eviction before loading heavyweight caption models.
Ollama-facing stages are different. Ollama models live in the Ollama daemon, not inside the CaptionForge Python model cache. Before handing work to Ollama, the Ollama Caption node and the CaptionForge capstone clear any resident CaptionForge Python/HF caption models if needed. After that handoff, Ollama owns Ollama model residency.
In short:
The defaults remain:
```text
Joy/Qwen engines:
manage Python-hosted caption models through captionforge_model_cache
Ollama Caption and CaptionForge capstone:
clear Python-hosted models before calling the Ollama daemon
Ollama daemon:
owns Ollama model loading and residency
Distiller -> mistral-small:24b
Validator -> gemma4:26b
Formatter -> mistral-small:24b
Caption -> gemma4:26b
```
## Model locations
Custom Ollama model tags can also be exposed through the node configuration.
Large model weights are intentionally not stored in this repository.
CaptionForge does **not** ship model weights.
Python-based witness models are expected under ComfyUI model folders, for example:
---
## Model handoff and memory management
CaptionForge coordinates Python/Hugging Face caption models and Ollama-backed stages within the same workflow.
Joy and Qwen models are evicted before Ollama handoff when required, reducing stale GPU-memory pressure during long mixed-engine runs.
Large local models may take significant time to load, especially on first use. Later calls are generally faster once the selected backend models are resident.
CaptionForge is intentionally heavier than a one-model caption node. The design target is caption quality and auditability rather than minimum inference count.
---
## Seed behavior
### Pass A
The Pipeline Planner owns Pass A seed scheduling.
The same planned run-seed schedule is reused across images.
### Pass B / C / D
Distiller, Validator, and Formatter each use one stage seed for the run.
For Ollama-backed stages:
```text
ComfyUI/models/LLM/JLC_JoyCaption/
ComfyUI/models/LLM/JLC_QwenCaption/
seed = -1
```
Ollama models must be installed and runnable through Ollama outside this repository.
means no explicit seed is sent.
CaptionForge does not require every supported backend to be installed for every workflow. Users can test smaller subsets first.
Use explicit seeds when reproducibility is required.
## Ollama model dropdown configuration
---
The file:
## Running the full workflow
1. Load [`CaptionForge_FullWorkflow_Rel_v1.0.2.json`](assets/workflows/CaptionForge_FullWorkflow_Rel_v1.0.2.json).
2. Select an image or dataset folder in the **Pipeline Planner**.
3. Select the output folder and run name.
4. Start with the recommended witness counts:
- Joy: 2
- Qwen: 1
- Ollama: 1 per connected Ollama witness
5. Leave Distiller, Validator, and Formatter settings at their defaults unless you have a reason to experiment.
6. Queue the workflow.
CaptionForge will produce the final caption sidecars and preserve the run's structured audit trail.
The optional IMAGE socket is intended for a quick single-image workflow. For multiple images, use **Input - image path** with a folder. CaptionForge 1.0.2 does not claim heterogeneous mixed-aspect IMAGE-list support: generic ComfyUI IMAGE batchers may resize or crop images to a common tensor shape before CaptionForge receives them. Native heterogeneous image-list handling is deferred to future/v2 work.
---
## Validation for 1.0.1
The final 1.0.1 release-preparation pass included:
- Planner ownership/default tests
- Pass A artifact and source-identity tests
- Orchestrator output-contract tests
- SHORT/Formatter behavior tests
- configuration and release-polish tests
- canonical workflow structural verification
- full real-model smoke testing with Joy, Qwen, Ollama, Distiller, Validator, and Formatter
The final audited smoke test processed both Planner folder input and optional IMAGE input through the complete pipeline with:
```text
config/captionforge_ollama_models.json
final_ok = 6
final_failed = 0
```
defines user-editable Ollama model tags for dropdowns used by distiller, validator, formatter, and Ollama caption-witness nodes.
The final release-preparation verification completed with **51 focused tests passing**.
Example:
---
```json
{
"distiller_models": [
"mistral-small:24b",
"VladimirGav/gemma4-26b-16GB-VRAM-Uncensored",
"deepseek-r1:32b",
"tarruda/neuraldaredevil-8b-abliterated:fp16",
"gpt-oss:20b"
],
"validator_models": [
"gemma4:26b",
"qwen3.6:35B-A3B",
"huihui_ai/gemma-4-abliterated:26b"
],
"format_models": [
"mistral-small:24b",
"VladimirGav/gemma4-26b-16GB-VRAM-Uncensored",
"gpt-oss:20b",
"deepseek-r1:32b"
],
"caption_models": [
"gemma4:26b",
"qwen3.6:35B-A3B",
"huihui_ai/gemma-4-abliterated:26b"
],
"defaults": {
"distiller_model": "mistral-small:24b",
"validator_model": "gemma4:26b",
"format_model": "mistral-small:24b",
"caption_model": "gemma4:26b"
},
"include_custom": true
}
```
## Caption quality expectations
Terminology:
CaptionForge reduces many common single-captioner failure modes, but it does not make captioning perfect.
```text
distiller_model text-only LLM for Pass B distillation
validator_model image-aware VLM for Pass C validation
format_model text-only LLM for formatting/taggy conversion when used
caption_model Ollama-backed Pass A image-caption witness model
```
Vision-language models can still miss subtle colors, confuse nearby objects, preserve a plausible but incorrect consensus, omit details, or phrase the same visible fact differently between runs.
Values should be concrete Ollama model tags used exactly as written.
For large LoRA datasets, the goal is to improve the bulk quality and auditability of generated captions enough that remaining small errors are manageable.
## Output layout
---
CaptionForge writes auditable run artifacts and final sidecars during planned runs.
## License
A typical planned run uses this structure:
CaptionForge is released under the **MIT License**.
```text
<output_root>/
opt_images/
comfy_image_0000.png
comfy_image_0000_long.txt
comfy_image_0000_short.txt
comfy_image_0000_taggy.txt
comfy_image_0001.png
comfy_image_0001_long.txt
comfy_image_0001_short.txt
comfy_image_0001_taggy.txt
See [`LICENSE`](LICENSE) for details.
<run_name>__working/
<run_name>__A_RAW_CAPTIONS.jsonl
<run_name>__B_DISTILL.jsonl
<run_name>__B_DISTILL_readable.jsonl
<run_name>__B_DISTILL_readable.json
<run_name>__B_DISTILL_prompts.jsonl
<run_name>__C_VLM_VALIDATED.jsonl
<run_name>__C_VLM_VALIDATED_readable/
<run_name>__C_VLM_VALIDATOR_prompts.jsonl
<run_name>__D_FINAL_EXPORT.jsonl
<run_name>__output_paths.json
<run_name>__run_config.json
```
Folder-input images keep their source locations, and final TXT sidecars are written beside those original images.
Optional direct `IMAGE` inputs are copied into a visible output-root folder:
```text
<output_root>/opt_images/
```
Final caption sidecars are written beside the resolved source image. For folder-input images, that means beside the original image. For optional direct images, that means beside the saved optional image inside `opt_images/`.
Final sidecars currently include:
```text
<image_stem>_long.txt
<image_stem>_short.txt
<image_stem>_taggy.txt
```
Meaning:
```text
_long.txt the authoritative VLM-validated natural caption
_short.txt a shorter LoRA-length caption derived from the long caption
_taggy.txt a compact comma-separated taggy caption derived from the long caption
```
Long captions are intentional in v0.1.x. The current release-candidate strategy favors preserving visible, trainable detail in the validated long caption, then deriving shorter and taggy outputs from that result.
Exact JSONL schemas may evolve during the preview phase.
## Dependencies
Python dependencies are declared in `pyproject.toml` where applicable.
Typical local use may involve:
```text
torch
transformers
accelerate
huggingface-hub
pillow
numpy
safetensors
qwen-vl-utils
```
Optional quantization support may involve:
```text
bitsandbytes
```
Ollama-backed stages require a working local Ollama installation and installed Ollama model tags.
## Hardware notes
CaptionForge is designed for local workflows, but strong results may require large local models.
Practical performance depends on:
- GPU VRAM
- system RAM
- model size
- quantization mode
- Ollama version
- context length
- image size
- number of Pass A witness runs
- whether models are kept loaded or unloaded between runs
The author's active development environment includes an RTX 4090 Laptop GPU with 16 GB VRAM. Larger models may be slow, may require careful quantization, or may need more capable hardware.
## Experimental branches
Some experimental or unsupported code may exist in the repository for future A/B testing or research.
Experimental branches should be:
- clearly labeled
- kept out of the normal ComfyUI registration path
- not imported by `__init__.py`
- not shown as mainline nodes unless deliberately enabled
- treated as unsupported starting points rather than stable user features
The active public workflow should be the main Planner → Pass A witnesses → Distiller → VLM Validator → Export path.
## Development principles
CaptionForge currently prioritizes:
- local execution
- auditable intermediate records
- JSONL sidecars
- reusable engines separated from ComfyUI node wrappers
- planner-driven workflows
- model cache and VRAM hygiene
- strong defaults for LoRA captioning
- explicit prompt roles
- model-agnostic backends
- visible, trainable detail over generic caption prose
- practical feedback from real datasets
## Feedback wanted
Useful feedback includes:
- comparisons against standalone JoyCaption, Qwen, or other captioners
- examples where CaptionForge improves caption quality
- examples where CaptionForge makes captions worse
- hallucination reports
- missed-detail reports
- model recommendations
- prompt improvements
- broken node reports
- workflow usability feedback
- VRAM/performance observations
- JSONL/audit trail suggestions
Please include enough context to reproduce the issue or evaluate the result: selected nodes, model tags, relevant settings, whether the run used direct IMAGE input or a folder path, and a small sample of generated captions when possible.
## Attribution & License
Concept and implementation by **J. L. Córdova**, with development assistance from **ChatGPT (OpenAI)**.
CaptionForge's Joy/template-option workflow is locally adapted and was inspired in part by the practical template interface pattern used by the public JoyCaption Beta One Hugging Face Space:
```text
https://huggingface.co/spaces/fffiloni/JoyCaption-Beta-One
```
Copyright (c) 2026 J. L. Córdova
Released under the **MIT License**. See [`LICENSE`](./LICENSE) for details.
+9 -21
View File
@@ -8,11 +8,9 @@ CaptionForge — ComfyUI Package Entry Point
- Repository
https://github.com/Damkohler/CaptionForge
- CaptionForge focuses on practical dataset-captioning infrastructure for
LoRA dataset preparation, using multi-engine caption generation, JSONL
audit trails, claim extraction and refinement, text-LLM distillation,
image-aware VLM validation, and consensus-oriented caption improvement
to produce grounded, auditable training captions.
- CaptionForge 1.0 provides a production A/B/C/D caption pipeline for LoRA
dataset preparation: independent witness captions, text-LLM synthesis,
image-aware VLM validation, SHORT/TAGGY formatting, and JSONL audit trails.
- Package Purpose
- This file is the ComfyUI registration entry point for the CaptionForge
@@ -27,12 +25,8 @@ CaptionForge — ComfyUI Package Entry Point
• display the nodes in the Add Node menu
• register the package as a unified CaptionForge node collection
- The package currently registers:
• JLC Qwen Caption
• JLC Joy Caption
• JLC Qwen Caption (Lite)
• JLC Joy Caption (Lite)
• JLC CaptionForge Claim Extractor
- The package registers the Pipeline Planner, Template Options helper,
Joy/Qwen/Ollama Pass-A caption nodes, and the CaptionForge Orchestrator.
- Package Structure
- CaptionForge keeps ComfyUI-facing node wrappers separate from reusable
@@ -46,12 +40,8 @@ CaptionForge — ComfyUI Package Entry Point
export, model cache behavior, and Pass B claim extraction.
- Web / Icon Assets
- JavaScript/icon registration is intentionally disabled in this entry point
for now to avoid frontend branding conflicts while the package structure
stabilizes.
- A future version may re-enable WEB_DIRECTORY after confirming that the
CaptionForge frontend assets do not conflict with other JLC node packages.
- ``WEB_DIRECTORY`` exposes the package's current ComfyUI frontend assets.
Node registration remains entirely explicit through the mappings below.
- Attribution & License
- Concept and implementation by **J. L. Córdova**
@@ -136,12 +126,10 @@ NODE_DISPLAY_NAME_MAPPINGS.update(CAPTIONFORGE_DISPLAY_NAME_MAPPINGS)
WEB_DIRECTORY = "./web"
CAPTIONFORGE_ICON = "⚒"
CAPTIONFORGE_PREFIX = f"{CAPTIONFORGE_ICON} CaptionForge"
print(f"{CAPTIONFORGE_PREFIX} loaded ({len(NODE_CLASS_MAPPINGS)} nodes)")
print(f"CaptionForge loaded ({len(NODE_CLASS_MAPPINGS)} nodes)")
__all__ = [
"NODE_CLASS_MAPPINGS",
"NODE_DISPLAY_NAME_MAPPINGS",
"WEB_DIRECTORY",
]
]
File diff suppressed because it is too large Load Diff
Binary file not shown.

Before

Width:  |  Height:  |  Size: 1.8 MiB

@@ -0,0 +1,403 @@
{
"51": {
"inputs": {
"model": "llama-joycaption-beta-one-hf-llava",
"memory_mode": "Balanced (8-bit)",
"keep_loaded": true,
"caption_template_mode": true,
"caption_type": "JLC LoRA Literal",
"caption_length": "any",
"custom_prompt_mode": false,
"prompt_preset": "default_literal",
"system_prompt": "You are a helpful image-captioning assistant. Describe only what is visible in the image. Do not invent unseen context.",
"custom_prompt": "",
"max_new_tokens": 384,
"temperature": 0.75,
"top_p": 0.9,
"top_k": 50,
"repetition_penalty": 1,
"max_size": 1024,
"forbidden_phrases": "",
"replace_pairs": "",
"download_probe_only": false,
"image": [
"148",
0
],
"pipeline_plan": [
"148",
1
],
"template_options": [
"55",
0
]
},
"class_type": "JLC_CaptionForgeJoy",
"_meta": {
"title": " JLC CaptionForge Joy Caption"
}
},
"52": {
"inputs": {
"model": "Qwen2.5-VL-7B-NSFW-Caption-V3-abliterated",
"qwen_quantization": "Balanced (8-bit)",
"keep_loaded": true,
"caption_template_mode": true,
"caption_type": "LoRA Literal",
"caption_length": "any",
"custom_prompt_mode": false,
"prompt_preset": "default_literal",
"system_prompt": "You are a helpful image-captioning assistant. Describe only what is visible in the image. Do not invent unseen context.",
"custom_prompt": "",
"max_new_tokens": 384,
"temperature": 0.75,
"top_p": 0.9,
"top_k": 50,
"repetition_penalty": 1.08,
"max_size": 1024,
"forbidden_phrases": "",
"replace_pairs": "",
"download_probe_only": false,
"image": [
"51",
0
],
"pipeline_plan": [
"51",
1
],
"template_options": [
"51",
2
]
},
"class_type": "JLC_CaptionForgeQwen",
"_meta": {
"title": " JLC CaptionForge Qwen Caption"
}
},
"55": {
"inputs": {
"name_input__replacement_for_name": "",
"option_01__character_name_required": false,
"option_02__literal_visible_details_only": true,
"option_03__no_roleplay_or_dialogue": true,
"option_04__no_ambiguous_wording": true,
"option_05__avoid_meta_lead_ins": true,
"option_06__describe_facial_features": true,
"option_07__eye_color_vs_makeup": true,
"option_08__describe_makeup_separately": true,
"option_09__describe_expression": true,
"option_10__hair_color_and_style": true,
"option_11__skin_tone_and_texture": true,
"option_12__body_shape_and_proportions": true,
"option_13__visible_anatomy_and_coverage": true,
"option_14__stylized_material_traits": true,
"option_15__clothing_pieces_clearly": true,
"option_16__clothing_materials_textures": true,
"option_17__precise_colors": false,
"option_18__accessories_and_jewelry": false,
"option_19__pose_and_stance": true,
"option_20__hands_and_arms": true,
"option_21__framing_shot_distance": true,
"option_22__viewing_angle": true,
"option_23__background_and_setting": true,
"option_24__lighting": false,
"option_25__image_style_medium": true,
"option_26__watermark_only_if_present": false,
"option_27__visible_text_if_relevant": true,
"option_28__sfw_suggestive_nsfw_rating": false,
"option_29__neutral_sensual_details": true,
"option_30__vulgar_wording_allowed": true,
"option_31__no_polite_euphemisms": true
},
"class_type": "JLC_CaptionForgeExtraOptions",
"_meta": {
"title": " JLC CaptionForge Template Options"
}
},
"56": {
"inputs": {
"model": "custom",
"custom_model_tag": "starnodes/qwen3.8-vl-27b-promptrefiner-abliterated",
"ollama_url": "http://127.0.0.1:11434",
"keep_loaded": true,
"caption_template_mode": true,
"caption_type": "Descriptive",
"caption_length": "any",
"custom_prompt_mode": false,
"prompt_preset": "female_character_lora",
"system_prompt": "You are a multimodal image captioning engine for female character NSFW LoRA dataset preparation.\n\nYour job is to inspect the image and write one strong standalone NSFW caption. Use sexual explicit language without euphemisms. When clearly visible, describe anatomy or actions in explicit terms, using words like pussy, blowjob, masturbating, dildo, fingering, spread, pubic hair, shaved pussy, horny, big breasts, anal, or similar. Do not moralize. No censorship.\n\nRules:\n- Describe only visible image content.\n- Output exactly one paragraph and nothing else.\n- Do not output analysis, reasoning, notes, bullet points, labels, or a thinking trace.\n- Do not roleplay, address the user, ask follow-up questions, or include safety disclaimers.\n- If a detail is uncertain, omit it rather than hedge.\n- Use direct dataset-caption language, not chatty commentary.",
"custom_prompt": "Write one clean caption for a female character LoRA dataset.\n\nCaption priorities:\n- Be specific rather than generic.\n- Prioritize female-character-LoRA-relevant visual traits, preserving NSFW content and explicit language: subject type, pose, facial structure, expression, hair color and hairstyle, eye color, makeup color as distinct from eye color, lip shape and fullness, skin appearance, neck length, shoulder line, bust/chest shape, waist definition, hip width, body proportions, silhouette, legs, clothing construction, accessories, dominant colors, lighting, background, framing, and visual style.\n- Include visible anatomy and figure information when clear and relevant. Do not omit body-shape or facial-structure details merely because they are sensual, revealing, or anatomy-related.\n- If the image presents glamour, sensuality, revealing clothing, lingerie, swimwear, cleavage, underboob, sideboob, bare midriff, exposed hips, thighs, or other sexualized styling, describe that directly and matter-of-factly when visible.\n- Use strong, open dataset-caption language rather than timid euphemisms, but do not turn the caption into roleplay, commentary, or erotic prose.\n- Do not invent nudity, explicit acts, hidden anatomy, or hidden clothing details.\n- Do not force photo, render, doll, anime, realistic, or stylized language unless supported by the image.\n- Avoid meta phrases like \"this image shows\" or \"the picture depicts.\"\n- One paragraph only.\n\nFinal caption:",
"max_new_tokens": 6000,
"temperature": 0.9,
"top_p": 0.6,
"top_k": 80,
"repetition_penalty": 1.03,
"max_size": 1024,
"request_timeout_seconds": 1800,
"forbidden_phrases": "",
"replace_pairs": "",
"download_probe_only": false,
"image": [
"52",
0
],
"pipeline_plan": [
"52",
1
],
"template_options": [
"52",
2
]
},
"class_type": "JLC_CaptionForgeOllamaCaption",
"_meta": {
"title": " JLC CaptionForge Ollama Caption"
}
},
"90": {
"inputs": {
"model": "custom",
"custom_model_tag": "huihui_ai/qwen3.6-abliterated:35b",
"ollama_url": "http://127.0.0.1:11434",
"keep_loaded": true,
"caption_template_mode": true,
"caption_type": "Descriptive",
"caption_length": "any",
"custom_prompt_mode": false,
"prompt_preset": "female_character_lora",
"system_prompt": "You are a multimodal image captioning engine for female character NSFW LoRA dataset preparation.\n\nYour job is to inspect the image and write one strong standalone NSFW caption. Use sexual explicit language without euphemisms. When clearly visible, describe anatomy or actions in explicit terms, using words like pussy, blowjob, masturbating, dildo, fingering, spread, pubic hair, shaved pussy, horny, big breasts, anal, or similar. Do not moralize. No censorship.\n\nRules:\n- Describe only visible image content.\n- Output exactly one paragraph and nothing else.\n- Do not output analysis, reasoning, notes, bullet points, labels, or a thinking trace.\n- Do not roleplay, address the user, ask follow-up questions, or include safety disclaimers.\n- If a detail is uncertain, omit it rather than hedge.\n- Use direct dataset-caption language, not chatty commentary.",
"custom_prompt": "Write one clean caption for a female character LoRA dataset.\n\nCaption priorities:\n- Be specific rather than generic.\n- Prioritize female-character-LoRA-relevant visual traits, preserving NSFW content and explicit language: subject type, pose, facial structure, expression, hair color and hairstyle, eye color, makeup color as distinct from eye color, lip shape and fullness, skin appearance, neck length, shoulder line, bust/chest shape, waist definition, hip width, body proportions, silhouette, legs, clothing construction, accessories, dominant colors, lighting, background, framing, and visual style.\n- Include visible anatomy and figure information when clear and relevant. Do not omit body-shape or facial-structure details merely because they are sensual, revealing, or anatomy-related.\n- If the image presents glamour, sensuality, revealing clothing, lingerie, swimwear, cleavage, underboob, sideboob, bare midriff, exposed hips, thighs, or other sexualized styling, describe that directly and matter-of-factly when visible.\n- Use strong, open dataset-caption language rather than timid euphemisms, but do not turn the caption into roleplay, commentary, or erotic prose.\n- Do not invent nudity, explicit acts, hidden anatomy, or hidden clothing details.\n- Do not force photo, render, doll, anime, realistic, or stylized language unless supported by the image.\n- Avoid meta phrases like \"this image shows\" or \"the picture depicts.\"\n- One paragraph only.\n\nFinal caption:",
"max_new_tokens": 6000,
"temperature": 0.9,
"top_p": 0.6,
"top_k": 80,
"repetition_penalty": 1.03,
"max_size": 1024,
"request_timeout_seconds": 1800,
"forbidden_phrases": "",
"replace_pairs": "",
"download_probe_only": false,
"image": [
"56",
0
],
"pipeline_plan": [
"56",
1
],
"template_options": [
"56",
2
]
},
"class_type": "JLC_CaptionForgeOllamaCaption",
"_meta": {
"title": " JLC CaptionForge Ollama Caption"
}
},
"147": {
"inputs": {
"Input - captions JSONL": "",
"Input - image path": "",
"Input - include caption families": "joy,qwen,ollama",
"Input - max captions per family": 5,
"Input - max total captions": 20,
"Output - folder": "",
"Output - run name": "captionforge_run",
"Output - overwrite outputs": true,
"Ollama - URL": "http://127.0.0.1:11434",
"Ollama - keep loaded": true,
"Ollama - request timeout seconds": 1800,
"LoRA - trigger word": "",
"LoRA - user caption anchor": "",
"Cleanup - forbidden phrases": "",
"Cleanup - replace pairs": "",
"Fat Draft - model": "mistral-small:24b",
"Fat Draft - custom Ollama model": "",
"Fat Draft - prompt": "/no_think\n\nYou are a detail-preserving caption merger for LoRA dataset preparation.\n\nYou receive multiple captions of the same image. You do NOT see the image.\n\nTask:\nMerge all non-contradictory caption details into one deliberately over-complete draft caption.\n\nRules:\n- Do not validate against the image.\n- Do not decide that details are false just because they appear once.\n- Do not summarize aggressively.\n- Preserve concrete details from all captions.\n- Split contradictions by choosing cautious wording or listing the alternative only when needed.\n- Prefer specific visual language over generic language.\n- Keep visible body, clothing, material, accessory, color, pose, lighting, style, and framing details.\n- Preserve doll-like, glossy/plastic-like, material, garment-construction, body-shape, and facial-feature details when present.\n- Use neutral dataset-caption language, including visible sensual styling or revealing clothing when present.\n- Do not add details absent from the captions.\n- Treat subject names or trigger-like identity tokens as optional identity labels. Preserve them only when they appear consistently in the captions; do not let them replace visible description.\n- Output only one paragraph, no notes, no JSON.",
"Fat Draft - max caption chars": 1536,
"Fat Draft - max new tokens": 3096,
"Fat Draft - temperature": 0.24,
"Fat Draft - top p": 0.9,
"Fat Draft - top k": 60,
"Validator - model": "gemma4:26b",
"Validator - custom Ollama model": "",
"Validator - system prompt": "/no_think\nYou are a direct image validation engine. Inspect the image and answer only with the requested caption.",
"Validator - prompt": "/no_think\n\nLook at the image and validate this draft caption.\n\nTask:\nReturn a corrected caption paragraph that keeps only image-supported details.\n\nRules:\n- Output only the corrected caption.\n- One paragraph.\n- No reasoning, no notes, no JSON.\n- Keep all true visible details from the draft.\n- Delete unsupported details.\n- Correct small visible errors.\n- Do not add new details unless needed to correct an error already present.\n- Preserve useful LoRA details: subject, face, hair, eyes, makeup, lips, skin texture, pose, body shape, outfit, accessories, materials, colors, lighting, background, framing, and visual style.\n- Visible sensual styling, revealing clothing, cleavage, thighs, bare skin, swimwear, lingerie, or body-shape details may be described neutrally when present.\n- Do not invent hidden anatomy, unseen clothing, explicit acts, or details contradicted by the image.",
"Validator - max new tokens": 2112,
"Validator - max image size": 1024,
"Validator - temperature": 0,
"Validator - top p": 0.92,
"Validator - top k": 80,
"Formatter - model": "mistral-small:24b",
"Formatter - custom Ollama model": "",
"Formatter - prompt": "/no_think\n\nYou are a LoRA caption format converter. The validated paragraph is your only source of truth.\n\nOutput exactly two labeled lines:\n\nSHORT: <a concise natural-language caption, typically around 100 words, that preserves all LoRA-useful validated details>\n\nTAGGY: <one compact comma-separated caption>\n\nSHORT must preserve the image's distinctive training identity across the whole source:\n1. subject, defining face/hair/body traits, and every major outfit piece/material;\n2. pose/action and key accessories or unusual visible details;\n3. setting, lighting, framing, and visual medium/style.\n\nOmit a category only when absent. Use only source details; never add, infer, euphemize, or correct. Compress wording, not category coverage. Do not copy only the source opening.\n\nAim for roughly 100 words. Keep it concise, but allow modest variation when needed to preserve important information and finish the caption naturally. Do not cut off a sentence merely to satisfy a word-count target.\n\nTAGGY must preserve all concrete LoRA-useful source details as compact comma-separated phrases.\n\nNo markdown, reasoning, notes, or other labels.",
"Formatter - max new tokens": 3200,
"Formatter - temperature": 0.12,
"Formatter - top p": 0.88,
"Formatter - top k": 50,
"Audit - write prompt JSONL": false,
"Audit - preserve raw responses": false,
"Final - TXT export format": "natural",
"Final - write TXT sidecars": true,
"Final - write JSONL": true,
"Dataset - export image and caption": false,
"Dataset - output folder": "",
"Dataset - max image size": 0,
"Dataset - dimension divisor": 16,
"Dataset - image format": "PNG",
"Dataset - JPEG quality": 95,
"Dataset - caption": "short",
"Input - single image": [
"90",
0
],
"pipeline_plan": [
"90",
1
]
},
"class_type": "JLC_CaptionForge",
"_meta": {
"title": " JLC CaptionForge Orchestrator"
}
},
"148": {
"inputs": {
"Planner - enabled": true,
"Input - image path": "",
"Input - recursive": false,
"Input - filename glob": "*",
"Output - folder": "",
"Output - run name": "captionforge_run",
"Output - overwrite outputs": true,
"Ollama - URL": "http://127.0.0.1:11434",
"Ollama - keep loaded": true,
"Ollama - request timeout seconds": 1800,
"LoRA - trigger word": "Fairy",
"LoRA - user caption anchor": "",
"Cleanup - forbidden phrases": "",
"Cleanup - replace pairs": "",
"Caption - Joy runs/image": "2",
"Caption - Qwen runs/image": "1",
"Caption - Ollama runs/image": "1",
"Caption - base seed": 1,
"Caption - seed mode": "increment",
"Caption - temperature schedule": "0.75,0.85,0.95",
"Caption - top p schedule": "0.60",
"Caption - top k schedule": "80",
"Caption - max image size": 1024,
"Caption - max new tokens": 4096,
"Distiller - model": "mistral-small:24b",
"Distiller - custom Ollama model": "",
"Distiller - prompt": "/no_think\n\nYou are a detail-preserving caption merger for LoRA dataset preparation.\n\nYou receive multiple captions of the same image. You do NOT see the image.\n\nTask:\nMerge all non-contradictory caption details into one deliberately over-complete draft caption.\n\nRules:\n- Do not validate against the image.\n- Do not decide that details are false just because they appear once.\n- Do not summarize aggressively.\n- Preserve concrete details from all captions.\n- Split contradictions by choosing cautious wording or listing the alternative only when needed.\n- Prefer specific visual language over generic language.\n- Keep visible body, clothing, material, accessory, color, pose, lighting, style, and framing details.\n- Preserve doll-like, glossy/plastic-like, material, garment-construction, body-shape, and facial-feature details when present.\n- Use neutral dataset-caption language, including visible sensual styling or revealing clothing when present.\n- Do not add details absent from the captions.\n- Treat subject names or trigger-like identity tokens as optional identity labels. Preserve them only when they appear consistently in the captions; do not let them replace visible description.\n- Output only one paragraph, no notes, no JSON.",
"Distiller - seed": -1,
"Distiller - max caption chars for LLM": 1536,
"Distiller - num predict": 3096,
"Distiller - temperature": 0.35,
"Distiller - top p": 0.92,
"Distiller - top k": 80,
"Distiller - write prompt JSONL": false,
"Distiller - preserve raw response": false,
"Validator - model": "gemma4:26b",
"Validator - custom Ollama model": "",
"Validator - system prompt": "/no_think\nYou are a direct image validation engine. Inspect the image and answer only with the requested caption.",
"Validator - prompt": "/no_think\n\nLook at the image and validate this draft caption.\n\nTask:\nReturn a corrected caption paragraph that keeps only image-supported details.\n\nRules:\n- Output only the corrected caption.\n- One paragraph.\n- No reasoning, no notes, no JSON.\n- Keep all true visible details from the draft.\n- Delete unsupported details.\n- Correct small visible errors.\n- Do not add new details unless needed to correct an error already present.\n- Preserve useful LoRA details: subject, face, hair, eyes, makeup, lips, skin texture, pose, body shape, outfit, accessories, materials, colors, lighting, background, framing, and visual style.\n- Visible sensual styling, revealing clothing, cleavage, thighs, bare skin, swimwear, lingerie, or body-shape details may be described neutrally when present.\n- Do not invent hidden anatomy, unseen clothing, explicit acts, or details contradicted by the image.",
"Validator - seed": -1,
"Validator - num predict": 2112,
"Validator - temperature": 0,
"Validator - top p": 0.92,
"Validator - top k": 80,
"Validator - write prompt JSONL": false,
"Validator - preserve raw VLM response": false,
"Formatter - model": "mistral-small:24b",
"Formatter - custom Ollama model": "",
"Formatter - prompt": "/no_think\n\nYou are a LoRA caption format converter. The validated paragraph is your only source of truth.\n\nOutput exactly two labeled lines:\n\nSHORT: <a concise natural-language caption, typically around 100 words, that preserves all LoRA-useful validated details>\n\nTAGGY: <one compact comma-separated caption>\n\nSHORT must preserve the image's distinctive training identity across the whole source:\n1. subject, defining face/hair/body traits, and every major outfit piece/material;\n2. pose/action and key accessories or unusual visible details;\n3. setting, lighting, framing, and visual medium/style.\n\nOmit a category only when absent. Use only source details; never add, infer, euphemize, or correct. Compress wording, not category coverage. Do not copy only the source opening.\n\nAim for roughly 100 words. Keep it concise, but allow modest variation when needed to preserve important information and finish the caption naturally. Do not cut off a sentence merely to satisfy a word-count target.\n\nTAGGY must preserve all concrete LoRA-useful source details as compact comma-separated phrases.\n\nNo markdown, reasoning, notes, or other labels.",
"Formatter - seed": -1,
"Formatter - num predict": 3200,
"Formatter - temperature": 0,
"Formatter - top p": 0.9,
"Formatter - top k": 60,
"Formatter - write prompt JSONL": false,
"Formatter - preserve raw response": false,
"Final - write TXT sidecars": true,
"Final - write JSONL": true,
"Dataset - export image and caption": true,
"Dataset - output folder": "",
"Dataset - max image size": 1536,
"Dataset - dimension divisor": 16,
"Dataset - image format": "PNG",
"Dataset - JPEG quality": 95,
"Dataset - caption": "short",
"Input - single image": [
"151",
0
]
},
"class_type": "JLC_CaptionForge_Pipeline_Planner",
"_meta": {
"title": " JLC CaptionForge Pipeline Planner"
}
},
"151": {
"inputs": {
"image": "jlc_CaptionForge.jpg",
"resize_by": "scale longer dimension",
"multiplier": 1,
"longer_size": 1536,
"shorter_size": 1024,
"width": 1024,
"height": 1024,
"megapixels": 1,
"scale_method": "area",
"divisible_by": 16
},
"class_type": "JLC_LoadAndResizeImage",
"_meta": {
"title": " JLC Load, Resize & Encode Image"
}
},
"156": {
"inputs": {
"mode": "raw value",
"input": [
"147",
0
]
},
"class_type": "DisplayAny",
"_meta": {
"title": "🔧 Display Any"
}
},
"157": {
"inputs": {
"mode": "raw value",
"input": [
"147",
1
]
},
"class_type": "DisplayAny",
"_meta": {
"title": "🔧 Display Any"
}
},
"158": {
"inputs": {
"mode": "raw value",
"input": [
"147",
2
]
},
"class_type": "DisplayAny",
"_meta": {
"title": "🔧 Display Any"
}
}
}
File diff suppressed because one or more lines are too long
Binary file not shown.

After

Width:  |  Height:  |  Size: 2.3 MiB

+3 -2
View File
@@ -1,2 +1,3 @@
# captionforge_version.py
CAPTIONFORGE_VERSION = "0.1.0"
"""Single authoritative package/release version for CaptionForge."""
CAPTIONFORGE_VERSION = "1.0.2"
+8 -10
View File
@@ -1,28 +1,27 @@
{
"_meta": {
"name": "CaptionForge Ollama Model Dropdowns",
"version": "0.1.0",
"version": "1.0.2",
"description": "User-editable Ollama model dropdown configuration for CaptionForge nodes and engines.",
"consumed_by": [
"nodes/captionforge_ollama_model_dropdowns.py",
"CaptionForge Pipeline Planner",
"JLC CaptionForge capstone",
"JLC CaptionForge Orchestrator",
"JLC CaptionForge Ollama Caption"
],
"notes": [
"Values should be concrete Ollama model tags used exactly as written.",
"distiller_models are used for text-only LLM distillation and formatting stages.",
"distiller_models are used for Pass B text-only LLM synthesis.",
"validator_models are used for image-aware VLM validation.",
"format_models are used for Pass D text-only SHORT and TAGGY formatting.",
"caption_models are used by Ollama-backed Pass A caption witness nodes.",
"Set include_custom to true to expose a custom model-tag entry in supported nodes."
]
},
"distiller_models": [
"mistral-small:24b",
"VladimirGav/gemma4-26b-16GB-VRAM-Uncensored",
"deepseek-r1:32b",
"tarruda/neuraldaredevil-8b-abliterated:fp16",
"gpt-oss:20b"
"gpt-oss:20b",
"VladimirGav/gemma4-26b-16GB-VRAM-Uncensored"
],
"validator_models": [
"gemma4:26b",
@@ -31,9 +30,8 @@
],
"format_models": [
"mistral-small:24b",
"VladimirGav/gemma4-26b-16GB-VRAM-Uncensored",
"gpt-oss:20b",
"deepseek-r1:32b"
"VladimirGav/gemma4-26b-16GB-VRAM-Uncensored"
],
"caption_models": [
"gemma4:26b",
@@ -47,4 +45,4 @@
"caption_model": "gemma4:26b"
},
"include_custom": true
}
}
+99
View File
@@ -0,0 +1,99 @@
# Dataset export prototype
This opt-in export mode extends the existing Pipeline Planner and Orchestrator.
## Try it
1. Reload your user-managed ComfyUI instance at port **8189** to load the Python changes.
2. Load `assets/workflows/CaptionForge_FullWorkflow_Rel_v1.0.2.json`, or add fresh
Planner and Orchestrator nodes to an existing workflow.
3. Set the Planner input path and output folder. The prototype follows a
**1536-pixel** caption/Validator maximum, with no enlargement.
4. Enable **Dataset - export image and caption** on the Planner. The supplied
canonical workflow exposes this control; ordinary node defaults leave it off.
5. Choose the divisor, image format, and training caption. Queue a small dataset.
When connected, the Planner owns **every** Dataset setting, including disabled,
zero, and blank values. Orchestrator controls apply only in standalone mode.
Older plans with no Dataset settings keep export disabled.
| Control on both nodes | Default | Behavior |
| --- | --- | --- |
| Dataset - export image and caption | Off | Enables paired image and plain TXT export |
| Dataset - output folder | Blank | Parent folder; blank uses the main Output folder |
| Dataset - max image size | 0 | Follows configured Validator size; positive values override export size only |
| Dataset - dimension divisor | 16 | Integer; rounds both edges down. 1 disables alignment |
| Dataset - image format | PNG | PNG or JPEG, RGB |
| Dataset - JPEG quality | 95 | Applies only to JPEG |
| Dataset - caption | short | short, long, or taggy for the matching plain TXT |
In planned runs, the effective Validator maximum comes from the Planner's
**Caption - max image size**. Standalone runs use **Validator - max image size**.
If both the export override and effective Validator limit are zero, there is no
long-edge cap; divisor alignment still applies. Neither mode enlarges images.
## Output and protection
For an input `portraits/photo.jpg`, PNG export creates:
```text
<selected output parent>/training_dataset/
.captionforge-dataset.json
files/portraits/
photo.jpg.png
photo.jpg.txt
photo.jpg.captionforge.json
photo.jpg_long.txt
photo.jpg_short.txt
photo.jpg_taggy.txt
```
The last three files follow **Final - write TXT sidecars**. The selected plain
training TXT is always written when Dataset export is enabled. With export on,
caption variants are written beside the exported image rather than into the
source archive. With export off, existing v1.0.1 sidecar behavior is preserved.
Original extensions remain in the export stem, so `photo.jpg` and `photo.png`
produce distinct image/caption pairs. Relative folders are retained. Optional
IMAGE inputs use a separate `optional/` namespace. Sources outside the declared
input root use an `external/` namespace derived from their parent directory.
The dataset folder must be empty or already owned by CaptionForge. Untracked
files are not overwritten even when overwrite is enabled. The input path may
not be inside the selected export dataset. Resolved destination paths must stay
within the dataset and cannot refer to the source image, including hard links.
All three witness scanners exclude marked datasets on subsequent runs, even
when export has since been disabled. Do not remove the ownership marker.
## Resize and resume
Both output dimensions are rounded down after proportional size calculation.
This introduces a small aspect-ratio adjustment without cropping. If either
edge would become zero, export reports an error and keeps the captions in the
final record; reduce the divisor to handle such images.
Validator pixels are reused when their dimensions exactly match the requested
export dimensions. Otherwise, export resizes directly from the original to
avoid repeated resampling. Validator retry sizes do not change export size.
With overwrite disabled, completed caption records can receive a new dataset
export without repeating B/C/D model calls. Existing pairs resume only when
source metadata, settings, caption, and output hashes match their receipt.
Changed or incomplete pairs require overwrite to regenerate.
When changing PNG/JPEG format for an existing dataset, choose another output
parent; the prototype refuses to leave duplicate training images sharing one caption.
Export failures
retain caption text and are reported in the final record. Each file is published
by replacement from a temporary file; a receipt written last records completion.
An interrupted pair is not treated as complete.
## Scope
This prototype does not encode training latents or caption embeddings. It does
not change release version numbers, production captioning defaults, or public
node outputs. Export paths and dimensions are included in `final_records` under
`dataset_export`. The existing JSONL audit pipeline remains available.
CPU checks cover resize limits, divisor errors, pairing, collisions, recursive
exclusion, Planner ownership, and caption-preserving resume. Live canvas and
real-model validation remain pending; port 8189 was not responding during development.
+117
View File
@@ -0,0 +1,117 @@
# Joy 8-bit dtype and clean installation
## Why Balanced (8-bit) retains BF16
`Balanced (8-bit)::bf16` describes quantized weights plus Joy's floating-point
configuration, not an all-INT8 computation. Joy's quantized loader uses
`torch_dtype="auto"` (the checkpoint dtype), excludes the vision tower and
multimodal projector from INT8 conversion, and uses BF16 autocast where supported.
The configuration dtype is also part of the cache key; it does not override
`auto` in the quantized loader. This existing behavior is unchanged.
JoyCaption's upstream ComfyUI implementation uses the same `auto` loading and
BF16 autocast. In bitsandbytes 0.46.1, `MatMul8bitLt.forward` explicitly casts
activations to FP16 for INT8 quantization, while retaining the input dtype for
other computation/output handling. This is an expected internal conversion,
not sufficient evidence that the entire Joy model should switch to FP16.
Changing the vision tower, projector and surrounding floating-point computation
just to remove the warning would change numerical behavior without a quality
validation basis.
References:
- [JoyCaption native BF16 usage](https://github.com/fpgaminer/joycaption/blob/main/README.md)
- [Upstream JoyCaption ComfyUI implementation](https://github.com/fpgaminer/joycaption_comfyui/blob/main/nodes.py)
- [bitsandbytes 0.46.1 implementation](https://github.com/bitsandbytes-foundation/bitsandbytes/blob/0.46.1/bitsandbytes/autograd/_functions.py)
The desktop reproduction also showed FP32 input activations. The BF16 cache key
is not a promise that every intermediate tensor is BF16: autocast applies per
operation, and normalization or type promotion can produce FP32 intermediates.
The exact producing layer cannot be identified from the supplied log.
Bitsandbytes accepts these through the same explicit FP16 quantization conversion,
so both observed messages are handled without changing numerical behavior.
See the [current bitsandbytes implementation](https://github.com/bitsandbytes-foundation/bitsandbytes/blob/main/bitsandbytes/autograd/_functions.py).
## Warning scope
The old import-time filters were removed. During Balanced (8-bit) generation
only, CaptionForge ignores these exact `UserWarning` messages or WARNING log records from
`bitsandbytes.autograd._functions`:
```text
MatMul8bitLt: inputs will be cast from torch.bfloat16 to float16 during quantization
MatMul8bitLt: inputs will be cast from torch.float32 to float16 during quantization
```
Other cast dtypes, different messages, other modules, other warning categories,
processor/loading/cleanup diagnostics and Default-mode warnings remain visible.
The caller's filters are restored even if generation fails. Recent bitsandbytes versions emit this through `logger.warning` instead of
`warnings.warn`; both routes are covered. The logging filter matches the fully
formatted message and originating logger, and applies only to the inference
thread. It is removed on normal exit or failure, without changing logger levels,
handlers, propagation, or pre-existing filters. This does not alter tensors,
quantization parameters or model outputs.
The same scope is used around quantized Qwen generation. Qwen's previous
load-time warning filter was process-wide, BF16-only, and did not cover the
bitsandbytes logging route; it has been removed. Non-quantized Qwen generation
does not install the scope.
Python 3.10-3.12 warning filters are process-wide while a `catch_warnings` scope
is active. An identical warning from concurrent bitsandbytes work could therefore
also be suppressed during that interval. This is scoped filtering, not a claim
of thread-local warning isolation.
## Installation declarations
`accelerate` was already required by `pyproject.toml`; declaring it there alone
did not cover Manager's requirements-file installation path. CaptionForge now
ships `requirements.txt` with the same runtime dependencies as
`project.dependencies`, including `accelerate` and `bitsandbytes>=0.46.1`.
The `quantization` extra is retained for compatibility with existing install
commands, but bitsandbytes no longer requires opting into an extra.
[ComfyUI Manager's installation code](https://github.com/Comfy-Org/ComfyUI-Manager/blob/main/glob/manager_core.py)
reads `requirements.txt` and installs its entries. Both declarations are kept
explicit so Manager can process individual requirement lines. A regression test
prevents the lists from drifting. The lower bound does not guarantee compatibility
with every future torch/CUDA/bitsandbytes combination.
For an existing installation, run this from CaptionForge's directory using
**the Python interpreter belonging to ComfyUI Desktop's environment**, then
restart ComfyUI:
```powershell
python -m pip install -r requirements.txt
python -m pip check
```
This affects new Registry installs after a release containing these changes is
published; editing the repository does not update an already-published archive.
## Validation
CPU-only focused checks (Python 3.11/3.12; Python 3.10 also needs `tomli`):
```text
python -m unittest discover -s tests -p test_joy_hardening.py -v
```
They cover six repeated caption bursts through both warnings and logging, near-match diagnostics,
Default mode, filter restoration, exceptions, and matching dependency lists.
Generation tests execute the real `caption_pil` method with test doubles for
processor, model and torch; they do not test numerical inference.
Before releasing, validate on a clean supported ComfyUI Desktop environment:
1. Install the updated package through Manager without manually installing
accelerate/bitsandbytes. Confirm both are present using that environment's
`python -m pip show accelerate bitsandbytes`, then run `python -m pip check`.
2. Run Joy Balanced (8-bit), two runs over three ordinary test images. Confirm
six successful captions, normal progress/error diagnostics, and no BF16 cast
warning bursts. Reuse the loaded model for a subsequent run as well.
3. Check Default mode if sufficient VRAM is available. No warning policy or
numerical settings should change there.
A real clean Desktop install and CUDA caption generation cannot be established
by the CPU-only checks; they remain release validation steps.
+157
View File
@@ -0,0 +1,157 @@
# Fixed-corpus caption quality study
Date: 2026-09-09
Baseline: `5568f74dbb103b99bd6603481a247aee6033f753`
Release disposition: the dual-output Pass-D refinement described below was
adopted for CaptionForge 1.0.0. The corpus name is retained as historical study
provenance; it is not the package version.
## Question
Evaluate the current `_long.txt`, `_short.txt`, and `_taggy.txt` outputs before
the 1.0 release. In particular, determine whether short captions retain useful
LoRA-training detail and whether taggy captions are compact, faithful, and
readable enough to keep without redesign.
## Corpus and method
The fixed corpus is the 19-image local `Release_1.0.1_BigTest01` set. Despite
the historical folder name, this is pre-1.0 release-candidate evidence. It
contains real photographs, digital illustrations, anime-style art, glossy 3D
characters, a ball-jointed doll, portraits, full-body images, stage/runway
scenes, indoor and outdoor settings, simple and detailed backgrounds, varied
poses, and varied clothing/material descriptions.
The stored run used the production model path:
- Pass A: one Joy and one Qwen caption per image
- Pass B: `mistral-small:24b`
- Pass C: `gemma4:26b`
- Pass D: `mistral-small:24b`
The study combined mechanical measurements with a manual image/caption review.
The proposed refinement was then tested directly against all 19 validated long
captions with `mistral-small:24b`, seed 3, temperature 0.12, top-p 0.88, and
top-k 50.
## Existing output measurements
| Metric | Result |
| --- | ---: |
| Images with complete long/short/taggy triplets | 19 |
| Mean long length | 127.3 words |
| Mean short length | 81.6 words |
| Mean short/long ratio | 66.6% |
| Shorts that are exact prefixes of long | 19/19 |
| Shorts that drop content | 16/19 |
| Shorts that drop at least 20 words | 14/19 |
| Shorts identical to long | 3/19 |
| Mean taggy length | 67.8 words / 28.5 items |
| Taggy range | 18–36 items |
| Largest taggy output | 648 characters |
| Exact duplicate taggy items | 0 |
| Taggy outputs reaching a compaction limit | 0 |
## Findings
### Long
`_long` is appropriately treated as authoritative. It consistently contains
the richest set of subject, appearance, clothing, pose, setting, lighting,
framing, and style details. This study did not identify a downstream-formatting
reason to change Pass C.
Manual image review did find isolated Pass-C wording errors (for example, a
doll support stand described as a microphone stand). Those are validator
accuracy issues, not short/taggy divergence, and are outside this formatting
study.
### Short: material problem found
The deterministic implementation is sentence-prefix extraction, not semantic
compression. Every stored short is the beginning of its long caption. Because
the validator commonly places setting, lighting, framing, style, accessories,
and some pose/body details near the end, those categories are systematically
more likely to disappear.
Representative losses include:
- the entire outfit, exposed midriff/thighs, pose, and shiny fabric detail from
the pink bikini/sarong illustration;
- mirror-selfie action, phone/hand placement, background, lighting, and
anime/semi-realistic rendering details;
- body shape, side-view pose, runway, crowd, sign text, reflective floor,
full-body framing, and doll-like style;
- sailor hat, high heels, salute, relaxed arm, full-body framing, and studio
lighting;
- castle, lanterns, trees, moon, stars, river, lighting, and 3D style from the
fairy image.
There is also a concrete limit defect: when a validated long caption is a
single sentence longer than 90 words, the current helper keeps the entire
sentence. Two corpus shorts therefore contain 113 and 119 words despite the
advertised 90-word maximum.
Conclusion: the existing `_short` is materially worse than `_long` for
training whenever useful categories occur late in the validated paragraph. It
is not reliable as the intended LoRA-length semantic summary.
### Taggy: acceptable with minor cleanup
`_taggy` is compact enough for the studied corpus and remains semantically
close to `_long`. The deterministic compactor did not truncate any result, so
it removed no useful content through item or character limits. Exact duplicate
items were absent.
Minor defects remain: occasional dangling fragments (`sides`, `left`,
`visible`), terminal sentence punctuation, and near-duplicates such as
`glossy skin`/`smooth skin` or `bokeh lights`/`bokeh effect`. These do not
justify a new taggy architecture. Terminal punctuation can be safely removed
during deterministic cleanup; semantic near-deduplication should not be made
aggressive because similar phrases can carry distinct material or appearance
information.
## Refinement experiment
The existing Pass-D formatter was asked to return two labeled derivatives in
the same call. The short output was constrained to three concise sentences
covering, when present:
1. subject/appearance and major clothing/material;
2. pose/body/accessories or unusual visible details;
3. setting/lighting/framing and medium/style.
Results:
| Metric | Result |
| --- | ---: |
| Dual responses parsed | 19/19 |
| Shorts within 90 words | 19/19 |
| Short word range | 47–77 |
| Mean short length | 60.9 words |
| Added model calls | 0 |
| Added image-validation calls | 0 |
Manual review found that the refined shorts consistently selected information
from across the whole validated caption rather than stopping after its opening.
They are meaningfully shorter while retaining a balanced training identity.
As expected for compression, not every minor detail survives; the full long and
taggy variants remain available when maximum recall is preferred.
## Decision
Adopt the dual-output Pass-D prompt as the smallest defensible fix:
- `_long` remains the untouched Pass-C authority.
- One existing Pass-D text-only call derives both `_short` and `_taggy`.
- The formatter may only compress validated content; it may not add, infer, or
correct visual claims.
- Deterministic cleanup enforces the short limit and normalizes taggy output.
- Legacy/custom taggy-only formatter responses remain supported, using the
corrected deterministic short helper as a fallback.
- Do not add another VLM pass or another LLM call.
This resolves the demonstrated short-caption failure without redesigning the
successful semantic pipeline or the acceptable taggy path.
+6 -1
View File
@@ -1 +1,6 @@
# empty file
"""Production backend and support engines for CaptionForge.
Active ComfyUI execution is orchestrated by the node wrappers. The standalone
Distiller and Validator modules are retained as reference/CLI implementations,
not as the authoritative production B/C/D path.
"""
+12 -15
View File
@@ -8,19 +8,16 @@ CaptionForge Caption Prompt Kit
- Repository:
https://github.com/Damkohler/CaptionForge
- CaptionForge focuses on practical dataset-captioning infrastructure for
LoRA dataset preparation, using multi-engine caption generation, JSONL
audit trails, claim extraction and refinement, text-LLM distillation,
image-aware VLM validation, and consensus-oriented caption improvement
to produce grounded, auditable training captions.
- CaptionForge 1.0 uses independent Pass-A witnesses, text-LLM synthesis,
image-aware validation, SHORT/TAGGY formatting, and JSONL audit trails to
produce grounded LoRA dataset captions.
- Module Purpose
- The **CaptionForge Caption Prompt Kit** is a shared, dependency-free
prompt builder for non-Joy CaptionForge caption witnesses.
- The **CaptionForge Caption Prompt Kit** is the shared, dependency-free
prompt builder for active Qwen and Ollama caption witnesses.
- It provides a common caption-prompt vocabulary for Qwen, SmolVLM, and
future lightweight or generic caption engines while still allowing
model-dialect-specific wording.
- It retains generic and model-dialect-specific wording helpers so active
witnesses can share controls without sharing identical prompts.
- It defines:
• caption length choices
@@ -65,10 +62,10 @@ CaptionForge Caption Prompt Kit
JSONL audit trails so users can understand what each caption witness was
asked to do.
- Development Status
- CaptionForge v0.1.0 experimental developer-preview infrastructure.
- Caption types, extra options, and dialect wording may evolve as additional
caption engines are tested.
- Production Status
- Active CaptionForge 1.0 Pass-A support code used by Qwen and Ollama
witnesses. ``PROMPT_KIT_VERSION`` below identifies its internal prompt
contract and is intentionally separate from the package release version.
- Attribution & License
- Concept and implementation by **J. L. Córdova**
@@ -94,7 +91,7 @@ MANIFEST = {
"Shared dependency-free prompt builder for non-Joy CaptionForge caption "
"witnesses. Provides caption length choices, caption type choices, "
"extra-option text, dialect normalization, prompt construction helpers, "
"and prompt metadata for Qwen, SmolVLM, Ollama, and future generic "
"and prompt metadata for Qwen, Ollama, and future generic "
"caption engines."
),
}
+141
View File
@@ -0,0 +1,141 @@
"""Shared forbidden-phrase matching helpers for CaptionForge cleanup paths."""
from __future__ import annotations
import re
from collections.abc import Iterable
from typing import Any
def phrase_boundary_pattern(
phrase: str,
*,
case_insensitive: bool = True,
) -> re.Pattern[str] | None:
"""Compile a phrase matcher that respects token boundaries.
Boundary checks are added only when the corresponding phrase edge is a
word character. This preserves literal punctuation in configured phrases
while preventing tokens such as old from matching inside holding or bold.
"""
text = str(phrase or "").strip()
if not text:
return None
pattern = re.escape(text)
if re.match(r"\w", text[0], flags=re.UNICODE):
pattern = r"(?<!\w)" + pattern
if re.match(r"\w", text[-1], flags=re.UNICODE):
pattern = pattern + r"(?!\w)"
flags = re.IGNORECASE if case_insensitive else 0
return re.compile(pattern, flags=flags)
def contains_forbidden_phrase(text: str, forbidden_phrases: Iterable[str]) -> bool:
"""Return True when any configured forbidden phrase matches at boundaries."""
haystack = str(text or "")
for phrase in forbidden_phrases:
pattern = phrase_boundary_pattern(phrase)
if pattern is not None and pattern.search(haystack):
return True
return False
def remove_forbidden_phrases(text: str, forbidden_phrases: Iterable[str]) -> str:
"""Remove configured forbidden phrases without corrupting containing words."""
result = str(text or "")
for phrase in forbidden_phrases:
pattern = phrase_boundary_pattern(phrase)
if pattern is not None:
result = pattern.sub("", result)
return result
def replace_phrases(
text: str,
replacement_rules: Iterable[tuple[str, str]],
*,
case_insensitive: bool = True,
) -> str:
"""Apply replacement rules only at whole-word/phrase boundaries."""
result = str(text or "")
for old, new in replacement_rules:
pattern = phrase_boundary_pattern(old, case_insensitive=case_insensitive)
if pattern is not None:
result = pattern.sub(str(new or ""), result)
return result
def normalize_forbidden_phrases(value: Any) -> list[str]:
"""Normalize UI or plan values into an ordered forbidden-phrase list."""
if value is None:
return []
if isinstance(value, str):
items = value.splitlines()
elif isinstance(value, (list, tuple)):
items = value
else:
items = [value]
return [text for item in items if (text := str(item or "").strip())]
def normalize_replace_pairs(value: Any) -> list[tuple[str, str]]:
"""Normalize ``old=>new`` UI text or serialized plan replacement pairs."""
if value is None:
return []
if isinstance(value, str):
items: Iterable[Any] = value.splitlines()
elif isinstance(value, (list, tuple)):
items = value
else:
items = [value]
pairs: list[tuple[str, str]] = []
for item in items:
if isinstance(item, dict):
old, new = item.get("old", ""), item.get("new", "")
elif isinstance(item, (list, tuple)) and len(item) >= 2:
old, new = item[0], item[1]
else:
line = str(item or "").strip()
if not line or line.startswith("#") or "=>" not in line:
continue
old, new = line.split("=>", 1)
old_text = str(old or "").strip()
if old_text:
pairs.append((old_text, str(new or "").strip()))
return pairs
def resolve_cleanup_settings(
pipeline_plan: Any,
standalone_forbidden_phrases: Any,
standalone_replace_pairs: Any,
) -> tuple[list[str], list[tuple[str, str]]]:
"""Resolve Planner-owned cleanup values, preserving standalone node use."""
plan = pipeline_plan if isinstance(pipeline_plan, dict) else {}
cleanup = plan.get("cleanup") if isinstance(plan.get("cleanup"), dict) else None
if cleanup is not None:
forbidden_value = cleanup.get("forbidden_phrases", [])
replace_value = cleanup.get("replace_pairs", [])
else:
forbidden_value = standalone_forbidden_phrases
replace_value = standalone_replace_pairs
return normalize_forbidden_phrases(forbidden_value), normalize_replace_pairs(replace_value)
def apply_cleanup_contract(
text: str,
forbidden_phrases: Iterable[str],
replace_pairs: Iterable[tuple[str, str]],
) -> str:
"""Apply the shared boundary-safe cleanup contract and repair separators."""
result = replace_phrases(text, replace_pairs)
result = remove_forbidden_phrases(result, forbidden_phrases)
result = re.sub(r"\s+([,.;:!?])", r"\1", result)
result = re.sub(r",\s*,+", ",", result)
result = re.sub(r"([.;:!?])(?:\s*[,.;:!?])+", r"\1", result)
result = re.sub(r"\s+", " ", result)
return result.strip(" ,")
+227
View File
@@ -0,0 +1,227 @@
"""Optional, non-enlarging training image/caption export shared by both nodes."""
from __future__ import annotations
import hashlib
import json
import os
import tempfile
from pathlib import Path
from PIL import Image
from .captionforge_source_identity import optional_image_filename
EXPORT_MARKER = ".captionforge-dataset.json"
MARKER_CONTENT = {"type": "captionforge_training_dataset", "version": 1}
EXPORT_DEFAULTS = {
"enabled": False,
"output_folder": "",
"max_size": 0,
"divisor": 16,
"image_format": "PNG",
"jpeg_quality": 95,
"caption": "short",
}
EXPORT_WIDGETS = {
"enabled": "Dataset - export image and caption",
"output_folder": "Dataset - output folder",
"max_size": "Dataset - max image size",
"divisor": "Dataset - dimension divisor",
"image_format": "Dataset - image format",
"jpeg_quality": "Dataset - JPEG quality",
"caption": "Dataset - caption",
}
def dataset_export_inputs() -> dict:
"""Append optional widgets so older API workflows remain valid."""
specs = {
"enabled": ("BOOLEAN", {"tooltip": "Export a resized image and matching training TXT. Planner owns these controls when connected."}),
"output_folder": ("STRING", {"tooltip": "Parent folder for training_dataset. Blank uses Output - folder. Originals are never replaced."}),
"max_size": ("INT", {"min": 0, "max": 8192, "step": 1, "tooltip": "Maximum long edge; never enlarges. 0 follows the configured Validator size (Planner Caption - max image size)."}),
"divisor": ("INT", {"min": 1, "max": 512, "step": 1, "tooltip": "Round both dimensions DOWN to this multiple after resizing. 1 disables alignment. Images too small for the divisor fail export without enlargement."}),
"image_format": (["PNG", "JPEG"], {"tooltip": "Format of the exported RGB training image."}),
"jpeg_quality": ("INT", {"min": 1, "max": 100, "step": 1, "tooltip": "JPEG quality; ignored for PNG."}),
"caption": (["short", "long", "taggy"], {"tooltip": "Caption written to the matching plain .txt file. Other caption variants remain available."}),
}
return {EXPORT_WIDGETS[key]: (kind, {"default": EXPORT_DEFAULTS[key], **options})
for key, (kind, options) in specs.items()}
def export_settings_from_widgets(widgets: dict) -> dict:
return {key: widgets.get(name, EXPORT_DEFAULTS[key]) for key, name in EXPORT_WIDGETS.items()}
def normalize_export_settings(values: dict | None) -> dict:
result = {**EXPORT_DEFAULTS, **(values or {})}
result["enabled"] = str(result["enabled"]).strip().lower() in {"true", "1", "yes", "on"}
result["output_folder"] = str(result["output_folder"] or "").strip()
for key, low, high in (("max_size", 0, 8192), ("divisor", 1, 512), ("jpeg_quality", 1, 100)):
value = float(result[key])
if not value.is_integer() or not low <= value <= high:
raise ValueError(f"Dataset {key} must be an integer from {low} to {high}.")
result[key] = int(value)
if result["image_format"] not in {"PNG", "JPEG"} or result["caption"] not in {"short", "long", "taggy"}:
raise ValueError("Unsupported dataset image format or caption choice.")
return result
def dataset_root(settings: dict, output_folder: str | Path) -> Path:
return (Path(settings["output_folder"] or output_folder).expanduser() / "training_dataset").resolve()
def is_dataset_export(path: Path) -> bool:
"""Recognize generated datasets, including after export has been disabled."""
resolved = path.resolve()
return any((parent / EXPORT_MARKER).is_file() for parent in (resolved, *resolved.parents))
def prepare_dataset_root(root: Path, input_path: str | Path = "") -> None:
"""Claim only an empty directory or a previously managed export directory."""
root = root.resolve()
if input_path and Path(input_path).resolve().is_relative_to(root):
raise ValueError("Dataset destination contains the input path. Choose another output folder.")
marker = root / EXPORT_MARKER
if marker.exists():
if marker.is_symlink() or json.loads(marker.read_text(encoding="utf-8")) != MARKER_CONTENT:
raise ValueError("Dataset folder has an invalid CaptionForge marker.")
return
if root.exists() and any(root.iterdir()):
raise ValueError(f"Dataset folder is not empty and is not managed by CaptionForge: {root}")
root.mkdir(parents=True, exist_ok=True)
with marker.open("x", encoding="utf-8") as stream:
json.dump(MARKER_CONTENT, stream)
def export_dimensions(source_size: tuple[int, int], max_size: int, divisor: int) -> tuple[int, int]:
"""Use integer arithmetic so exact multiples never lose a pixel to float error."""
width, height = source_size
longest = max(width, height)
scaled = tuple(edge * max_size // longest if 0 < max_size < longest else edge for edge in source_size)
size = tuple((edge // divisor) * divisor for edge in scaled)
if min(size) < 1:
raise ValueError(f"Image {width}x{height} is too small for divisor {divisor} at max size {max_size}; use a smaller divisor.")
return size
def resize_for_export(image: Image.Image, max_size: int, divisor: int) -> Image.Image:
"""Calculate dimensions first and resample once; never clamp an edge upward."""
size = export_dimensions(image.size, max_size, divisor)
return image.resize(size, Image.Resampling.LANCZOS) if size != image.size else image
def export_paths(root: Path, source: Path, input_root: str | Path, image_key: str, image_format: str) -> tuple[Path, Path, Path]:
"""Keep the original extension in the stem to avoid conversion collisions."""
optional_name = optional_image_filename(image_key)
if optional_name:
relative = Path("optional") / optional_name
else:
base = Path(input_root).resolve() if input_root else source.resolve().parent
if base.is_file():
base = base.parent
try:
relative = Path("files") / source.resolve().relative_to(base)
except ValueError:
namespace = hashlib.sha256(str(source.resolve().parent).encode()).hexdigest()[:16]
relative = Path("external") / namespace / source.name
target = root / relative.parent / (relative.name + (".png" if image_format == "PNG" else ".jpg"))
caption = target.with_suffix(".txt")
receipt = target.with_suffix(".captionforge.json")
check_export_destinations(root, source, (target, caption, receipt))
return target, caption, receipt
def check_export_destinations(root: Path, source: Path, paths) -> None:
for path in paths:
if path.is_symlink() or not path.resolve().is_relative_to(root.resolve()):
raise ValueError("Dataset destination escapes its protected folder.")
if path.resolve() == source.resolve() or (path.exists() and path.samefile(source)):
raise ValueError("Dataset export would overwrite a source image.")
def _digest(path: Path) -> str:
return hashlib.sha256(path.read_bytes()).hexdigest()
def export_pair(*, source: Path, image_key: str, input_root: str | Path, root: Path,
settings: dict, validator_max_size: int, captions: dict, overwrite: bool,
prepared_image: Image.Image | None = None, write_variants: bool = False) -> dict:
"""Publish a pair with a receipt written last; incomplete pairs never resume."""
source = source.resolve()
if is_dataset_export(source):
raise ValueError("A generated dataset image cannot be used as its own archive source.")
prepare_dataset_root(root, input_root)
target, caption_path, receipt_path = export_paths(root, source, input_root, image_key, settings["image_format"])
max_size = settings["max_size"] or validator_max_size
caption = str(captions.get(settings["caption"]) or "").strip()
if not caption:
raise ValueError("Selected training caption is empty.")
variants = {style: str(captions.get(style) or "").strip() for style in ("long", "short", "taggy")} if write_variants else {}
variant_paths = {style: target.with_name(f"{target.stem}_{style}.txt") for style in variants}
check_export_destinations(root, source, variant_paths.values())
stat = source.stat()
signature = {"source": str(source), "image_key": image_key, "source_size": stat.st_size,
"source_mtime_ns": stat.st_mtime_ns, "max_size": max_size,
"divisor": settings["divisor"], "format": settings["image_format"],
"jpeg_quality": settings["jpeg_quality"], "caption": caption, "variants": variants}
previous = json.loads(receipt_path.read_text(encoding="utf-8")) if receipt_path.exists() else {}
owned_variants = set(previous.get("owned_variants", previous.get("signature", {}).get("variants", {})))
for style, path in variant_paths.items():
if path.exists() and style not in owned_variants:
raise FileExistsError(f"Dataset caption variant is untracked: {path}")
previous_image = previous.get("image") or previous.get("export", {}).get("image")
if previous_image and Path(previous_image) != target and Path(previous_image).exists():
raise FileExistsError("This source was exported in another image format. Choose a different Dataset output folder to avoid duplicate training images.")
if any(path.exists() for path in (target, caption_path, receipt_path)):
if previous.get("signature", {}).get("source") != str(source) or previous.get("signature", {}).get("image_key") != image_key:
raise FileExistsError(f"Dataset filename is already occupied by another or untracked source: {target}")
if not overwrite:
if (previous.get("signature") == signature and target.is_file() and caption_path.is_file()
and previous.get("image_sha256") == _digest(target)
and previous.get("caption_sha256") == _digest(caption_path)
and all(path.is_file() and previous.get("variant_sha256", {}).get(style) == _digest(path)
for style, path in variant_paths.items())):
return {**previous["export"], "resumed": True}
raise FileExistsError("Dataset pair is incomplete, changed, or uses different settings; enable overwrite to regenerate it.")
if prepared_image is None:
with Image.open(source) as original:
prepared_image = original.convert("RGB")
resized = resize_for_export(prepared_image, max_size, settings["divisor"])
result = {"status": "ok", "image": str(target), "caption": str(caption_path),
"width": resized.width, "height": resized.height, "caption_style": settings["caption"],
"max_size": max_size, "divisor": settings["divisor"], "resumed": False,
"variants": {style: str(path) for style, path in variant_paths.items()}}
target.parent.mkdir(parents=True, exist_ok=True)
temporary: list[Path] = []
try:
for _ in range(3 + len(variants)):
descriptor, name = tempfile.mkstemp(prefix=".captionforge-", dir=target.parent)
os.close(descriptor)
temporary.append(Path(name))
options = {"quality": settings["jpeg_quality"], "subsampling": 0} if settings["image_format"] == "JPEG" else {}
resized.save(temporary[0], format=settings["image_format"], **options)
temporary[1].write_text(caption + "\n", encoding="utf-8")
variant_hashes = {}
for index, (style, text) in enumerate(variants.items(), start=2):
temporary[index].write_text(text + "\n", encoding="utf-8")
variant_hashes[style] = _digest(temporary[index])
receipt = {"signature": signature, "export": result,
"image_sha256": _digest(temporary[0]), "caption_sha256": _digest(temporary[1]),
"variant_sha256": variant_hashes, "owned_variants": sorted(owned_variants | set(variants))}
# Reserve ownership before publishing, without claiming untracked files.
reservation = {"signature": signature, "image": str(target), "owned_variants": receipt["owned_variants"]}
if previous:
temporary[-1].write_text(json.dumps(reservation), encoding="utf-8")
os.replace(temporary[-1], receipt_path)
else:
with receipt_path.open("x", encoding="utf-8") as stream:
json.dump(reservation, stream)
temporary[-1].write_text(json.dumps(receipt, ensure_ascii=False, indent=2), encoding="utf-8")
for staging, destination in zip(temporary, (target, caption_path, *variant_paths.values(), receipt_path)):
os.replace(staging, destination)
finally:
for path in temporary:
path.unlink(missing_ok=True)
return result
+8 -27
View File
@@ -8,11 +8,9 @@ CaptionForge Distiller Engine
- Repository:
https://github.com/Damkohler/CaptionForge
- CaptionForge focuses on practical dataset-captioning infrastructure for
LoRA dataset preparation, using multi-engine caption generation, JSONL
audit trails, claim extraction and refinement, text-LLM distillation,
image-aware VLM validation, and consensus-oriented caption improvement
to produce grounded, auditable training captions.
- CaptionForge 1.0 uses independent Pass-A witnesses, text-LLM synthesis,
image-aware validation, SHORT/TAGGY formatting, and JSONL audit trails to
produce grounded LoRA dataset captions.
- Engine Purpose
- The **CaptionForge Distiller Engine** is the Pass B text-LLM
@@ -72,10 +70,10 @@ CaptionForge Distiller Engine
- Draft captions should be rich enough for LoRA dataset preparation while
remaining auditable through their source claims and metadata.
- Development Status
- CaptionForge v0.1.0 experimental developer-preview infrastructure.
- Prompt contracts, audit fields, and parser behavior may evolve before a
stable CaptionForge release.
- Reference Status
- This CLI-oriented engine is retained for reference, diagnostics, and
controlled experiments. It is not imported or registered by the active
ComfyUI pipeline; production Pass B runs inside ``jlc_captionforge_node``.
- Attribution & License
- Concept and implementation by **J. L. Córdova**
@@ -127,23 +125,6 @@ from pathlib import Path
from typing import Any, Optional
ENGINE_NAME = "captionforge_distiller_engine"
ENGINE_VERSION = "0.2.0"
CAPTIONFORGE_PASS = "B_DISTILL"
DEFAULT_OLLAMA_BASE_URL = "http://127.0.0.1:11434"
MANIFEST = {
"name": "CaptionForge Distiller Engine",
"version": ENGINE_VERSION,
"author": "J. L. Córdova",
"description": (
"CLI-first CaptionForge Pass B pollster/copywriter engine. Consumes Pass A "
"caption JSONL records, groups captions by image, asks a text LLM to "
"vote/organize visual claims, and emits accepted evidence, singleton "
"candidates, rejected conflicts, and rich/taggy draft captions."
),
}
DEFAULT_DISTILLER_INSTRUCTIONS = (
"You are CaptionForge Pass B: a caption ballot pollster and rich-caption copywriter. "
"Your job is not to summarize, prune for brevity, or compress the source captions. "
@@ -1355,7 +1336,7 @@ def process_batch(batch: BatchConfig, config: DistillerConfig) -> int:
def build_parser() -> argparse.ArgumentParser:
p = argparse.ArgumentParser(description="CaptionForge Distiller Engine v0.2.0")
p = argparse.ArgumentParser(description=f"CaptionForge Distiller Engine v{ENGINE_VERSION}")
p.add_argument("--input-jsonl", required=True, help="Pass A captions JSONL input.")
p.add_argument("--output-jsonl", default="", help="Full output JSONL path. Default: <input>_distilled.jsonl")
+7 -9
View File
@@ -8,11 +8,9 @@ CaptionForge Joy Space Prompt Kit
- Repository:
https://github.com/Damkohler/CaptionForge
- CaptionForge focuses on practical dataset-captioning infrastructure for
LoRA dataset preparation, using multi-engine caption generation, JSONL
audit trails, claim extraction and refinement, text-LLM distillation,
image-aware VLM validation, and consensus-oriented caption improvement
to produce grounded, auditable training captions.
- CaptionForge 1.0 uses independent Pass-A witnesses, text-LLM synthesis,
image-aware validation, SHORT/TAGGY formatting, and JSONL audit trails to
produce grounded LoRA dataset captions.
- Module Purpose
- The **CaptionForge Joy Space Prompt Kit** is a small, dependency-free
@@ -69,10 +67,10 @@ CaptionForge Joy Space Prompt Kit
- Prompt metadata should make it clear which caption type, length, options,
name input, and system prompt were used for a Joy caption run.
- Development Status
- CaptionForge v0.1.0 experimental developer-preview infrastructure.
- Joy prompt templates and option lists may evolve as the local CaptionForge
Joy node matures.
- Production Status
- Active CaptionForge 1.0 Pass-A support code. It owns the Joy-specific
prompt contract while model loading and inference remain in the Joy
engine.
- Attribution & License
- Concept and implementation by **J. L. Córdova**
+84
View File
@@ -0,0 +1,84 @@
"""Scoped handling of expected bitsandbytes activation-cast diagnostics."""
from contextlib import contextmanager
from importlib import metadata
import logging
import re
import threading
import warnings
_CAST_MESSAGES = frozenset(
f"MatMul8bitLt: inputs will be cast from torch.{dtype} "
"to float16 during quantization"
for dtype in ("bfloat16", "float32")
)
_BNB_LOGGER = "bitsandbytes.autograd._functions"
_MINIMUM_BNB_VERSION = (0, 46, 1)
def warn_if_suspicious_8bit_stack(node_name: str) -> None:
"""Warn, without blocking inference, when bnb predates the project floor."""
try:
version = metadata.version("bitsandbytes")
except Exception:
return
parts = tuple(int(piece) for piece in re.findall(r"\d+", version)[:3])
normalized = parts + (0,) * (3 - len(parts))
if normalized < _MINIMUM_BNB_VERSION:
warnings.warn(
f"CaptionForge's {node_name} node detected bitsandbytes {version}; "
"this older 8-bit inference stack may cause severe slowdowns or compatibility issues. "
"CaptionForge will continue without modifying packages.",
RuntimeWarning,
stacklevel=2,
)
class _QuantizedCastLogFilter(logging.Filter):
def __init__(self):
super().__init__()
self.thread_id = threading.get_ident()
def filter(self, record):
return not (
record.name == _BNB_LOGGER
and record.levelno == logging.WARNING
and record.thread == self.thread_id
and record.getMessage() in _CAST_MESSAGES
)
@contextmanager
def quantized_inference_warnings(enabled: bool):
"""Preserve diagnostics except known BF16/FP32 casts during 8-bit inference.
No filter is installed at import time or for Default mode. catch_warnings
restores the caller's filters even when generation raises an exception.
On Python 3.10-3.12 warning filters are process-wide during this brief scope;
the exact message, category and module keep the suppression narrowly bounded.
"""
if not enabled:
yield
return
with warnings.catch_warnings():
warnings.filterwarnings(
"ignore",
message=(
r"\AMatMul8bitLt: inputs will be cast from torch\.(?:bfloat16|float32) "
r"to float16 during quantization\Z"
),
category=UserWarning,
module=r"\Abitsandbytes\.autograd\._functions\Z",
)
# Recent bitsandbytes versions use logging rather than warnings.warn.
# Attach to the emitting logger so propagation/ComfyUI handlers remain
# untouched. Limit this route to the current inference thread as well.
logger = logging.getLogger(_BNB_LOGGER)
log_filter = _QuantizedCastLogFilter()
logger.addFilter(log_filter)
try:
yield
finally:
logger.removeFilter(log_filter)
+7 -9
View File
@@ -8,11 +8,9 @@ CaptionForge Global Model Cache Manager
- Repository:
https://github.com/Damkohler/CaptionForge
- CaptionForge focuses on practical dataset-captioning infrastructure for
LoRA dataset preparation, using multi-engine caption generation, JSONL
audit trails, claim extraction and refinement, text-LLM distillation,
image-aware VLM validation, and consensus-oriented caption improvement
to produce grounded, auditable training captions.
- CaptionForge 1.0 uses independent Pass-A witnesses, text-LLM synthesis,
image-aware validation, SHORT/TAGGY formatting, and JSONL audit trails to
produce grounded LoRA dataset captions.
- Engine Purpose
- The **CaptionForge Global Model Cache Manager** provides shared,
@@ -75,10 +73,10 @@ CaptionForge Global Model Cache Manager
- The module favors predictable behavior over aggressive automatic memory
management.
- Development Status
- CaptionForge v0.1.0 experimental developer-preview infrastructure.
- Cache policy and diagnostics may evolve as CaptionForge's supported engine
set matures.
- Production Status
- Active CaptionForge 1.0 runtime support. Its conservative single-model
residency policy coordinates heavyweight local Hugging Face witnesses;
it does not own caption semantics.
- Attribution & License
- Concept and implementation by **J. L. Córdova**
+113 -46
View File
@@ -8,11 +8,9 @@ CaptionForge Pipeline Planner Engine
- Repository:
https://github.com/Damkohler/CaptionForge
- CaptionForge focuses on practical dataset-captioning infrastructure for
LoRA dataset preparation, using multi-engine caption generation, JSONL
audit trails, claim extraction and refinement, text-LLM distillation,
image-aware VLM validation, and consensus-oriented caption improvement
to produce grounded, auditable training captions.
- CaptionForge 1.0 uses independent Pass-A witnesses, text-LLM synthesis,
image-aware validation, SHORT/TAGGY formatting, and JSONL audit trails to
produce grounded LoRA dataset captions.
- Engine Purpose
- The **CaptionForge Pipeline Planner Engine** builds reusable
@@ -71,10 +69,10 @@ CaptionForge Pipeline Planner Engine
- The planner favors explicit dictionaries, predictable filenames, and
auditable JSONL-oriented handoff between passes.
- Development Status
- CaptionForge v0.1.0 experimental developer-preview infrastructure.
- Plan schema, supported witness families, and downstream defaults may evolve
before a stable CaptionForge release.
- Production Status
- Active CaptionForge 1.0 planning support. The Planner is authoritative in
connected workflows, and its shared B/C/D defaults are kept in parity with
the standalone Orchestrator controls.
- Attribution & License
- Concept and implementation by **J. L. Córdova**
@@ -106,17 +104,27 @@ MANIFEST = {
}
import hashlib
import json
import random
import re
from dataclasses import dataclass
from pathlib import Path
from typing import Any
from .captionforge_prompt_defaults import (
DEFAULT_FAT_DRAFT_INSTRUCTIONS,
DEFAULT_TAGGY_FORMATTER_INSTRUCTIONS,
DEFAULT_VALIDATOR_INSTRUCTIONS,
DEFAULT_VALIDATOR_SYSTEM_PROMPT,
)
from .captionforge_cleanup import normalize_forbidden_phrases, normalize_replace_pairs
MAX_SEED_32 = 0xFFFFFFFF
PIPELINE_PLAN_TYPE = "captionforge_pipeline_plan"
PIPELINE_PLAN_VERSION = 5
PIPELINE_PLAN_VERSION = 6
DEFAULT_OLLAMA_URL = "http://127.0.0.1:11434"
MAX_PASS_A_RUNS_PER_MODEL = 5
PASS_A_SEED_NAMESPACE = b"captionforge-pass-a-seed-v1"
@dataclass(frozen=True)
@@ -213,22 +221,26 @@ def _normalize_seed_mode(value: Any) -> str:
def _seed_for_run(base_seed: int, seed_mode: str, index: int) -> int | None:
if base_seed < 0:
if seed_mode == "random":
return random.SystemRandom().randint(0, MAX_SEED_32)
return None
base_seed = max(0, min(int(base_seed), MAX_SEED_32))
index = int(index)
if index < 0:
raise ValueError("CaptionForge Pass-A run index must be nonnegative")
if seed_mode == "fixed":
return base_seed
if seed_mode == "increment":
return min(MAX_SEED_32, base_seed + index)
return (base_seed + index) & MAX_SEED_32
if seed_mode == "decrement":
return max(0, base_seed - index)
return (base_seed - index) & MAX_SEED_32
if seed_mode == "random":
rng = random.Random(base_seed)
out = base_seed
for _ in range(index + 1):
out = rng.randint(0, MAX_SEED_32)
return out
payload = b"\0".join(
(
PASS_A_SEED_NAMESPACE,
str(base_seed).encode("ascii"),
str(index).encode("ascii"),
)
)
return int.from_bytes(hashlib.blake2s(payload, digest_size=4).digest(), "big")
return base_seed
@@ -364,8 +376,8 @@ def build_captionforge_pipeline_plan(
run_name: str = "captionforge_run",
overwrite_outputs: bool = True,
joy_runs_per_image: Any = 2,
qwen_runs_per_image: Any = 2,
ollama_runs_per_image: Any = "Disabled",
qwen_runs_per_image: Any = 1,
ollama_runs_per_image: Any = 1,
ollama_caption_runs_per_image: Any | None = None,
caption_ollama_runs_per_image: Any | None = None,
ollama_vlm_runs_per_image: Any | None = None,
@@ -377,12 +389,21 @@ def build_captionforge_pipeline_plan(
top_p_schedule: str = "",
top_k_schedule: str = "",
max_size: int = 1024,
max_new_tokens: int = 512,
max_new_tokens: int = 4096,
trigger_word: str = "",
user_caption_anchor: str = "",
forbidden_phrases: Any = "",
replace_pairs: Any = "",
ollama_url: str = DEFAULT_OLLAMA_URL,
ollama_keep_loaded: bool = True,
ollama_request_timeout_seconds: int = 1800,
distiller_model_family: str = "Llama",
distiller_base_seed: int | None = None,
distiller_prompt: str = DEFAULT_FAT_DRAFT_INSTRUCTIONS,
# Compatibility-only. Pass B owns one fixed seed; modes are ignored.
distiller_seed_mode: str = "fixed",
# Compatibility-only argument for older callers. Production Pass B is
# always one global fat-draft call per image.
distiller_strategy: str = "single_pass",
distiller_max_caption_chars_for_llm: int = 1536,
distiller_num_predict: int = 3096,
@@ -393,21 +414,36 @@ def build_captionforge_pipeline_plan(
distiller_preserve_raw_response: bool = False,
validator_model_family: str = "Llama Vision",
validator_base_seed: int | None = None,
validator_system_prompt: str = DEFAULT_VALIDATOR_SYSTEM_PROMPT,
validator_prompt: str = DEFAULT_VALIDATOR_INSTRUCTIONS,
# Compatibility-only. Pass C owns one fixed seed; modes are ignored.
validator_seed_mode: str = "fixed",
validator_num_predict: int = 2200,
validator_num_predict: int = 2112,
validator_temperature: float = 0.0,
validator_top_p: float = 0.92,
validator_top_k: int = 80,
validator_write_prompt_jsonl: bool = False,
validator_preserve_raw_vlm_response: bool = False,
formatter_model: str = "mistral-small:24b",
formatter_prompt: str = DEFAULT_TAGGY_FORMATTER_INSTRUCTIONS,
formatter_num_predict: int = 3200,
formatter_temperature: float = 0.12,
formatter_top_p: float = 0.88,
formatter_top_k: int = 50,
formatter_write_prompt_jsonl: bool = False,
formatter_preserve_raw_response: bool = False,
# Compatibility-only argument for older callers. Production always exports
# the invariant long/short/taggy sidecar set.
final_caption_style: str = "narrative",
final_write_txt_sidecars: bool = True,
final_write_jsonl: bool = True,
dataset_export: dict[str, Any] | None = None,
# Legacy compatibility aliases retained for older callers.
distiller_seed: int | None = None,
validator_seed: int | None = None,
distiller_model: str = "llama3.1:8b",
validator_model: str = "llama3.2-vision:11b",
formatter_seed: int | None = None,
distiller_model: str = "mistral-small:24b",
validator_model: str = "gemma4:26b",
captions_per_image: int | None = None,
# Deprecated compatibility-only parameters. Intentionally ignored.
pass_c_deterministic: bool = True,
@@ -415,9 +451,9 @@ def build_captionforge_pipeline_plan(
semantic_profile: str = "",
) -> dict[str, Any]:
joy_runs = _normalize_runs_per_image(joy_runs_per_image, 2)
qwen_runs = _normalize_runs_per_image(qwen_runs_per_image, 2)
qwen_runs = _normalize_runs_per_image(qwen_runs_per_image, 1)
ollama_runs = max(
_normalize_runs_per_image(value, "Disabled")
_normalize_runs_per_image(value, 1)
for value in (
ollama_runs_per_image,
ollama_caption_runs_per_image,
@@ -440,14 +476,14 @@ def build_captionforge_pipeline_plan(
base_seed_n = _coerce_int(base_seed, -1, -1, MAX_SEED_32)
seed_mode_n = _normalize_seed_mode(seed_mode)
if distiller_base_seed is None:
distiller_base_seed = distiller_seed
if validator_base_seed is None:
validator_base_seed = validator_seed
if distiller_seed is None:
distiller_seed = distiller_base_seed
if validator_seed is None:
validator_seed = validator_base_seed
d_seed = base_seed_n if distiller_base_seed is None else _coerce_int(distiller_base_seed, base_seed_n, -1, MAX_SEED_32)
v_seed_default = -1 if base_seed_n < 0 else min(MAX_SEED_32, base_seed_n + 1000003)
v_seed = v_seed_default if validator_base_seed is None else _coerce_int(validator_base_seed, v_seed_default, -1, MAX_SEED_32)
d_seed = _coerce_int(distiller_seed, -1, -1, MAX_SEED_32)
v_seed = _coerce_int(validator_seed, -1, -1, MAX_SEED_32)
f_seed = _coerce_int(formatter_seed, -1, -1, MAX_SEED_32)
run_name_n = _clean_name(run_name)
input_path_n = str(input_path or "").strip()
@@ -458,7 +494,7 @@ def build_captionforge_pipeline_plan(
"top_p_schedule": str(top_p_schedule or "").strip(),
"top_k_schedule": str(top_k_schedule or "").strip(),
"max_size": _coerce_int(max_size, 1024, 0, 4096),
"max_new_tokens": _coerce_int(max_new_tokens, 512, 16, 4096),
"max_new_tokens": _coerce_int(max_new_tokens, 4096, 16, 4096),
"trigger_word": str(trigger_word or "").strip(),
"user_caption_anchor": str(user_caption_anchor or "").strip(),
"output_dir": str(output_dir or "").strip(),
@@ -470,6 +506,14 @@ def build_captionforge_pipeline_plan(
"run_name": run_name_n,
"overwrite_outputs": _coerce_bool(overwrite_outputs, True),
}
forbidden = normalize_forbidden_phrases(forbidden_phrases)
replacements = normalize_replace_pairs(replace_pairs)
cleanup = {
"forbidden_phrases": forbidden,
"replace_pairs": [{"old": old, "new": new} for old, new in replacements],
"matching": "boundary_safe_case_insensitive",
"order": ["replace_pairs", "forbidden_phrases", "normalize_whitespace_punctuation"],
}
shared["captions_per_image"] = (
_coerce_int(captions_per_image, max(joy_runs, qwen_runs, ollama_runs, florence_runs, llama_runs, 1), 1, 100)
if captions_per_image is not None
@@ -483,14 +527,18 @@ def build_captionforge_pipeline_plan(
# Pass A JSONL/audit artifacts, so point it at the run working directory.
shared["output_dir"] = paths["output_dir"]
ollama = {
"url": str(ollama_url or DEFAULT_OLLAMA_URL).strip() or DEFAULT_OLLAMA_URL,
"keep_loaded": _coerce_bool(ollama_keep_loaded, True),
"request_timeout_seconds": _coerce_int(ollama_request_timeout_seconds, 1800, 10, 7200),
}
distiller = {
"backend": "ollama",
"model_family": str(distiller_model_family or "Llama").strip() or "Llama",
"model": str(distiller_model or "llama3.1:8b").strip() or "llama3.1:8b",
"base_seed": d_seed,
"model": str(distiller_model or "mistral-small:24b").strip() or "mistral-small:24b",
"prompt": str(distiller_prompt or DEFAULT_FAT_DRAFT_INSTRUCTIONS),
"seed": d_seed,
"seed_mode": _normalize_seed_mode(distiller_seed_mode),
"strategy": str(distiller_strategy or "single_pass").strip() or "single_pass",
"max_caption_chars_for_llm": _coerce_int(distiller_max_caption_chars_for_llm, 1536, 0, 12000),
"num_predict": _coerce_int(distiller_num_predict, 3096, 64, 12000),
"temperature": _coerce_float(distiller_temperature, 0.24, 0.0, 2.0),
@@ -503,12 +551,12 @@ def build_captionforge_pipeline_plan(
validator = {
"backend": "ollama",
"model_family": str(validator_model_family or "Llama Vision").strip() or "Llama Vision",
"model": str(validator_model or "llama3.2-vision:11b").strip() or "llama3.2-vision:11b",
"base_seed": v_seed,
"model": str(validator_model or "gemma4:26b").strip() or "gemma4:26b",
"system_prompt": str(validator_system_prompt or DEFAULT_VALIDATOR_SYSTEM_PROMPT),
"prompt": str(validator_prompt or DEFAULT_VALIDATOR_INSTRUCTIONS),
"seed": v_seed,
"seed_mode": _normalize_seed_mode(validator_seed_mode),
"image_root": input_path_n,
"num_predict": _coerce_int(validator_num_predict, 2200, 64, 12000),
"num_predict": _coerce_int(validator_num_predict, 2112, 64, 12000),
"temperature": _coerce_float(validator_temperature, 0.0, 0.0, 2.0),
"top_p": _coerce_float(validator_top_p, 0.92, 0.0, 1.0),
"top_k": _coerce_int(validator_top_k, 80, 0, 500),
@@ -516,8 +564,20 @@ def build_captionforge_pipeline_plan(
"preserve_raw_vlm_response": _coerce_bool(validator_preserve_raw_vlm_response, False),
"role": "image_aware_precision_validation",
}
formatter = {
"backend": "ollama",
"model": str(formatter_model or "mistral-small:24b").strip() or "mistral-small:24b",
"prompt": str(formatter_prompt or DEFAULT_TAGGY_FORMATTER_INSTRUCTIONS),
"seed": f_seed,
"num_predict": _coerce_int(formatter_num_predict, 3200, 64, 12000),
"temperature": _coerce_float(formatter_temperature, 0.12, 0.0, 2.0),
"top_p": _coerce_float(formatter_top_p, 0.88, 0.0, 1.0),
"top_k": _coerce_int(formatter_top_k, 50, 0, 500),
"write_prompt_jsonl": _coerce_bool(formatter_write_prompt_jsonl, False),
"preserve_raw_response": _coerce_bool(formatter_preserve_raw_response, False),
"role": "text_only_taggy_formatting",
}
final = {
"caption_style": str(final_caption_style or "narrative").strip() or "narrative",
"write_txt_sidecars": _coerce_bool(final_write_txt_sidecars, True),
"write_jsonl": _coerce_bool(final_write_jsonl, True),
"overwrite_outputs": _coerce_bool(overwrite_outputs, True),
@@ -525,10 +585,15 @@ def build_captionforge_pipeline_plan(
"large_model_passes_after_validator": False,
}
from .captionforge_dataset_export import normalize_export_settings
return {
"dataset_export": normalize_export_settings(dataset_export),
"captionforge_config_type": PIPELINE_PLAN_TYPE,
"captionforge_config_version": PIPELINE_PLAN_VERSION,
"shared": shared,
"cleanup": cleanup,
"ollama": ollama,
"paths": paths,
"pass_a": {
"joy": {"model_key": "joy", "enabled": joy_runs > 0, "runs_per_image": joy_runs, "role": "rich_descriptive_caption_witness"},
@@ -556,9 +621,11 @@ def build_captionforge_pipeline_plan(
},
"distiller": distiller,
"validator": validator,
"formatter": formatter,
"final": final,
"pass_b_distiller": distiller,
"pass_c_vlm_validator": validator,
"pass_d_formatter": formatter,
"final_export": final,
}
@@ -572,7 +639,7 @@ def build_captionforge_run_config(
top_p_schedule: str = "",
top_k_schedule: str = "",
max_size: int = 1024,
max_new_tokens: int = 512,
max_new_tokens: int = 4096,
trigger_word: str = "",
output_dir: str = "",
input_path: str = "",
+78
View File
@@ -0,0 +1,78 @@
"""Frozen CaptionForge 1.0 prompt defaults shared by Planner and Orchestrator.
These strings define the production B/C/D contract: a recall-oriented text
draft, image-grounded LONG validation, and one text-only SHORT/TAGGY format
call. Keeping them here prevents the two control surfaces from drifting.
"""
from __future__ import annotations
DEFAULT_FAT_DRAFT_INSTRUCTIONS = """/no_think
You are a detail-preserving caption merger for LoRA dataset preparation.
You receive multiple captions of the same image. You do NOT see the image.
Task:
Merge all non-contradictory caption details into one deliberately over-complete draft caption.
Rules:
- Do not validate against the image.
- Do not decide that details are false just because they appear once.
- Do not summarize aggressively.
- Preserve concrete details from all captions.
- Split contradictions by choosing cautious wording or listing the alternative only when needed.
- Prefer specific visual language over generic language.
- Keep visible body, clothing, material, accessory, color, pose, lighting, style, and framing details.
- Preserve doll-like, glossy/plastic-like, material, garment-construction, body-shape, and facial-feature details when present.
- Use neutral dataset-caption language, including visible sensual styling or revealing clothing when present.
- Do not add details absent from the captions.
- Treat subject names or trigger-like identity tokens as optional identity labels. Preserve them only when they appear consistently in the captions; do not let them replace visible description.
- Output only one paragraph, no notes, no JSON."""
DEFAULT_VALIDATOR_SYSTEM_PROMPT = (
"/no_think\n"
"You are a direct image validation engine. Inspect the image and answer only with the requested caption."
)
DEFAULT_VALIDATOR_INSTRUCTIONS = """/no_think
Look at the image and validate this draft caption.
Task:
Return a corrected caption paragraph that keeps only image-supported details.
Rules:
- Output only the corrected caption.
- One paragraph.
- No reasoning, no notes, no JSON.
- Keep all true visible details from the draft.
- Delete unsupported details.
- Correct small visible errors.
- Do not add new details unless needed to correct an error already present.
- Preserve useful LoRA details: subject, face, hair, eyes, makeup, lips, skin texture, pose, body shape, outfit, accessories, materials, colors, lighting, background, framing, and visual style.
- Visible sensual styling, revealing clothing, cleavage, thighs, bare skin, swimwear, lingerie, or body-shape details may be described neutrally when present.
- Do not invent hidden anatomy, unseen clothing, explicit acts, or details contradicted by the image."""
DEFAULT_TAGGY_FORMATTER_INSTRUCTIONS = """/no_think
You are a LoRA caption format converter. The validated paragraph is your only source of truth.
Output exactly two labeled lines:
SHORT: <a concise natural-language caption, typically around 100 words, that preserves all LoRA-useful validated details>
TAGGY: <one compact comma-separated caption>
SHORT must preserve the image's distinctive training identity across the whole source:
1. subject, defining face/hair/body traits, and every major outfit piece/material;
2. pose/action and key accessories or unusual visible details;
3. setting, lighting, framing, and visual medium/style.
Omit a category only when absent. Use only source details; never add, infer, euphemize, or correct. Compress wording, not category coverage. Do not copy only the source opening.
Aim for roughly 100 words. Keep it concise, but allow modest variation when needed to preserve important information and finish the caption naturally. Do not cut off a sentence merely to satisfy a word-count target.
TAGGY must preserve all concrete LoRA-useful source details as compact comma-separated phrases.
No markdown, reasoning, notes, or other labels."""
+38
View File
@@ -0,0 +1,38 @@
"""Canonical source-image identities shared by CaptionForge Pass-A witnesses."""
from __future__ import annotations
from pathlib import Path
OPTIONAL_IMAGE_KEY_PREFIX = "captionforge-optional-image://"
def file_source_identity(image_path: Path, input_root: Path) -> tuple[str, str]:
"""Return the readable stem and portable, collision-safe dataset key."""
path = Path(image_path)
root = Path(input_root)
image = path.stem or "image"
if root.is_dir():
try:
return image, path.relative_to(root).as_posix()
except ValueError:
pass
return image, path.name
def optional_image_identity(index: int) -> tuple[str, str]:
"""Return deterministic display/key values for a filename-less IMAGE tensor."""
filename = f"comfy_image_{int(index):04d}.png"
return Path(filename).stem, f"{OPTIONAL_IMAGE_KEY_PREFIX}{filename}"
def optional_image_filename(image_key: object) -> str:
"""Return the synthetic PNG name encoded by an optional-image key, if any."""
text = str(image_key or "")
if not text.startswith(OPTIONAL_IMAGE_KEY_PREFIX):
return ""
filename = text[len(OPTIONAL_IMAGE_KEY_PREFIX):]
if not filename or "/" in filename or "\\" in filename:
return ""
return filename
+11 -14
View File
@@ -8,11 +8,9 @@ CaptionForge VLM Validator Engine
- Repository:
https://github.com/Damkohler/CaptionForge
- CaptionForge focuses on practical dataset-captioning infrastructure for
LoRA dataset preparation, using multi-engine caption generation, JSONL
audit trails, claim extraction and refinement, text-LLM distillation,
image-aware VLM validation, and consensus-oriented caption improvement
to produce grounded, auditable training captions.
- CaptionForge 1.0 uses independent Pass-A witnesses, text-LLM synthesis,
image-aware validation, SHORT/TAGGY formatting, and JSONL audit trails to
produce grounded LoRA dataset captions.
- Engine Purpose
- The **CaptionForge VLM Validator Engine** is the Pass C image-aware
@@ -58,9 +56,8 @@ CaptionForge VLM Validator Engine
- The engine writes structured JSONL records and optional readable sidecars.
- The module also contains an experimental cleaner function used by the
quarantined reversed-pipeline branch; that path is not the recommended
mainline v0.1.0 workflow.
- The module also contains an experimental cleaner retained for historical
diagnostics; it is not part of the production CaptionForge 1.0 workflow.
- Design Philosophy
- The distiller is recall-oriented; the VLM validator is grounding-oriented.
@@ -77,10 +74,10 @@ CaptionForge VLM Validator Engine
- Final candidate captions should remain rich, visually grounded, and
auditable.
- Development Status
- CaptionForge v0.1.0 experimental developer-preview infrastructure.
- Prompt contracts, parser behavior, image-resolution heuristics, and audit
fields may evolve before a stable CaptionForge release.
- Reference Status
- This CLI-oriented engine is retained for reference, diagnostics, and
controlled experiments. It is not imported or registered by the active
ComfyUI pipeline; production Pass C runs inside ``jlc_captionforge_node``.
- Attribution & License
- Concept and implementation by **J. L. Córdova**
@@ -134,7 +131,7 @@ from typing import Any, Iterable, Optional
@dataclass
class VLMValidatorConfig:
engine_name: str = "CaptionForge VLM Validator Engine"
engine_version: str = "0.2.0"
engine_version: str = CAPTIONFORGE_VERSION
vlm_backend: str = "ollama" # ollama, manual_json, prompt_only
vlm_model: str = ""
@@ -2148,7 +2145,7 @@ def _cf_cleaner_record(
return {
"captionforge_pass": CLEANER_PASS,
"engine": "CaptionForge VLM Validator Engine",
"engine_version": getattr(config, "engine_version", "0.2.0"),
"engine_version": getattr(config, "engine_version", CAPTIONFORGE_VERSION),
"contract": "vlm_statement_cleaner_v0.1",
"image_key": _cf_cleaner_record_key(source_record),
"image": str(source_record.get("image") or ""),
+28 -46
View File
@@ -9,12 +9,9 @@ JLC Joy Caption Engine
- Repository
https://github.com/Damkohler/CaptionForge
- CaptionForge focuses on practical dataset-captioning infrastructure for:
• LoRA dataset preparation
• multi-engine caption generation
• JSONL audit trails
• claim extraction and refinement
• consensus-oriented caption improvement
- CaptionForge 1.0 uses independent Pass-A witnesses, text-LLM synthesis,
image-aware validation, SHORT/TAGGY formatting, and JSONL audit trails to
produce grounded LoRA dataset captions.
- Module Purpose
- The **JLC Joy Caption Engine** provides the shared importable backend for
@@ -101,12 +98,10 @@ JLC Joy Caption Engine
- The engine therefore prioritizes reproducibility, auditability, and clean
separation between model-specific inference code and ComfyUI node wrappers.
- ⚠️ Development Status
- This is early CaptionForge Pass A engine infrastructure.
- Registry entries are intentionally conservative and should be expanded only
after processor/model compatibility is tested.
- Memory behavior, prompt presets, and audit fields may evolve as the
multi-pass CaptionForge pipeline matures.
- Production Status
- Active CaptionForge 1.0 Joy witness backend. Its registry remains
deliberately conservative because processor/model compatibility and VRAM
behavior are model-specific.
- Attribution & License
- Concept and implementation by **J. L. Córdova**
@@ -150,7 +145,6 @@ from dataclasses import asdict, dataclass, field
from datetime import datetime
from pathlib import Path
from typing import Any, Iterable, Optional
import warnings
from PIL import Image
import torch
@@ -162,22 +156,10 @@ from .captionforge_model_cache import (
prepare_for_model_load,
unload_after_run,
)
# -------------------------------------------------------------------------
# Eliminate noise from transformers: UserWarning:
# MatMul8bitLt: inputs will be cast from torch.float32 to float16 during quantization
# -------------------------------------------------------------------------
warnings.filterwarnings(
"ignore",
message=r".*MatMul8bitLt: inputs will be cast.*",
category=UserWarning,
)
warnings.filterwarnings(
"ignore",
message=r".*torchvision backend image processor with LANCZOS resample.*",
from .captionforge_joy_warnings import quantized_inference_warnings, warn_if_suspicious_8bit_stack
from .captionforge_cleanup import (
remove_forbidden_phrases as _remove_forbidden_phrases_boundary_safe,
replace_phrases as _replace_phrases_boundary_safe,
)
@@ -389,7 +371,7 @@ class CleanupConfig:
forbidden_phrases: list[str] = field(default_factory=list)
replacement_rules: list[tuple[str, str]] = field(default_factory=list)
replace_case_insensitive: bool = True
replace_whole_words_only: bool = False
replace_whole_words_only: bool = True
strip_boilerplate_prefixes: bool = True
strip_trailing_period: bool = True
@@ -801,7 +783,7 @@ def apply_replacements(
caption: str,
rules: list[tuple[str, str]],
case_insensitive: bool = True,
whole_words_only: bool = False,
whole_words_only: bool = True,
) -> str:
if not rules:
return caption
@@ -815,13 +797,15 @@ def apply_replacements(
if not old:
continue
flags = re.IGNORECASE if case_insensitive else 0
pattern = re.escape(old)
if whole_words_only:
pattern = r"\b" + pattern + r"\b"
result = re.sub(pattern, new, result, flags=flags)
result = _replace_phrases_boundary_safe(
result,
[(old, new)],
case_insensitive=case_insensitive,
)
else:
flags = re.IGNORECASE if case_insensitive else 0
result = re.sub(re.escape(old), new, result, flags=flags)
return result
@@ -830,14 +814,7 @@ def remove_forbidden_phrases(caption: str, forbidden_phrases: list[str]) -> str:
if not forbidden_phrases:
return caption
result = caption
for phrase in forbidden_phrases:
phrase = phrase.strip()
if not phrase:
continue
result = re.sub(re.escape(phrase), "", result, flags=re.IGNORECASE)
result = _remove_forbidden_phrases_boundary_safe(caption, forbidden_phrases)
result = re.sub(r"\s+,", ",", result)
result = re.sub(r",\s*,+", ",", result)
result = re.sub(r"\s+", " ", result)
@@ -1348,6 +1325,7 @@ class JoyCaptionEngine:
self._free_memory(self.model_size_bytes, self.offload_device)
self.model.to(self.offload_device)
else:
warn_if_suspicious_8bit_stack("Joy Caption")
print(f"[JLC Joy Engine] Loading model in {self.config.memory_mode}: {local_path}")
try:
from transformers import BitsAndBytesConfig
@@ -1582,7 +1560,11 @@ class JoyCaptionEngine:
generation_kwargs["eos_token_id"] = eos_token_id
try:
with torch.autocast(
# Keep Joy's native BF16 path. LLM.int8 casts activations to FP16
# internally; silence only that exact diagnostic during 8-bit generation.
with quantized_inference_warnings(
self.config.memory_mode == "Balanced (8-bit)"
), torch.autocast(
device_type=device_type,
dtype=torch.bfloat16,
enabled=autocast_enabled and bf16_supported,
+254 -51
View File
@@ -9,12 +9,9 @@ JLC Qwen Caption Engine
- Repository
https://github.com/Damkohler/CaptionForge
- CaptionForge focuses on practical dataset-captioning infrastructure for:
• LoRA dataset preparation
• multi-engine caption generation
• JSONL audit trails
• claim extraction and refinement
• consensus-oriented caption improvement
- CaptionForge 1.0 uses independent Pass-A witnesses, text-LLM synthesis,
image-aware validation, SHORT/TAGGY formatting, and JSONL audit trails to
produce grounded LoRA dataset captions.
- Module Purpose
- The **JLC Qwen Caption Engine** provides the shared importable backend for
@@ -112,12 +109,10 @@ JLC Qwen Caption Engine
- The engine therefore prioritizes reproducibility, auditability, and clean
separation between model-specific inference code and ComfyUI node wrappers.
- ⚠️ Development Status
- This is early CaptionForge Pass A engine infrastructure.
- Registry entries should be expanded only after model-class, processor, and
memory behavior are tested.
- Quantization, prompt presets, compatibility patches, and audit fields may
evolve as the multi-pass CaptionForge pipeline matures.
- Production Status
- Active CaptionForge 1.0 Qwen witness backend. Registry expansion remains
conservative because processor/model compatibility, quantization, and
VRAM behavior are model-specific.
- Attribution & License
- Concept and implementation by **J. L. Córdova**
@@ -151,9 +146,11 @@ MANIFEST = {
),
}
import ctypes
import fnmatch
import json
import logging
import os
import random
import re
import shutil
@@ -161,7 +158,6 @@ from dataclasses import asdict, dataclass, field
from datetime import datetime
from pathlib import Path
from typing import Any, Iterable, Optional
import warnings
from PIL import Image
import torch
@@ -173,6 +169,11 @@ from .captionforge_model_cache import (
prepare_for_model_load,
unload_after_run,
)
from .captionforge_joy_warnings import quantized_inference_warnings, warn_if_suspicious_8bit_stack
from .captionforge_cleanup import (
remove_forbidden_phrases as _remove_forbidden_phrases_boundary_safe,
replace_phrases as _replace_phrases_boundary_safe,
)
try:
from .captionforge_caption_prompt_kit import (
@@ -339,7 +340,7 @@ class CleanupConfig:
forbidden_phrases: list[str] = field(default_factory=list)
replacement_rules: list[tuple[str, str]] = field(default_factory=list)
replace_case_insensitive: bool = True
replace_whole_words_only: bool = False
replace_whole_words_only: bool = True
strip_boilerplate_prefixes: bool = True
strip_trailing_period: bool = True
@@ -786,7 +787,7 @@ def apply_replacements(
caption: str,
rules: list[tuple[str, str]],
case_insensitive: bool = True,
whole_words_only: bool = False,
whole_words_only: bool = True,
) -> str:
if not rules:
return caption
@@ -800,13 +801,15 @@ def apply_replacements(
if not old:
continue
flags = re.IGNORECASE if case_insensitive else 0
pattern = re.escape(old)
if whole_words_only:
pattern = r"\b" + pattern + r"\b"
result = re.sub(pattern, new, result, flags=flags)
result = _replace_phrases_boundary_safe(
result,
[(old, new)],
case_insensitive=case_insensitive,
)
else:
flags = re.IGNORECASE if case_insensitive else 0
result = re.sub(re.escape(old), new, result, flags=flags)
return result
@@ -815,14 +818,7 @@ def remove_forbidden_phrases(caption: str, forbidden_phrases: list[str]) -> str:
if not forbidden_phrases:
return caption
result = caption
for phrase in forbidden_phrases:
phrase = phrase.strip()
if not phrase:
continue
result = re.sub(re.escape(phrase), "", result, flags=re.IGNORECASE)
result = _remove_forbidden_phrases_boundary_safe(caption, forbidden_phrases)
result = re.sub(r"\s+,", ",", result)
result = re.sub(r",\s*,+", ",", result)
result = re.sub(r"\s+", " ", result)
@@ -1079,6 +1075,175 @@ def _cuda_diagnostic_line() -> str:
except Exception as exc:
return f"CUDA diagnostics unavailable: {exc}"
def _available_system_memory_bytes() -> int | None:
"""Best-effort available physical RAM without adding a new dependency."""
try:
if os.name == "nt":
class MEMORYSTATUSEX(ctypes.Structure):
_fields_ = [
("dwLength", ctypes.c_ulong),
("dwMemoryLoad", ctypes.c_ulong),
("ullTotalPhys", ctypes.c_ulonglong),
("ullAvailPhys", ctypes.c_ulonglong),
("ullTotalPageFile", ctypes.c_ulonglong),
("ullAvailPageFile", ctypes.c_ulonglong),
("ullTotalVirtual", ctypes.c_ulonglong),
("ullAvailVirtual", ctypes.c_ulonglong),
("ullAvailExtendedVirtual", ctypes.c_ulonglong),
]
status = MEMORYSTATUSEX()
status.dwLength = ctypes.sizeof(MEMORYSTATUSEX)
if ctypes.windll.kernel32.GlobalMemoryStatusEx(ctypes.byref(status)):
return int(status.ullAvailPhys)
page_size = os.sysconf("SC_PAGE_SIZE")
avail_pages = os.sysconf("SC_AVPHYS_PAGES")
return int(page_size * avail_pages)
except Exception:
return None
def _qwen_memory_budget(headroom: float = 0.20) -> dict[Any, int]:
"""Return conservative Accelerate max_memory budgets from currently free memory."""
usable_fraction = max(0.10, min(1.0, 1.0 - float(headroom)))
budgets: dict[Any, int] = {}
if torch.cuda.is_available():
try:
index = torch.cuda.current_device()
free_bytes, _total_bytes = torch.cuda.mem_get_info(index)
budgets[index] = max(1, int(free_bytes * usable_fraction))
except Exception:
pass
available_ram = _available_system_memory_bytes()
if available_ram:
# infer_auto_device_map below estimates the skeleton with dtype=int8.
# bitsandbytes CPU-offloaded modules, however, remain FP32. Reduce the
# CPU capacity by 4x so an int8-sized placement estimate corresponds to
# the actual four-byte-per-parameter CPU residency requirement.
budgets["cpu"] = max(
1,
int((available_ram * usable_fraction) / 4.0),
)
return budgets
def _format_memory_budget(max_memory: dict[Any, int] | None) -> str:
if not max_memory:
return "automatic"
return ", ".join(
f"{device}={_format_bytes(value)}"
for device, value in max_memory.items()
)
def _summarize_device_map(device_map: dict[str, Any] | None) -> str:
if not isinstance(device_map, dict) or not device_map:
return "none"
counts: dict[str, int] = {}
for mapped_device in device_map.values():
if isinstance(mapped_device, int):
label = f"cuda:{mapped_device}"
else:
label = str(mapped_device)
counts[label] = counts.get(label, 0) + 1
return ", ".join(
f"{device}: {count} module(s)"
for device, count in sorted(counts.items())
)
def _resolve_model_execution_device(model: Any) -> torch.device:
"""Choose the device that should receive inference inputs for dispatched models."""
hook = getattr(model, "_hf_hook", None)
hook_device = getattr(hook, "execution_device", None)
if hook_device is not None:
return torch.device(hook_device)
device_map = getattr(model, "hf_device_map", None)
if isinstance(device_map, dict):
for mapped_device in device_map.values():
if mapped_device in {"cpu", "disk", "meta"}:
continue
if isinstance(mapped_device, int):
return torch.device(f"cuda:{mapped_device}")
try:
candidate = torch.device(mapped_device)
except Exception:
continue
if candidate.type not in {"cpu", "meta"}:
return candidate
return next(model.parameters()).device
def _build_qwen_8bit_device_map(
model_cls: Any,
local_path: Path,
trust_remote_code: bool,
) -> tuple[dict[str, Any] | str, dict[Any, int]]:
"""
Build a conservative Accelerate placement map without materializing weights.
GPU placement is estimated as int8. CPU overflow remains FP32 at real load
time through bitsandbytes CPU offload. Twenty percent of currently available
GPU/RAM is intentionally left outside the placement budget.
"""
if not torch.cuda.is_available():
return "auto", {}
max_memory = _qwen_memory_budget(headroom=0.20)
if not max_memory:
return "auto", {}
try:
from accelerate import infer_auto_device_map, init_empty_weights
from transformers import AutoConfig
model_config = AutoConfig.from_pretrained(
str(local_path),
trust_remote_code=trust_remote_code,
)
with init_empty_weights():
empty_model = model_cls(model_config)
no_split_modules = getattr(empty_model, "_no_split_modules", None)
device_map = infer_auto_device_map(
empty_model,
max_memory=max_memory,
no_split_module_classes=no_split_modules,
dtype=torch.int8,
)
if any(str(device).lower() == "disk" for device in device_map.values()):
raise RuntimeError(
"CaptionForge Qwen 8-bit placement would require disk offload. "
"Disk spill is intentionally not enabled because it is extremely slow "
"and can make ComfyUI inference impractical. More GPU VRAM or available "
"system RAM is required for this model."
)
return device_map, max_memory
except RuntimeError:
raise
except Exception as exc:
print(
"[JLC Qwen Engine] Adaptive placement probe was unavailable; "
f"falling back to Accelerate device_map='auto': {exc}"
)
return "auto", max_memory
def json_safe(value):
if isinstance(value, set):
return sorted(value)
@@ -1247,6 +1412,9 @@ class QwenCaptionEngine:
self.model = cached["model"]
print(f"[JLC Qwen Engine] Reusing cached model: {local_path}")
return
if quantization == "bnb_8bit":
warn_if_suspicious_8bit_stack("Qwen Caption")
cache_policy = getattr(
self.config,
@@ -1319,19 +1487,26 @@ class QwenCaptionEngine:
"Install/verify compatible packages before using quantization='bnb_8bit'."
) from exc
warnings.filterwarnings(
"ignore",
message=r"MatMul8bitLt: inputs will be cast from torch\.bfloat16 to float16 during quantization",
category=UserWarning,
module=r"bitsandbytes\.autograd\._functions",
model_kwargs["quantization_config"] = BitsAndBytesConfig(
load_in_8bit=True,
llm_int8_enable_fp32_cpu_offload=True,
)
model_kwargs["quantization_config"] = BitsAndBytesConfig(load_in_8bit=True)
# bitsandbytes quantized models should be loaded through Accelerate dispatch.
# Keep this explicit so users do not accidentally request a later .to(device).
if not effective_device_map:
effective_device_map = "auto"
# Balanced 8-bit is intended to remain usable when the complete
# quantized model does not fit in VRAM. Build an adaptive placement
# map with headroom and permit supported FP32 CPU overflow.
if not effective_device_map or effective_device_map == "auto":
effective_device_map, max_memory = _build_qwen_8bit_device_map(
model_cls,
local_path,
self.config.trust_remote_code,
)
if max_memory:
model_kwargs["max_memory"] = max_memory
print(
"[JLC Qwen Engine] Adaptive memory budget: "
f"{_format_memory_budget(max_memory)}"
)
if effective_device_map:
model_kwargs["device_map"] = effective_device_map
@@ -1342,14 +1517,40 @@ class QwenCaptionEngine:
f"device_map={effective_device_map!r}, quantization={quantization}"
)
self.model = model_cls.from_pretrained(
str(local_path),
**model_kwargs,
)
try:
self.model = model_cls.from_pretrained(
str(local_path),
**model_kwargs,
)
except Exception as exc:
message = str(exc)
if quantization == "bnb_8bit" and (
"Some modules are dispatched on the CPU or the disk" in message
or "llm_int8_enable_fp32_cpu_offload" in message
or "device_map" in message and "CPU" in message
):
raise RuntimeError(
"CaptionForge could not place this Qwen model within the "
"available GPU/CPU memory budget using Balanced (8-bit). "
f"{_cuda_diagnostic_line()}. "
f"Memory budget: {_format_memory_budget(model_kwargs.get('max_memory'))}. "
"The 8-bit path supports FP32 CPU offload, but this configuration "
"still could not be loaded. Close other GPU/RAM-heavy applications, "
"use a smaller Qwen model, or free additional system memory."
) from exc
raise
device_map = getattr(self.model, "hf_device_map", None)
if device_map:
print(f"[JLC Qwen Engine] hf_device_map: {device_map}")
if any(str(mapped).lower() == "disk" for mapped in device_map.values()):
raise RuntimeError(
"CaptionForge Qwen loaded with disk-offloaded modules. "
"This configuration is not supported for interactive captioning."
)
print(
"[JLC Qwen Engine] Placement: "
f"{_summarize_device_map(device_map)}"
)
else:
try:
print(f"[JLC Qwen Engine] first parameter device: {next(self.model.parameters()).device}")
@@ -1485,8 +1686,7 @@ class QwenCaptionEngine:
)
try:
device = next(self.model.parameters()).device
inputs = inputs.to(device)
inputs = inputs.to(_resolve_model_execution_device(self.model))
except Exception:
pass
@@ -1519,10 +1719,13 @@ class QwenCaptionEngine:
generation_kwargs["do_sample"] = False
generated_ids = self.model.generate(
**inputs,
**generation_kwargs,
)
with quantized_inference_warnings(
self._resolve_quantization(self.config.quantization) == "bnb_8bit"
):
generated_ids = self.model.generate(
**inputs,
**generation_kwargs,
)
generated_ids_trimmed = [
out_ids[len(in_ids):]
+1 -1
View File
@@ -1 +1 @@
# empty file
"""ComfyUI-facing CaptionForge nodes and production support helpers."""
+1 -1
View File
@@ -1 +1 @@
# empty file
"""Active Joy, Qwen, and Ollama Pass-A witness node wrappers."""
@@ -8,11 +8,9 @@ JLC CaptionForge Joy Caption — ComfyUI Node Wrapper
- Repository
https://github.com/Damkohler/CaptionForge
- CaptionForge focuses on practical dataset-captioning infrastructure for
LoRA dataset preparation, using multi-engine caption generation, JSONL
audit trails, claim extraction and refinement, text-LLM distillation,
image-aware VLM validation, and consensus-oriented caption improvement
to produce grounded, auditable training captions.
- CaptionForge 1.0 uses independent Pass-A witnesses, text-LLM synthesis,
image-aware validation, SHORT/TAGGY formatting, and JSONL audit trails to
produce grounded LoRA dataset captions.
- Node Purpose
- The **JLC CaptionForge Joy Caption** node provides a ComfyUI
@@ -30,7 +28,6 @@ JLC CaptionForge Joy Caption — ComfyUI Node Wrapper
• clear template-vs-custom prompt controls
• CaptionForge Pipeline Planner consumption through `pipeline_plan`
• CaptionForge Template Options consumption through `template_options`
• TXT audit sidecar writing in planned runs
• shared JSONL audit output in planned runs
• direct ComfyUI caption and resolved-prompt string outputs
• IMAGE and pipeline-plan passthrough for clean graph chaining
@@ -89,7 +86,6 @@ JLC CaptionForge Joy Caption — ComfyUI Node Wrapper
• shared LoRA trigger word
- In planned mode, the node can write:
• TXT audit sidecar captions
• JSONL audit records
• run-configuration JSON files
@@ -134,12 +130,10 @@ JLC CaptionForge Joy Caption — ComfyUI Node Wrapper
clean separation between the ComfyUI interface and the backend model
engine.
- ⚠️ Development Status
- This is early CaptionForge raw-caption infrastructure.
- The UI, model registry, prompt behavior, and output audit fields may
evolve as the multi-pass CaptionForge pipeline matures.
- The node is intended for local dataset preparation and controlled caption
audit workflows.
- Production Status
- Active CaptionForge 1.0 Pass-A witness node. In planned mode the Planner
owns run counts and sampling schedules; in standalone mode this node's
visible controls are authoritative.
- Attribution & License
- Concept and implementation by **J. L. Córdova**
@@ -173,7 +167,7 @@ MANIFEST = {
"consumed only through the template_options input from the CaptionForge "
"Template Options sidecar. The UI separates caption_template_mode and "
"custom_prompt_mode for clearer prompt routing while delegating model "
"loading, memory mode, generation, cleanup, TXT sidecars, and JSONL audit "
"loading, memory mode, generation, cleanup, and JSONL audit "
"records to the Joy caption engine."
),
}
@@ -206,9 +200,11 @@ from ...engines.jlc_joy_caption_engine import (
resolve_prompt,
timestamp,
write_run_config_json,
write_text_sidecar,
)
from ...engines.captionforge_pipeline_planner_engine import expand_captionforge_runs
from ...engines.captionforge_cleanup import resolve_cleanup_settings
from ...engines.captionforge_source_identity import file_source_identity, optional_image_identity
from ...engines.captionforge_dataset_export import is_dataset_export
from ..jlc_captionforge_template_options import resolve_effective_extra_options
@@ -250,20 +246,7 @@ def _tensor_to_pil(image_tensor) -> list[Image.Image]:
return images
def _safe_source_name(value: str) -> str:
cleaned = []
for ch in value.replace("\\", "/"):
if ch.isalnum() or ch in {"-", "_", "."}:
cleaned.append(ch)
elif ch == "/":
cleaned.append("__")
else:
cleaned.append("_")
out = "".join(cleaned).strip("._")
return out or "image"
def _iter_input_path_images(input_path: str, recursive: bool, filename_glob: str) -> list[tuple[str, Path]]:
def _iter_input_path_images(input_path: str, recursive: bool, filename_glob: str) -> list[tuple[str, str, Path]]:
root = Path(str(input_path or "").strip())
if not root:
return []
@@ -271,22 +254,26 @@ def _iter_input_path_images(input_path: str, recursive: bool, filename_glob: str
raise RuntimeError(f"CaptionForge input_path does not exist: {root}")
glob_text = (filename_glob or "*").strip() or "*"
explicit_dataset_source = is_dataset_export(root)
if root.is_file():
if root.suffix.lower() not in _SUPPORTED_IMAGE_SUFFIXES:
raise RuntimeError(f"CaptionForge input_path is not a supported image file: {root}")
return [(_safe_source_name(root.stem), root)]
image_name, image_key = file_source_identity(root, root)
return [(image_name, image_key, root)]
pattern_iter = root.rglob(glob_text) if recursive else root.glob(glob_text)
paths = sorted(
p for p in pattern_iter
if p.is_file() and p.suffix.lower() in _SUPPORTED_IMAGE_SUFFIXES
if p.is_file()
and p.suffix.lower() in _SUPPORTED_IMAGE_SUFFIXES
and (explicit_dataset_source or not is_dataset_export(p))
)
items: list[tuple[str, Path]] = []
items: list[tuple[str, str, Path]] = []
for path in paths:
rel = path.relative_to(root).with_suffix("")
items.append((_safe_source_name(str(rel)), path))
image_name, image_key = file_source_identity(path, root)
items.append((image_name, image_key, path))
return items
@@ -323,12 +310,6 @@ def _planned_caption_jsonl_path(pipeline_plan) -> Path | None:
return None
def _run_txt_path(output_dir: Path, source_name: str, run_count: int, run_index: int) -> Path:
if run_count <= 1:
return output_dir / f"{source_name}.txt"
return output_dir / f"{source_name}__cf_run_{run_index:02d}.txt"
def _use_template_mode(caption_template_mode: bool, custom_prompt_mode: bool) -> bool:
if bool(custom_prompt_mode):
return False
@@ -666,7 +647,7 @@ class JLC_CaptionForgeJoy:
"resolved_prompt",
)
FUNCTION = "caption"
CATEGORY = "Captioning/CaptionForge/Captioning Nodes"
CATEGORY = "Caption/CaptionForge/Caption Nodes"
@classmethod
def IS_CHANGED(cls, **kwargs):
@@ -715,7 +696,7 @@ class JLC_CaptionForgeJoy:
if download_probe_only:
result = probe_registry_model_download(model, JLC_JOY_MODEL_ROOT)
return (image, pipeline_plan, result, resolved_prompt)
return (image, pipeline_plan, template_options, result, resolved_prompt)
effective_seed = -1 if seed is None else int(seed)
@@ -739,9 +720,12 @@ class JLC_CaptionForgeJoy:
if run_plan_connected and not run_plan:
status = "[CaptionForge] Joy disabled by Pipeline Planner."
print(status)
return (image, pipeline_plan, status, resolved_prompt)
return (image, pipeline_plan, template_options, status, resolved_prompt)
first_run = run_plan[0]
effective_forbidden, effective_replacements = resolve_cleanup_settings(
_normalize_pipeline_plan(pipeline_plan), forbidden_phrases, replace_pairs
)
generation = GenerationConfig(
max_new_tokens=int(first_run.max_new_tokens),
@@ -756,8 +740,8 @@ class JLC_CaptionForgeJoy:
trigger="",
prefix=(f"{first_run.trigger_word}," if first_run.trigger_word else ""),
suffix="",
forbidden_phrases=_parse_forbidden_lines(forbidden_phrases),
replacement_rules=_parse_replace_pairs(replace_pairs),
forbidden_phrases=effective_forbidden,
replacement_rules=effective_replacements,
)
joy_config = JoyCaptionConfig(
@@ -780,8 +764,8 @@ class JLC_CaptionForgeJoy:
engine = JoyCaptionEngine(config=joy_config, generation=generation, cleanup=cleanup)
direct_images = [(f"comfy_image_{i:04d}", pil) for i, pil in enumerate(_tensor_to_pil(image))]
file_images: list[tuple[str, Path]] = []
direct_images = [(*optional_image_identity(i), pil) for i, pil in enumerate(_tensor_to_pil(image))]
file_images: list[tuple[str, str, Path]] = []
if first_run.input_path:
file_images = _iter_input_path_images(first_run.input_path, first_run.recursive, first_run.filename_glob)
@@ -819,7 +803,7 @@ class JLC_CaptionForgeJoy:
dry_run=False,
)
def process_one(source_name: str, pil: Image.Image):
def process_one(source_name: str, image_key: str, pil: Image.Image):
for run in run_plan:
engine.generation = GenerationConfig(
max_new_tokens=int(run.max_new_tokens),
@@ -833,8 +817,8 @@ class JLC_CaptionForgeJoy:
trigger="",
prefix=(f"{run.trigger_word}," if run.trigger_word else ""),
suffix="",
forbidden_phrases=_parse_forbidden_lines(forbidden_phrases),
replacement_rules=_parse_replace_pairs(replace_pairs),
forbidden_phrases=effective_forbidden,
replacement_rules=effective_replacements,
)
engine.config.max_size = int(run.max_size)
@@ -861,29 +845,22 @@ class JLC_CaptionForgeJoy:
captionforge_pass="A",
model_family="joy",
ensemble_run_index=run.ensemble_run_index,
image_key=source_name,
image_key=image_key,
)
all_records.append(record)
if run_plan_connected and jsonl_path is not None and output_dir is not None:
append_jsonl_records(jsonl_path, [record], dry_run=False)
write_text_sidecar(
_run_txt_path(output_dir, source_name, len(run_plan), run.ensemble_run_index),
record.caption,
overwrite=True,
backup_existing=False,
dry_run=False,
)
print(
f"[JLC CaptionForge Joy] Captioned {source_name} "
f"run {run.ensemble_run_index + 1}/{len(run_plan)}"
)
for source_name, pil in direct_images:
process_one(source_name, pil)
for source_name, image_key, pil in direct_images:
process_one(source_name, image_key, pil)
for source_name, path in file_images:
process_one(source_name, _open_image(path))
for source_name, image_key, path in file_images:
process_one(source_name, image_key, _open_image(path))
if not keep_loaded:
engine.unload()
@@ -8,11 +8,9 @@ JLC CaptionForge Ollama Caption — ComfyUI Node Wrapper
- Repository
https://github.com/Damkohler/CaptionForge
- CaptionForge focuses on practical dataset-captioning infrastructure for
LoRA dataset preparation, using multi-engine caption generation, JSONL
audit trails, claim extraction and refinement, text-LLM distillation,
image-aware VLM validation, and consensus-oriented caption improvement
to produce grounded, auditable training captions.
- CaptionForge 1.0 uses independent Pass-A witnesses, text-LLM synthesis,
image-aware validation, SHORT/TAGGY formatting, and JSONL audit trails to
produce grounded LoRA dataset captions.
- Node Purpose
- The **JLC CaptionForge Ollama Caption** node provides a ComfyUI
@@ -31,7 +29,6 @@ JLC CaptionForge Ollama Caption — ComfyUI Node Wrapper
• clear template-vs-custom prompt controls
• CaptionForge Pipeline Planner consumption through `pipeline_plan`
• CaptionForge Template Options consumption through `template_options`
• TXT audit sidecar writing in planned runs
• shared JSONL audit output in planned runs
• direct ComfyUI caption and resolved-prompt string outputs
• IMAGE and pipeline-plan passthrough for clean graph chaining
@@ -91,7 +88,6 @@ JLC CaptionForge Ollama Caption — ComfyUI Node Wrapper
• shared LoRA trigger word
- In planned mode, the node can write:
• TXT audit sidecar captions
• JSONL audit records
• run-configuration JSON files
@@ -142,12 +138,10 @@ JLC CaptionForge Ollama Caption — ComfyUI Node Wrapper
clean separation between the ComfyUI interface and the backend model
service.
- ⚠️ Development Status
- This is early CaptionForge Ollama-backed raw-caption infrastructure.
- The UI, model-tag config, prompt behavior, and output audit fields may
evolve as the multi-pass CaptionForge pipeline matures.
- The node is intended for local dataset preparation and controlled caption
audit workflows.
- Production Status
- Active CaptionForge 1.0 Pass-A witness node. The selected Ollama VLM tag
remains local to this node; a connected Planner owns shared run counts and
sampling schedules.
- Attribution & License
- Concept and implementation by **J. L. Córdova**
@@ -206,6 +200,10 @@ from PIL import Image
import folder_paths
from ...engines.captionforge_pipeline_planner_engine import expand_captionforge_runs
from ...engines.captionforge_cleanup import resolve_cleanup_settings
from ...engines.captionforge_source_identity import file_source_identity, optional_image_identity
from ...engines.captionforge_dataset_export import is_dataset_export
from ...engines.captionforge_cleanup import contains_forbidden_phrase, replace_phrases
from ...engines.captionforge_caption_prompt_kit import (
CAPTION_LENGTH_CHOICES,
CAPTION_TYPE_CHOICES,
@@ -355,7 +353,7 @@ class OllamaCaptionRecord:
model_path: str = ""
prompt: str = ""
system_prompt: str = ""
seed: int = -1
seed: int | None = None
temperature: float = 0.18
top_p: float = 0.92
top_k: int = 60
@@ -654,7 +652,7 @@ def _ollama_options(
top_p: float,
top_k: int,
repetition_penalty: float,
seed: int,
seed: int | None,
) -> dict[str, Any]:
"""Build Ollama generation options.
@@ -669,7 +667,7 @@ def _ollama_options(
"top_k": int(top_k),
"repeat_penalty": float(repetition_penalty),
}
if int(seed) >= 0:
if seed is not None and int(seed) >= 0:
options["seed"] = int(seed)
return options
@@ -780,7 +778,7 @@ def _ollama_generate_caption(
top_p: float,
top_k: int,
repetition_penalty: float,
seed: int,
seed: int | None,
max_size: int,
keep_loaded: bool,
timeout: float,
@@ -917,20 +915,7 @@ def _pil_to_base64_png(pil: Image.Image, max_size: int) -> str:
return base64.b64encode(buf.getvalue()).decode("ascii")
def _safe_source_name(value: str) -> str:
cleaned = []
for ch in value.replace("\\", "/"):
if ch.isalnum() or ch in {"-", "_", "."}:
cleaned.append(ch)
elif ch == "/":
cleaned.append("__")
else:
cleaned.append("_")
out = "".join(cleaned).strip("._")
return out or "image"
def _iter_input_path_images(input_path: str, recursive: bool, filename_glob: str) -> list[tuple[str, Path]]:
def _iter_input_path_images(input_path: str, recursive: bool, filename_glob: str) -> list[tuple[str, str, Path]]:
root = Path(str(input_path or "").strip())
if not root:
return []
@@ -938,22 +923,26 @@ def _iter_input_path_images(input_path: str, recursive: bool, filename_glob: str
raise RuntimeError(f"CaptionForge input_path does not exist: {root}")
glob_text = (filename_glob or "*").strip() or "*"
explicit_dataset_source = is_dataset_export(root)
if root.is_file():
if root.suffix.lower() not in _SUPPORTED_IMAGE_SUFFIXES:
raise RuntimeError(f"CaptionForge input_path is not a supported image file: {root}")
return [(_safe_source_name(root.stem), root)]
image_name, image_key = file_source_identity(root, root)
return [(image_name, image_key, root)]
pattern_iter = root.rglob(glob_text) if recursive else root.glob(glob_text)
paths = sorted(
p for p in pattern_iter
if p.is_file() and p.suffix.lower() in _SUPPORTED_IMAGE_SUFFIXES
if p.is_file()
and p.suffix.lower() in _SUPPORTED_IMAGE_SUFFIXES
and (explicit_dataset_source or not is_dataset_export(p))
)
items: list[tuple[str, Path]] = []
items: list[tuple[str, str, Path]] = []
for path in paths:
rel = path.relative_to(root).with_suffix("")
items.append((_safe_source_name(str(rel)), path))
image_name, image_key = file_source_identity(path, root)
items.append((image_name, image_key, path))
return items
@@ -990,12 +979,6 @@ def _planned_caption_jsonl_path(pipeline_plan) -> Path | None:
return None
def _run_txt_path(output_dir: Path, source_name: str, run_count: int, run_index: int) -> Path:
if run_count <= 1:
return output_dir / f"{source_name}.txt"
return output_dir / f"{source_name}__cf_run_{run_index:02d}.txt"
def _use_template_mode(caption_template_mode: bool, custom_prompt_mode: bool) -> bool:
if bool(custom_prompt_mode):
return False
@@ -1048,14 +1031,16 @@ def _clean_caption(
replacement_rules: list[tuple[str, str]],
) -> tuple[str, str]:
text = str(raw or "").strip().strip('"').strip()
for old, new in replacement_rules:
text = text.replace(old, new)
text = replace_phrases(
text,
replacement_rules,
case_insensitive=False,
)
if forbidden_phrases:
kept: list[str] = []
for line in text.splitlines() or [text]:
lowered = line.lower()
if any(phrase.lower() in lowered for phrase in forbidden_phrases if phrase):
if contains_forbidden_phrase(line, forbidden_phrases):
continue
kept.append(line)
text = "\n".join(line.strip() for line in kept if line.strip()).strip()
@@ -1074,11 +1059,6 @@ def _append_jsonl_records(path: Path, records: list[OllamaCaptionRecord]) -> Non
f.write(json.dumps(asdict(record), ensure_ascii=False) + "\n")
def _write_text_sidecar(path: Path, text: str) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(str(text or ""), encoding="utf-8")
def _write_run_config_json(path: Path, data: dict[str, Any]) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(json.dumps(data, indent=2, ensure_ascii=False) + "\n", encoding="utf-8")
@@ -1096,6 +1076,8 @@ def _build_run_config(
top_k: int,
repetition_penalty: float,
max_size: int,
forbidden_phrases: list[str],
replacement_rules: list[tuple[str, str]],
) -> dict[str, Any]:
return {
"backend": "ollama",
@@ -1111,6 +1093,14 @@ def _build_run_config(
"repetition_penalty": float(repetition_penalty),
"max_size": int(max_size),
},
"cleanup": {
"forbidden_phrases": list(forbidden_phrases),
"replacement_rules": [list(rule) for rule in replacement_rules],
"replacement_match_mode": "whole_word_or_phrase_boundary",
"replacement_case_insensitive": False,
"forbidden_match_mode": "whole_word_or_phrase_boundary",
"forbidden_action": "drop_matching_line",
},
"timestamp": datetime.now().isoformat(timespec="seconds"),
}
@@ -1455,7 +1445,7 @@ class JLC_CaptionForgeOllamaCaption:
"resolved_prompt",
)
FUNCTION = "caption"
CATEGORY = "Captioning/CaptionForge/Caption Nodes"
CATEGORY = "Caption/CaptionForge/Caption Nodes"
@classmethod
def IS_CHANGED(cls, **kwargs):
@@ -1510,15 +1500,6 @@ class JLC_CaptionForgeOllamaCaption:
resolved_prompt = _format_resolved_prompt(system_prompt, prompt)
_evict_python_models_before_ollama_if_needed("JLC CaptionForge Ollama Caption")
if download_probe_only:
result = _probe_ollama_model(ollama_url, model_tag, timeout=min(timeout, 60.0))
return (image, pipeline_plan, result, resolved_prompt)
_ensure_ollama_model(ollama_url, model_tag, timeout=timeout)
_persist_caption_model_if_possible(model_tag)
effective_seed = -1 if seed is None else int(seed)
run_plan, planner_key = _expand_ollama_runs_compat(
@@ -1534,7 +1515,16 @@ class JLC_CaptionForgeOllamaCaption:
if run_plan_connected and not run_plan:
status = "[CaptionForge] Ollama Caption disabled by Pipeline Planner or Planner has no Ollama caption count yet."
print(status)
return (image, pipeline_plan, status, resolved_prompt)
return (image, pipeline_plan, template_options, status, resolved_prompt)
_evict_python_models_before_ollama_if_needed("JLC CaptionForge Ollama Caption")
if download_probe_only:
result = _probe_ollama_model(ollama_url, model_tag, timeout=min(timeout, 60.0))
return (image, pipeline_plan, template_options, result, resolved_prompt)
_ensure_ollama_model(ollama_url, model_tag, timeout=timeout)
_persist_caption_model_if_possible(model_tag)
if not run_plan:
# Defensive fallback for standalone mode if expand_captionforge_runs ever changes behavior.
@@ -1557,9 +1547,12 @@ class JLC_CaptionForgeOllamaCaption:
run_plan = [standalone_run]
first_run = run_plan[0]
forbidden, replacements = resolve_cleanup_settings(
_normalize_pipeline_plan(pipeline_plan), forbidden_phrases, replace_pairs
)
direct_images = [(f"comfy_image_{i:04d}", pil) for i, pil in enumerate(_tensor_to_pil(image))]
file_images: list[tuple[str, Path]] = []
direct_images = [(*optional_image_identity(i), pil) for i, pil in enumerate(_tensor_to_pil(image))]
file_images: list[tuple[str, str, Path]] = []
if getattr(first_run, "input_path", ""):
file_images = _iter_input_path_images(first_run.input_path, first_run.recursive, first_run.filename_glob)
@@ -1599,14 +1592,13 @@ class JLC_CaptionForgeOllamaCaption:
top_k=int(first_run.top_k),
repetition_penalty=float(repetition_penalty),
max_size=int(first_run.max_size),
forbidden_phrases=forbidden,
replacement_rules=replacements,
),
)
all_records: list[OllamaCaptionRecord] = []
forbidden = _parse_forbidden_lines(forbidden_phrases)
replacements = _parse_replace_pairs(replace_pairs)
def process_one(source_name: str, pil: Image.Image):
def process_one(source_name: str, image_key: str, pil: Image.Image):
for run in run_plan:
t0 = time.perf_counter()
raw_caption = _ollama_generate_caption(
@@ -1620,7 +1612,7 @@ class JLC_CaptionForgeOllamaCaption:
top_p=float(run.top_p),
top_k=int(run.top_k),
repetition_penalty=float(repetition_penalty),
seed=int(run.seed),
seed=run.seed,
max_size=int(run.max_size),
keep_loaded=bool(keep_loaded),
timeout=timeout,
@@ -1643,7 +1635,7 @@ class JLC_CaptionForgeOllamaCaption:
model_path=model_tag,
prompt=prompt,
system_prompt=system_prompt,
seed=int(run.seed),
seed=run.seed,
temperature=float(run.temperature),
top_p=float(run.top_p),
top_k=int(run.top_k),
@@ -1654,7 +1646,7 @@ class JLC_CaptionForgeOllamaCaption:
captionforge_pass="A",
model_family="ollama",
ensemble_run_index=int(run.ensemble_run_index),
image_key=source_name,
image_key=image_key,
backend="ollama",
status=status,
)
@@ -1662,20 +1654,16 @@ class JLC_CaptionForgeOllamaCaption:
if run_plan_connected and jsonl_path is not None and output_dir is not None:
_append_jsonl_records(jsonl_path, [record])
_write_text_sidecar(
_run_txt_path(output_dir, source_name, len(run_plan), int(run.ensemble_run_index)),
record.caption,
)
print(
f"[JLC CaptionForge Ollama Caption] Captioned {source_name} "
f"run {int(run.ensemble_run_index) + 1}/{len(run_plan)} via Planner key '{planner_key}'"
)
for source_name, pil in direct_images:
process_one(source_name, pil)
for source_name, image_key, pil in direct_images:
process_one(source_name, image_key, pil)
for source_name, path in file_images:
process_one(source_name, _open_image(path))
for source_name, image_key, path in file_images:
process_one(source_name, image_key, _open_image(path))
# Ollama model residency is requested through keep_alive on generation calls;
# Ollama's server/runtime policy ultimately decides residency.
@@ -8,11 +8,9 @@ JLC CaptionForge Qwen Caption — ComfyUI Node Wrapper
- Repository
https://github.com/Damkohler/CaptionForge
- CaptionForge focuses on practical dataset-captioning infrastructure for
LoRA dataset preparation, using multi-engine caption generation, JSONL
audit trails, claim extraction and refinement, text-LLM distillation,
image-aware VLM validation, and consensus-oriented caption improvement
to produce grounded, auditable training captions.
- CaptionForge 1.0 uses independent Pass-A witnesses, text-LLM synthesis,
image-aware validation, SHORT/TAGGY formatting, and JSONL audit trails to
produce grounded LoRA dataset captions.
- Node Purpose
- The **JLC CaptionForge Qwen Caption** node provides a ComfyUI
@@ -30,7 +28,6 @@ JLC CaptionForge Qwen Caption — ComfyUI Node Wrapper
• clear template-vs-custom prompt controls
• CaptionForge Pipeline Planner consumption through `pipeline_plan`
• CaptionForge Template Options consumption through `template_options`
• TXT audit sidecar writing in planned runs
• shared JSONL audit output in planned runs
• direct ComfyUI caption and resolved-prompt string outputs
• IMAGE and pipeline-plan passthrough for clean graph chaining
@@ -89,7 +86,6 @@ JLC CaptionForge Qwen Caption — ComfyUI Node Wrapper
• shared LoRA trigger word
- In planned mode, the node can write:
• TXT audit sidecar captions
• JSONL audit records
• run-configuration JSON files
@@ -133,12 +129,10 @@ JLC CaptionForge Qwen Caption — ComfyUI Node Wrapper
clean separation between the ComfyUI interface and the backend model
engine.
- ⚠️ Development Status
- This is early CaptionForge raw-caption infrastructure.
- The UI, model registry, prompt behavior, and output audit fields may
evolve as the multi-pass CaptionForge pipeline matures.
- The node is intended for local dataset preparation and controlled caption
audit workflows.
- Production Status
- Active CaptionForge 1.0 Pass-A witness node. In planned mode the Planner
owns run counts and sampling schedules; in standalone mode this node's
visible controls are authoritative.
- Attribution & License
- Concept and implementation by **J. L. Córdova**
@@ -167,7 +161,7 @@ MANIFEST = {
"the template_options input from the CaptionForge Template Options sidecar. "
"The UI separates caption_template_mode and custom_prompt_mode for clearer "
"prompt routing while delegating model loading, quantization, generation, "
"cleanup, TXT sidecars, and JSONL audit records to the Qwen caption engine."
"cleanup and JSONL audit records to the Qwen caption engine."
),
}
@@ -194,9 +188,11 @@ from ...engines.jlc_qwen_caption_engine import (
probe_registry_model_download,
timestamp,
write_run_config_json,
write_text_sidecar,
)
from ...engines.captionforge_pipeline_planner_engine import expand_captionforge_runs
from ...engines.captionforge_cleanup import resolve_cleanup_settings
from ...engines.captionforge_source_identity import file_source_identity, optional_image_identity
from ...engines.captionforge_dataset_export import is_dataset_export
from ...engines.captionforge_caption_prompt_kit import (
CAPTION_LENGTH_CHOICES,
CAPTION_TYPE_CHOICES,
@@ -291,20 +287,7 @@ def _tensor_to_pil(image_tensor) -> list[Image.Image]:
return images
def _safe_source_name(value: str) -> str:
cleaned = []
for ch in value.replace("\\", "/"):
if ch.isalnum() or ch in {"-", "_", "."}:
cleaned.append(ch)
elif ch == "/":
cleaned.append("__")
else:
cleaned.append("_")
out = "".join(cleaned).strip("._")
return out or "image"
def _iter_input_path_images(input_path: str, recursive: bool, filename_glob: str) -> list[tuple[str, Path]]:
def _iter_input_path_images(input_path: str, recursive: bool, filename_glob: str) -> list[tuple[str, str, Path]]:
root = Path(str(input_path or "").strip())
if not root:
return []
@@ -312,22 +295,26 @@ def _iter_input_path_images(input_path: str, recursive: bool, filename_glob: str
raise RuntimeError(f"CaptionForge input_path does not exist: {root}")
glob_text = (filename_glob or "*").strip() or "*"
explicit_dataset_source = is_dataset_export(root)
if root.is_file():
if root.suffix.lower() not in _SUPPORTED_IMAGE_SUFFIXES:
raise RuntimeError(f"CaptionForge input_path is not a supported image file: {root}")
return [(_safe_source_name(root.stem), root)]
image_name, image_key = file_source_identity(root, root)
return [(image_name, image_key, root)]
pattern_iter = root.rglob(glob_text) if recursive else root.glob(glob_text)
paths = sorted(
p for p in pattern_iter
if p.is_file() and p.suffix.lower() in _SUPPORTED_IMAGE_SUFFIXES
if p.is_file()
and p.suffix.lower() in _SUPPORTED_IMAGE_SUFFIXES
and (explicit_dataset_source or not is_dataset_export(p))
)
items: list[tuple[str, Path]] = []
items: list[tuple[str, str, Path]] = []
for path in paths:
rel = path.relative_to(root).with_suffix("")
items.append((_safe_source_name(str(rel)), path))
image_name, image_key = file_source_identity(path, root)
items.append((image_name, image_key, path))
return items
@@ -364,12 +351,6 @@ def _planned_caption_jsonl_path(pipeline_plan) -> Path | None:
return None
def _run_txt_path(output_dir: Path, source_name: str, run_count: int, run_index: int) -> Path:
if run_count <= 1:
return output_dir / f"{source_name}.txt"
return output_dir / f"{source_name}__cf_run_{run_index:02d}.txt"
def _use_template_mode(caption_template_mode: bool, custom_prompt_mode: bool) -> bool:
if bool(custom_prompt_mode):
return False
@@ -689,7 +670,7 @@ class JLC_CaptionForgeQwen:
"resolved_prompt",
)
FUNCTION = "caption"
CATEGORY = "Captioning/CaptionForge/Captioning Nodes"
CATEGORY = "Caption/CaptionForge/Caption Nodes"
@classmethod
def IS_CHANGED(cls, **kwargs):
@@ -741,7 +722,7 @@ class JLC_CaptionForgeQwen:
if download_probe_only:
result = probe_registry_model_download(model, JLC_QWEN_MODEL_ROOT)
return (image, pipeline_plan, result, resolved_prompt)
return (image, pipeline_plan, template_options, result, resolved_prompt)
effective_seed = -1 if seed is None else int(seed)
@@ -765,9 +746,12 @@ class JLC_CaptionForgeQwen:
if run_plan_connected and not run_plan:
status = "[CaptionForge] Qwen disabled by Pipeline Planner."
print(status)
return (image, pipeline_plan, status, resolved_prompt)
return (image, pipeline_plan, template_options, status, resolved_prompt)
first_run = run_plan[0]
effective_forbidden, effective_replacements = resolve_cleanup_settings(
_normalize_pipeline_plan(pipeline_plan), forbidden_phrases, replace_pairs
)
qwen_quantization_value = "bnb_8bit" if qwen_quantization == "Balanced (8-bit)" else "none"
generation = GenerationConfig(
@@ -783,8 +767,8 @@ class JLC_CaptionForgeQwen:
trigger="",
prefix=(f"{first_run.trigger_word}," if first_run.trigger_word else ""),
suffix="",
forbidden_phrases=_parse_forbidden_lines(forbidden_phrases),
replacement_rules=_parse_replace_pairs(replace_pairs),
forbidden_phrases=effective_forbidden,
replacement_rules=effective_replacements,
)
qwen_config = QwenCaptionConfig(
@@ -807,8 +791,8 @@ class JLC_CaptionForgeQwen:
engine = QwenCaptionEngine(config=qwen_config, generation=generation, cleanup=cleanup)
direct_images = [(f"comfy_image_{i:04d}", pil) for i, pil in enumerate(_tensor_to_pil(image))]
file_images: list[tuple[str, Path]] = []
direct_images = [(*optional_image_identity(i), pil) for i, pil in enumerate(_tensor_to_pil(image))]
file_images: list[tuple[str, str, Path]] = []
if first_run.input_path:
file_images = _iter_input_path_images(first_run.input_path, first_run.recursive, first_run.filename_glob)
@@ -846,7 +830,7 @@ class JLC_CaptionForgeQwen:
dry_run=False,
)
def process_one(source_name: str, pil: Image.Image):
def process_one(source_name: str, image_key: str, pil: Image.Image):
for run in run_plan:
engine.generation = GenerationConfig(
max_new_tokens=int(run.max_new_tokens),
@@ -860,8 +844,8 @@ class JLC_CaptionForgeQwen:
trigger="",
prefix=(f"{run.trigger_word}," if run.trigger_word else ""),
suffix="",
forbidden_phrases=_parse_forbidden_lines(forbidden_phrases),
replacement_rules=_parse_replace_pairs(replace_pairs),
forbidden_phrases=effective_forbidden,
replacement_rules=effective_replacements,
)
engine.config.max_size = int(run.max_size)
@@ -887,29 +871,22 @@ class JLC_CaptionForgeQwen:
captionforge_pass="A",
model_family="qwen",
ensemble_run_index=run.ensemble_run_index,
image_key=source_name,
image_key=image_key,
)
all_records.append(record)
if run_plan_connected and jsonl_path is not None and output_dir is not None:
append_jsonl_records(jsonl_path, [record], dry_run=False)
write_text_sidecar(
_run_txt_path(output_dir, source_name, len(run_plan), run.ensemble_run_index),
record.caption,
overwrite=True,
backup_existing=False,
dry_run=False,
)
print(
f"[JLC CaptionForge Qwen] Captioned {source_name} "
f"run {run.ensemble_run_index + 1}/{len(run_plan)}"
)
for source_name, pil in direct_images:
process_one(source_name, pil)
for source_name, image_key, pil in direct_images:
process_one(source_name, image_key, pil)
for source_name, path in file_images:
process_one(source_name, _open_image(path))
for source_name, image_key, path in file_images:
process_one(source_name, image_key, _open_image(path))
if not keep_loaded:
engine.unload()
+8 -10
View File
@@ -8,11 +8,9 @@ CaptionForge Ollama Model Dropdown Helper
- Repository:
https://github.com/Damkohler/CaptionForge
- CaptionForge focuses on practical dataset-captioning infrastructure for
LoRA dataset preparation, using multi-engine caption generation, JSONL
audit trails, claim extraction and refinement, text-LLM distillation,
image-aware VLM validation, and consensus-oriented caption improvement
to produce grounded, auditable training captions.
- CaptionForge 1.0 uses independent Pass-A witnesses, text-LLM synthesis,
image-aware validation, SHORT/TAGGY formatting, and JSONL audit trails to
produce grounded LoRA dataset captions.
- Helper Purpose
- The **CaptionForge Ollama Model Dropdown Helper** centralizes loading of
@@ -57,17 +55,17 @@ CaptionForge Ollama Model Dropdown Helper
- CaptionForge should avoid duplicated dropdown constants spread across
multiple node wrappers.
- A single helper makes pre-release cleanup easier, keeps ComfyUI widgets
- A single helper keeps release configuration centralized, keeps ComfyUI widgets
consistent, and allows users to edit one JSON file rather than patching
several Python files.
- The helper favors explicitness, reproducibility, and predictable fallback
behavior over clever model discovery.
- Development Status
- CaptionForge v0.1.0 experimental developer-preview infrastructure.
- This helper is active shared infrastructure, but the exact JSON schema may
evolve before the first stable CaptionForge release.
- Production Status
- Active CaptionForge 1.0 configuration support. It reads concrete Ollama
model tags for Pass-A captioning and the production B/C/D stages without
performing inference itself.
- Attribution & License
- Concept and implementation by **J. L. Córdova**
File diff suppressed because it is too large Load Diff
+395 -101
View File
@@ -8,11 +8,9 @@ JLC CaptionForge Pipeline Planner — ComfyUI Node Wrapper
- Repository
https://github.com/Damkohler/CaptionForge
- CaptionForge focuses on practical dataset-captioning infrastructure for
LoRA dataset preparation, using multi-engine caption generation, JSONL
audit trails, claim extraction and refinement, text-LLM distillation,
image-aware VLM validation, and consensus-oriented caption improvement
to produce grounded, auditable training captions.
- CaptionForge 1.0 uses independent Pass-A witnesses, text-LLM synthesis,
image-aware validation, SHORT/TAGGY formatting, and JSONL audit trails to
produce grounded LoRA dataset captions.
- Node Purpose
- The **JLC CaptionForge Pipeline Planner** is the ordinary-run control
@@ -24,11 +22,12 @@ JLC CaptionForge Pipeline Planner — ComfyUI Node Wrapper
• optional IMAGE passthrough for quick single-image workflows
• shared input path, recursion, and filename-glob routing
• output folder and run-name policy
• LoRA trigger word and user caption anchor routing
• LoRA trigger word and persistent caption/training anchor routing
• raw-caption run counts for Joy, Qwen, and generic Ollama Caption nodes
• caption seed, sampling, image-size, and token policy
• Distiller model/settings selection
• Validator model/settings selection
• SHORT/TAGGY Formatter model/settings selection
• final export policy
• JSON serialization of the plan for audit/debugging
@@ -36,32 +35,38 @@ JLC CaptionForge Pipeline Planner — ComfyUI Node Wrapper
captionforge_pipeline_planner_engine.py
- Ollama Model Dropdowns
- Distiller and Validator dropdown values are explicit Ollama model tags.
- Distiller, Validator, and Formatter dropdown values are explicit Ollama
model tags.
- Caption-stage Ollama models are intentionally selected on each
**JLC CaptionForge Ollama Caption** node, not in this Planner.
- Dropdown choices for Distiller/Validator are loaded at node-import time
- Dropdown choices for Distiller/Validator/Formatter are loaded at node-import time
from:
config/captionforge_ollama_models.json
- If the JSON file is missing or malformed, the node falls back to:
Distiller: llama3.1:8b
Validator: gemma4:e4b
Distiller: mistral-small:24b
Validator: gemma4:26b
Formatter: mistral-small:24b
- The optional custom choice lets users enter any installed Ollama model tag
without editing Python.
- CaptionForge Pipeline Role
- The planner emits the CAPTIONFORGE_PIPELINE_PLAN consumed by caption
nodes and the JLC CaptionForge capstone node.
nodes and the JLC CaptionForge Orchestrator.
- The canonical graph flow is:
Pipeline Planner
-> Joy/Qwen/Ollama raw-caption nodes
-> JLC CaptionForge capstone node
-> Distiller Engine
-> VLM Validator Engine
-> final deterministic TXT/JSONL export
-> Joy/Qwen/Ollama Pass A caption witness nodes
-> JLC CaptionForge Orchestrator
-> Pass B fat draft (text Ollama call)
-> Pass C image-aware validator (Ollama VLM call)
-> Pass D SHORT + TAGGY formatter (text Ollama call)
-> final deterministic TXT/JSONL export
- SmolVLM is not exposed in the current mainline Planner UI. It may remain
available as a standalone/experimental node and can be revisited later.
- The standalone CLI-oriented distiller and validator engines are
prototype/reference implementations, not the production runtime path.
- Historical SmolVLM experiments are not registered or exposed by the
CaptionForge 1.0 production package.
- Design Philosophy
- CaptionForge is an original concept and implementation, not derived from
@@ -75,10 +80,10 @@ JLC CaptionForge Pipeline Planner — ComfyUI Node Wrapper
low UI ambiguity, and clean separation between ComfyUI UI and reusable
pipeline logic.
- ⚠️ Development Status
- This is release-candidate CaptionForge infrastructure.
- Widget names, output schema details, and downstream validation strategy may
evolve as CaptionForge matures.
- Production Status
- This is the active CaptionForge 1.0 project-level control surface. When
connected, Planner values override corresponding Orchestrator controls; the
frozen production defaults are aligned between both nodes.
- Attribution & License
- Concept and implementation by **J. L. Córdova**
@@ -99,10 +104,10 @@ MANIFEST = {
"description": (
"ComfyUI-facing Pipeline Planner node for CaptionForge. Builds the "
"CAPTIONFORGE_PIPELINE_PLAN consumed through the pipeline_plan pin by "
"caption nodes and the JLC CaptionForge capstone node. Exposes Joy, Qwen, "
"caption nodes and the JLC CaptionForge Orchestrator. Exposes Joy, Qwen, "
"and generic Ollama Caption run counts for the current supported Pass A set. "
"Caption-stage Ollama model tags are selected directly on each Ollama Caption "
"node. Loads explicit Ollama Distiller/Validator dropdown tags from "
"node. Loads explicit Ollama Distiller/Validator/Formatter dropdown tags from "
"config/captionforge_ollama_models.json, with no family aliases or shorthand "
"model substitutions. The selected output folder is treated as an output root; "
"the planner derives a run-specific working directory for JSON/JSONL artifacts. "
@@ -117,11 +122,34 @@ import re
from pathlib import Path
from typing import Any
from ..engines.captionforge_dataset_export import (
dataset_export_inputs,
dataset_root,
export_settings_from_widgets,
normalize_export_settings,
prepare_dataset_root,
)
try:
from .captionforge_ollama_model_dropdowns import load_ollama_model_dropdowns
except Exception: # pragma: no cover - useful for direct local smoke tests
from captionforge_ollama_model_dropdowns import load_ollama_model_dropdowns
try:
from ..engines.captionforge_prompt_defaults import (
DEFAULT_FAT_DRAFT_INSTRUCTIONS,
DEFAULT_TAGGY_FORMATTER_INSTRUCTIONS,
DEFAULT_VALIDATOR_INSTRUCTIONS,
DEFAULT_VALIDATOR_SYSTEM_PROMPT,
)
except Exception: # pragma: no cover - useful for direct local smoke tests
from engines.captionforge_prompt_defaults import (
DEFAULT_FAT_DRAFT_INSTRUCTIONS,
DEFAULT_TAGGY_FORMATTER_INSTRUCTIONS,
DEFAULT_VALIDATOR_INSTRUCTIONS,
DEFAULT_VALIDATOR_SYSTEM_PROMPT,
)
try:
import folder_paths
except Exception:
@@ -142,17 +170,16 @@ DEFAULT_CAPTION_TEMPERATURE_SCHEDULE = "0.90"
DEFAULT_CAPTION_TOP_P_SCHEDULE = "0.60"
DEFAULT_CAPTION_TOP_K_SCHEDULE = "80"
DEFAULT_CAPTION_MAX_IMAGE_SIZE = 1024
DEFAULT_CAPTION_MAX_NEW_TOKENS = 6000
DEFAULT_CAPTION_MAX_NEW_TOKENS = 4096
_MODEL_DROPDOWNS = load_ollama_model_dropdowns(__file__)
DISTILLER_MODEL_CHOICES = _MODEL_DROPDOWNS["distiller_models"]
VALIDATOR_MODEL_CHOICES = _MODEL_DROPDOWNS["validator_models"]
FORMAT_MODEL_CHOICES = _MODEL_DROPDOWNS["format_models"]
DEFAULT_DISTILLER_MODEL = _MODEL_DROPDOWNS["distiller_default"]
DEFAULT_VALIDATOR_MODEL = _MODEL_DROPDOWNS["validator_default"]
DISTILLER_STRATEGIES = ["single_pass", "by_model_then_global"]
FINAL_CAPTION_STYLES = ["narrative", "comma", "both"]
DEFAULT_FORMAT_MODEL = _MODEL_DROPDOWNS["format_default"]
DEFAULT_OLLAMA_URL = "http://127.0.0.1:11434"
def _default_output_dir() -> str:
if folder_paths is not None:
try:
@@ -174,6 +201,11 @@ def _as_bool(value: Any) -> bool:
return str(value).strip().lower() in {"1", "true", "yes", "on"}
def _value_or_default(value: Any, default: Any) -> Any:
"""Return a default only for an absent/empty value, preserving numeric zero."""
return default if value is None or value == "" else value
def _runs_per_image(value: Any, default: str = "Disabled") -> int:
"""Normalize caption witness count widgets. Disabled means exactly 0 runs."""
text = str(value if value is not None else default).strip()
@@ -222,10 +254,19 @@ def _call_build_captionforge_pipeline_plan_compat(**kwargs) -> dict[str, Any]:
shared.setdefault("single_image_connected", bool(kwargs.get("single_image_connected", False)))
shared.setdefault("overwrite_outputs", bool(kwargs.get("overwrite_outputs", True)))
plan["ollama"] = {
"url": kwargs.get("ollama_url", DEFAULT_OLLAMA_URL),
"keep_loaded": kwargs.get("ollama_keep_loaded", True),
"request_timeout_seconds": kwargs.get("ollama_request_timeout_seconds", 1800),
}
# Concrete model names are stored in several compatible locations because
# older and newer capstone nodes look in slightly different namespaces.
distiller_model = str(kwargs.get("distiller_model") or DEFAULT_DISTILLER_MODEL).strip() or DEFAULT_DISTILLER_MODEL
validator_model = str(kwargs.get("validator_model") or DEFAULT_VALIDATOR_MODEL).strip() or DEFAULT_VALIDATOR_MODEL
distiller_seed = kwargs.get("distiller_seed", kwargs.get("distiller_base_seed", -1))
validator_seed = kwargs.get("validator_seed", kwargs.get("validator_base_seed", -1))
formatter_seed = kwargs.get("formatter_seed", -1)
caption_common = {
"base_seed": kwargs.get("base_seed", -1),
@@ -249,7 +290,8 @@ def _call_build_captionforge_pipeline_plan_compat(**kwargs) -> dict[str, Any]:
"backend": "ollama",
"model": distiller_model,
"ollama_model": distiller_model,
"seed": kwargs.get("distiller_base_seed", -1),
"prompt": kwargs.get("distiller_prompt", DEFAULT_FAT_DRAFT_INSTRUCTIONS),
"seed": distiller_seed,
})
pass_c_vlm_validator = plan.setdefault("pass_c_vlm_validator", {})
@@ -258,16 +300,17 @@ def _call_build_captionforge_pipeline_plan_compat(**kwargs) -> dict[str, Any]:
"backend": "ollama",
"model": validator_model,
"ollama_model": validator_model,
"seed": kwargs.get("validator_base_seed", -1),
"system_prompt": kwargs.get("validator_system_prompt", DEFAULT_VALIDATOR_SYSTEM_PROMPT),
"prompt": kwargs.get("validator_prompt", DEFAULT_VALIDATOR_INSTRUCTIONS),
"seed": validator_seed,
})
distiller_common = {
"model": distiller_model,
"ollama_model": distiller_model,
"model_family": distiller_model,
"base_seed": kwargs.get("distiller_base_seed", -1),
"seed_mode": kwargs.get("distiller_seed_mode", "fixed"),
"strategy": kwargs.get("distiller_strategy", "single_pass"),
"prompt": kwargs.get("distiller_prompt", DEFAULT_FAT_DRAFT_INSTRUCTIONS),
"seed": distiller_seed,
"max_caption_chars_for_llm": kwargs.get("distiller_max_caption_chars_for_llm", 1536),
"num_predict": kwargs.get("distiller_num_predict", 3096),
"temperature": kwargs.get("distiller_temperature", 0.24),
@@ -283,9 +326,10 @@ def _call_build_captionforge_pipeline_plan_compat(**kwargs) -> dict[str, Any]:
"model": validator_model,
"ollama_model": validator_model,
"model_family": validator_model,
"base_seed": kwargs.get("validator_base_seed", -1),
"seed_mode": kwargs.get("validator_seed_mode", "fixed"),
"num_predict": kwargs.get("validator_num_predict", 2200),
"system_prompt": kwargs.get("validator_system_prompt", DEFAULT_VALIDATOR_SYSTEM_PROMPT),
"prompt": kwargs.get("validator_prompt", DEFAULT_VALIDATOR_INSTRUCTIONS),
"seed": validator_seed,
"num_predict": kwargs.get("validator_num_predict", 2112),
"temperature": kwargs.get("validator_temperature", 0.0),
"top_p": kwargs.get("validator_top_p", 0.92),
"top_k": kwargs.get("validator_top_k", 80),
@@ -295,13 +339,31 @@ def _call_build_captionforge_pipeline_plan_compat(**kwargs) -> dict[str, Any]:
plan["validator"] = dict(validator_common)
plan["pass_c"] = dict(validator_common)
formatter_model = str(kwargs.get("formatter_model") or DEFAULT_FORMAT_MODEL).strip() or DEFAULT_FORMAT_MODEL
formatter_common = {
"model": formatter_model,
"ollama_model": formatter_model,
"model_family": formatter_model,
"prompt": kwargs.get("formatter_prompt", DEFAULT_TAGGY_FORMATTER_INSTRUCTIONS),
"seed": formatter_seed,
"num_predict": kwargs.get("formatter_num_predict", 3200),
"temperature": kwargs.get("formatter_temperature", 0.12),
"top_p": kwargs.get("formatter_top_p", 0.88),
"top_k": kwargs.get("formatter_top_k", 50),
"write_prompt_jsonl": kwargs.get("formatter_write_prompt_jsonl", False),
"preserve_raw_response": kwargs.get("formatter_preserve_raw_response", False),
}
plan["formatter"] = dict(formatter_common)
plan["pass_d"] = dict(formatter_common)
plan["pass_d_formatter"] = dict(formatter_common)
plan["final"] = {
"caption_style": kwargs.get("final_caption_style", "narrative"),
"write_txt_sidecars": kwargs.get("final_write_txt_sidecars", True),
"write_jsonl": kwargs.get("final_write_jsonl", True),
"overwrite_outputs": kwargs.get("overwrite_outputs", True),
}
plan["output"] = {"overwrite_outputs": kwargs.get("overwrite_outputs", True)}
plan["dataset_export"] = normalize_export_settings(kwargs.get("dataset_export"))
return plan
@@ -419,7 +481,7 @@ def _patch_supported_caption_witnesses(
def _patch_v010_working_image_paths(plan: dict[str, Any]) -> dict[str, Any]:
"""Normalize v0.1.x optional-image directories in the emitted plan.
"""Normalize legacy optional-image directories in the emitted plan.
The planner engine owns most path derivation. This wrapper patch keeps
JSON/JSONL/audit artifacts inside the run working directory while promoting
@@ -608,6 +670,35 @@ class JLC_CaptionForge_Pipeline_Planner:
},
),
# -----------------------------------------------------------------
# Shared Ollama runtime controls for Passes B/C/D.
# -----------------------------------------------------------------
"Ollama - URL": (
"STRING",
{
"default": DEFAULT_OLLAMA_URL,
"multiline": False,
"tooltip": "Planner-owned local Ollama server URL for Passes B, C, and D.",
},
),
"Ollama - keep loaded": (
"BOOLEAN",
{
"default": True,
"tooltip": "Planner-owned keep_alive policy for Passes B, C, and D.",
},
),
"Ollama - request timeout seconds": (
"INT",
{
"default": 1800,
"min": 10,
"max": 7200,
"step": 10,
"tooltip": "Planner-owned HTTP patience for Passes B, C, and D; does not affect caption quality.",
},
),
# -----------------------------------------------------------------
# LoRA metadata immediately after outputs.
# -----------------------------------------------------------------
@@ -624,7 +715,27 @@ class JLC_CaptionForge_Pipeline_Planner:
{
"default": "",
"multiline": False,
"tooltip": "Optional user style/identity anchor passed to distiller and validator.",
"tooltip": (
"Optional persistent caption/training anchor. In CaptionForge 1.x, a non-empty "
"anchor is preserved in the final caption variants rather than treated as image "
"evidence that the Validator may remove."
),
},
),
"Cleanup - forbidden phrases": (
"STRING",
{
"default": "",
"multiline": True,
"tooltip": "One forbidden word or phrase per line. Planner values override caption-node and Orchestrator cleanup controls.",
},
),
"Cleanup - replace pairs": (
"STRING",
{
"default": "",
"multiline": True,
"tooltip": "One boundary-safe old=>new replacement per line. Planner values override standalone cleanup controls.",
},
),
@@ -641,14 +752,14 @@ class JLC_CaptionForge_Pipeline_Planner:
"Caption - Qwen runs/image": (
CAPTION_RUNS,
{
"default": "2",
"default": "1",
"tooltip": "Qwen Caption runs per image. Set to Disabled to omit Qwen from this run. Dropdown is capped at 5 to prevent accidental giant runs.",
},
),
"Caption - Ollama runs/image": (
CAPTION_RUNS,
{
"default": "Disabled",
"default": "1",
"tooltip": (
"Ollama Caption runs per image for each connected JLC CaptionForge Ollama Caption node. "
"The actual Ollama model tag is selected in each Ollama Caption node. Connecting multiple "
@@ -663,12 +774,19 @@ class JLC_CaptionForge_Pipeline_Planner:
"min": -1,
"max": MAX_SEED_32,
"step": 1,
"tooltip": "Base seed for caption generation. -1 means unseeded when supported.",
"tooltip": "Base seed for the reusable Pass-A run schedule. -1 means intentionally unseeded.",
},
),
"Caption - seed mode": (
SEED_MODES,
{"default": "fixed"},
{
"default": "fixed",
"tooltip": (
"How the base seed changes across witness runs: fixed reuses it, "
"increment/decrement step by one, and random creates a repeatable "
"hash-derived schedule. A base seed of -1 remains unseeded in every mode."
),
},
),
"Caption - temperature schedule": (
"STRING",
@@ -680,19 +798,48 @@ class JLC_CaptionForge_Pipeline_Planner:
),
"Caption - top p schedule": (
"STRING",
{"default": DEFAULT_CAPTION_TOP_P_SCHEDULE, "multiline": False},
{
"default": DEFAULT_CAPTION_TOP_P_SCHEDULE,
"multiline": False,
"tooltip": (
"Comma-separated nucleus-sampling values for Pass-A runs. Lower values "
"limit choices to more likely tokens; the final value repeats as needed."
),
},
),
"Caption - top k schedule": (
"STRING",
{"default": DEFAULT_CAPTION_TOP_K_SCHEDULE, "multiline": False},
{
"default": DEFAULT_CAPTION_TOP_K_SCHEDULE,
"multiline": False,
"tooltip": (
"Comma-separated token-choice limits for Pass-A runs. Lower values are "
"more restrictive; the final value repeats as needed."
),
},
),
"Caption - max image size": (
"INT",
{"default": DEFAULT_CAPTION_MAX_IMAGE_SIZE, "min": 0, "max": 4096, "step": 64},
{
"default": DEFAULT_CAPTION_MAX_IMAGE_SIZE,
"min": 0,
"max": 4096,
"step": 64,
"tooltip": (
"Maximum longest image side sent to Pass-A captioners. Larger images are "
"resized proportionally; 0 keeps their original size."
),
},
),
"Caption - max new tokens": (
"INT",
{"default": DEFAULT_CAPTION_MAX_NEW_TOKENS, "min": 16, "max": 12000, "step": 64},
{
"default": DEFAULT_CAPTION_MAX_NEW_TOKENS,
"min": 16,
"max": 4096,
"step": 64,
"tooltip": "Maximum Pass-A generation budget. CaptionForge supports up to 4096 tokens.",
},
),
# -----------------------------------------------------------------
@@ -716,51 +863,69 @@ class JLC_CaptionForge_Pipeline_Planner:
"tooltip": "Used only when Distiller - model is custom, e.g. my-model:latest.",
},
),
"Distiller - base seed": (
"Distiller - prompt": (
"STRING",
{
"default": DEFAULT_FAT_DRAFT_INSTRUCTIONS,
"multiline": True,
"tooltip": (
"Planner-owned instructions for the text-only fat draft LLM. "
"Pass A captions are appended automatically."
),
},
),
"Distiller - seed": (
"INT",
{
"default": -1,
"min": -1,
"max": MAX_SEED_32,
"step": 1,
"tooltip": "Base seed for the distiller. -1 means omit seed.",
"tooltip": "Fixed seed used by the distiller for every image. -1 means omit seed.",
},
),
"Distiller - seed mode": (
SEED_MODES,
{"default": "fixed"},
),
"Distiller - strategy": (
DISTILLER_STRATEGIES,
{"default": "single_pass"},
),
"Distiller - max caption chars for LLM": (
"INT",
{"default": 1536, "min": 0, "max": 12000, "step": 64},
{
"default": 1536,
"min": 0,
"max": 12000,
"step": 64,
"tooltip": (
"Maximum characters retained from each Pass-A source caption before "
"building the Pass-B prompt. 0 keeps the complete caption."
),
},
),
"Distiller - num predict": (
"INT",
{"default": 3096, "min": 64, "max": 12000, "step": 64},
{
"default": 3096,
"min": 64,
"max": 12000,
"step": 64,
"tooltip": "Planner token budget for Pass B; maps to Ollama num_predict.",
},
),
"Distiller - temperature": (
"FLOAT",
{"default": 0.24, "min": 0.0, "max": 2.0, "step": 0.01},
{"default": 0.24, "min": 0.0, "max": 2.0, "step": 0.01, "tooltip": "Pass-B variation level. Lower values are steadier; higher values permit more varied wording."},
),
"Distiller - top p": (
"FLOAT",
{"default": 0.90, "min": 0.0, "max": 1.0, "step": 0.01},
{"default": 0.90, "min": 0.0, "max": 1.0, "step": 0.01, "tooltip": "Pass-B nucleus-sampling limit. Lower values restrict the model to more likely tokens."},
),
"Distiller - top k": (
"INT",
{"default": 60, "min": 0, "max": 500, "step": 1},
{"default": 60, "min": 0, "max": 500, "step": 1, "tooltip": "Pass-B token-choice limit. Lower values are more restrictive; 0 lets the backend disable top-k filtering."},
),
"Distiller - write prompt JSONL": (
"BOOLEAN",
{"default": False},
{"default": False, "tooltip": "Write the complete Pass-B prompt to a separate JSONL audit file. This can substantially increase output size."},
),
"Distiller - preserve raw response": (
"BOOLEAN",
{"default": False},
{"default": False, "tooltip": "Keep the unparsed Pass-B model response in audit records for troubleshooting."},
),
# -----------------------------------------------------------------
@@ -784,52 +949,142 @@ class JLC_CaptionForge_Pipeline_Planner:
"tooltip": "Used only when Validator - model is custom, e.g. gemma4:e4b or another installed VLM tag.",
},
),
"Validator - base seed": (
"Validator - system prompt": (
"STRING",
{
"default": DEFAULT_VALIDATOR_SYSTEM_PROMPT,
"multiline": True,
"tooltip": "Planner-owned system prompt for the image-aware VLM validator.",
},
),
"Validator - prompt": (
"STRING",
{
"default": DEFAULT_VALIDATOR_INSTRUCTIONS,
"multiline": True,
"tooltip": (
"Planner-owned instructions for the image-aware VLM validator. "
"The fat draft is appended automatically."
),
},
),
"Validator - seed": (
"INT",
{
"default": -1,
"min": -1,
"max": MAX_SEED_32,
"step": 1,
"tooltip": "Base seed for the VLM validator. -1 means omit seed.",
"tooltip": "Fixed seed used by the VLM validator for every image. -1 means omit seed.",
},
),
"Validator - seed mode": (
SEED_MODES,
{"default": "fixed"},
),
"Validator - num predict": (
"INT",
{"default": 2200, "min": 64, "max": 12000, "step": 64},
{
"default": 2112,
"min": 64,
"max": 12000,
"step": 64,
"tooltip": "Planner token budget for Pass C; maps to Ollama num_predict.",
},
),
"Validator - temperature": (
"FLOAT",
{"default": 0.0, "min": 0.0, "max": 2.0, "step": 0.01},
{"default": 0.0, "min": 0.0, "max": 2.0, "step": 0.01, "tooltip": "Pass-C variation level. Zero requests the most deterministic image-validation result."},
),
"Validator - top p": (
"FLOAT",
{"default": 0.92, "min": 0.0, "max": 1.0, "step": 0.01},
{"default": 0.92, "min": 0.0, "max": 1.0, "step": 0.01, "tooltip": "Pass-C nucleus-sampling limit. Lower values restrict the validator to more likely tokens."},
),
"Validator - top k": (
"INT",
{"default": 80, "min": 0, "max": 500, "step": 1},
{"default": 80, "min": 0, "max": 500, "step": 1, "tooltip": "Pass-C token-choice limit. Lower values are more restrictive; 0 lets the backend disable top-k filtering."},
),
"Validator - write prompt JSONL": (
"BOOLEAN",
{"default": False},
{"default": False, "tooltip": "Write the complete image-validation prompt to a separate JSONL audit file."},
),
"Validator - preserve raw VLM response": (
"BOOLEAN",
{"default": False},
{"default": False, "tooltip": "Keep the unparsed Pass-C VLM response in audit records for troubleshooting."},
),
# -----------------------------------------------------------------
# Formatter controls.
# -----------------------------------------------------------------
"Formatter - model": (
FORMAT_MODEL_CHOICES,
{
"default": DEFAULT_FORMAT_MODEL,
"tooltip": (
"Concrete Ollama text model tag for the SHORT + TAGGY formatter. "
"Use custom to enter any other installed Ollama text model."
),
},
),
"Formatter - custom Ollama model": (
"STRING",
{
"default": "",
"multiline": False,
"tooltip": "Used only when Formatter - model is custom.",
},
),
"Formatter - prompt": (
"STRING",
{
"default": DEFAULT_TAGGY_FORMATTER_INSTRUCTIONS,
"multiline": True,
"tooltip": (
"Planner-owned SHORT + TAGGY instructions for the text-only formatter. "
"The validated LONG caption is appended automatically."
),
},
),
"Formatter - seed": (
"INT",
{
"default": -1,
"min": -1,
"max": MAX_SEED_32,
"step": 1,
"tooltip": "Fixed seed used by the SHORT/TAGGY formatter for every image. -1 means omit seed.",
},
),
"Formatter - num predict": (
"INT",
{
"default": 3200,
"min": 64,
"max": 12000,
"step": 64,
"tooltip": "Planner token budget for Pass D; maps to Ollama num_predict.",
},
),
"Formatter - temperature": (
"FLOAT",
{"default": 0.12, "min": 0.0, "max": 2.0, "step": 0.01, "tooltip": "Pass-D variation level. Lower values make SHORT/TAGGY formatting more consistent."},
),
"Formatter - top p": (
"FLOAT",
{"default": 0.88, "min": 0.0, "max": 1.0, "step": 0.01, "tooltip": "Pass-D nucleus-sampling limit. Lower values restrict the formatter to more likely tokens."},
),
"Formatter - top k": (
"INT",
{"default": 50, "min": 0, "max": 500, "step": 1, "tooltip": "Pass-D token-choice limit. Lower values are more restrictive; 0 lets the backend disable top-k filtering."},
),
"Formatter - write prompt JSONL": (
"BOOLEAN",
{"default": False, "tooltip": "Write the complete Pass-D SHORT/TAGGY prompt to a separate JSONL audit file."},
),
"Formatter - preserve raw response": (
"BOOLEAN",
{"default": False, "tooltip": "Keep the unparsed Pass-D model response in audit records for troubleshooting."},
),
# -----------------------------------------------------------------
# Final export controls.
# -----------------------------------------------------------------
"Final - caption style": (
FINAL_CAPTION_STYLES,
{"default": "narrative"},
),
"Final - write TXT sidecars": (
"BOOLEAN",
{
@@ -841,10 +1096,11 @@ class JLC_CaptionForge_Pipeline_Planner:
),
"Final - write JSONL": (
"BOOLEAN",
{"default": True},
{"default": True, "tooltip": "Write the final run-level JSONL containing LONG, SHORT, and TAGGY captions for every processed image."},
),
},
"optional": {
**dataset_export_inputs(),
"Input - single image": (
"IMAGE",
{
@@ -860,7 +1116,7 @@ class JLC_CaptionForge_Pipeline_Planner:
RETURN_TYPES = ("IMAGE", "CAPTIONFORGE_PIPELINE_PLAN", "STRING")
RETURN_NAMES = ("single_image", "pipeline_plan", "pipeline_plan_json")
FUNCTION = "plan"
CATEGORY = "Captioning/CaptionForge"
CATEGORY = "Caption/CaptionForge"
@classmethod
def IS_CHANGED(cls, **kwargs):
@@ -882,12 +1138,16 @@ class JLC_CaptionForge_Pipeline_Planner:
kwargs.get("Validator - custom Ollama model", ""),
DEFAULT_VALIDATOR_MODEL,
)
formatter_model = _resolve_ollama_model_name(
kwargs.get("Formatter - model", DEFAULT_FORMAT_MODEL),
kwargs.get("Formatter - custom Ollama model", ""),
DEFAULT_FORMAT_MODEL,
)
planner_enabled = _as_bool(kwargs.get("Planner - enabled", True))
joy_runs = _runs_per_image(kwargs.get("Caption - Joy runs/image", "2"), "2")
qwen_runs = _runs_per_image(kwargs.get("Caption - Qwen runs/image", "2"), "2")
ollama_runs = _runs_per_image(kwargs.get("Caption - Ollama runs/image", "Disabled"), "Disabled")
smolvlm_runs = 0
qwen_runs = _runs_per_image(kwargs.get("Caption - Qwen runs/image", "1"), "1")
ollama_runs = _runs_per_image(kwargs.get("Caption - Ollama runs/image", "1"), "1")
if not planner_enabled:
plan: dict[str, Any] = {}
@@ -912,40 +1172,68 @@ class JLC_CaptionForge_Pipeline_Planner:
smol_runs_per_image=0,
florence_runs_per_image=0,
llama_vision_runs_per_image=0,
base_seed=int(kwargs.get("Caption - base seed", -1) or -1),
base_seed=int(_value_or_default(kwargs.get("Caption - base seed", -1), -1)),
seed_mode=str(kwargs.get("Caption - seed mode", "fixed") or "fixed"),
temperature_schedule=str(kwargs.get("Caption - temperature schedule", DEFAULT_CAPTION_TEMPERATURE_SCHEDULE) or DEFAULT_CAPTION_TEMPERATURE_SCHEDULE),
top_p_schedule=str(kwargs.get("Caption - top p schedule", DEFAULT_CAPTION_TOP_P_SCHEDULE) or DEFAULT_CAPTION_TOP_P_SCHEDULE),
top_k_schedule=str(kwargs.get("Caption - top k schedule", DEFAULT_CAPTION_TOP_K_SCHEDULE) or DEFAULT_CAPTION_TOP_K_SCHEDULE),
max_size=int(kwargs.get("Caption - max image size", DEFAULT_CAPTION_MAX_IMAGE_SIZE) or DEFAULT_CAPTION_MAX_IMAGE_SIZE),
max_new_tokens=int(kwargs.get("Caption - max new tokens", DEFAULT_CAPTION_MAX_NEW_TOKENS) or DEFAULT_CAPTION_MAX_NEW_TOKENS),
max_size=int(_value_or_default(kwargs.get("Caption - max image size", DEFAULT_CAPTION_MAX_IMAGE_SIZE), DEFAULT_CAPTION_MAX_IMAGE_SIZE)),
max_new_tokens=int(_value_or_default(kwargs.get("Caption - max new tokens", DEFAULT_CAPTION_MAX_NEW_TOKENS), DEFAULT_CAPTION_MAX_NEW_TOKENS)),
trigger_word=str(kwargs.get("LoRA - trigger word", "") or "").strip(),
user_caption_anchor=str(kwargs.get("LoRA - user caption anchor", "") or "").strip(),
forbidden_phrases=str(kwargs.get("Cleanup - forbidden phrases", "") or ""),
replace_pairs=str(kwargs.get("Cleanup - replace pairs", "") or ""),
ollama_url=str(kwargs.get("Ollama - URL", DEFAULT_OLLAMA_URL) or DEFAULT_OLLAMA_URL),
ollama_keep_loaded=_as_bool(kwargs.get("Ollama - keep loaded", True)),
ollama_request_timeout_seconds=int(
_value_or_default(kwargs.get("Ollama - request timeout seconds", 1800), 1800)
),
distiller_model=distiller_model,
distiller_model_family=distiller_model,
distiller_base_seed=int(kwargs.get("Distiller - base seed", -1) or -1),
distiller_seed_mode=str(kwargs.get("Distiller - seed mode", "fixed") or "fixed"),
distiller_strategy=str(kwargs.get("Distiller - strategy", "single_pass") or "single_pass"),
distiller_max_caption_chars_for_llm=int(kwargs.get("Distiller - max caption chars for LLM", 1536) or 1536),
distiller_prompt=str(
kwargs.get("Distiller - prompt", DEFAULT_FAT_DRAFT_INSTRUCTIONS)
or DEFAULT_FAT_DRAFT_INSTRUCTIONS
),
distiller_seed=int(_value_or_default(kwargs.get("Distiller - seed", -1), -1)),
distiller_max_caption_chars_for_llm=int(_value_or_default(kwargs.get("Distiller - max caption chars for LLM", 1536), 1536)),
distiller_num_predict=int(kwargs.get("Distiller - num predict", 3096) or 3096),
distiller_temperature=float(kwargs.get("Distiller - temperature", 0.24) or 0.0),
distiller_top_p=float(kwargs.get("Distiller - top p", 0.90) or 0.90),
distiller_top_k=int(kwargs.get("Distiller - top k", 60) or 60),
distiller_top_p=float(_value_or_default(kwargs.get("Distiller - top p", 0.90), 0.90)),
distiller_top_k=int(_value_or_default(kwargs.get("Distiller - top k", 60), 60)),
distiller_write_prompt_jsonl=_as_bool(kwargs.get("Distiller - write prompt JSONL", False)),
distiller_preserve_raw_response=_as_bool(kwargs.get("Distiller - preserve raw response", False)),
validator_model=validator_model,
validator_model_family=validator_model,
validator_base_seed=int(kwargs.get("Validator - base seed", -1) or -1),
validator_seed_mode=str(kwargs.get("Validator - seed mode", "fixed") or "fixed"),
validator_num_predict=int(kwargs.get("Validator - num predict", 2200) or 2200),
validator_system_prompt=str(
kwargs.get("Validator - system prompt", DEFAULT_VALIDATOR_SYSTEM_PROMPT)
or DEFAULT_VALIDATOR_SYSTEM_PROMPT
),
validator_prompt=str(
kwargs.get("Validator - prompt", DEFAULT_VALIDATOR_INSTRUCTIONS)
or DEFAULT_VALIDATOR_INSTRUCTIONS
),
validator_seed=int(_value_or_default(kwargs.get("Validator - seed", -1), -1)),
validator_num_predict=int(kwargs.get("Validator - num predict", 2112) or 2112),
validator_temperature=float(kwargs.get("Validator - temperature", 0.0) or 0.0),
validator_top_p=float(kwargs.get("Validator - top p", 0.92) or 0.92),
validator_top_k=int(kwargs.get("Validator - top k", 80) or 80),
validator_top_p=float(_value_or_default(kwargs.get("Validator - top p", 0.92), 0.92)),
validator_top_k=int(_value_or_default(kwargs.get("Validator - top k", 80), 80)),
validator_write_prompt_jsonl=_as_bool(kwargs.get("Validator - write prompt JSONL", False)),
validator_preserve_raw_vlm_response=_as_bool(kwargs.get("Validator - preserve raw VLM response", False)),
final_caption_style=str(kwargs.get("Final - caption style", "narrative") or "narrative"),
formatter_model=formatter_model,
formatter_prompt=str(
kwargs.get("Formatter - prompt", DEFAULT_TAGGY_FORMATTER_INSTRUCTIONS)
or DEFAULT_TAGGY_FORMATTER_INSTRUCTIONS
),
formatter_seed=int(_value_or_default(kwargs.get("Formatter - seed", -1), -1)),
formatter_num_predict=int(kwargs.get("Formatter - num predict", 3200) or 3200),
formatter_temperature=float(kwargs.get("Formatter - temperature", 0.12) or 0.0),
formatter_top_p=float(_value_or_default(kwargs.get("Formatter - top p", 0.88), 0.88)),
formatter_top_k=int(_value_or_default(kwargs.get("Formatter - top k", 50), 50)),
formatter_write_prompt_jsonl=_as_bool(kwargs.get("Formatter - write prompt JSONL", False)),
formatter_preserve_raw_response=_as_bool(kwargs.get("Formatter - preserve raw response", False)),
final_write_txt_sidecars=_as_bool(kwargs.get("Final - write TXT sidecars", True)),
final_write_jsonl=_as_bool(kwargs.get("Final - write JSONL", True)),
dataset_export=export_settings_from_widgets(kwargs),
overwrite_outputs=_as_bool(kwargs.get("Output - overwrite outputs", True)),
)
plan = _patch_supported_caption_witnesses(
@@ -956,6 +1244,12 @@ class JLC_CaptionForge_Pipeline_Planner:
)
plan = _patch_v010_working_image_paths(plan)
export = plan["dataset_export"]
if export["enabled"]:
root = dataset_root(export, output_dir)
prepare_dataset_root(root, str(kwargs.get("Input - image path", "") or ""))
plan["paths"]["training_dataset_dir"] = str(root)
overwrite_outputs = _as_bool(kwargs.get("Output - overwrite outputs", True))
_reset_pass_a_jsonl_for_overwrite(plan, overwrite_outputs=overwrite_outputs)
+8 -12
View File
@@ -8,11 +8,9 @@ JLC CaptionForge Template Options — ComfyUI Node Wrapper
- Repository
https://github.com/Damkohler/CaptionForge
- CaptionForge focuses on practical dataset-captioning infrastructure for
LoRA dataset preparation, using multi-engine caption generation, JSONL
audit trails, claim extraction and refinement, text-LLM distillation,
image-aware VLM validation, and consensus-oriented caption improvement
to produce grounded, auditable training captions.
- CaptionForge 1.0 uses independent Pass-A witnesses, text-LLM synthesis,
image-aware validation, SHORT/TAGGY formatting, and JSONL audit trails to
produce grounded LoRA dataset captions.
- Node Purpose
- The **JLC CaptionForge Template Options** node provides a shared
@@ -85,12 +83,10 @@ JLC CaptionForge Template Options — ComfyUI Node Wrapper
Joy, Qwen, Ollama VLMs, or future caption models without duplicating
every checkbox in every caption node.
- ⚠️ Development Status
- This is active CaptionForge raw-caption infrastructure.
- The UI, option taxonomy, payload schema, and future configuration-file
support may evolve as CaptionForge matures.
- The node is intended for local dataset preparation and controlled caption
audit workflows.
- Production Status
- Active CaptionForge 1.0 Pass-A support node. It supplies shared prompt
modifiers only; each witness backend retains its own prompt dialect and
model behavior.
- Attribution & License
- Concept and implementation by **J. L. Córdova**
@@ -339,7 +335,7 @@ class JLC_CaptionForgeExtraOptions:
RETURN_TYPES = ("CAPTIONFORGE_EXTRA_OPTIONS", "STRING")
RETURN_NAMES = ("template_options", "template_options_json")
FUNCTION = "build"
CATEGORY = "Captioning/CaptionForge"
CATEGORY = "Caption/CaptionForge"
def build(self, **kwargs):
# Accept the old name_input key as a courtesy during hot-swaps, but the
+20 -14
View File
@@ -4,8 +4,8 @@ build-backend = "setuptools.build_meta"
[project]
name = "captionforge"
version = "0.1.1"
description = "CaptionForge is a local ComfyUI captioning toolkit for preparing better LoRA-ready image datasets. It runs multiple caption witnesses, compares their agreements and contradictions, distills the strongest visual claims, validates them against the image with a VLM, and exports long, short, and taggy TXT captions alongside JSONL audit trails. CaptionForge does not ship model weights: Python-backed Hugging Face models such as Joy and Qwen are downloaded on demand and cached locally, while Ollama-backed stages require a local Ollama installation with the selected model tags installed or pullable."
version = "1.0.2"
description = "Local, auditable ComfyUI captioning for LoRA datasets: multi-witness evidence, text-LLM synthesis, image-aware VLM validation, and LONG/SHORT/TAGGY exports."
readme = "README.md"
requires-python = ">=3.10,<3.13"
license = { text = "MIT" }
@@ -52,6 +52,7 @@ dependencies = [
"torch",
"transformers",
"accelerate",
"bitsandbytes>=0.46.1",
"huggingface-hub",
"pillow",
"numpy",
@@ -61,7 +62,7 @@ dependencies = [
[project.optional-dependencies]
quantization = [
"bitsandbytes",
"bitsandbytes>=0.46.1",
]
dev = [
@@ -78,25 +79,30 @@ Issues = "https://github.com/Damkohler/CaptionForge/issues"
supported_comfyui_version = ">=0.2"
PublisherId = "damkohler"
DisplayName = "CaptionForge"
Icon = "https://raw.githubusercontent.com/Damkohler/jlc-comfyui-nodes/main/assets/icons/jlc-comfyui-nodes_Logo-0512.png"
Icon = "https://raw.githubusercontent.com/Damkohler/CaptionForge/main/assets/icons/jlc-comfyui-nodes_Logo-0512.png"
Category = "nodes"
Homepage = "https://github.com/Damkohler/CaptionForge"
[tool.setuptools]
include-package-data = true
[tool.setuptools.packages.find]
where = ["."]
include = [
"engines*",
"nodes*",
include-package-data = false
packages = [
"CaptionForge",
"CaptionForge.engines",
"CaptionForge.nodes",
"CaptionForge.nodes.caption_nodes",
]
[tool.setuptools.package-dir]
CaptionForge = "."
"CaptionForge.engines" = "engines"
"CaptionForge.nodes" = "nodes"
"CaptionForge.nodes.caption_nodes" = "nodes/caption_nodes"
[tool.setuptools.package-data]
"*" = [
CaptionForge = [
"config/*.json",
"config/**/*.json",
"assets/**/*",
"assets/icons/*",
"assets/workflows/*",
"web/**/*",
]
+11
View File
@@ -0,0 +1,11 @@
# ComfyUI Manager installs this file. Keep aligned with project.dependencies
# in pyproject.toml; tests/test_joy_hardening.py checks both install paths.
torch
transformers
accelerate
bitsandbytes>=0.46.1
huggingface-hub
pillow
numpy
safetensors
qwen-vl-utils
+231
View File
@@ -0,0 +1,231 @@
"""Regression tests for boundary-safe Pass-A forbidden-phrase cleanup."""
from __future__ import annotations
import importlib
import sys
import types
import unittest
from pathlib import Path
ROOT = Path(__file__).resolve().parents[1]
def _install_namespace(name: str, path: Path) -> None:
if name in sys.modules:
return
module = types.ModuleType(name)
module.__path__ = [str(path)]
sys.modules[name] = module
_install_namespace("CaptionForge", ROOT)
_install_namespace("CaptionForge.engines", ROOT / "engines")
_install_namespace("CaptionForge.nodes", ROOT / "nodes")
_install_namespace("CaptionForge.nodes.caption_nodes", ROOT / "nodes" / "caption_nodes")
if "folder_paths" not in sys.modules:
folder_paths = types.ModuleType("folder_paths")
folder_paths.models_dir = str(ROOT / "models")
folder_paths.get_output_directory = lambda: str(ROOT / "output")
sys.modules["folder_paths"] = folder_paths
cleanup = importlib.import_module("CaptionForge.engines.captionforge_cleanup")
planner_engine = importlib.import_module("CaptionForge.engines.captionforge_pipeline_planner_engine")
joy = importlib.import_module("CaptionForge.engines.jlc_joy_caption_engine")
qwen = importlib.import_module("CaptionForge.engines.jlc_qwen_caption_engine")
ollama = importlib.import_module(
"CaptionForge.nodes.caption_nodes.jlc_captionforge_ollama_caption_node"
)
class SharedForbiddenPhraseContractTests(unittest.TestCase):
def test_pipeline_cleanup_settings_propagate_and_override_standalone_values(self) -> None:
plan = planner_engine.build_captionforge_pipeline_plan(
forbidden_phrases="old\nsafety disclaimer",
replace_pairs="former=>current\nred car=>blue car",
)
self.assertEqual(plan["cleanup"]["forbidden_phrases"], ["old", "safety disclaimer"])
self.assertEqual(
plan["cleanup"]["replace_pairs"],
[{"old": "former", "new": "current"}, {"old": "red car", "new": "blue car"}],
)
forbidden, pairs = cleanup.resolve_cleanup_settings(
plan, "standalone forbidden", "standalone old=>standalone new"
)
self.assertEqual(forbidden, ["old", "safety disclaimer"])
self.assertEqual(pairs, [("former", "current"), ("red car", "blue car")])
def test_standalone_cleanup_settings_survive_without_planner(self) -> None:
forbidden, pairs = cleanup.resolve_cleanup_settings({}, "old", "former=>current")
self.assertEqual(forbidden, ["old"])
self.assertEqual(pairs, [("former", "current")])
def test_end_to_end_cleanup_keeps_boundaries_and_normalizes_spacing(self) -> None:
value = cleanup.apply_cleanup_contract(
"bold holding gold, old, former wording.", ["old"], [("former", "current")]
)
self.assertEqual(value, "bold holding gold, current wording.")
def test_substrings_inside_legitimate_words_do_not_match(self) -> None:
text = "A bold subject is holding a gold accessory."
self.assertFalse(cleanup.contains_forbidden_phrase(text, ["old"]))
self.assertEqual(cleanup.remove_forbidden_phrases(text, ["old"]), text)
def test_boundary_safe_replacements_preserve_containing_words(self) -> None:
source = "bold pose, holding a gold prop, old stone wall"
expected = "bold pose, holding a gold prop, young stone wall"
self.assertEqual(
cleanup.replace_phrases(source, [("old", "young")]),
expected,
)
def test_true_word_and_phrase_matches_are_boundary_aware(self) -> None:
self.assertTrue(cleanup.contains_forbidden_phrase("an old stone wall", ["old"]))
self.assertTrue(
cleanup.contains_forbidden_phrase(
"caption includes safety disclaimer text",
["safety disclaimer"],
)
)
self.assertFalse(
cleanup.contains_forbidden_phrase(
"caption includes safety disclaimers",
["safety disclaimer"],
)
)
def test_joy_and_qwen_replace_pairs_are_boundary_safe_by_default(self) -> None:
source = "bold pose, holding a gold prop, old stone wall"
expected = "bold pose, holding a gold prop, young stone wall"
self.assertTrue(joy.CleanupConfig().replace_whole_words_only)
self.assertTrue(qwen.CleanupConfig().replace_whole_words_only)
self.assertEqual(
joy.apply_replacements(source, [("old", "young")]),
expected,
)
self.assertEqual(
qwen.apply_replacements(source, [("old", "young")]),
expected,
)
def test_joy_and_qwen_preserve_containing_words_but_remove_true_match(self) -> None:
source = "bold pose, holding a gold prop, old stone wall"
expected = "bold pose, holding a gold prop, stone wall"
self.assertEqual(joy.remove_forbidden_phrases(source, ["old"]), expected)
self.assertEqual(qwen.remove_forbidden_phrases(source, ["old"]), expected)
def test_ollama_does_not_drop_paragraph_for_substring_false_positive(self) -> None:
raw = "A bold figure is holding a gold accessory in dramatic light."
cleaned, status = ollama._clean_caption(
raw,
trigger_word="",
forbidden_phrases=["old"],
replacement_rules=[],
)
self.assertEqual(status, "ok")
self.assertEqual(cleaned, raw)
def test_ollama_replace_pairs_are_boundary_safe(self) -> None:
raw = "A bold figure is holding a gold prop near an old wall."
cleaned, status = ollama._clean_caption(
raw,
trigger_word="",
forbidden_phrases=[],
replacement_rules=[("old", "young")],
)
self.assertEqual(status, "ok")
self.assertEqual(
cleaned,
"A bold figure is holding a gold prop near an young wall.",
)
def test_ollama_replacement_case_sensitivity_is_preserved(self) -> None:
raw = "Old wall beside an old wall."
cleaned, status = ollama._clean_caption(
raw,
trigger_word="",
forbidden_phrases=[],
replacement_rules=[("old", "young")],
)
self.assertEqual(status, "ok")
self.assertEqual(cleaned, "Old wall beside an young wall.")
def test_ollama_still_drops_line_for_true_forbidden_match(self) -> None:
raw = "First safe line.\nAn old line.\nFinal safe line."
cleaned, status = ollama._clean_caption(
raw,
trigger_word="",
forbidden_phrases=["old"],
replacement_rules=[],
)
self.assertEqual(status, "ok")
self.assertEqual(cleaned, "First safe line.\nFinal safe line.")
def test_ollama_run_config_audits_cleanup_settings(self) -> None:
config = ollama._build_run_config(
model_tag="example:model",
ollama_url="http://127.0.0.1:11434",
system_prompt="system",
prompt="prompt",
max_new_tokens=100,
temperature=0.2,
top_p=0.9,
top_k=40,
repetition_penalty=1.03,
max_size=1024,
forbidden_phrases=["old", "safety disclaimer"],
replacement_rules=[("foo", "bar")],
)
self.assertEqual(
config["cleanup"]["forbidden_phrases"],
["old", "safety disclaimer"],
)
self.assertEqual(config["cleanup"]["replacement_rules"], [["foo", "bar"]])
self.assertEqual(
config["cleanup"]["replacement_match_mode"],
"whole_word_or_phrase_boundary",
)
self.assertFalse(config["cleanup"]["replacement_case_insensitive"])
self.assertEqual(
config["cleanup"]["forbidden_match_mode"],
"whole_word_or_phrase_boundary",
)
self.assertEqual(config["cleanup"]["forbidden_action"], "drop_matching_line")
def test_test01_style_seven_witnesses_all_remain_eligible(self) -> None:
"""Reproduce the Test_01 false-positive signature across seven Pass-A witnesses."""
witnesses = [
("joy", "A subject is holding a prop with both hands."),
("joy", "A bold composition with detailed clothing."),
("qwen", "The subject is holding an accessory against a plain background."),
("ollama", "A bold figure is holding a gold accessory in dramatic light."),
("ollama", "The subject is holding a pose with bold styling."),
("ollama", "A gold ornament is visible while the subject is holding the garment."),
("ollama", "A detailed portrait with a confident pose and studio lighting."),
]
survivors = []
for family, caption in witnesses:
if family == "joy":
cleaned = joy.remove_forbidden_phrases(caption, ["old"])
status = "ok" if cleaned else "filtered"
elif family == "qwen":
cleaned = qwen.remove_forbidden_phrases(caption, ["old"])
status = "ok" if cleaned else "filtered"
else:
cleaned, status = ollama._clean_caption(
caption,
trigger_word="",
forbidden_phrases=["old"],
replacement_rules=[],
)
if status == "ok" and cleaned:
survivors.append((family, cleaned))
self.assertEqual(len(survivors), 7)
if __name__ == "__main__":
unittest.main()
File diff suppressed because it is too large Load Diff
+305
View File
@@ -0,0 +1,305 @@
"""CPU checks for paired exports, source protection, and Planner ownership."""
from __future__ import annotations
import importlib
import io
import json
import sys
import tempfile
import types
import unittest
import urllib.error
from pathlib import Path
from unittest import mock
from PIL import Image
ROOT = Path(__file__).resolve().parents[1]
for name, path in (("CaptionForge", ROOT), ("CaptionForge.engines", ROOT / "engines"),
("CaptionForge.nodes", ROOT / "nodes"),
("CaptionForge.nodes.caption_nodes", ROOT / "nodes" / "caption_nodes")):
if name not in sys.modules:
module = types.ModuleType(name)
module.__path__ = [str(path)]
sys.modules[name] = module
if "folder_paths" not in sys.modules:
sys.modules["folder_paths"] = types.ModuleType("folder_paths")
sys.modules["folder_paths"].models_dir = str(ROOT / "models")
sys.modules["folder_paths"].get_output_directory = lambda: str(ROOT / "output")
exporter = importlib.import_module("CaptionForge.engines.captionforge_dataset_export")
planner = importlib.import_module("CaptionForge.nodes.jlc_captionforge_pipeline_planner_node")
orchestrator = importlib.import_module("CaptionForge.nodes.jlc_captionforge_node")
class DatasetExportTests(unittest.TestCase):
def setUp(self):
temporary = tempfile.TemporaryDirectory()
self.addCleanup(temporary.cleanup)
self.root = Path(temporary.name)
self.source = self.root / "portraits" / "photo.jpg"
self.source.parent.mkdir()
Image.new("RGB", (900, 600), "orange").save(self.source)
self.original_bytes = self.source.read_bytes()
self.settings = exporter.normalize_export_settings({"enabled": True, "max_size": 256})
def pair(self, **overrides):
arguments = dict(source=self.source, image_key="portraits/photo.jpg", input_root=self.root,
root=self.root / "training_dataset", settings=self.settings,
validator_max_size=1536, captions={"short": "A portrait.", "long": "Long portrait.", "taggy": "portrait"},
overwrite=False)
arguments.update(overrides)
return exporter.export_pair(**arguments)
def test_downscale_and_divisibility_never_enlarge(self):
result = self.pair()
with Image.open(result["image"]) as image:
self.assertEqual(image.size, (256, 160))
self.assertEqual(Path(result["caption"]).read_text().strip(), "A portrait.")
self.assertEqual(self.source.read_bytes(), self.original_bytes)
self.assertFalse(self.source.with_suffix(".txt").exists())
def test_small_image_and_divisor_one(self):
image = Image.new("RGB", (99, 65))
self.assertEqual(exporter.resize_for_export(image, 1536, 1).size, (99, 65))
self.assertEqual(exporter.resize_for_export(image, 1536, 16).size, (96, 64))
with self.assertRaises(ValueError):
exporter.resize_for_export(Image.new("RGB", (10, 9)), 1536, 16)
def test_fractional_divisor_rejected(self):
with self.assertRaises(ValueError):
exporter.normalize_export_settings({"divisor": 16.5})
def test_original_extensions_and_folders_cannot_collide(self):
first = self.pair()
second_source = self.source.with_suffix(".png")
Image.new("RGB", (900, 600)).save(second_source)
second = self.pair(source=second_source, image_key="portraits/photo.png")
self.assertNotEqual(first["image"], second["image"])
self.assertNotEqual(first["caption"], second["caption"])
self.assertEqual(Path(first["image"]).relative_to(self.root).as_posix(), "training_dataset/files/portraits/photo.jpg.png")
def test_resume_and_changed_settings(self):
first = self.pair()
original_mtime = Path(first["image"]).stat().st_mtime_ns
self.assertTrue(self.pair()["resumed"])
self.assertEqual(Path(first["image"]).stat().st_mtime_ns, original_mtime)
with self.assertRaises(FileExistsError):
self.pair(settings={**self.settings, "divisor": 32})
regenerated = self.pair(settings={**self.settings, "divisor": 32}, overwrite=True)
self.assertFalse(regenerated["resumed"])
def test_foreign_folder_and_foreign_file_are_protected(self):
destination = self.root / "training_dataset"
destination.mkdir()
original = destination / "archive.png"
original.write_bytes(b"archive")
with self.assertRaises(ValueError):
self.pair(overwrite=True)
self.assertEqual(original.read_bytes(), b"archive")
original.unlink()
exporter.prepare_dataset_root(destination)
target, _, _ = exporter.export_paths(destination, self.source, self.root, "portraits/photo.jpg", "PNG")
target.parent.mkdir(parents=True)
target.write_bytes(b"untracked")
with self.assertRaises(FileExistsError):
self.pair(overwrite=True)
self.assertEqual(target.read_bytes(), b"untracked")
def test_destination_containing_input_is_rejected(self):
with self.assertRaises(ValueError):
exporter.prepare_dataset_root(self.root, self.source)
def test_generated_images_excluded_from_every_witness_scan(self):
self.pair()
for family in ("joy", "qwen", "ollama"):
module = importlib.import_module(f"CaptionForge.nodes.caption_nodes.jlc_captionforge_{family}_caption_node")
paths = module._iter_input_path_images(str(self.root), True, "*")
self.assertEqual([entry[2] for entry in paths], [self.source])
def test_explicit_generated_dataset_root_is_valid_witness_input(self):
exported = self.pair()
dataset_root = self.root / "training_dataset"
exported_image = Path(exported["image"])
for family in ("joy", "qwen", "ollama"):
module = importlib.import_module(f"CaptionForge.nodes.caption_nodes.jlc_captionforge_{family}_caption_node")
# A generated dataset discovered under a broader source root remains excluded.
parent_scan = module._iter_input_path_images(str(self.root), True, "*")
self.assertEqual([entry[2] for entry in parent_scan], [self.source])
# But explicitly selecting that generated dataset is an intentional user action
# and must make its images available for recaptioning/reprocessing.
explicit_scan = module._iter_input_path_images(str(dataset_root), True, "*")
self.assertEqual([entry[2] for entry in explicit_scan], [exported_image])
def test_jpeg_and_optional_image_namespace(self):
result = self.pair(image_key="captionforge-optional-image://comfy_image_0000.png",
settings={**self.settings, "image_format": "JPEG", "caption": "taggy"})
self.assertIn("optional", Path(result["image"]).parts)
with Image.open(result["image"]) as image:
self.assertEqual(image.format, "JPEG")
self.assertEqual(Path(result["caption"]).read_text().strip(), "portrait")
def run_pipeline(self, *, plan=None, overwrite=True, validator_side_effect=None, **extra):
caption_path = self.root / "witness.jsonl"
caption_path.write_text(json.dumps({"image": str(self.source), "image_key": "portraits/photo.jpg",
"caption": "A portrait.", "model_family": "joy", "status": "ok"}) + "\n")
kwargs = {"Input - captions JSONL": str(caption_path), "Input - image path": str(self.root),
"Output - folder": str(self.root), "Output - run name": "prototype",
"Output - overwrite outputs": overwrite, "Validator - max image size": 256,
"Dataset - export image and caption": True, "pipeline_plan": plan}
kwargs.update(extra)
with mock.patch.object(orchestrator, "_evict_python_models_before_ollama_if_needed"), \
mock.patch.object(orchestrator, "_ollama_generate_text", return_value=("SHORT: A portrait.\nTAGGY: portrait", {})) as text, \
mock.patch.object(orchestrator, "_ollama_chat_image", return_value=("A long portrait.", {}), side_effect=validator_side_effect) as validator:
result = orchestrator.JLC_CaptionForge().forge(**kwargs)
return json.loads(result[3])["records"][0], text.call_count, validator.call_count
def test_standalone_export_and_resume_without_model_calls(self):
record, _, _ = self.run_pipeline()
self.assertEqual(record["status"], "ok", record)
self.assertEqual(record["dataset_export"]["width"], 256)
self.assertTrue(Path(record["outputs"]["short"]).is_relative_to(self.root / "training_dataset"))
self.assertFalse(self.source.with_name("photo_short.txt").exists())
resumed, text_calls, validator_calls = self.run_pipeline(overwrite=False)
self.assertEqual((text_calls, validator_calls), (0, 0))
self.assertTrue(resumed["dataset_export"]["resumed"])
def test_planner_controls_override_every_local_setting_including_blank_folder(self):
_, plan, _ = planner.JLC_CaptionForge_Pipeline_Planner().plan(**{
"Input - image path": str(self.root), "Output - folder": str(self.root),
"Output - run name": "prototype", "Caption - max image size": 384,
"Dataset - export image and caption": True, "Dataset - dimension divisor": 32,
"Dataset - caption": "long", "Dataset - image format": "JPEG", "Dataset - JPEG quality": 88,
})
plan["paths"]["a_raw_captions_jsonl"] = str(self.root / "witness.jsonl")
# The canonical plan may expose more than one compatible ledger key.
for key in plan["paths"]:
if "caption" in key.lower() and "jsonl" in key.lower():
plan["paths"][key] = str(self.root / "witness.jsonl")
record, _, _ = self.run_pipeline(plan=plan, **{
"Dataset - export image and caption": False,
"Dataset - output folder": str(self.root / "wrong"), "Dataset - max image size": 128,
"Dataset - dimension divisor": 1, "Dataset - caption": "taggy",
"Dataset - image format": "PNG", "Dataset - JPEG quality": 1,
})
self.assertEqual(record["status"], "ok", record)
exported = record["dataset_export"]
self.assertTrue(Path(exported["image"]).is_relative_to(self.root / "training_dataset"))
self.assertEqual((exported["width"], exported["height"]), (384, 256))
self.assertEqual(exported["caption_style"], "long")
self.assertTrue(exported["image"].endswith(".jpg"))
self.assertFalse((self.root / "wrong").exists())
receipt = json.loads(Path(exported["image"]).with_suffix(".captionforge.json").read_text())
self.assertEqual(receipt["signature"]["jpeg_quality"], 88)
def test_old_or_disabled_planner_cannot_enable_local_export(self):
record, _, _ = self.run_pipeline(plan={"captionforge_config_type": "captionforge_pipeline_plan"})
self.assertEqual(record["status"], "ok", record)
self.assertNotIn("dataset_export", record)
self.assertFalse((self.root / "training_dataset").exists())
def test_export_backfills_completed_captions_without_model_calls(self):
first, _, _ = self.run_pipeline(**{"Dataset - export image and caption": False})
self.assertNotIn("dataset_export", first)
record, text_calls, validator_calls = self.run_pipeline(overwrite=False)
self.assertEqual(record["dataset_export"]["status"], "ok")
self.assertEqual((text_calls, validator_calls), (0, 0))
def test_failed_export_preserves_captions_and_can_retry_without_models(self):
record, _, _ = self.run_pipeline(**{"Dataset - dimension divisor": 512})
self.assertEqual(record["status"], "error")
self.assertEqual(record["error_stage"], "dataset_export")
self.assertEqual(record["long"], "A long portrait.")
recovered, text_calls, validator_calls = self.run_pipeline(overwrite=False)
self.assertEqual(recovered["status"], "ok", recovered)
self.assertEqual((text_calls, validator_calls), (0, 0))
def test_validator_retry_does_not_reduce_training_export(self):
error = urllib.error.HTTPError("http://localhost/api/chat", 413, "Too large", {}, io.BytesIO())
record, _, calls = self.run_pipeline(
validator_side_effect=[error, ("A long portrait.", {})],
**{"Validator - max image size": 768},
)
self.assertEqual(calls, 2)
self.assertEqual((record["dataset_export"]["width"], record["dataset_export"]["height"]), (768, 512))
def test_interrupted_publication_requires_overwrite_and_recovers(self):
replace = exporter.os.replace
counter = 0
def interrupted(source, target):
nonlocal counter
counter += 1
if counter == 2:
raise OSError("Interrupted caption publication")
return replace(source, target)
with mock.patch.object(exporter.os, "replace", side_effect=interrupted):
with self.assertRaises(OSError):
self.pair()
with self.assertRaises(FileExistsError):
self.pair()
self.assertEqual(self.pair(overwrite=True)["status"], "ok")
self.assertEqual(self.source.read_bytes(), self.original_bytes)
def test_hardlink_to_source_cannot_be_overwritten(self):
destination = self.root / "training_dataset"
exporter.prepare_dataset_root(destination)
target, _, _ = exporter.export_paths(destination, self.source, self.root, "portraits/photo.jpg", "PNG")
target.parent.mkdir(parents=True)
target.hardlink_to(self.source)
with self.assertRaises(ValueError):
self.pair(overwrite=True)
self.assertEqual(self.source.read_bytes(), self.original_bytes)
def test_caption_variant_hardlinks_and_untracked_files_are_protected(self):
destination = self.root / "training_dataset"
exporter.prepare_dataset_root(destination)
target, _, _ = exporter.export_paths(destination, self.source, self.root, "portraits/photo.jpg", "PNG")
target.parent.mkdir(parents=True)
variant = target.with_name(target.stem + "_long.txt")
variant.hardlink_to(self.source)
with self.assertRaises(ValueError):
self.pair(overwrite=True, write_variants=True)
self.assertEqual(self.source.read_bytes(), self.original_bytes)
variant.unlink()
variant.write_text("My existing caption")
with self.assertRaises(FileExistsError):
self.pair(overwrite=True, write_variants=True)
self.assertEqual(variant.read_text(), "My existing caption")
def test_format_change_cannot_leave_duplicate_training_images(self):
original = self.pair()
with self.assertRaises(FileExistsError):
self.pair(settings={**self.settings, "image_format": "JPEG"}, overwrite=True)
self.assertTrue(Path(original["image"]).exists())
self.assertFalse(Path(original["image"]).with_suffix(".jpg").exists())
def test_changed_variant_cannot_silently_resume(self):
first = self.pair(write_variants=True)
Path(first["variants"]["long"]).write_text("Changed caption")
with self.assertRaises(FileExistsError):
self.pair(write_variants=True)
self.assertEqual(self.pair(write_variants=True, overwrite=True)["status"], "ok")
def test_canonical_workflows_serialize_dataset_widgets(self):
directory = ROOT / "assets" / "workflows"
ui = json.loads((directory / "CaptionForge_FullWorkflow_Rel_v1.0.2.json").read_text(encoding="utf-8"))
api = json.loads((directory / "CaptionForge_FullWorkflow_API_Rel_v1.0.2.json").read_text(encoding="utf-8"))
for name, cls in (("JLC_CaptionForge", orchestrator.JLC_CaptionForge),
("JLC_CaptionForge_Pipeline_Planner", planner.JLC_CaptionForge_Pipeline_Planner)):
node = next(item for item in ui["nodes"] if item["type"] == name)
api_node = next(item for item in api.values() if item["class_type"] == name)
names = list(cls.INPUT_TYPES()["required"]) + list(exporter.dataset_export_inputs())
self.assertEqual(node["widgets_values"], [node["widgets_values_named"][key] for key in names])
for key in names:
self.assertEqual(api_node["inputs"][key], node["widgets_values_named"][key])
if __name__ == "__main__":
unittest.main()
+258
View File
@@ -0,0 +1,258 @@
"""CPU-only warning and installation-contract tests; no model downloads required."""
import ast
from contextlib import nullcontext
import importlib.util
import logging
import threading
from pathlib import Path
import types
import unittest
from unittest import mock
from unittest.mock import Mock
import warnings
try:
import tomllib
except ModuleNotFoundError:
import tomli as tomllib
ROOT = Path(__file__).resolve().parents[1]
SPEC = importlib.util.spec_from_file_location(
'joy_warnings', ROOT / 'engines/captionforge_joy_warnings.py'
)
joy_warnings = importlib.util.module_from_spec(SPEC)
SPEC.loader.exec_module(joy_warnings)
scope = joy_warnings.quantized_inference_warnings
MESSAGE = 'MatMul8bitLt: inputs will be cast from torch.bfloat16 to float16 during quantization'
FP32_MESSAGE = MESSAGE.replace('bfloat16', 'float32')
MODULE = 'bitsandbytes.autograd._functions'
def emit(message=MESSAGE, category=UserWarning, module=MODULE):
warnings.warn_explicit(message, category, filename='mock_bnb.py', lineno=1, module=module)
class WarningScopeTests(unittest.TestCase):
def test_old_8bit_stack_warns_for_joy_and_qwen_without_blocking(self):
with mock.patch.object(joy_warnings.metadata, "version", return_value="0.45.5"):
with warnings.catch_warnings(record=True) as caught:
warnings.simplefilter("always")
joy_warnings.warn_if_suspicious_8bit_stack("Joy Caption")
joy_warnings.warn_if_suspicious_8bit_stack("Qwen Caption")
messages = [str(item.message) for item in caught]
self.assertTrue(any("CaptionForge's Joy Caption node detected" in item for item in messages))
self.assertTrue(any("CaptionForge's Qwen Caption node detected" in item for item in messages))
self.assertTrue(all("continue without modifying packages" in item for item in messages))
def test_current_8bit_stack_does_not_warn(self):
with mock.patch.object(joy_warnings.metadata, "version", return_value="0.46.1"):
with warnings.catch_warnings(record=True) as caught:
warnings.simplefilter("always")
joy_warnings.warn_if_suspicious_8bit_stack("Joy Caption")
joy_warnings.warn_if_suspicious_8bit_stack("Qwen Caption")
self.assertEqual(caught, [])
def test_import_does_not_change_filters(self):
before = list(warnings.filters)
SPEC.loader.exec_module(joy_warnings)
self.assertEqual(warnings.filters, before)
def test_six_caption_bursts_and_useful_diagnostics(self):
with warnings.catch_warnings(record=True) as caught:
warnings.simplefilter('always')
before = list(warnings.filters)
for _ in range(6):
with scope(True):
for _ in range(100):
emit()
emit(FP32_MESSAGE)
emit('Unrelated bitsandbytes warning')
self.assertEqual(warnings.filters, before)
self.assertEqual(len(caught), 6)
self.assertTrue(all(str(w.message) == 'Unrelated bitsandbytes warning' for w in caught))
def test_near_matches_remain_visible(self):
with warnings.catch_warnings(record=True) as caught:
warnings.simplefilter('always')
with scope(True):
emit(MESSAGE.replace('bfloat16', 'float64'))
emit(MESSAGE + ' extra context')
emit('prefix ' + MESSAGE)
emit(module='another_engine')
emit(FP32_MESSAGE, module='another_engine')
emit(category=RuntimeWarning)
emit(module=MODULE + '.other')
self.assertEqual(len(caught), 7)
def test_default_mode_and_after_scope_remain_visible(self):
with warnings.catch_warnings(record=True) as caught:
warnings.simplefilter('always')
with scope(False):
emit()
with scope(True):
emit()
emit()
self.assertEqual(len(caught), 2)
def test_exception_restores_callers_error_policy(self):
with warnings.catch_warnings():
warnings.simplefilter('error')
before = list(warnings.filters)
with self.assertRaisesRegex(ValueError, 'generation failed'):
with scope(True):
emit()
raise ValueError('generation failed')
self.assertEqual(warnings.filters, before)
with self.assertRaises(UserWarning):
emit()
with scope(True):
with self.assertRaisesRegex(UserWarning, 'unrelated'):
emit('unrelated')
class LoggingScopeTests(unittest.TestCase):
def setUp(self):
self.logger = logging.getLogger(MODULE)
self.before = (list(self.logger.filters), self.logger.level, list(self.logger.handlers), self.logger.propagate)
def tearDown(self):
self.assertEqual((list(self.logger.filters), self.logger.level, list(self.logger.handlers), self.logger.propagate), self.before)
def test_six_logging_bursts_keep_other_diagnostics(self):
with self.assertLogs(MODULE, level='WARNING') as caught:
for _ in range(6):
with scope(True):
for _ in range(100):
self.logger.warning('MatMul8bitLt: inputs will be cast from %s to float16 during quantization', 'torch.bfloat16')
self.logger.warning('MatMul8bitLt: inputs will be cast from %s to float16 during quantization', 'torch.float32')
self.logger.warning('Useful diagnostic')
self.assertEqual([r.getMessage() for r in caught.records], ['Useful diagnostic'] * 6)
def test_near_matches_levels_child_logger_and_other_thread(self):
with self.assertLogs(MODULE, level='INFO') as caught:
with scope(True):
self.logger.warning(MESSAGE.replace('bfloat16', 'float64'))
self.logger.warning(MESSAGE + ' extra')
self.logger.error(MESSAGE)
self.logger.info(MESSAGE)
logging.getLogger(MODULE + '.other').warning(MESSAGE)
thread = threading.Thread(target=lambda: self.logger.warning(MESSAGE))
thread.start()
thread.join()
self.assertEqual(len(caught.records), 6)
def test_default_after_scope_and_failure(self):
with self.assertLogs(MODULE, level='WARNING') as caught:
with scope(False):
self.logger.warning(MESSAGE)
with self.assertRaises(ValueError):
with scope(True):
self.logger.warning(MESSAGE)
raise ValueError('generation failed')
self.logger.warning(MESSAGE)
self.assertEqual(len(caught.records), 2)
def test_existing_filters_preserved_in_nested_scope(self):
existing = logging.Filter()
self.logger.addFilter(existing)
try:
with self.assertLogs(MODULE, level='WARNING') as caught:
with scope(True):
with scope(True):
self.logger.warning(MESSAGE)
self.logger.warning(MESSAGE)
self.assertIn(existing, self.logger.filters)
self.logger.warning(MESSAGE)
self.assertEqual(len(caught.records), 1)
self.assertEqual(self.logger.filters, self.before[0] + [existing])
finally:
self.logger.removeFilter(existing)
class GenerationScopeTests(unittest.TestCase):
"""Execute the real caption_pil method with processor/torch/model test doubles."""
def run_generation(self, mode, fail=False):
tree = ast.parse((ROOT / 'engines/jlc_joy_caption_engine.py').read_text(encoding='utf-8'))
cls = next(n for n in tree.body if isinstance(n, ast.ClassDef) and n.name == 'JoyCaptionEngine')
method = next(n for n in cls.body if isinstance(n, ast.FunctionDef) and n.name == 'caption_pil')
method.decorator_list = []
namespace = {
'quantized_inference_warnings': scope,
'torch': types.SimpleNamespace(
bfloat16='bf16', dtype=type(None), autocast=Mock(return_value=nullcontext())
),
'resize_for_model': lambda image, size: image,
'cleanup_caption': lambda text, config: text,
}
# Postponed annotations avoid importing PIL/torch just to test control flow.
module = ast.Module(body=[ast.ImportFrom(module='__future__', names=[ast.alias(name='annotations')], level=0), method], type_ignores=[])
exec(compile(ast.fix_missing_locations(module), '<caption_pil>', 'exec'), namespace)
class Inputs(dict):
def to(self, device):
return self
processor = Mock(return_value=Inputs(input_ids=types.SimpleNamespace(shape=(1, 2))))
processor.tokenizer.eos_token_id = 2
processor.tokenizer.pad_token_id = 2
processor.tokenizer.decode.return_value = 'caption'
def generate(**kwargs):
emit()
emit(FP32_MESSAGE)
logging.getLogger(MODULE).warning(
'MatMul8bitLt: inputs will be cast from %s to float16 during quantization',
'torch.bfloat16',
)
logging.getLogger(MODULE).warning(FP32_MESSAGE)
emit('Useful generation diagnostic')
if fail:
raise ValueError('generation failed')
return [[1, 2, 3]]
engine = types.SimpleNamespace(
generation=types.SimpleNamespace(seed=None, max_new_tokens=10, temperature=0, repetition_penalty=1),
config=types.SimpleNamespace(system_prompt='system', prompt='prompt', max_size=1024, memory_mode=mode),
cleanup=None, inference_device='cpu', processor=processor,
model=types.SimpleNamespace(dtype=None, generate=generate),
prepare_for_inference=Mock(), cleanup_after_inference=Mock(side_effect=lambda: emit('Cleanup diagnostic')),
_autocast_device_type=lambda: 'cpu', _autocast_enabled=lambda device: False,
)
with warnings.catch_warnings(record=True) as caught:
warnings.simplefilter('always')
before = list(warnings.filters)
if fail:
with self.assertRaisesRegex(ValueError, 'generation failed'):
namespace['caption_pil'](engine, Mock())
else:
self.assertEqual(namespace['caption_pil'](engine, Mock()), ('caption', 'caption'))
self.assertEqual(warnings.filters, before)
engine.cleanup_after_inference.assert_called_once()
self.assertEqual(namespace['torch'].autocast.call_args.kwargs['dtype'], 'bf16')
return [str(w.message) for w in caught]
def test_balanced_generation(self):
self.assertEqual(self.run_generation('Balanced (8-bit)'), ['Useful generation diagnostic', 'Cleanup diagnostic'])
def test_default_generation(self):
self.assertEqual(self.run_generation('Default'), [MESSAGE, FP32_MESSAGE, 'Useful generation diagnostic', 'Cleanup diagnostic'])
def test_failed_generation(self):
self.assertEqual(self.run_generation('Balanced (8-bit)', fail=True), ['Useful generation diagnostic', 'Cleanup diagnostic'])
class DependencyContractTests(unittest.TestCase):
def test_manager_and_package_dependencies_match(self):
project = tomllib.loads((ROOT / 'pyproject.toml').read_text(encoding='utf-8'))['project']
requirements = {
line.strip() for line in (ROOT / 'requirements.txt').read_text().splitlines()
if line.strip() and not line.lstrip().startswith('#')
}
self.assertEqual(requirements, set(project['dependencies']))
self.assertIn('accelerate', requirements)
self.assertIn('bitsandbytes>=0.46.1', requirements)
self.assertEqual(project['optional-dependencies']['quantization'], ['bitsandbytes>=0.46.1'])
if __name__ == '__main__':
unittest.main()
+406
View File
@@ -0,0 +1,406 @@
"""Focused tests for the CaptionForge Orchestrator output contract."""
from __future__ import annotations
import base64
import importlib
import io
import json
import sys
import tempfile
import types
import unittest
from contextlib import redirect_stdout
from pathlib import Path
from unittest import mock
from PIL import Image
ROOT = Path(__file__).resolve().parents[1]
def _install_namespace(name: str, path: Path) -> None:
if name in sys.modules:
return
module = types.ModuleType(name)
module.__path__ = [str(path)]
sys.modules[name] = module
_install_namespace("CaptionForge", ROOT)
_install_namespace("CaptionForge.engines", ROOT / "engines")
_install_namespace("CaptionForge.nodes", ROOT / "nodes")
if "folder_paths" not in sys.modules:
folder_paths = types.ModuleType("folder_paths")
folder_paths.get_output_directory = lambda: str(ROOT / "output")
sys.modules["folder_paths"] = folder_paths
planner_engine = importlib.import_module("CaptionForge.engines.captionforge_pipeline_planner_engine")
orchestrator = importlib.import_module("CaptionForge.nodes.jlc_captionforge_node")
class ValidatorImagePreparationTests(unittest.TestCase):
def _prepare(self, size: tuple[int, int], max_size: int) -> tuple[tuple[int, int], str, Path]:
temp_dir = tempfile.TemporaryDirectory()
self.addCleanup(temp_dir.cleanup)
image_path = Path(temp_dir.name) / "source.jpg"
Image.new("RGB", size, "white").save(image_path, format="JPEG", quality=95)
output = io.StringIO()
with redirect_stdout(output):
encoded = orchestrator._pil_to_base64_png(image_path, max_size=max_size)
with Image.open(io.BytesIO(base64.b64decode(encoded))) as transmitted:
transmitted_size = transmitted.size
transmitted_format = transmitted.format
self.assertEqual(transmitted_format, "PNG")
return transmitted_size, output.getvalue(), image_path
def test_large_image_is_reduced_to_longest_side_maximum(self) -> None:
transmitted_size, log, image_path = self._prepare((400, 200), max_size=100)
self.assertEqual(transmitted_size, (100, 50))
with Image.open(image_path) as source:
self.assertEqual(source.size, (400, 200))
self.assertIn("source=400x200", log)
self.assertIn("validator=100x50", log)
self.assertIn("encoded_payload=", log)
self.assertIn("MiB", log)
def test_resize_preserves_aspect_ratio(self) -> None:
transmitted_size, _, _ = self._prepare((300, 500), max_size=100)
self.assertEqual(transmitted_size, (60, 100))
self.assertEqual(transmitted_size[0] / transmitted_size[1], 300 / 500)
def test_image_below_maximum_is_not_enlarged(self) -> None:
transmitted_size, _, _ = self._prepare((80, 40), max_size=100)
self.assertEqual(transmitted_size, (80, 40))
def test_zero_maximum_disables_resizing(self) -> None:
transmitted_size, _, _ = self._prepare((400, 200), max_size=0)
self.assertEqual(transmitted_size, (400, 200))
class OrchestratorOutputContractTests(unittest.TestCase):
def test_public_output_surface_is_exactly_five_named_strings(self) -> None:
self.assertEqual(
orchestrator.JLC_CaptionForge.RETURN_NAMES,
("long_captions", "short_captions", "taggy_captions", "final_records", "status"),
)
self.assertEqual(orchestrator.JLC_CaptionForge.RETURN_TYPES, ("STRING",) * 5)
def _run(self, *, planner_connected: bool) -> tuple[tuple[str, ...], Path]:
temp_dir = tempfile.TemporaryDirectory()
self.addCleanup(temp_dir.cleanup)
root = Path(temp_dir.name)
image_names = ("one.png", "two.png")
for name in image_names:
Image.new("RGB", (2, 2), "white").save(root / name)
caption_path = root / "captions.jsonl"
caption_path.write_text(
"\n".join(
json.dumps(
{
"image": name,
"image_key": name,
"caption": f"Pass-A caption for {name}",
"model_family": "joy",
"status": "ok",
}
)
for name in image_names
)
+ "\n",
encoding="utf-8",
)
kwargs = {
"Input - captions JSONL": str(caption_path),
"Input - image path": str(root),
"Output - folder": str(root / "standalone-output"),
"Output - run name": "output-contract",
"Output - overwrite outputs": True,
"Final - write TXT sidecars": True,
"Final - write JSONL": True,
}
if planner_connected:
plan = planner_engine.build_captionforge_pipeline_plan(
output_dir=str(root / "planned-output"),
input_path=str(root),
run_name="output-contract-planned",
final_write_txt_sidecars=True,
final_write_jsonl=True,
)
plan["paths"]["caption_jsonl"] = str(caption_path)
plan["paths"]["pass_a_jsonl"] = str(caption_path)
kwargs["pipeline_plan"] = plan
text_results = iter(
(
"fat draft one",
"SHORT: Short one.\nTAGGY: tag one, detail one",
"fat draft two",
"SHORT: Short two.\nTAGGY: tag two, detail two",
)
)
validator_results = iter(("Long one.", "Long two."))
def fake_text(**_kwargs):
return next(text_results), {"ok": True}
def fake_image(**_kwargs):
return next(validator_results), {"ok": True}
with mock.patch.object(
orchestrator, "_evict_python_models_before_ollama_if_needed"
), mock.patch.object(
orchestrator, "_ollama_generate_text", side_effect=fake_text
), mock.patch.object(
orchestrator, "_ollama_chat_image", side_effect=fake_image
):
result = orchestrator.JLC_CaptionForge().forge(**kwargs)
return result, root
def _assert_semantic_outputs(self, *, planner_connected: bool) -> None:
result, root = self._run(planner_connected=planner_connected)
self.assertEqual(len(result), 5)
long_captions, short_captions, taggy_captions, final_records_json, status = result
self.assertEqual(long_captions.split("\n\n"), ["Long one.", "Long two."])
self.assertEqual(short_captions.split("\n\n"), ["Short one.", "Short two."])
self.assertEqual(
taggy_captions.split("\n\n"),
["tag one, detail one", "tag two, detail two"],
)
payload = json.loads(final_records_json)
self.assertEqual(set(payload), {"records", "run_outputs"})
self.assertEqual(len(payload["records"]), 2)
self.assertEqual(
[(record["long"], record["short"], record["taggy"]) for record in payload["records"]],
[
("Long one.", "Short one.", "tag one, detail one"),
("Long two.", "Short two.", "tag two, detail two"),
],
)
for image_name, record in zip(("one.png", "two.png"), payload["records"], strict=True):
self.assertEqual(record["image_key"], image_name)
self.assertEqual(set(record["outputs"]), {"long", "short", "taggy"})
self.assertEqual(len(record["sidecar_paths"]), 3)
for path in record["outputs"].values():
self.assertTrue(Path(path).is_file())
self.assertEqual(Path(path).parent, root)
run_outputs = payload["run_outputs"]
for key in (
"caption_jsonl",
"pass_a_jsonl",
"fat_draft_jsonl",
"validator_jsonl",
"taggy_jsonl",
"final_jsonl",
"output_paths_json",
"image_root",
"image_roots",
):
self.assertIn(key, run_outputs)
self.assertEqual(run_outputs["planner_connected"], planner_connected)
self.assertIsInstance(status, str)
self.assertIn("CaptionForge Orchestrator", status)
self.assertIn(f"planner_connected={planner_connected}", status)
self.assertIn("images=2", status)
self.assertIn("final_ok=2 final_failed=0", status)
def test_standalone_returns_aligned_semantic_products_and_records(self) -> None:
self._assert_semantic_outputs(planner_connected=False)
def test_planner_connected_returns_the_same_semantic_products(self) -> None:
self._assert_semantic_outputs(planner_connected=True)
def test_failed_item_keeps_an_empty_aligned_semantic_position(self) -> None:
temp_dir = tempfile.TemporaryDirectory()
self.addCleanup(temp_dir.cleanup)
root = Path(temp_dir.name)
image_names = ("one.png", "two.png", "three.png")
for name in image_names:
Image.new("RGB", (2, 2), "white").save(root / name)
caption_path = root / "captions.jsonl"
source_records = (
{
"image": "one.png",
"image_key": "one.png",
"caption": "usable caption one",
"model_family": "joy",
"status": "ok",
},
{
"image": "two.png",
"image_key": "two.png",
"caption": "FAILED: exception text must never become a caption",
"model_family": "joy",
"status": "error",
},
{
"image": "three.png",
"image_key": "three.png",
"caption": "usable caption three",
"model_family": "joy",
"status": "ok",
},
)
caption_path.write_text(
"\n".join(json.dumps(record) for record in source_records) + "\n",
encoding="utf-8",
)
text_results = iter(
(
"fat draft one",
"SHORT: Short one.\nTAGGY: tag one",
"fat draft three",
"SHORT: Short three.\nTAGGY: tag three",
)
)
validator_results = iter(("Long one.", "Long three."))
with mock.patch.object(
orchestrator, "_evict_python_models_before_ollama_if_needed"
), mock.patch.object(
orchestrator,
"_ollama_generate_text",
side_effect=lambda **_kwargs: (next(text_results), {"ok": True}),
), mock.patch.object(
orchestrator,
"_ollama_chat_image",
side_effect=lambda **_kwargs: (next(validator_results), {"ok": True}),
):
result = orchestrator.JLC_CaptionForge().forge(
**{
"Input - captions JSONL": str(caption_path),
"Input - image path": str(root),
"Output - folder": str(root / "output"),
"Output - run name": "failure-output-contract",
"Output - overwrite outputs": True,
"Final - write TXT sidecars": True,
"Final - write JSONL": True,
}
)
long_captions, short_captions, taggy_captions, final_records_json, status = result
semantic_entries = (
long_captions.split("\n\n"),
short_captions.split("\n\n"),
taggy_captions.split("\n\n"),
)
self.assertEqual(semantic_entries[0], ["Long one.", "", "Long three."])
self.assertEqual(semantic_entries[1], ["Short one.", "", "Short three."])
self.assertEqual(semantic_entries[2], ["tag one", "", "tag three"])
for entries in semantic_entries:
self.assertEqual(len(entries), 3)
self.assertEqual(entries[1], "")
joined = " ".join(entries).lower()
for diagnostic in ("failed", "failure", "error", "exception", "traceback"):
self.assertNotIn(diagnostic, joined)
records = json.loads(final_records_json)["records"]
self.assertEqual(len(records), 3)
self.assertEqual(records[1]["status"], "error")
self.assertEqual(records[1]["error"], "no_usable_captions_selected")
self.assertEqual(
(records[1]["long"], records[1]["short"], records[1]["taggy"]),
("", "", ""),
)
self.assertIn("final_ok=2 final_failed=1", status)
def test_downstream_reintroduction_is_cleaned_with_planner_precedence_and_audited(self) -> None:
temp_dir = tempfile.TemporaryDirectory()
self.addCleanup(temp_dir.cleanup)
root = Path(temp_dir.name)
Image.new("RGB", (2, 2), "white").save(root / "one.png")
captions = root / "captions.jsonl"
captions.write_text(json.dumps({
"image": "one.png", "image_key": "one.png", "caption": "clean witness",
"model_family": "joy", "status": "ok",
}) + "\n", encoding="utf-8")
plan = planner_engine.build_captionforge_pipeline_plan(
output_dir=str(root / "out"), input_path=str(root), run_name="cleanup",
forbidden_phrases="old", replace_pairs="former=>current",
)
plan["paths"]["caption_jsonl"] = str(captions)
plan["paths"]["pass_a_jsonl"] = str(captions)
generated = iter((
"former draft with old but bold holding gold",
"SHORT: former short old bold.\nTAGGY: former, old, gold",
))
with mock.patch.object(orchestrator, "_evict_python_models_before_ollama_if_needed"), \
mock.patch.object(orchestrator, "_ollama_generate_text", side_effect=lambda **_: (next(generated), {})), \
mock.patch.object(orchestrator, "_ollama_chat_image", return_value=("former long old bold holding gold.", {})):
result = orchestrator.JLC_CaptionForge().forge(**{
"Input - captions JSONL": str(captions), "Input - image path": str(root),
"Output - folder": str(root / "standalone"), "Output - run name": "ignored",
"Output - overwrite outputs": True, "Cleanup - forbidden phrases": "gold",
"Cleanup - replace pairs": "former=>wrong", "Final - write TXT sidecars": False,
"Final - write JSONL": True, "pipeline_plan": plan,
})
long_text, short_text, taggy_text, payload_text, _ = result
for text_value in (long_text, short_text, taggy_text):
self.assertNotIn(" old", f" {text_value.lower()}")
self.assertNotIn("former", text_value.lower())
self.assertIn("current", text_value.lower())
self.assertIn("bold", long_text)
self.assertIn("holding", long_text)
self.assertIn("gold", long_text)
payload = json.loads(payload_text)
self.assertEqual(payload["run_outputs"]["cleanup"], plan["cleanup"])
self.assertEqual(payload["records"][0]["cleanup"], plan["cleanup"])
run_config = json.loads(Path(plan["paths"]["run_config_json"]).read_text(encoding="utf-8"))
self.assertEqual(run_config["cleanup"], plan["cleanup"])
def test_standalone_orchestrator_cleanup_values_are_effective_and_audited(self) -> None:
temp_dir = tempfile.TemporaryDirectory()
self.addCleanup(temp_dir.cleanup)
root = Path(temp_dir.name)
Image.new("RGB", (2, 2), "white").save(root / "one.png")
captions = root / "captions.jsonl"
captions.write_text(json.dumps({
"image": "one.png", "image_key": "one.png", "caption": "clean witness",
"model_family": "joy", "status": "ok",
}) + "\n", encoding="utf-8")
generated = iter(("former draft old", "SHORT: former short old.\nTAGGY: former, old"))
with mock.patch.object(orchestrator, "_evict_python_models_before_ollama_if_needed"), \
mock.patch.object(orchestrator, "_ollama_generate_text", side_effect=lambda **_: (next(generated), {})), \
mock.patch.object(orchestrator, "_ollama_chat_image", return_value=("former long old.", {})):
result = orchestrator.JLC_CaptionForge().forge(**{
"Input - captions JSONL": str(captions), "Input - image path": str(root),
"Output - folder": str(root / "out"), "Output - run name": "standalone-cleanup",
"Output - overwrite outputs": True, "Cleanup - forbidden phrases": "old",
"Cleanup - replace pairs": "former=>current", "Final - write TXT sidecars": False,
"Final - write JSONL": True,
})
for text_value in result[:3]:
self.assertNotIn("old", text_value.lower())
self.assertIn("current", text_value.lower())
payload = json.loads(result[3])
expected = {
"forbidden_phrases": ["old"],
"replace_pairs": [{"old": "former", "new": "current"}],
"matching": "boundary_safe_case_insensitive",
"order": ["replace_pairs", "forbidden_phrases", "normalize_whitespace_punctuation"],
}
self.assertEqual(payload["records"][0]["cleanup"], expected)
run_config_path = Path(payload["run_outputs"]["run_config_json"])
self.assertEqual(json.loads(run_config_path.read_text(encoding="utf-8"))["cleanup"], expected)
if __name__ == "__main__":
unittest.main()
+297
View File
@@ -0,0 +1,297 @@
"""Focused resilience tests for the CaptionForge Orchestrator."""
from __future__ import annotations
import importlib
import io
import json
import sys
import tempfile
import types
import unittest
import urllib.error
from pathlib import Path
from unittest import mock
from PIL import Image
ROOT = Path(__file__).resolve().parents[1]
def _install_namespace(name: str, path: Path) -> None:
if name in sys.modules:
return
module = types.ModuleType(name)
module.__path__ = [str(path)]
sys.modules[name] = module
_install_namespace("CaptionForge", ROOT)
_install_namespace("CaptionForge.engines", ROOT / "engines")
_install_namespace("CaptionForge.nodes", ROOT / "nodes")
if "folder_paths" not in sys.modules:
folder_paths = types.ModuleType("folder_paths")
folder_paths.get_output_directory = lambda: str(ROOT / "output")
sys.modules["folder_paths"] = folder_paths
orchestrator = importlib.import_module("CaptionForge.nodes.jlc_captionforge_node")
def _http_error(status: int) -> urllib.error.HTTPError:
return urllib.error.HTTPError(
"http://127.0.0.1:11434/api/chat",
status,
"Request Entity Too Large" if status == 413 else "HTTP failure",
{},
io.BytesIO(b"error"),
)
def _wrapped_http_error(status: int) -> RuntimeError:
error = RuntimeError(f"HTTP {status} from Ollama")
error.__cause__ = _http_error(status)
return error
def _completed_record(image_key: str, *, status: str = "ok") -> dict:
complete = status == "ok"
return {
"captionforge_pass": "D_FINAL_EXPORT",
"image_key": image_key,
"status": status,
"final_caption": "Existing long." if complete else "",
"long": "Existing long." if complete else "",
"short": "Existing short." if complete else "",
"taggy": "existing, tags" if complete else "",
}
class OrchestratorResilienceTests(unittest.TestCase):
def _run(
self,
image_keys: tuple[str, ...],
*,
overwrite: bool = True,
max_size: int | None = None,
existing_final_records: tuple[dict, ...] = (),
text_side_effect=None,
validator_side_effect=None,
) -> tuple[tuple[str, ...], Path, mock.Mock, mock.Mock, mock.Mock]:
temp_dir = tempfile.TemporaryDirectory()
self.addCleanup(temp_dir.cleanup)
root = Path(temp_dir.name)
for image_key in image_keys:
Image.new("RGB", (4, 2), "white").save(root / f"{image_key}.png")
caption_path = root / "captions.jsonl"
caption_path.write_text(
"\n".join(
json.dumps(
{
"image": f"{image_key}.png",
"image_key": image_key,
"caption": f"source caption for {image_key}",
"model_family": "joy",
"status": "ok",
}
)
for image_key in image_keys
)
+ "\n",
encoding="utf-8",
)
output_dir = root / "output"
run_name = "resilience"
plan: dict = {}
if max_size is not None:
plan = {
"captionforge_config_type": "captionforge_pipeline_plan",
"caption_settings": {"max_size": max_size},
}
paths = orchestrator._derive_paths(output_dir, run_name, plan)
final_path = Path(paths["final_jsonl"])
if existing_final_records:
orchestrator._write_jsonl(final_path, list(existing_final_records), append=True)
if text_side_effect is None:
text_side_effect = lambda **_kwargs: (
"SHORT: Generated short.\nTAGGY: generated, tags",
{"ok": True},
)
if validator_side_effect is None:
validator_side_effect = lambda **_kwargs: ("Generated long caption.", {"ok": True})
kwargs = {
"Input - captions JSONL": str(caption_path),
"Input - image path": str(root),
"Output - folder": str(output_dir),
"Output - run name": run_name,
"Output - overwrite outputs": overwrite,
"Final - write TXT sidecars": False,
"Final - write JSONL": True,
}
if plan:
kwargs["pipeline_plan"] = plan
original_prepare = orchestrator._pil_to_base64_png
with mock.patch.object(
orchestrator, "_evict_python_models_before_ollama_if_needed"
), mock.patch.object(
orchestrator, "_ollama_generate_text", side_effect=text_side_effect
) as text_mock, mock.patch.object(
orchestrator, "_ollama_chat_image", side_effect=validator_side_effect
) as validator_mock, mock.patch.object(
orchestrator, "_pil_to_base64_png", wraps=original_prepare
) as prepare_mock:
result = orchestrator.JLC_CaptionForge().forge(**kwargs)
return result, final_path, text_mock, validator_mock, prepare_mock
def test_middle_image_failure_is_isolated_and_later_image_completes(self) -> None:
def text_response(**kwargs):
if "source caption for middle" in kwargs["prompt"]:
raise RuntimeError("synthetic distiller failure")
return "SHORT: Generated short.\nTAGGY: generated, tags", {"ok": True}
result, _, _, validator_mock, _ = self._run(
("first", "middle", "third"),
text_side_effect=text_response,
)
payload = json.loads(result[3])
records = payload["records"]
self.assertEqual([record["status"] for record in records], ["ok", "error", "ok"])
self.assertEqual(records[1]["image_key"], "middle")
self.assertEqual(records[1]["error_stage"], "distiller")
self.assertEqual(records[1]["error_type"], "RuntimeError")
self.assertEqual(records[1]["error_message"], "synthetic distiller failure")
self.assertEqual(validator_mock.call_count, 2)
self.assertIn("final_ok=2 final_failed=1", result[4])
def test_resume_skips_only_latest_complete_final_records(self) -> None:
existing = (
_completed_record("a", status="error"),
_completed_record("a", status="ok"),
_completed_record("b", status="ok"),
_completed_record("b", status="error"),
)
result, final_path, text_mock, validator_mock, _ = self._run(
("a", "b", "c"),
overwrite=False,
existing_final_records=existing,
)
self.assertEqual(text_mock.call_count, 4)
self.assertEqual(validator_mock.call_count, 2)
self.assertIn("resume_skipped=1", result[4])
self.assertEqual(json.loads(result[3])["run_outputs"]["resume_skipped"], 1)
ledger = orchestrator._read_jsonl(final_path)
successful_a = [
record
for record in ledger
if record.get("image_key") == "a" and record.get("status") == "ok"
]
self.assertEqual(len(successful_a), 1)
self.assertEqual(ledger[-2]["image_key"], "b")
self.assertEqual(ledger[-1]["image_key"], "c")
def test_resume_completion_uses_latest_usable_record_and_ignores_bad_line(self) -> None:
temp_dir = tempfile.TemporaryDirectory()
self.addCleanup(temp_dir.cleanup)
ledger_path = Path(temp_dir.name) / "final.jsonl"
partial_b = _completed_record("b")
partial_b["short"] = ""
records = (
_completed_record("a", status="error"),
_completed_record("a"),
_completed_record("b"),
partial_b,
_completed_record("c"),
_completed_record("c", status="error"),
)
orchestrator._write_jsonl(ledger_path, list(records), append=True)
with ledger_path.open("a", encoding="utf-8") as ledger:
ledger.write("{interrupted")
completed = orchestrator._completed_final_records(ledger_path)
self.assertEqual(set(completed), {"a"})
def test_overwrite_true_ignores_existing_completion_and_runs_fresh(self) -> None:
result, final_path, text_mock, validator_mock, _ = self._run(
("a", "b", "c"),
overwrite=True,
existing_final_records=(_completed_record("a"),),
)
self.assertEqual(text_mock.call_count, 6)
self.assertEqual(validator_mock.call_count, 3)
self.assertIn("resume_skipped=0", result[4])
self.assertEqual(len(orchestrator._read_jsonl(final_path)), 3)
def test_validator_413_retries_1024_then_succeeds_at_768(self) -> None:
result, _, _, validator_mock, prepare_mock = self._run(
("image",),
max_size=1024,
validator_side_effect=(_wrapped_http_error(413), ("Recovered long caption.", {"ok": True})),
)
self.assertEqual(result[4].count("final_ok=1"), 1)
self.assertEqual(validator_mock.call_count, 2)
self.assertEqual(
[call.kwargs["max_size"] for call in prepare_mock.call_args_list],
[1024, 768],
)
def test_unrelated_validator_http_error_is_not_retried(self) -> None:
result, _, _, validator_mock, prepare_mock = self._run(
("image",),
max_size=1024,
validator_side_effect=_wrapped_http_error(500),
)
record = json.loads(result[3])["records"][0]
self.assertEqual(record["status"], "error")
self.assertEqual(record["error_stage"], "validator")
self.assertEqual(validator_mock.call_count, 1)
self.assertEqual(len(prepare_mock.call_args_list), 1)
def test_exhausted_413_retries_fail_one_image_and_continue(self) -> None:
result, _, _, validator_mock, prepare_mock = self._run(
("first", "second"),
max_size=768,
validator_side_effect=(
_wrapped_http_error(413),
_wrapped_http_error(413),
("Second image completed.", {"ok": True}),
),
)
records = json.loads(result[3])["records"]
self.assertEqual([record["status"] for record in records], ["error", "ok"])
self.assertEqual(records[0]["error_stage"], "validator")
self.assertEqual(records[1]["image_key"], "second")
self.assertEqual(validator_mock.call_count, 3)
self.assertEqual(
[call.kwargs["max_size"] for call in prepare_mock.call_args_list],
[768, 512, 768],
)
self.assertIn("final_ok=1 final_failed=1", result[4])
def test_validator_retry_ladders_are_strictly_smaller(self) -> None:
self.assertEqual(orchestrator._validator_image_retry_caps(0), (0, 1024, 768, 512))
self.assertEqual(orchestrator._validator_image_retry_caps(1536), (1536, 1024, 768, 512))
self.assertEqual(orchestrator._validator_image_retry_caps(1024), (1024, 768, 512))
self.assertEqual(orchestrator._validator_image_retry_caps(768), (768, 512))
self.assertEqual(orchestrator._validator_image_retry_caps(512), (512,))
def test_explicit_user_abort_is_not_isolated(self) -> None:
with self.assertRaises(KeyboardInterrupt):
self._run(("image",), validator_side_effect=KeyboardInterrupt())
if __name__ == "__main__":
unittest.main()
+203
View File
@@ -0,0 +1,203 @@
"""Focused tests for Pass-A source identity and artifact persistence."""
from __future__ import annotations
import importlib
import json
import sys
import tempfile
import types
import unittest
from pathlib import Path
from unittest import mock
from PIL import Image
ROOT = Path(__file__).resolve().parents[1]
def _install_namespace(name: str, path: Path) -> None:
if name in sys.modules:
return
module = types.ModuleType(name)
module.__path__ = [str(path)]
sys.modules[name] = module
_install_namespace("CaptionForge", ROOT)
_install_namespace("CaptionForge.engines", ROOT / "engines")
_install_namespace("CaptionForge.nodes", ROOT / "nodes")
_install_namespace("CaptionForge.nodes.caption_nodes", ROOT / "nodes" / "caption_nodes")
if "folder_paths" not in sys.modules:
folder_paths = types.ModuleType("folder_paths")
folder_paths.models_dir = str(ROOT / "models")
folder_paths.get_output_directory = lambda: str(ROOT / "output")
sys.modules["folder_paths"] = folder_paths
elif not hasattr(sys.modules["folder_paths"], "models_dir"):
sys.modules["folder_paths"].models_dir = str(ROOT / "models")
planner_engine = importlib.import_module("CaptionForge.engines.captionforge_pipeline_planner_engine")
source_identity = importlib.import_module("CaptionForge.engines.captionforge_source_identity")
orchestrator = importlib.import_module("CaptionForge.nodes.jlc_captionforge_node")
joy = importlib.import_module("CaptionForge.nodes.caption_nodes.jlc_captionforge_joy_caption_node")
qwen = importlib.import_module("CaptionForge.nodes.caption_nodes.jlc_captionforge_qwen_caption_node")
ollama = importlib.import_module("CaptionForge.nodes.caption_nodes.jlc_captionforge_ollama_caption_node")
FILE_SOURCE_IDENTITIES = {
"_DSC2094-Edit-2.jpg": "_DSC2094-Edit-2",
"ChatGPT Image May 26, 2026, 08_21_31 AM.png": "ChatGPT Image May 26, 2026, 08_21_31 AM",
"Set A/Sub Folder/_DSC2094-Edit-2.jpg": "_DSC2094-Edit-2",
"Set A/photo.jpg": "photo",
"Set A/photo.png": "photo",
"Set B/photo.jpg": "photo",
"comfy_image_0000.png": "comfy_image_0000",
}
SOURCE_IDENTITIES = {
**FILE_SOURCE_IDENTITIES,
"captionforge-optional-image://comfy_image_0000.png": "comfy_image_0000",
}
def _required_defaults(node_class) -> dict[str, object]:
return {
name: spec[1]["default"]
for name, spec in node_class.INPUT_TYPES()["required"].items()
}
class PassAArtifactIdentityTests(unittest.TestCase):
def test_each_witness_writes_collision_safe_raw_identity_without_txt_sidecars(self) -> None:
with tempfile.TemporaryDirectory() as temp_dir:
root = Path(temp_dir)
input_dir = root / "inputs"
output_dir = root / "outputs"
input_dir.mkdir()
for relative_name in FILE_SOURCE_IDENTITIES:
image_path = input_dir / Path(*relative_name.split("/"))
image_path.parent.mkdir(parents=True, exist_ok=True)
Image.new("RGB", (2, 2), "white").save(image_path)
optional_pil = Image.new("RGB", (2, 2), "black")
plan = planner_engine.build_captionforge_pipeline_plan(
output_dir=str(output_dir),
input_path=str(input_dir),
run_name="captionforge_run",
joy_runs_per_image=1,
qwen_runs_per_image=1,
ollama_runs_per_image=1,
)
joy_engine = mock.Mock()
joy_engine.config = types.SimpleNamespace(max_size=0)
joy_engine.local_model_path = ""
joy_engine.caption_pil.return_value = ("joy caption", "joy raw")
joy_engine.build_run_config.return_value = {}
joy_kwargs = _required_defaults(joy.JLC_CaptionForgeJoy)
joy_kwargs["pipeline_plan"] = plan
with mock.patch.object(joy, "JoyCaptionEngine", return_value=joy_engine), mock.patch.object(
joy, "_tensor_to_pil", return_value=[optional_pil]
):
joy.JLC_CaptionForgeJoy().caption(**joy_kwargs)
qwen_engine = mock.Mock()
qwen_engine.config = types.SimpleNamespace(max_size=0)
qwen_engine.local_model_path = ""
qwen_engine.caption_pil.return_value = ("qwen caption", "qwen raw")
qwen_engine.build_run_config.return_value = {}
qwen_kwargs = _required_defaults(qwen.JLC_CaptionForgeQwen)
qwen_kwargs["pipeline_plan"] = plan
with mock.patch.object(qwen, "QwenCaptionEngine", return_value=qwen_engine), mock.patch.object(
qwen, "_tensor_to_pil", return_value=[optional_pil]
):
qwen.JLC_CaptionForgeQwen().caption(**qwen_kwargs)
ollama_kwargs = _required_defaults(ollama.JLC_CaptionForgeOllamaCaption)
ollama_kwargs["pipeline_plan"] = plan
with mock.patch.object(ollama, "_ensure_ollama_model"), mock.patch.object(
ollama, "_persist_caption_model_if_possible"
), mock.patch.object(
ollama, "_evict_python_models_before_ollama_if_needed"
), mock.patch.object(
ollama, "_ollama_generate_caption", return_value="ollama caption"
), mock.patch.object(
ollama, "_tensor_to_pil", return_value=[optional_pil]
):
ollama.JLC_CaptionForgeOllamaCaption().caption(**ollama_kwargs)
raw_path = Path(plan["paths"]["caption_jsonl"])
records = [json.loads(line) for line in raw_path.read_text(encoding="utf-8").splitlines()]
self.assertEqual(len(records), len(SOURCE_IDENTITIES) * 3)
for family in ("joy", "qwen", "ollama"):
family_records = [record for record in records if record["model_family"] == family]
actual = {record["image_key"]: record["image"] for record in family_records}
self.assertEqual(actual, SOURCE_IDENTITIES)
grouped = orchestrator._group_records_by_image(records)
self.assertEqual(set(grouped), set(SOURCE_IDENTITIES))
self.assertTrue(all(len(group) == 3 for group in grouped.values()))
self.assertEqual(list(root.rglob("*.txt")), [])
opt_images_dir = root / "opt_images"
opt_images_dir.mkdir()
optional_path = opt_images_dir / "comfy_image_0000.png"
optional_pil.save(optional_path)
optional_key = source_identity.optional_image_identity(0)[1]
self.assertEqual(optional_key, "captionforge-optional-image://comfy_image_0000.png")
self.assertEqual(
orchestrator._resolve_image_path_for_group(
grouped[optional_key],
[input_dir, opt_images_dir],
optional_image_root=opt_images_dir,
),
optional_path,
)
self.assertEqual(
orchestrator._resolve_image_path_for_group(
grouped["Set A/Sub Folder/_DSC2094-Edit-2.jpg"],
[input_dir, opt_images_dir],
optional_image_root=opt_images_dir,
),
input_dir / "Set A" / "Sub Folder" / "_DSC2094-Edit-2.jpg",
)
self.assertEqual(
orchestrator._resolve_image_path_for_group(
grouped["Set A/photo.jpg"],
[input_dir, opt_images_dir],
optional_image_root=opt_images_dir,
),
input_dir / "Set A" / "photo.jpg",
)
empty_dataset = root / "empty-inputs"
empty_dataset.mkdir()
self.assertEqual(
orchestrator._resolve_image_path_for_group(
[{"image": "comfy_image_0000", "image_key": "comfy_image_0000"}],
[empty_dataset, opt_images_dir],
optional_image_root=opt_images_dir,
),
optional_path,
)
self.assertEqual(
orchestrator._resolve_image_path_for_group(
[{"image": "photo", "image_key": r"Set A\photo.jpg"}],
[input_dir, opt_images_dir],
optional_image_root=opt_images_dir,
),
input_dir / "Set A" / "photo.jpg",
)
def test_raw_response_filenames_do_not_collapse_relative_keys(self) -> None:
with tempfile.TemporaryDirectory() as temp_dir:
raw_dir = Path(temp_dir)
first = orchestrator._save_raw_response(raw_dir, "Set A/photo.jpg", "stage", {"ok": 1})
second = orchestrator._save_raw_response(raw_dir, "Set B/photo.jpg", "stage", {"ok": 2})
extension_variant = orchestrator._save_raw_response(raw_dir, "Set A/photo.png", "stage", {"ok": 3})
self.assertEqual(len({first, second, extension_variant}), 3)
self.assertTrue(all(Path(path).exists() for path in (first, second, extension_variant)))
if __name__ == "__main__":
unittest.main()
+264
View File
@@ -0,0 +1,264 @@
"""Regression tests for Qwen 8-bit low-VRAM CPU offload behavior."""
from __future__ import annotations
import ast
import importlib
import sys
import tempfile
import types
import unittest
from types import SimpleNamespace
from pathlib import Path
from unittest import mock
import torch
ROOT = Path(__file__).resolve().parents[1]
def _install_namespace(name: str, path: Path) -> None:
if name in sys.modules:
return
module = types.ModuleType(name)
module.__path__ = [str(path)]
sys.modules[name] = module
_install_namespace("CaptionForge", ROOT)
_install_namespace("CaptionForge.engines", ROOT / "engines")
qwen = importlib.import_module("CaptionForge.engines.jlc_qwen_caption_engine")
class QwenOffloadTests(unittest.TestCase):
def test_generation_uses_scoped_bnb_diagnostic_handling(self):
tree = ast.parse(
(ROOT / "engines/jlc_qwen_caption_engine.py").read_text(encoding="utf-8")
)
engine_class = next(
node
for node in tree.body
if isinstance(node, ast.ClassDef) and node.name == "QwenCaptionEngine"
)
caption_method = next(
node
for node in engine_class.body
if isinstance(node, ast.FunctionDef) and node.name == "caption_pil"
)
scoped_generate_calls = []
for node in ast.walk(caption_method):
if not isinstance(node, ast.With):
continue
uses_scope = any(
isinstance(item.context_expr, ast.Call)
and isinstance(item.context_expr.func, ast.Name)
and item.context_expr.func.id == "quantized_inference_warnings"
for item in node.items
)
if uses_scope:
scoped_generate_calls.extend(
child
for child in ast.walk(node)
if isinstance(child, ast.Call)
and isinstance(child.func, ast.Attribute)
and child.func.attr == "generate"
)
self.assertEqual(len(scoped_generate_calls), 1)
self.assertNotIn("filterwarnings", ast.unparse(caption_method))
def test_memory_budget_reserves_headroom_and_fp32_cpu_cost(self):
gib = 1024 ** 3
with mock.patch.object(qwen.torch.cuda, "is_available", return_value=True), \
mock.patch.object(qwen.torch.cuda, "current_device", return_value=0), \
mock.patch.object(qwen.torch.cuda, "mem_get_info", return_value=(10 * gib, 12 * gib)), \
mock.patch.object(qwen, "_available_system_memory_bytes", return_value=40 * gib):
budget = qwen._qwen_memory_budget(headroom=0.20)
# GPU receives 80% of currently free VRAM.
self.assertEqual(budget[0], int(10 * gib * 0.80))
# CPU placement is estimated with dtype=int8 but actual bitsandbytes
# CPU overflow remains FP32, so CPU capacity is divided by four.
self.assertEqual(budget["cpu"], int(40 * gib * 0.80 / 4.0))
def test_execution_device_prefers_accelerate_hook(self):
model = SimpleNamespace(
_hf_hook=SimpleNamespace(execution_device="cuda:0"),
hf_device_map={"": "cpu"},
)
device = qwen._resolve_model_execution_device(model)
self.assertEqual(device, torch.device("cuda:0"))
def test_execution_device_prefers_gpu_from_device_map(self):
model = SimpleNamespace(
_hf_hook=None,
hf_device_map={
"visual": "cpu",
"model.layers.0": 0,
"model.layers.1": "cpu",
},
)
device = qwen._resolve_model_execution_device(model)
self.assertEqual(device, torch.device("cuda:0"))
def test_execution_device_falls_back_to_parameter_device(self):
parameter = torch.nn.Parameter(torch.zeros(1))
model = torch.nn.Linear(1, 1)
model.weight = parameter
device = qwen._resolve_model_execution_device(model)
self.assertEqual(device, parameter.device)
def test_device_map_summary(self):
summary = qwen._summarize_device_map(
{
"visual": 0,
"layer0": 0,
"layer1": "cpu",
}
)
self.assertIn("cuda:0: 2 module(s)", summary)
self.assertIn("cpu: 1 module(s)", summary)
def test_load_uses_supported_8bit_cpu_offload_and_preserves_preload_eviction(self):
events = []
captured = {}
class FakeBitsAndBytesConfig:
def __init__(self, **kwargs):
captured["bnb"] = dict(kwargs)
class FakeProcessor:
@classmethod
def from_pretrained(cls, *_args, **_kwargs):
return cls()
class FakeModel:
hf_device_map = {
"visual": 0,
"model.layers.0": 0,
"model.layers.1": "cpu",
}
def eval(self):
return self
class FakeModelClass:
@classmethod
def from_pretrained(cls, *_args, **kwargs):
events.append("load")
captured["model_kwargs"] = dict(kwargs)
return FakeModel()
fake_transformers = types.ModuleType("transformers")
fake_transformers.AutoProcessor = FakeProcessor
fake_transformers.BitsAndBytesConfig = FakeBitsAndBytesConfig
with tempfile.TemporaryDirectory() as temp_dir:
local_path = Path(temp_dir)
config = qwen.QwenCaptionConfig(
model_path=str(local_path),
quantization="bnb_8bit",
device_map="auto",
keep_loaded=True,
allow_download=False,
)
engine = qwen.QwenCaptionEngine(config)
def fake_prepare(*_args, **_kwargs):
events.append("prepare")
with mock.patch.dict(sys.modules, {"transformers": fake_transformers}), \
mock.patch.object(qwen, "get_cached_model", return_value=None), \
mock.patch.object(qwen, "prepare_for_model_load", side_effect=fake_prepare), \
mock.patch.object(qwen, "register_model"), \
mock.patch.object(engine, "_load_model_class", return_value=FakeModelClass), \
mock.patch.object(engine, "_detect_model_type", return_value="qwen2_5_vl"), \
mock.patch.object(
qwen,
"_build_qwen_8bit_device_map",
return_value=(
{
"visual": 0,
"model.layers.0": 0,
"model.layers.1": "cpu",
},
{0: 8 * 1024**3, "cpu": 8 * 1024**3},
),
):
engine.load()
self.assertEqual(events[:2], ["prepare", "load"])
self.assertTrue(captured["bnb"]["load_in_8bit"])
self.assertTrue(
captured["bnb"]["llm_int8_enable_fp32_cpu_offload"]
)
self.assertEqual(
captured["model_kwargs"]["device_map"]["model.layers.1"],
"cpu",
)
self.assertIn("max_memory", captured["model_kwargs"])
def test_adaptive_mapper_rejects_disk_spill(self):
class FakeModel:
_no_split_modules = []
class FakeModelClass:
def __new__(cls, *_args, **_kwargs):
return FakeModel()
fake_accelerate = SimpleNamespace(
init_empty_weights=mock.MagicMock(),
infer_auto_device_map=mock.MagicMock(
return_value={
"visual": 0,
"model.layers.0": "cpu",
"model.layers.1": "disk",
}
),
)
class FakeContext:
def __enter__(self):
return None
def __exit__(self, exc_type, exc, tb):
return False
fake_accelerate.init_empty_weights.return_value = FakeContext()
fake_transformers = SimpleNamespace(
AutoConfig=SimpleNamespace(
from_pretrained=mock.MagicMock(return_value=object())
)
)
modules = {
"accelerate": fake_accelerate,
"transformers": fake_transformers,
}
with mock.patch.dict("sys.modules", modules), \
mock.patch.object(qwen.torch.cuda, "is_available", return_value=True), \
mock.patch.object(qwen, "_qwen_memory_budget", return_value={0: 1, "cpu": 1}):
with self.assertRaisesRegex(RuntimeError, "disk offload"):
qwen._build_qwen_8bit_device_map(
FakeModelClass,
qwen.Path("."),
True,
)
if __name__ == "__main__":
unittest.main()
+126
View File
@@ -0,0 +1,126 @@
"""CPU-only release metadata, module documentation, and node-help checks."""
from __future__ import annotations
import ast
import importlib
import json
import re
import sys
import types
import unittest
from pathlib import Path
ROOT = Path(__file__).resolve().parents[1]
def _install_namespace(name: str, path: Path) -> None:
if name in sys.modules:
return
module = types.ModuleType(name)
module.__path__ = [str(path)]
sys.modules[name] = module
_install_namespace("CaptionForge", ROOT)
_install_namespace("CaptionForge.engines", ROOT / "engines")
_install_namespace("CaptionForge.nodes", ROOT / "nodes")
_install_namespace("CaptionForge.nodes.caption_nodes", ROOT / "nodes" / "caption_nodes")
if "folder_paths" not in sys.modules:
folder_paths = types.ModuleType("folder_paths")
folder_paths.models_dir = str(ROOT / "models")
folder_paths.get_output_directory = lambda: str(ROOT / "output")
sys.modules["folder_paths"] = folder_paths
elif not hasattr(sys.modules["folder_paths"], "models_dir"):
sys.modules["folder_paths"].models_dir = str(ROOT / "models")
version_module = importlib.import_module("CaptionForge.captionforge_version")
planner = importlib.import_module("CaptionForge.nodes.jlc_captionforge_pipeline_planner_node")
capstone = importlib.import_module("CaptionForge.nodes.jlc_captionforge_node")
template_options = importlib.import_module("CaptionForge.nodes.jlc_captionforge_template_options")
joy = importlib.import_module("CaptionForge.nodes.caption_nodes.jlc_captionforge_joy_caption_node")
qwen = importlib.import_module("CaptionForge.nodes.caption_nodes.jlc_captionforge_qwen_caption_node")
ollama = importlib.import_module("CaptionForge.nodes.caption_nodes.jlc_captionforge_ollama_caption_node")
class ReleaseMetadataTests(unittest.TestCase):
def test_current_release_versions_are_1_0_2(self) -> None:
self.assertEqual(version_module.CAPTIONFORGE_VERSION, "1.0.2")
pyproject = (ROOT / "pyproject.toml").read_text(encoding="utf-8")
match = re.search(r'^version\s*=\s*"([^"]+)"', pyproject, re.MULTILINE)
self.assertIsNotNone(match)
self.assertEqual(match.group(1), version_module.CAPTIONFORGE_VERSION)
config = json.loads(
(ROOT / "config" / "captionforge_ollama_models.json").read_text(encoding="utf-8")
)
self.assertEqual(config["_meta"]["version"], version_module.CAPTIONFORGE_VERSION)
self.assertEqual(capstone.CAPTIONFORGE_NODE_VERSION, version_module.CAPTIONFORGE_VERSION)
def test_anchor_tooltips_describe_frozen_1_x_behavior(self) -> None:
expected = (
"Optional persistent caption/training anchor. In CaptionForge 1.x, a non-empty "
"anchor is preserved in the final caption variants rather than treated as image "
"evidence that the Validator may remove."
)
self.assertEqual(
planner.JLC_CaptionForge_Pipeline_Planner.INPUT_TYPES()["required"]
["LoRA - user caption anchor"][1]["tooltip"],
expected,
)
self.assertEqual(
capstone.JLC_CaptionForge.INPUT_TYPES()["required"]
["LoRA - user caption anchor"][1]["tooltip"],
expected,
)
def test_package_discovery_is_explicit_and_production_only(self) -> None:
pyproject = (ROOT / "pyproject.toml").read_text(encoding="utf-8")
self.assertNotIn("[tool.setuptools.packages.find]", pyproject)
for package in (
"CaptionForge",
"CaptionForge.engines",
"CaptionForge.nodes",
"CaptionForge.nodes.caption_nodes",
):
self.assertIn(f'"{package}"', pyproject)
for excluded in (".backups", ".tests", ".bak", ".deprecated", "experimental"):
self.assertNotIn(f'"{excluded}', pyproject)
def test_active_python_modules_have_module_docstrings(self) -> None:
files = [ROOT / "__init__.py", ROOT / "captionforge_version.py"]
files.extend((ROOT / "engines").glob("*.py"))
files.extend((ROOT / "nodes").glob("*.py"))
files.extend((ROOT / "nodes" / "caption_nodes").glob("*.py"))
files.extend((ROOT / "tests").glob("*.py"))
for path in files:
with self.subTest(path=path.relative_to(ROOT)):
tree = ast.parse(path.read_text(encoding="utf-8"), filename=str(path))
self.assertTrue(ast.get_docstring(tree))
class ActiveNodeHelpTests(unittest.TestCase):
def test_every_active_input_has_plain_help_text(self) -> None:
node_classes = (
planner.JLC_CaptionForge_Pipeline_Planner,
capstone.JLC_CaptionForge,
template_options.JLC_CaptionForgeExtraOptions,
joy.JLC_CaptionForgeJoy,
qwen.JLC_CaptionForgeQwen,
ollama.JLC_CaptionForgeOllamaCaption,
)
for node_class in node_classes:
inputs = node_class.INPUT_TYPES()
for section in ("required", "optional"):
for name, spec in inputs.get(section, {}).items():
with self.subTest(node=node_class.__name__, input=name):
self.assertGreaterEqual(len(spec), 2)
metadata = spec[1]
self.assertIsInstance(metadata, dict)
self.assertTrue(str(metadata.get("tooltip", "")).strip())
if __name__ == "__main__":
unittest.main()
+300
View File
@@ -0,0 +1,300 @@
"""CPU tests for the shared Pass-A and standalone seed contracts."""
from __future__ import annotations
import importlib
import json
import random
import sys
import tempfile
import types
import unittest
from pathlib import Path
from unittest import mock
from PIL import Image
ROOT = Path(__file__).resolve().parents[1]
def _install_namespace(name: str, path: Path) -> None:
if name in sys.modules:
return
module = types.ModuleType(name)
module.__path__ = [str(path)]
sys.modules[name] = module
_install_namespace("CaptionForge", ROOT)
_install_namespace("CaptionForge.engines", ROOT / "engines")
_install_namespace("CaptionForge.nodes", ROOT / "nodes")
_install_namespace("CaptionForge.nodes.caption_nodes", ROOT / "nodes" / "caption_nodes")
if "folder_paths" not in sys.modules:
folder_paths = types.ModuleType("folder_paths")
folder_paths.get_output_directory = lambda: str(ROOT / "output")
sys.modules["folder_paths"] = folder_paths
planner_engine = importlib.import_module("CaptionForge.engines.captionforge_pipeline_planner_engine")
planner_node = importlib.import_module("CaptionForge.nodes.jlc_captionforge_pipeline_planner_node")
capstone = importlib.import_module("CaptionForge.nodes.jlc_captionforge_node")
ollama_caption = importlib.import_module("CaptionForge.nodes.caption_nodes.jlc_captionforge_ollama_caption_node")
class PassASeedContractTests(unittest.TestCase):
def _schedule(self, base_seed: int, seed_mode: str, runs: int = 3) -> list[int | None]:
plan = planner_engine.build_captionforge_pipeline_plan(
output_dir="unused",
joy_runs_per_image=runs,
qwen_runs_per_image=0,
ollama_runs_per_image=0,
base_seed=base_seed,
seed_mode=seed_mode,
)
return [run.seed for run in planner_engine.expand_captionforge_runs(plan, model_key="joy")]
def test_fixed_increment_and_decrement(self) -> None:
self.assertEqual(self._schedule(100, "fixed"), [100, 100, 100])
self.assertEqual(self._schedule(100, "increment"), [100, 101, 102])
self.assertEqual(self._schedule(100, "decrement"), [100, 99, 98])
def test_uint32_wraparound(self) -> None:
maximum = planner_engine.MAX_SEED_32
self.assertEqual(self._schedule(maximum - 1, "increment"), [maximum - 1, maximum, 0])
self.assertEqual(self._schedule(0, "decrement"), [0, maximum, maximum - 1])
def test_literal_zero_survives(self) -> None:
self.assertEqual(self._schedule(0, "fixed"), [0, 0, 0])
self.assertEqual(self._schedule(0, "increment"), [0, 1, 2])
self.assertEqual(self._schedule(0, "random"), self._schedule(0, "random"))
self.assertTrue(all(seed is not None for seed in self._schedule(0, "random")))
def test_hash_random_has_a_stable_golden_schedule(self) -> None:
expected = [90425488, 1231494799, 4241276224, 2828971892]
self.assertEqual(self._schedule(123, "random", 4), expected)
self.assertEqual(self._schedule(123, "random", 4), expected)
self.assertEqual(len(set(expected)), len(expected))
def test_hash_random_is_independent_of_global_random_state(self) -> None:
state = random.getstate()
try:
random.seed(1)
first = self._schedule(321, "random", 5)
for _ in range(100):
random.random()
random.seed(999999)
second = self._schedule(321, "random", 5)
finally:
random.setstate(state)
self.assertEqual(first, second)
def test_schedule_is_image_ordinal_independent_and_reusable(self) -> None:
expected = self._schedule(100, "increment", 3)
per_image = [self._schedule(100, "increment", 3) for _ in range(4)]
self.assertEqual(per_image, [expected, expected, expected, expected])
def test_minus_one_is_unseeded_for_every_mode(self) -> None:
for mode in ("fixed", "increment", "decrement", "random"):
with self.subTest(mode=mode):
self.assertEqual(self._schedule(-1, mode), [None, None, None])
def test_joy_qwen_and_ollama_share_the_planner_schedule(self) -> None:
plan = planner_engine.build_captionforge_pipeline_plan(
output_dir="unused",
joy_runs_per_image=3,
qwen_runs_per_image=3,
ollama_runs_per_image=3,
base_seed=77,
seed_mode="random",
)
schedules = [
[run.seed for run in planner_engine.expand_captionforge_runs(plan, model_key=family)]
for family in ("joy", "qwen", "ollama")
]
self.assertEqual(schedules[0], schedules[1])
self.assertEqual(schedules[1], schedules[2])
class OllamaPassAUnsetSeedTests(unittest.TestCase):
def test_unset_seed_is_omitted_from_ollama_options(self) -> None:
options = ollama_caption._ollama_options(
max_new_tokens=10,
temperature=0.5,
top_p=0.9,
top_k=20,
repetition_penalty=1.0,
seed=None,
)
self.assertNotIn("seed", options)
def test_planned_unset_seed_reaches_caption_call_without_int_none(self) -> None:
with tempfile.TemporaryDirectory() as temp_dir:
root = Path(temp_dir)
Image.new("RGB", (2, 2), "white").save(root / "image.png")
plan = planner_engine.build_captionforge_pipeline_plan(
output_dir=temp_dir,
input_path=temp_dir,
joy_runs_per_image=0,
qwen_runs_per_image=0,
ollama_runs_per_image=1,
base_seed=-1,
seed_mode="random",
)
required = ollama_caption.JLC_CaptionForgeOllamaCaption.INPUT_TYPES()["required"]
kwargs = {name: spec[1]["default"] for name, spec in required.items()}
kwargs["pipeline_plan"] = plan
with mock.patch.object(ollama_caption, "_ensure_ollama_model"), mock.patch.object(
ollama_caption, "_persist_caption_model_if_possible"
), mock.patch.object(
ollama_caption, "_evict_python_models_before_ollama_if_needed"
), mock.patch.object(
ollama_caption, "_ollama_generate_caption", return_value="caption"
) as generate:
ollama_caption.JLC_CaptionForgeOllamaCaption().caption(**kwargs)
self.assertIsNone(generate.call_args.kwargs["seed"])
class StageSeedContractTests(unittest.TestCase):
def _run_capstone(
self,
*,
standalone_seeds: tuple[int | None, int | None, int | None],
plan_seeds: tuple[int, int, int] | None = None,
) -> tuple[list[int | None], list[int | None]]:
with tempfile.TemporaryDirectory() as temp_dir:
root = Path(temp_dir)
for name in ("one.png", "two.png"):
Image.new("RGB", (2, 2), "white").save(root / name)
caption_path = root / "captions.jsonl"
caption_path.write_text(
"\n".join(
json.dumps(
{
"image": name,
"image_key": Path(name).stem,
"caption": f"caption for {name}",
"model_family": "joy",
"status": "ok",
}
)
for name in ("one.png", "two.png")
)
+ "\n",
encoding="utf-8",
)
kwargs = {
"Input - captions JSONL": str(caption_path),
"Input - image path": temp_dir,
"Output - folder": str(root / "output"),
"Output - run name": "seed-test",
"Output - overwrite outputs": True,
"Distiller seed": standalone_seeds[0],
"Validator seed": standalone_seeds[1],
"Formatter seed": standalone_seeds[2],
}
if plan_seeds is not None:
kwargs["pipeline_plan"] = {
"captionforge_config_type": "captionforge_pipeline_plan",
"distiller": {"seed": plan_seeds[0]},
"validator": {"seed": plan_seeds[1]},
"formatter": {"seed": plan_seeds[2]},
}
text_seeds: list[int | None] = []
image_seeds: list[int | None] = []
def fake_text(**call_kwargs):
text_seeds.append(call_kwargs["seed"])
return "generated text", {"ok": True}
def fake_image(**call_kwargs):
image_seeds.append(call_kwargs["seed"])
return "validated natural caption", {"ok": True}
with mock.patch.object(capstone, "_evict_python_models_before_ollama_if_needed"), mock.patch.object(
capstone, "_ollama_generate_text", side_effect=fake_text
), mock.patch.object(capstone, "_ollama_chat_image", side_effect=fake_image):
capstone.JLC_CaptionForge().forge(**kwargs)
return text_seeds, image_seeds
def test_standalone_exact_stage_seeds_repeat_for_every_image(self) -> None:
text_seeds, image_seeds = self._run_capstone(standalone_seeds=(200, 300, 400))
self.assertEqual(text_seeds, [200, 400, 200, 400])
self.assertEqual(image_seeds, [300, 300])
def test_planner_overrides_all_standalone_stage_seed_inputs(self) -> None:
text_seeds, image_seeds = self._run_capstone(
standalone_seeds=(1, 2, 3),
plan_seeds=(200, 300, 400),
)
self.assertEqual(text_seeds, [200, 400, 200, 400])
self.assertEqual(image_seeds, [300, 300])
def test_literal_zero_survives_all_stage_paths(self) -> None:
text_seeds, image_seeds = self._run_capstone(
standalone_seeds=(9, 9, 9),
plan_seeds=(0, 0, 0),
)
self.assertEqual(text_seeds, [0, 0, 0, 0])
self.assertEqual(image_seeds, [0, 0])
def test_omitted_standalone_seeds_are_unseeded_and_do_not_crash(self) -> None:
text_seeds, image_seeds = self._run_capstone(standalone_seeds=(None, None, None))
self.assertEqual(text_seeds, [None, None, None, None])
self.assertEqual(image_seeds, [None, None])
class SeedOwnershipSchemaTests(unittest.TestCase):
def test_planner_preserves_three_exact_stage_seeds_without_modes(self) -> None:
plan = planner_engine.build_captionforge_pipeline_plan(
output_dir="unused",
joy_runs_per_image=1,
qwen_runs_per_image=0,
distiller_seed=200,
validator_seed=300,
formatter_seed=400,
)
self.assertEqual(plan["distiller"]["seed"], 200)
self.assertEqual(plan["validator"]["seed"], 300)
self.assertEqual(plan["formatter"]["seed"], 400)
self.assertNotIn("seed_mode", plan["distiller"])
self.assertNotIn("seed_mode", plan["validator"])
self.assertNotIn("seed_mode", plan["formatter"])
def test_planner_input_schema_has_fixed_stage_seeds_only(self) -> None:
required = planner_node.JLC_CaptionForge_Pipeline_Planner.INPUT_TYPES()["required"]
self.assertIn("Distiller - seed", required)
self.assertIn("Validator - seed", required)
self.assertIn("Formatter - seed", required)
self.assertNotIn("Distiller - seed mode", required)
self.assertNotIn("Validator - seed mode", required)
self.assertNotIn("Formatter - seed mode", required)
def test_capstone_exposes_only_optional_stage_seed_inputs(self) -> None:
input_types = capstone.JLC_CaptionForge.INPUT_TYPES()
required = input_types["required"]
optional = input_types["optional"]
self.assertIn("Distiller seed", optional)
self.assertIn("Validator seed", optional)
self.assertIn("Formatter seed", optional)
for name in ("Distiller seed", "Validator seed", "Formatter seed"):
self.assertTrue(optional[name][1]["forceInput"])
for old_name in (
"Fat Draft - base seed",
"Fat Draft - seed mode",
"Validator - base seed",
"Validator - seed mode",
"Formatter - base seed",
"Formatter - seed mode",
):
self.assertNotIn(old_name, required)
self.assertNotIn(old_name, optional)
if __name__ == "__main__":
unittest.main()
+7 -6
View File
@@ -3,11 +3,12 @@ import { app } from "/scripts/app.js";
const ICON_SIZE = 12;
const CAPTIONFORGE_NODE_NAMES = new Set([
"JLC_QwenCaption",
"JLC_JoyCaption",
"JLC_QwenCaptionLite",
"JLC_JoyCaptionLite",
"JLC_CaptionForgeClaimExtractor",
"JLC_CaptionForge_Pipeline_Planner",
"JLC_CaptionForgeExtraOptions",
"JLC_CaptionForgeJoy",
"JLC_CaptionForgeQwen",
"JLC_CaptionForgeOllamaCaption",
"JLC_CaptionForge",
]);
const iconImage = new Image();
@@ -69,4 +70,4 @@ app.registerExtension({
}
};
},
});
});