feat: add Ming-Image architecture & Prompt Enhancer and update upstream deps
Browse files- README.md +4 -0
- chain_injectors/ming_image_prompt_enhancer_injector.py +262 -0
- chain_injectors/qwen_image_2_1_prompt_enhancer_injector.py +1 -1
- core/pipelines/pipeline_input_processor.py +15 -1
- core/pipelines/sd_image_pipeline.py +2 -0
- core/pipelines/workflow_executor.py +4 -0
- core/pipelines/workflow_recipes/_partials/conditioning/ming-image.yaml +108 -0
- mcp_tools/common.py +29 -2
- requirements.txt +1 -1
- ui/events/change_handlers.py +4 -2
- ui/events/main.py +7 -2
- ui/events/run_handlers.py +9 -0
- ui/imagegen_ui.py +2 -1
- ui/shared/ui_components.py +54 -2
- yaml/chain_features.yaml +27 -2
- yaml/constants.yaml +8 -0
- yaml/file_list.yaml +21 -1
- yaml/model_architecture_features.yaml +6 -0
- yaml/model_architectures.yaml +3 -0
- yaml/model_defaults.yaml +7 -0
- yaml/model_list.yaml +8 -0
- yaml/task_features.yaml +5 -0
README.md
CHANGED
|
@@ -52,10 +52,12 @@ models:
|
|
| 52 |
- Comfy-Org/LongCat-Image
|
| 53 |
- Comfy-Org/Lumina_Image_2.0_Repackaged
|
| 54 |
- Comfy-Org/Mage-Flow
|
|
|
|
| 55 |
- Comfy-Org/NewBie-image-Exp0.1_repackaged
|
| 56 |
- Comfy-Org/Omnigen2_ComfyUI_repackaged
|
| 57 |
- Comfy-Org/Ovis-Image
|
| 58 |
- Comfy-Org/PixelDiT
|
|
|
|
| 59 |
- Comfy-Org/Qwen-Image_ComfyUI
|
| 60 |
- Comfy-Org/Qwen-Image-2.1
|
| 61 |
- Comfy-Org/Qwen-Image-Edit_ComfyUI
|
|
@@ -151,6 +153,7 @@ models:
|
|
| 151 |
- HiDream-ai/HiDream-O1-Image
|
| 152 |
- HiDream-ai/HiDream-O1-Image-Dev
|
| 153 |
- ideogram-ai/ideogram-4-fp8
|
|
|
|
| 154 |
- kohya-ss/Anima-LLLite
|
| 155 |
- krea/Krea-2-Raw
|
| 156 |
- krea/Krea-2-Turbo
|
|
@@ -169,6 +172,7 @@ models:
|
|
| 169 |
- nvidia/PiD
|
| 170 |
- nvidia/PixelDiT-1300M-1024px
|
| 171 |
- OmniGen2/OmniGen2
|
|
|
|
| 172 |
- Qwen/Qwen-Image
|
| 173 |
- Qwen/Qwen-Image-2512
|
| 174 |
- Qwen/Qwen-Image-2.1
|
|
|
|
| 52 |
- Comfy-Org/LongCat-Image
|
| 53 |
- Comfy-Org/Lumina_Image_2.0_Repackaged
|
| 54 |
- Comfy-Org/Mage-Flow
|
| 55 |
+
- Comfy-Org/Ming-Image
|
| 56 |
- Comfy-Org/NewBie-image-Exp0.1_repackaged
|
| 57 |
- Comfy-Org/Omnigen2_ComfyUI_repackaged
|
| 58 |
- Comfy-Org/Ovis-Image
|
| 59 |
- Comfy-Org/PixelDiT
|
| 60 |
+
- Comfy-Org/Qwen3.8-27B
|
| 61 |
- Comfy-Org/Qwen-Image_ComfyUI
|
| 62 |
- Comfy-Org/Qwen-Image-2.1
|
| 63 |
- Comfy-Org/Qwen-Image-Edit_ComfyUI
|
|
|
|
| 153 |
- HiDream-ai/HiDream-O1-Image
|
| 154 |
- HiDream-ai/HiDream-O1-Image-Dev
|
| 155 |
- ideogram-ai/ideogram-4-fp8
|
| 156 |
+
- inclusionAI/Ming-Image-0.1-Design
|
| 157 |
- kohya-ss/Anima-LLLite
|
| 158 |
- krea/Krea-2-Raw
|
| 159 |
- krea/Krea-2-Turbo
|
|
|
|
| 172 |
- nvidia/PiD
|
| 173 |
- nvidia/PixelDiT-1300M-1024px
|
| 174 |
- OmniGen2/OmniGen2
|
| 175 |
+
- Qwen/Qwen3.8-27B
|
| 176 |
- Qwen/Qwen-Image
|
| 177 |
- Qwen/Qwen-Image-2512
|
| 178 |
- Qwen/Qwen-Image-2.1
|
chain_injectors/ming_image_prompt_enhancer_injector.py
ADDED
|
@@ -0,0 +1,262 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Ming-Image Prompt Enhancer Chain Injector.
|
| 2 |
+
|
| 3 |
+
This injector automatically integrates the Qwen3.8-27B prompt enhancement model into
|
| 4 |
+
the Ming-Image generation workflow:
|
| 5 |
+
- Detects whether the task is pure text-to-image (T2I) or involves images (I2I, inpainting,
|
| 6 |
+
outpainting, hires.fix, or multi-reference images).
|
| 7 |
+
- Uses the Qwen3.8-27B PE text encoder: qwen3.8_27b_w4a8.safetensors (type: qwen_image).
|
| 8 |
+
- Handles single or multi-reference inputs: multiple reference images are stitched into
|
| 9 |
+
a structured grid layout via ComfyUI native ImageStitch nodes and scaled via ImageScaleToTotalPixels.
|
| 10 |
+
- Supports reasoning mode (thinking=True) leveraging ComfyUI TextGenerate native reasoning separation
|
| 11 |
+
(output slot 0 outputs clean prompt text, while output slot 1 isolates reasoning chain).
|
| 12 |
+
"""
|
| 13 |
+
|
| 14 |
+
try:
|
| 15 |
+
from utils.app_utils import ensure_file_downloaded
|
| 16 |
+
except ImportError:
|
| 17 |
+
ensure_file_downloaded = None
|
| 18 |
+
|
| 19 |
+
|
| 20 |
+
DEFAULT_MING_IMAGE_SYSTEM_PROMPT = (
|
| 21 |
+
"You are a senior visual designer and image-prompt engineer. Expand the user's request into one precise, high-resolution Figma-style caption. Return only one JSON object. "
|
| 22 |
+
"Use exactly two top-level keys. `canvas_settings` contains exactly `aspect_ratio`, `ambient_lighting`, and `image_style`. `layers` lists visible groups from background to topmost overlay. Every layer contains exactly `description`, `coordinates`, `hierarchy_and_relation`, and `color_specs`; `color_specs` is an array of hex colors. "
|
| 23 |
+
'`coordinates` MUST be one string, never an object or array, in exactly this form: `"cx: 0.500, cy: 0.500, w: 1.000, h: 1.000"`. Values are normalized; each bbox encloses its complete owned object and stays inside the canvas. '
|
| 24 |
+
"A layer is one selectable visible semantic group: background, full person, coherent object, panel, card, row, or text block. Prefer the fewest groups that preserve the layout. Keep people and objects intact. Never create invisible parents, guides, placeholders, empty layers, duplicate summaries, or multiple owners for one element. "
|
| 25 |
+
"Preserve every user-supplied rendered string character-for-character and as one contiguous string. Unless multiple visible copies are requested, it must occur exactly once across all `description` fields and zero times in `hierarchy_and_relation`. Quote it only where describing its visible rendering; refer to the related subject elsewhere with unquoted semantic wording. Enumerate intended copy, invent extra copy sparingly, and never hide content behind \"other text\", \"remaining labels\", or \"etc.\" "
|
| 26 |
+
"Describe concrete composition, typography, materials, texture, lighting, pose, and camera treatment without literary filler. Use `hierarchy_and_relation` only for ownership, alignment, containment, stacking, and occlusion. "
|
| 27 |
+
"Infer structured layouts first. Use one complete layer per card and state its row and column. A compact secondary table may be one layer only if every header and cell is listed; otherwise use a visible shared frame when present, one complete header, and one complete layer per body row, binding values to columns and stating blanks. Enumerate sequences, schedules, spans, gaps, and vacant tracks in visual order. Do not mistake ordinary alignment for a table. "
|
| 28 |
+
"Silently verify schema, string coordinates, Z-order, exact-text counts, geometry, bbox validity, and completeness."
|
| 29 |
+
)
|
| 30 |
+
|
| 31 |
+
|
| 32 |
+
def create_node(assembler, class_type, title):
|
| 33 |
+
"""Create a workflow node dictionary with fallback template generation."""
|
| 34 |
+
try:
|
| 35 |
+
node = assembler._get_node_template(class_type)
|
| 36 |
+
except Exception:
|
| 37 |
+
node = {
|
| 38 |
+
"inputs": {},
|
| 39 |
+
"class_type": class_type,
|
| 40 |
+
"_meta": {"title": title}
|
| 41 |
+
}
|
| 42 |
+
if "_meta" not in node:
|
| 43 |
+
node["_meta"] = {}
|
| 44 |
+
node['_meta']['title'] = title
|
| 45 |
+
return node
|
| 46 |
+
|
| 47 |
+
|
| 48 |
+
def inject(assembler, chain_definition, chain_items):
|
| 49 |
+
"""Inject Ming-Image prompt enhancer nodes into the target generation workflow."""
|
| 50 |
+
if not chain_items:
|
| 51 |
+
return
|
| 52 |
+
|
| 53 |
+
is_enabled = False
|
| 54 |
+
thinking_enabled = False
|
| 55 |
+
max_length = 4096
|
| 56 |
+
system_prompt = DEFAULT_MING_IMAGE_SYSTEM_PROMPT
|
| 57 |
+
for item in chain_items:
|
| 58 |
+
if isinstance(item, dict):
|
| 59 |
+
if item.get("enable", False) or item.get("enabled", False) or item.get("value", False):
|
| 60 |
+
is_enabled = True
|
| 61 |
+
if item.get("thinking", False) in (True, "true", "on", "yes", "1", 1):
|
| 62 |
+
thinking_enabled = True
|
| 63 |
+
if "max_length" in item and item["max_length"]:
|
| 64 |
+
try:
|
| 65 |
+
max_length = int(item["max_length"])
|
| 66 |
+
except (ValueError, TypeError):
|
| 67 |
+
pass
|
| 68 |
+
if "system_prompt" in item and item["system_prompt"]:
|
| 69 |
+
system_prompt = str(item["system_prompt"])
|
| 70 |
+
elif item is True or str(item).lower() in ("on", "true", "yes", "1"):
|
| 71 |
+
is_enabled = True
|
| 72 |
+
break
|
| 73 |
+
|
| 74 |
+
if not is_enabled:
|
| 75 |
+
return
|
| 76 |
+
|
| 77 |
+
target_node_name = chain_definition.get('target_node', 'prompt')
|
| 78 |
+
target_node_id = assembler.node_map.get(target_node_name)
|
| 79 |
+
|
| 80 |
+
if not target_node_id or target_node_id not in assembler.workflow:
|
| 81 |
+
for node_id, node in assembler.workflow.items():
|
| 82 |
+
if isinstance(node, dict) and node.get('class_type') == 'TextEncodeMingImageEdit':
|
| 83 |
+
target_node_id = node_id
|
| 84 |
+
break
|
| 85 |
+
|
| 86 |
+
if not target_node_id or target_node_id not in assembler.workflow:
|
| 87 |
+
print(f"Warning: Target node '{target_node_name}' (TextEncodeMingImageEdit) not found for Ming-Image Prompt Enhancer. Skipping.")
|
| 88 |
+
return
|
| 89 |
+
|
| 90 |
+
target_node = assembler.workflow[target_node_id]
|
| 91 |
+
original_prompt = target_node.get('inputs', {}).get('prompt', '')
|
| 92 |
+
|
| 93 |
+
# 1. Check if any LoadImage nodes exist in the workflow
|
| 94 |
+
load_image_nodes = [
|
| 95 |
+
node_id for node_id, node in assembler.workflow.items()
|
| 96 |
+
if isinstance(node, dict) and node.get('class_type') == 'LoadImage'
|
| 97 |
+
]
|
| 98 |
+
has_load_image = len(load_image_nodes) > 0
|
| 99 |
+
|
| 100 |
+
target_image_conn = None
|
| 101 |
+
clip_model_name = "qwen3.8_27b_w4a8.safetensors"
|
| 102 |
+
|
| 103 |
+
if has_load_image:
|
| 104 |
+
# Collect image connections
|
| 105 |
+
# Prioritize extracting all images.image_* slots from target_node['inputs'] (injected by reference image injectors)
|
| 106 |
+
indexed_slots = []
|
| 107 |
+
target_inputs = target_node.get('inputs', {})
|
| 108 |
+
for slot_key, slot_val in target_inputs.items():
|
| 109 |
+
if slot_key.startswith('images.image_') and isinstance(slot_val, (list, tuple)):
|
| 110 |
+
try:
|
| 111 |
+
idx = int(slot_key.split('_')[-1])
|
| 112 |
+
except Exception:
|
| 113 |
+
idx = 999
|
| 114 |
+
indexed_slots.append((idx, slot_val))
|
| 115 |
+
|
| 116 |
+
indexed_slots.sort(key=lambda x: x[0])
|
| 117 |
+
ref_image_conns = [s[1] for s in indexed_slots]
|
| 118 |
+
|
| 119 |
+
# If no reference image slots exist (e.g. pure img2img / inpaint / outpaint), retrieve from LoadImage nodes
|
| 120 |
+
if not ref_image_conns:
|
| 121 |
+
for lid in load_image_nodes:
|
| 122 |
+
scaled_conn = None
|
| 123 |
+
for nid, node in assembler.workflow.items():
|
| 124 |
+
if isinstance(node, dict) and node.get('class_type') == 'ImageScaleToTotalPixels':
|
| 125 |
+
img_input = node.get('inputs', {}).get('image')
|
| 126 |
+
if isinstance(img_input, (list, tuple)) and str(img_input[0]) == str(lid):
|
| 127 |
+
scaled_conn = [nid, 0]
|
| 128 |
+
break
|
| 129 |
+
if scaled_conn:
|
| 130 |
+
ref_image_conns.append(scaled_conn)
|
| 131 |
+
else:
|
| 132 |
+
scale_id = assembler._get_unique_id()
|
| 133 |
+
scale_node = create_node(assembler, "ImageScaleToTotalPixels", "Scale Input Image")
|
| 134 |
+
scale_node['inputs'] = {
|
| 135 |
+
"upscale_method": "nearest-exact",
|
| 136 |
+
"megapixels": 1,
|
| 137 |
+
"resolution_steps": 1,
|
| 138 |
+
"image": [lid, 0]
|
| 139 |
+
}
|
| 140 |
+
assembler.workflow[scale_id] = scale_node
|
| 141 |
+
ref_image_conns.append([scale_id, 0])
|
| 142 |
+
|
| 143 |
+
# Organize multiple reference images into grid stitching
|
| 144 |
+
if len(ref_image_conns) == 1:
|
| 145 |
+
target_image_conn = ref_image_conns[0]
|
| 146 |
+
elif len(ref_image_conns) >= 2:
|
| 147 |
+
num_cols = 2 if len(ref_image_conns) <= 4 else 3
|
| 148 |
+
|
| 149 |
+
rows = []
|
| 150 |
+
for i in range(0, len(ref_image_conns), num_cols):
|
| 151 |
+
rows.append(ref_image_conns[i:i + num_cols])
|
| 152 |
+
|
| 153 |
+
# Horizontally stitch images in each row
|
| 154 |
+
stitched_rows = []
|
| 155 |
+
for r_idx, row_imgs in enumerate(rows):
|
| 156 |
+
if len(row_imgs) == 1:
|
| 157 |
+
stitched_rows.append(row_imgs[0])
|
| 158 |
+
else:
|
| 159 |
+
curr_stitch = row_imgs[0]
|
| 160 |
+
for c_idx in range(1, len(row_imgs)):
|
| 161 |
+
st_id = assembler._get_unique_id()
|
| 162 |
+
st_node = create_node(assembler, "ImageStitch", f"Stitch Images (Row {r_idx+1}-{c_idx})")
|
| 163 |
+
st_node['inputs'] = {
|
| 164 |
+
"direction": "right",
|
| 165 |
+
"match_image_size": True,
|
| 166 |
+
"spacing_width": 0,
|
| 167 |
+
"spacing_color": "white",
|
| 168 |
+
"image1": curr_stitch,
|
| 169 |
+
"image2": row_imgs[c_idx]
|
| 170 |
+
}
|
| 171 |
+
assembler.workflow[st_id] = st_node
|
| 172 |
+
curr_stitch = [st_id, 0]
|
| 173 |
+
stitched_rows.append(curr_stitch)
|
| 174 |
+
|
| 175 |
+
# Vertically stitch multiple rows
|
| 176 |
+
if len(stitched_rows) == 1:
|
| 177 |
+
final_stitched = stitched_rows[0]
|
| 178 |
+
else:
|
| 179 |
+
curr_vertical = stitched_rows[0]
|
| 180 |
+
for r_idx in range(1, len(stitched_rows)):
|
| 181 |
+
st_id = assembler._get_unique_id()
|
| 182 |
+
st_node = create_node(assembler, "ImageStitch", f"Stitch Images (Vertical {r_idx})")
|
| 183 |
+
st_node['inputs'] = {
|
| 184 |
+
"direction": "down",
|
| 185 |
+
"match_image_size": True,
|
| 186 |
+
"spacing_width": 0,
|
| 187 |
+
"spacing_color": "white",
|
| 188 |
+
"image1": curr_vertical,
|
| 189 |
+
"image2": stitched_rows[r_idx]
|
| 190 |
+
}
|
| 191 |
+
assembler.workflow[st_id] = st_node
|
| 192 |
+
curr_vertical = [st_id, 0]
|
| 193 |
+
final_stitched = curr_vertical
|
| 194 |
+
|
| 195 |
+
# Scale final stitched image to ~1024 level via ImageScaleToTotalPixels
|
| 196 |
+
final_scale_id = assembler._get_unique_id()
|
| 197 |
+
final_scale_node = create_node(assembler, "ImageScaleToTotalPixels", "Scale Stitched Image")
|
| 198 |
+
final_scale_node['inputs'] = {
|
| 199 |
+
"upscale_method": "nearest-exact",
|
| 200 |
+
"megapixels": 1,
|
| 201 |
+
"resolution_steps": 1,
|
| 202 |
+
"image": final_stitched
|
| 203 |
+
}
|
| 204 |
+
assembler.workflow[final_scale_id] = final_scale_node
|
| 205 |
+
target_image_conn = [final_scale_id, 0]
|
| 206 |
+
|
| 207 |
+
# 2. Ensure model is downloaded in CPU stage before workflow execution
|
| 208 |
+
if ensure_file_downloaded:
|
| 209 |
+
try:
|
| 210 |
+
ensure_file_downloaded(clip_model_name)
|
| 211 |
+
except Exception as e:
|
| 212 |
+
print(f"Warning: Failed to ensure '{clip_model_name}' downloaded: {e}")
|
| 213 |
+
|
| 214 |
+
# 3. Create CLIPLoader node (using qwen_image type for Qwen3.8-27B PE)
|
| 215 |
+
clip_node_id = assembler._get_unique_id()
|
| 216 |
+
clip_node = create_node(assembler, "CLIPLoader", "Load CLIP")
|
| 217 |
+
clip_node['inputs'] = {
|
| 218 |
+
"clip_name": clip_model_name,
|
| 219 |
+
"type": "qwen_image",
|
| 220 |
+
"device": "default"
|
| 221 |
+
}
|
| 222 |
+
assembler.workflow[clip_node_id] = clip_node
|
| 223 |
+
|
| 224 |
+
# 4. Create PrimitiveString node for System Prompt (Text node)
|
| 225 |
+
system_prompt_node_id = assembler._get_unique_id()
|
| 226 |
+
system_prompt_node = create_node(assembler, "PrimitiveString", "Text")
|
| 227 |
+
system_prompt_node['inputs'] = {
|
| 228 |
+
"value": system_prompt
|
| 229 |
+
}
|
| 230 |
+
assembler.workflow[system_prompt_node_id] = system_prompt_node
|
| 231 |
+
|
| 232 |
+
# 5. Create TextGenerate node
|
| 233 |
+
text_gen_node_id = assembler._get_unique_id()
|
| 234 |
+
text_gen_node = create_node(assembler, "TextGenerate", "Generate Text")
|
| 235 |
+
text_gen_inputs = {
|
| 236 |
+
"prompt": original_prompt,
|
| 237 |
+
"max_length": max_length,
|
| 238 |
+
"sampling_mode": "on",
|
| 239 |
+
"sampling_mode.temperature": 0.7,
|
| 240 |
+
"sampling_mode.top_k": 64,
|
| 241 |
+
"sampling_mode.top_p": 0.95,
|
| 242 |
+
"sampling_mode.min_p": 0.05,
|
| 243 |
+
"sampling_mode.repetition_penalty": 1.05,
|
| 244 |
+
"sampling_mode.seed": 0,
|
| 245 |
+
"sampling_mode.presence_penalty": 0,
|
| 246 |
+
"thinking": thinking_enabled,
|
| 247 |
+
"use_default_template": True,
|
| 248 |
+
"mtp": "auto",
|
| 249 |
+
"clip": [clip_node_id, 0],
|
| 250 |
+
"system_prompt": [system_prompt_node_id, 0]
|
| 251 |
+
}
|
| 252 |
+
if target_image_conn is not None:
|
| 253 |
+
text_gen_inputs["image"] = target_image_conn
|
| 254 |
+
|
| 255 |
+
text_gen_node['inputs'] = text_gen_inputs
|
| 256 |
+
assembler.workflow[text_gen_node_id] = text_gen_node
|
| 257 |
+
|
| 258 |
+
# 6. Connect TextGenerate output directly to target node prompt
|
| 259 |
+
# ComfyUI TextGenerate node natively strips reasoning tags (<think>...</think>) into output slot 1,
|
| 260 |
+
# leaving clean enhanced prompt in output slot 0.
|
| 261 |
+
target_node['inputs']['prompt'] = [text_gen_node_id, 0]
|
| 262 |
+
print(f"[Injector] Ming-Image Prompt Enhancer applied successfully ({'I2I' if has_load_image else 'T2I'}, clip='{clip_model_name}', thinking={thinking_enabled}).")
|
chain_injectors/qwen_image_2_1_prompt_enhancer_injector.py
CHANGED
|
@@ -42,7 +42,7 @@ def inject(assembler, chain_definition, chain_items):
|
|
| 42 |
|
| 43 |
is_enabled = False
|
| 44 |
thinking_enabled = False
|
| 45 |
-
max_length =
|
| 46 |
for item in chain_items:
|
| 47 |
if isinstance(item, dict):
|
| 48 |
if item.get("enable", False) or item.get("enabled", False) or item.get("value", False):
|
|
|
|
| 42 |
|
| 43 |
is_enabled = False
|
| 44 |
thinking_enabled = False
|
| 45 |
+
max_length = 4096
|
| 46 |
for item in chain_items:
|
| 47 |
if isinstance(item, dict):
|
| 48 |
if item.get("enable", False) or item.get("enabled", False) or item.get("value", False):
|
core/pipelines/pipeline_input_processor.py
CHANGED
|
@@ -463,9 +463,22 @@ def process_pipeline_inputs(ui_inputs: Dict[str, Any], progress: gr.Progress, wo
|
|
| 463 |
active_qwen_image_2_1_prompt_enhancer.append({
|
| 464 |
"enable": True,
|
| 465 |
"thinking": bool(ui_inputs.get('qwen_image_2_1_prompt_enhancer_thinking', False)),
|
| 466 |
-
"max_length": int(ui_inputs.get('qwen_image_2_1_prompt_enhancer_max_length',
|
| 467 |
})
|
| 468 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 469 |
return {
|
| 470 |
"active_loras_for_gpu": active_loras_for_gpu,
|
| 471 |
"active_loras_for_meta": active_loras_for_meta,
|
|
@@ -487,5 +500,6 @@ def process_pipeline_inputs(ui_inputs: Dict[str, Any], progress: gr.Progress, wo
|
|
| 487 |
"active_reference_images": active_reference_images,
|
| 488 |
"active_conditioning": active_conditioning,
|
| 489 |
"active_qwen_image_2_1_prompt_enhancer": active_qwen_image_2_1_prompt_enhancer,
|
|
|
|
| 490 |
"temp_files_to_clean": temp_files_to_clean
|
| 491 |
}
|
|
|
|
| 463 |
active_qwen_image_2_1_prompt_enhancer.append({
|
| 464 |
"enable": True,
|
| 465 |
"thinking": bool(ui_inputs.get('qwen_image_2_1_prompt_enhancer_thinking', False)),
|
| 466 |
+
"max_length": int(ui_inputs.get('qwen_image_2_1_prompt_enhancer_max_length', 4096))
|
| 467 |
})
|
| 468 |
|
| 469 |
+
ming_pe_enable = ui_inputs.get('ming_image_prompt_enhancer_enable', False)
|
| 470 |
+
active_ming_image_prompt_enhancer = []
|
| 471 |
+
if 'ming_image_prompt_enhancer' in enabled_chains and ming_pe_enable:
|
| 472 |
+
item = {
|
| 473 |
+
"enable": True,
|
| 474 |
+
"thinking": bool(ui_inputs.get('ming_image_prompt_enhancer_thinking', False)),
|
| 475 |
+
"max_length": int(ui_inputs.get('ming_image_prompt_enhancer_max_length', 4096))
|
| 476 |
+
}
|
| 477 |
+
sys_p = ui_inputs.get('ming_image_prompt_enhancer_system_prompt')
|
| 478 |
+
if sys_p:
|
| 479 |
+
item["system_prompt"] = str(sys_p)
|
| 480 |
+
active_ming_image_prompt_enhancer.append(item)
|
| 481 |
+
|
| 482 |
return {
|
| 483 |
"active_loras_for_gpu": active_loras_for_gpu,
|
| 484 |
"active_loras_for_meta": active_loras_for_meta,
|
|
|
|
| 500 |
"active_reference_images": active_reference_images,
|
| 501 |
"active_conditioning": active_conditioning,
|
| 502 |
"active_qwen_image_2_1_prompt_enhancer": active_qwen_image_2_1_prompt_enhancer,
|
| 503 |
+
"active_ming_image_prompt_enhancer": active_ming_image_prompt_enhancer,
|
| 504 |
"temp_files_to_clean": temp_files_to_clean
|
| 505 |
}
|
core/pipelines/sd_image_pipeline.py
CHANGED
|
@@ -123,6 +123,7 @@ class SdImagePipeline(BasePipeline):
|
|
| 123 |
active_reference_images = processed.get("active_reference_images", [])
|
| 124 |
active_conditioning = processed["active_conditioning"]
|
| 125 |
active_qwen_image_2_1_prompt_enhancer = processed.get("active_qwen_image_2_1_prompt_enhancer", [])
|
|
|
|
| 126 |
|
| 127 |
loras_string = f"LoRAs: [{', '.join(active_loras_for_meta)}]" if active_loras_for_meta else ""
|
| 128 |
|
|
@@ -184,6 +185,7 @@ class SdImagePipeline(BasePipeline):
|
|
| 184 |
"qwen_image_edit_chain": active_qwen_image_edit,
|
| 185 |
"reference_image_chain": active_reference_images,
|
| 186 |
"qwen_image_2_1_prompt_enhancer_chain": active_qwen_image_2_1_prompt_enhancer,
|
|
|
|
| 187 |
"vae_chain": [ui_inputs.get('vae_name')] if (ui_inputs.get('vae_name') and 'vae' in enabled_chains) else [],
|
| 188 |
"hidream_o1_smoothing_chain": hidream_o1_smoothing_data,
|
| 189 |
"pid_chain": [ui_inputs.get('pid_settings', 'OFF')] if is_pid_enabled else [],
|
|
|
|
| 123 |
active_reference_images = processed.get("active_reference_images", [])
|
| 124 |
active_conditioning = processed["active_conditioning"]
|
| 125 |
active_qwen_image_2_1_prompt_enhancer = processed.get("active_qwen_image_2_1_prompt_enhancer", [])
|
| 126 |
+
active_ming_image_prompt_enhancer = processed.get("active_ming_image_prompt_enhancer", [])
|
| 127 |
|
| 128 |
loras_string = f"LoRAs: [{', '.join(active_loras_for_meta)}]" if active_loras_for_meta else ""
|
| 129 |
|
|
|
|
| 185 |
"qwen_image_edit_chain": active_qwen_image_edit,
|
| 186 |
"reference_image_chain": active_reference_images,
|
| 187 |
"qwen_image_2_1_prompt_enhancer_chain": active_qwen_image_2_1_prompt_enhancer,
|
| 188 |
+
"ming_image_prompt_enhancer_chain": active_ming_image_prompt_enhancer,
|
| 189 |
"vae_chain": [ui_inputs.get('vae_name')] if (ui_inputs.get('vae_name') and 'vae' in enabled_chains) else [],
|
| 190 |
"hidream_o1_smoothing_chain": hidream_o1_smoothing_data,
|
| 191 |
"pid_chain": [ui_inputs.get('pid_settings', 'OFF')] if is_pid_enabled else [],
|
core/pipelines/workflow_executor.py
CHANGED
|
@@ -113,6 +113,10 @@ class WorkflowExecutor:
|
|
| 113 |
else:
|
| 114 |
kwargs[param_name] = actual_value
|
| 115 |
|
|
|
|
|
|
|
|
|
|
|
|
|
| 116 |
function_name = getattr(node_class, 'FUNCTION')
|
| 117 |
execution_method = getattr(node_instance, function_name)
|
| 118 |
|
|
|
|
| 113 |
else:
|
| 114 |
kwargs[param_name] = actual_value
|
| 115 |
|
| 116 |
+
# Safety guard for nodes expecting dynamic autogrow dictionary (e.g. TextEncodeMingImageEdit)
|
| 117 |
+
if class_type == 'TextEncodeMingImageEdit' and ('images' not in kwargs or kwargs.get('images') is None):
|
| 118 |
+
kwargs['images'] = {}
|
| 119 |
+
|
| 120 |
function_name = getattr(node_class, 'FUNCTION')
|
| 121 |
execution_method = getattr(node_instance, function_name)
|
| 122 |
|
core/pipelines/workflow_recipes/_partials/conditioning/ming-image.yaml
ADDED
|
@@ -0,0 +1,108 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
nodes:
|
| 2 |
+
prompt:
|
| 3 |
+
class_type: TextEncodeMingImageEdit
|
| 4 |
+
title: "Text Encode Ming Image Edit"
|
| 5 |
+
params:
|
| 6 |
+
images: {}
|
| 7 |
+
unet_loader:
|
| 8 |
+
class_type: UNETLoader
|
| 9 |
+
title: "Load Diffusion Model"
|
| 10 |
+
params:
|
| 11 |
+
weight_dtype: "default"
|
| 12 |
+
vae_loader:
|
| 13 |
+
class_type: VAELoader
|
| 14 |
+
title: "Load VAE"
|
| 15 |
+
clip_loader:
|
| 16 |
+
class_type: CLIPLoader
|
| 17 |
+
title: "Load CLIP"
|
| 18 |
+
params:
|
| 19 |
+
type: "qwen_image"
|
| 20 |
+
device: "default"
|
| 21 |
+
guider:
|
| 22 |
+
class_type: BasicGuider
|
| 23 |
+
title: "Basic Guider"
|
| 24 |
+
ksampler_select:
|
| 25 |
+
class_type: KSamplerSelect
|
| 26 |
+
title: "KSamplerSelect"
|
| 27 |
+
scheduler_node:
|
| 28 |
+
class_type: BasicScheduler
|
| 29 |
+
title: "BasicScheduler"
|
| 30 |
+
params:
|
| 31 |
+
scheduler: "simple"
|
| 32 |
+
steps: 12
|
| 33 |
+
denoise: 1.0
|
| 34 |
+
random_noise:
|
| 35 |
+
class_type: RandomNoise
|
| 36 |
+
title: "RandomNoise"
|
| 37 |
+
ksampler:
|
| 38 |
+
class_type: SamplerCustomAdvanced
|
| 39 |
+
title: "SamplerCustomAdvanced"
|
| 40 |
+
|
| 41 |
+
connections:
|
| 42 |
+
- from: "unet_loader:0"
|
| 43 |
+
to: "guider:model"
|
| 44 |
+
- from: "unet_loader:0"
|
| 45 |
+
to: "scheduler_node:model"
|
| 46 |
+
- from: "clip_loader:0"
|
| 47 |
+
to: "prompt:clip"
|
| 48 |
+
- from: "prompt:0"
|
| 49 |
+
to: "guider:conditioning"
|
| 50 |
+
- from: "random_noise:0"
|
| 51 |
+
to: "ksampler:noise"
|
| 52 |
+
- from: "guider:0"
|
| 53 |
+
to: "ksampler:guider"
|
| 54 |
+
- from: "ksampler_select:0"
|
| 55 |
+
to: "ksampler:sampler"
|
| 56 |
+
- from: "scheduler_node:0"
|
| 57 |
+
to: "ksampler:sigmas"
|
| 58 |
+
- from: "vae_loader:0"
|
| 59 |
+
to: "vae_decode:vae"
|
| 60 |
+
- from: "vae_loader:0"
|
| 61 |
+
to: "vae_encode:vae"
|
| 62 |
+
|
| 63 |
+
dynamic_lora_chains:
|
| 64 |
+
lora_chain:
|
| 65 |
+
template: "LoraLoader"
|
| 66 |
+
output_map:
|
| 67 |
+
"unet_loader:0": "model"
|
| 68 |
+
"clip_loader:0": "clip"
|
| 69 |
+
input_map:
|
| 70 |
+
"model": "model"
|
| 71 |
+
"clip": "clip"
|
| 72 |
+
end_input_map:
|
| 73 |
+
"model": ["guider:model", "scheduler_node:model"]
|
| 74 |
+
"clip": ["prompt:clip"]
|
| 75 |
+
|
| 76 |
+
dynamic_conditioning_chains:
|
| 77 |
+
conditioning_chain:
|
| 78 |
+
guider_node: "guider"
|
| 79 |
+
guider_target_inputs: ["conditioning"]
|
| 80 |
+
clip_source: "clip_loader:0"
|
| 81 |
+
|
| 82 |
+
dynamic_pid_chains:
|
| 83 |
+
pid_chain:
|
| 84 |
+
ksampler_node: "ksampler"
|
| 85 |
+
|
| 86 |
+
dynamic_reference_image_chains:
|
| 87 |
+
reference_image_chain:
|
| 88 |
+
text_encode_node: "prompt"
|
| 89 |
+
vae_node: "vae_loader"
|
| 90 |
+
max_images: 8
|
| 91 |
+
|
| 92 |
+
dynamic_ming_image_prompt_enhancer_chains:
|
| 93 |
+
ming_image_prompt_enhancer_chain:
|
| 94 |
+
target_node: "prompt"
|
| 95 |
+
|
| 96 |
+
|
| 97 |
+
ui_map:
|
| 98 |
+
positive_prompt: "prompt:prompt"
|
| 99 |
+
negative_prompt: "dummy:negative_prompt"
|
| 100 |
+
unet_name: "unet_loader:unet_name"
|
| 101 |
+
clip_name: "clip_loader:clip_name"
|
| 102 |
+
vae_name: "vae_loader:vae_name"
|
| 103 |
+
seed: "random_noise:noise_seed"
|
| 104 |
+
steps: "scheduler_node:steps"
|
| 105 |
+
cfg: "dummy:cfg"
|
| 106 |
+
sampler_name: "ksampler_select:sampler_name"
|
| 107 |
+
scheduler: "scheduler_node:scheduler"
|
| 108 |
+
denoise: "scheduler_node:denoise"
|
mcp_tools/common.py
CHANGED
|
@@ -679,7 +679,19 @@ def _execute_imagegen_pipeline(task_id: str, params: dict):
|
|
| 679 |
if isinstance(is_thinking, str):
|
| 680 |
is_thinking = is_thinking.upper() in ("ON", "TRUE", "1")
|
| 681 |
ui_inputs["qwen_image_2_1_prompt_enhancer_thinking"] = bool(is_thinking)
|
| 682 |
-
ui_inputs["qwen_image_2_1_prompt_enhancer_max_length"] = int(item.get("max_length",
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 683 |
elif itype == "krea2_identity_edit":
|
| 684 |
img = _parse_image_param(item.get("image"))
|
| 685 |
if img:
|
|
@@ -790,12 +802,27 @@ def _execute_imagegen_pipeline(task_id: str, params: dict):
|
|
| 790 |
elif isinstance(pe_val, dict):
|
| 791 |
ui_inputs["qwen_image_2_1_prompt_enhancer_enable"] = bool(pe_val.get("enable", pe_val.get("enabled", True)))
|
| 792 |
ui_inputs["qwen_image_2_1_prompt_enhancer_thinking"] = bool(pe_val.get("thinking", False))
|
| 793 |
-
ui_inputs["qwen_image_2_1_prompt_enhancer_max_length"] = int(pe_val.get("max_length",
|
| 794 |
elif str(pe_val).upper() in ("ON", "TRUE", "1"):
|
| 795 |
ui_inputs["qwen_image_2_1_prompt_enhancer_enable"] = True
|
| 796 |
else:
|
| 797 |
ui_inputs["qwen_image_2_1_prompt_enhancer_enable"] = False
|
| 798 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 799 |
_TASKS_DB[task_id]["progress"] = 50
|
| 800 |
|
| 801 |
# Execute Pipeline
|
|
|
|
| 679 |
if isinstance(is_thinking, str):
|
| 680 |
is_thinking = is_thinking.upper() in ("ON", "TRUE", "1")
|
| 681 |
ui_inputs["qwen_image_2_1_prompt_enhancer_thinking"] = bool(is_thinking)
|
| 682 |
+
ui_inputs["qwen_image_2_1_prompt_enhancer_max_length"] = int(item.get("max_length", 4096))
|
| 683 |
+
elif itype == "ming_image_prompt_enhancer":
|
| 684 |
+
is_enabled = item.get("enable", item.get("enabled", True))
|
| 685 |
+
if isinstance(is_enabled, str):
|
| 686 |
+
is_enabled = is_enabled.upper() in ("ON", "TRUE", "1")
|
| 687 |
+
ui_inputs["ming_image_prompt_enhancer_enable"] = bool(is_enabled)
|
| 688 |
+
is_thinking = item.get("thinking", False)
|
| 689 |
+
if isinstance(is_thinking, str):
|
| 690 |
+
is_thinking = is_thinking.upper() in ("ON", "TRUE", "1")
|
| 691 |
+
ui_inputs["ming_image_prompt_enhancer_thinking"] = bool(is_thinking)
|
| 692 |
+
ui_inputs["ming_image_prompt_enhancer_max_length"] = int(item.get("max_length", 4096))
|
| 693 |
+
if "system_prompt" in item:
|
| 694 |
+
ui_inputs["ming_image_prompt_enhancer_system_prompt"] = str(item["system_prompt"])
|
| 695 |
elif itype == "krea2_identity_edit":
|
| 696 |
img = _parse_image_param(item.get("image"))
|
| 697 |
if img:
|
|
|
|
| 802 |
elif isinstance(pe_val, dict):
|
| 803 |
ui_inputs["qwen_image_2_1_prompt_enhancer_enable"] = bool(pe_val.get("enable", pe_val.get("enabled", True)))
|
| 804 |
ui_inputs["qwen_image_2_1_prompt_enhancer_thinking"] = bool(pe_val.get("thinking", False))
|
| 805 |
+
ui_inputs["qwen_image_2_1_prompt_enhancer_max_length"] = int(pe_val.get("max_length", 4096))
|
| 806 |
elif str(pe_val).upper() in ("ON", "TRUE", "1"):
|
| 807 |
ui_inputs["qwen_image_2_1_prompt_enhancer_enable"] = True
|
| 808 |
else:
|
| 809 |
ui_inputs["qwen_image_2_1_prompt_enhancer_enable"] = False
|
| 810 |
|
| 811 |
+
ming_pe_val = params.get("ming_image_prompt_enhancer") or params.get("ming_image_prompt_enhancer_enable")
|
| 812 |
+
if ming_pe_val is not None:
|
| 813 |
+
if isinstance(ming_pe_val, bool):
|
| 814 |
+
ui_inputs["ming_image_prompt_enhancer_enable"] = ming_pe_val
|
| 815 |
+
elif isinstance(ming_pe_val, dict):
|
| 816 |
+
ui_inputs["ming_image_prompt_enhancer_enable"] = bool(ming_pe_val.get("enable", ming_pe_val.get("enabled", True)))
|
| 817 |
+
ui_inputs["ming_image_prompt_enhancer_thinking"] = bool(ming_pe_val.get("thinking", False))
|
| 818 |
+
ui_inputs["ming_image_prompt_enhancer_max_length"] = int(ming_pe_val.get("max_length", 4096))
|
| 819 |
+
if "system_prompt" in ming_pe_val:
|
| 820 |
+
ui_inputs["ming_image_prompt_enhancer_system_prompt"] = str(ming_pe_val["system_prompt"])
|
| 821 |
+
elif str(ming_pe_val).upper() in ("ON", "TRUE", "1"):
|
| 822 |
+
ui_inputs["ming_image_prompt_enhancer_enable"] = True
|
| 823 |
+
else:
|
| 824 |
+
ui_inputs["ming_image_prompt_enhancer_enable"] = False
|
| 825 |
+
|
| 826 |
_TASKS_DB[task_id]["progress"] = 50
|
| 827 |
|
| 828 |
# Execute Pipeline
|
requirements.txt
CHANGED
|
@@ -1,5 +1,5 @@
|
|
| 1 |
comfyui-frontend-package==1.53.6
|
| 2 |
-
comfyui-workflow-templates==0.11.
|
| 3 |
comfyui-embedded-docs==0.5.12
|
| 4 |
torch
|
| 5 |
torchsde
|
|
|
|
| 1 |
comfyui-frontend-package==1.53.6
|
| 2 |
+
comfyui-workflow-templates==0.11.69
|
| 3 |
comfyui-embedded-docs==0.5.12
|
| 4 |
torch
|
| 5 |
torchsde
|
ui/events/change_handlers.py
CHANGED
|
@@ -18,7 +18,7 @@ from .config_loaders import (
|
|
| 18 |
load_ipadapter_config
|
| 19 |
)
|
| 20 |
|
| 21 |
-
def make_update_fn(m_comp, cat_comp, ar_comp, width_comp, height_comp, cn_types, cn_series, cn_filepaths, anima_cn_types, anima_cn_series, anima_cn_filepaths, diffsynth_cn_types, diffsynth_cn_series, diffsynth_cn_filepaths, krea2_cn_types, krea2_cn_series, krea2_cn_filepaths, ipa_preset, lora_acc, cn_acc, anima_cn_acc, diffsynth_cn_acc, krea2_cn_acc, ipa_acc, sd3_ipa_acc, flux1_ipa_acc, style_acc, embed_acc, cond_acc, ref_latent_acc, hidream_o1_ref_acc, prompt_comp, neg_prompt_comp, steps_comp, cfg_comp, sampler_comp, scheduler_comp, pid_acc=None, vae_acc=None, joyai_ref_acc=None, krea2_identity_edit_acc=None, krea2_reference_edit_acc=None, qwen_image_edit_acc=None, ref_img_acc=None, sensenova_ref_acc=None, task_type=None, type_comp=None, qwen_pe_acc=None):
|
| 22 |
def update_fn(*args):
|
| 23 |
arch = args[0]
|
| 24 |
category = args[1]
|
|
@@ -91,6 +91,7 @@ def make_update_fn(m_comp, cat_comp, ar_comp, width_comp, height_comp, cn_types,
|
|
| 91 |
if pid_acc: updates[pid_acc] = gr.update(visible=('pid' in enabled_chains))
|
| 92 |
if vae_acc: updates[vae_acc] = gr.update(visible=('vae' in enabled_chains))
|
| 93 |
if qwen_pe_acc: updates[qwen_pe_acc] = gr.update(visible=('qwen_image_2_1_prompt_enhancer' in enabled_chains))
|
|
|
|
| 94 |
|
| 95 |
if ar_comp:
|
| 96 |
res_key = arch_model_type
|
|
@@ -158,7 +159,7 @@ def make_update_fn(m_comp, cat_comp, ar_comp, width_comp, height_comp, cn_types,
|
|
| 158 |
return update_fn
|
| 159 |
|
| 160 |
|
| 161 |
-
def make_model_change_fn(cat_comp_ref, ar_comp, width_comp, height_comp, cn_types, cn_series, cn_filepaths, anima_cn_types, anima_cn_series, anima_cn_filepaths, diffsynth_cn_types, diffsynth_cn_series, diffsynth_cn_filepaths, krea2_cn_types, krea2_cn_series, krea2_cn_filepaths, arch_comp_ref, ipa_preset, lora_acc, cn_acc, anima_cn_acc, diffsynth_cn_acc, krea2_cn_acc, ipa_acc, sd3_ipa_acc, flux1_ipa_acc, style_acc, embed_acc, cond_acc, ref_latent_acc, hidream_o1_ref_acc, prompt_comp, neg_prompt_comp, steps_comp, cfg_comp, sampler_comp, scheduler_comp, pid_acc=None, vae_acc=None, joyai_ref_acc=None, krea2_identity_edit_acc=None, krea2_reference_edit_acc=None, qwen_image_edit_acc=None, ref_img_acc=None, sensenova_ref_acc=None, task_type=None, type_comp=None, qwen_pe_acc=None):
|
| 162 |
def change_fn(*args):
|
| 163 |
model_name = args[0]
|
| 164 |
idx = 1
|
|
@@ -239,6 +240,7 @@ def make_model_change_fn(cat_comp_ref, ar_comp, width_comp, height_comp, cn_type
|
|
| 239 |
if pid_acc: updates[pid_acc] = gr.update(visible=('pid' in enabled_chains))
|
| 240 |
if vae_acc: updates[vae_acc] = gr.update(visible=('vae' in enabled_chains))
|
| 241 |
if qwen_pe_acc: updates[qwen_pe_acc] = gr.update(visible=('qwen_image_2_1_prompt_enhancer' in enabled_chains))
|
|
|
|
| 242 |
|
| 243 |
if ar_comp:
|
| 244 |
res_key = arch_model_type
|
|
|
|
| 18 |
load_ipadapter_config
|
| 19 |
)
|
| 20 |
|
| 21 |
+
def make_update_fn(m_comp, cat_comp, ar_comp, width_comp, height_comp, cn_types, cn_series, cn_filepaths, anima_cn_types, anima_cn_series, anima_cn_filepaths, diffsynth_cn_types, diffsynth_cn_series, diffsynth_cn_filepaths, krea2_cn_types, krea2_cn_series, krea2_cn_filepaths, ipa_preset, lora_acc, cn_acc, anima_cn_acc, diffsynth_cn_acc, krea2_cn_acc, ipa_acc, sd3_ipa_acc, flux1_ipa_acc, style_acc, embed_acc, cond_acc, ref_latent_acc, hidream_o1_ref_acc, prompt_comp, neg_prompt_comp, steps_comp, cfg_comp, sampler_comp, scheduler_comp, pid_acc=None, vae_acc=None, joyai_ref_acc=None, krea2_identity_edit_acc=None, krea2_reference_edit_acc=None, qwen_image_edit_acc=None, ref_img_acc=None, sensenova_ref_acc=None, task_type=None, type_comp=None, qwen_pe_acc=None, ming_pe_acc=None):
|
| 22 |
def update_fn(*args):
|
| 23 |
arch = args[0]
|
| 24 |
category = args[1]
|
|
|
|
| 91 |
if pid_acc: updates[pid_acc] = gr.update(visible=('pid' in enabled_chains))
|
| 92 |
if vae_acc: updates[vae_acc] = gr.update(visible=('vae' in enabled_chains))
|
| 93 |
if qwen_pe_acc: updates[qwen_pe_acc] = gr.update(visible=('qwen_image_2_1_prompt_enhancer' in enabled_chains))
|
| 94 |
+
if ming_pe_acc: updates[ming_pe_acc] = gr.update(visible=('ming_image_prompt_enhancer' in enabled_chains))
|
| 95 |
|
| 96 |
if ar_comp:
|
| 97 |
res_key = arch_model_type
|
|
|
|
| 159 |
return update_fn
|
| 160 |
|
| 161 |
|
| 162 |
+
def make_model_change_fn(cat_comp_ref, ar_comp, width_comp, height_comp, cn_types, cn_series, cn_filepaths, anima_cn_types, anima_cn_series, anima_cn_filepaths, diffsynth_cn_types, diffsynth_cn_series, diffsynth_cn_filepaths, krea2_cn_types, krea2_cn_series, krea2_cn_filepaths, arch_comp_ref, ipa_preset, lora_acc, cn_acc, anima_cn_acc, diffsynth_cn_acc, krea2_cn_acc, ipa_acc, sd3_ipa_acc, flux1_ipa_acc, style_acc, embed_acc, cond_acc, ref_latent_acc, hidream_o1_ref_acc, prompt_comp, neg_prompt_comp, steps_comp, cfg_comp, sampler_comp, scheduler_comp, pid_acc=None, vae_acc=None, joyai_ref_acc=None, krea2_identity_edit_acc=None, krea2_reference_edit_acc=None, qwen_image_edit_acc=None, ref_img_acc=None, sensenova_ref_acc=None, task_type=None, type_comp=None, qwen_pe_acc=None, ming_pe_acc=None):
|
| 163 |
def change_fn(*args):
|
| 164 |
model_name = args[0]
|
| 165 |
idx = 1
|
|
|
|
| 240 |
if pid_acc: updates[pid_acc] = gr.update(visible=('pid' in enabled_chains))
|
| 241 |
if vae_acc: updates[vae_acc] = gr.update(visible=('vae' in enabled_chains))
|
| 242 |
if qwen_pe_acc: updates[qwen_pe_acc] = gr.update(visible=('qwen_image_2_1_prompt_enhancer' in enabled_chains))
|
| 243 |
+
if ming_pe_acc: updates[ming_pe_acc] = gr.update(visible=('ming_image_prompt_enhancer' in enabled_chains))
|
| 244 |
|
| 245 |
if ar_comp:
|
| 246 |
res_key = arch_model_type
|
ui/events/main.py
CHANGED
|
@@ -81,6 +81,7 @@ def attach_event_handlers(ui_components, demo):
|
|
| 81 |
pid_accordion = ui_components.get(f'pid_accordion_{prefix}')
|
| 82 |
vae_accordion = ui_components.get(f'vae_accordion_{prefix}')
|
| 83 |
qwen_pe_accordion = ui_components.get(f'qwen_image_2_1_prompt_enhancer_accordion_{prefix}')
|
|
|
|
| 84 |
|
| 85 |
ipa_preset_list = ui_components.get(f'ipadapter_final_preset_{prefix}')
|
| 86 |
|
|
@@ -123,6 +124,7 @@ def attach_event_handlers(ui_components, demo):
|
|
| 123 |
if pid_accordion: outputs.append(pid_accordion)
|
| 124 |
if vae_accordion: outputs.append(vae_accordion)
|
| 125 |
if qwen_pe_accordion: outputs.append(qwen_pe_accordion)
|
|
|
|
| 126 |
if ipa_preset_list: outputs.append(ipa_preset_list)
|
| 127 |
|
| 128 |
outputs.extend(valid_extra_comps)
|
|
@@ -136,7 +138,7 @@ def attach_event_handlers(ui_components, demo):
|
|
| 136 |
ipa_preset_list, lora_accordion, cn_accordion, anima_cn_accordion, diffsynth_cn_accordion, krea2_cn_accordion, ipa_accordion, sd3_ipa_accordion, flux1_ipa_accordion, style_accordion, embedding_accordion, conditioning_accordion,
|
| 137 |
ref_latent_accordion, hidream_o1_ref_accordion, prompt_comp, neg_prompt_comp, steps_comp, cfg_comp, sampler_comp, scheduler_comp,
|
| 138 |
pid_acc=pid_accordion, vae_acc=vae_accordion, joyai_ref_acc=joyai_ref_accordion, krea2_identity_edit_acc=krea2_identity_edit_accordion, krea2_reference_edit_acc=krea2_reference_edit_accordion, qwen_image_edit_acc=qwen_image_edit_accordion, ref_img_acc=ref_img_accordion, sensenova_ref_acc=sensenova_ref_accordion,
|
| 139 |
-
task_type=task_type, type_comp=type_comp, qwen_pe_acc=qwen_pe_accordion
|
| 140 |
)
|
| 141 |
inputs = [arch_comp, cat_comp]
|
| 142 |
if aspect_ratio_comp:
|
|
@@ -177,6 +179,7 @@ def attach_event_handlers(ui_components, demo):
|
|
| 177 |
if pid_accordion: outputs2.append(pid_accordion)
|
| 178 |
if vae_accordion: outputs2.append(vae_accordion)
|
| 179 |
if qwen_pe_accordion: outputs2.append(qwen_pe_accordion)
|
|
|
|
| 180 |
if ipa_preset_list: outputs2.append(ipa_preset_list)
|
| 181 |
|
| 182 |
outputs2.extend(valid_extra_comps)
|
|
@@ -195,7 +198,7 @@ def attach_event_handlers(ui_components, demo):
|
|
| 195 |
arch_comp, ipa_preset_list, lora_accordion, cn_accordion, anima_cn_accordion, diffsynth_cn_accordion, krea2_cn_accordion, ipa_accordion, sd3_ipa_accordion, flux1_ipa_accordion, style_accordion, embedding_accordion, conditioning_accordion,
|
| 196 |
ref_latent_accordion, hidream_o1_ref_accordion, prompt_comp, neg_prompt_comp, steps_comp, cfg_comp, sampler_comp, scheduler_comp,
|
| 197 |
pid_acc=pid_accordion, vae_acc=vae_accordion, joyai_ref_acc=joyai_ref_accordion, krea2_identity_edit_acc=krea2_identity_edit_accordion, krea2_reference_edit_acc=krea2_reference_edit_accordion, qwen_image_edit_acc=qwen_image_edit_accordion, ref_img_acc=ref_img_accordion, sensenova_ref_acc=sensenova_ref_accordion,
|
| 198 |
-
task_type=task_type, type_comp=type_comp, qwen_pe_acc=qwen_pe_accordion
|
| 199 |
)
|
| 200 |
if type_comp:
|
| 201 |
inputs2.append(type_comp)
|
|
@@ -290,6 +293,7 @@ def attach_event_handlers(ui_components, demo):
|
|
| 290 |
f'qwen_image_edit_accordion_{prefix}': 'qwen_image_edit',
|
| 291 |
f'reference_image_accordion_{prefix}': 'reference_image',
|
| 292 |
f'qwen_image_2_1_prompt_enhancer_accordion_{prefix}': 'qwen_image_2_1_prompt_enhancer',
|
|
|
|
| 293 |
f'pid_accordion_{prefix}': 'pid',
|
| 294 |
f'vae_accordion_{prefix}': 'vae',
|
| 295 |
}
|
|
@@ -329,6 +333,7 @@ def attach_event_handlers(ui_components, demo):
|
|
| 329 |
'qwen_image_edit_accordion_imagegen',
|
| 330 |
'reference_image_accordion_imagegen',
|
| 331 |
'qwen_image_2_1_prompt_enhancer_accordion_imagegen',
|
|
|
|
| 332 |
'pid_accordion_imagegen', 'vae_accordion_imagegen',
|
| 333 |
]
|
| 334 |
for key in accordion_keys:
|
|
|
|
| 81 |
pid_accordion = ui_components.get(f'pid_accordion_{prefix}')
|
| 82 |
vae_accordion = ui_components.get(f'vae_accordion_{prefix}')
|
| 83 |
qwen_pe_accordion = ui_components.get(f'qwen_image_2_1_prompt_enhancer_accordion_{prefix}')
|
| 84 |
+
ming_pe_accordion = ui_components.get(f'ming_image_prompt_enhancer_accordion_{prefix}')
|
| 85 |
|
| 86 |
ipa_preset_list = ui_components.get(f'ipadapter_final_preset_{prefix}')
|
| 87 |
|
|
|
|
| 124 |
if pid_accordion: outputs.append(pid_accordion)
|
| 125 |
if vae_accordion: outputs.append(vae_accordion)
|
| 126 |
if qwen_pe_accordion: outputs.append(qwen_pe_accordion)
|
| 127 |
+
if ming_pe_accordion: outputs.append(ming_pe_accordion)
|
| 128 |
if ipa_preset_list: outputs.append(ipa_preset_list)
|
| 129 |
|
| 130 |
outputs.extend(valid_extra_comps)
|
|
|
|
| 138 |
ipa_preset_list, lora_accordion, cn_accordion, anima_cn_accordion, diffsynth_cn_accordion, krea2_cn_accordion, ipa_accordion, sd3_ipa_accordion, flux1_ipa_accordion, style_accordion, embedding_accordion, conditioning_accordion,
|
| 139 |
ref_latent_accordion, hidream_o1_ref_accordion, prompt_comp, neg_prompt_comp, steps_comp, cfg_comp, sampler_comp, scheduler_comp,
|
| 140 |
pid_acc=pid_accordion, vae_acc=vae_accordion, joyai_ref_acc=joyai_ref_accordion, krea2_identity_edit_acc=krea2_identity_edit_accordion, krea2_reference_edit_acc=krea2_reference_edit_accordion, qwen_image_edit_acc=qwen_image_edit_accordion, ref_img_acc=ref_img_accordion, sensenova_ref_acc=sensenova_ref_accordion,
|
| 141 |
+
task_type=task_type, type_comp=type_comp, qwen_pe_acc=qwen_pe_accordion, ming_pe_acc=ming_pe_accordion
|
| 142 |
)
|
| 143 |
inputs = [arch_comp, cat_comp]
|
| 144 |
if aspect_ratio_comp:
|
|
|
|
| 179 |
if pid_accordion: outputs2.append(pid_accordion)
|
| 180 |
if vae_accordion: outputs2.append(vae_accordion)
|
| 181 |
if qwen_pe_accordion: outputs2.append(qwen_pe_accordion)
|
| 182 |
+
if ming_pe_accordion: outputs2.append(ming_pe_accordion)
|
| 183 |
if ipa_preset_list: outputs2.append(ipa_preset_list)
|
| 184 |
|
| 185 |
outputs2.extend(valid_extra_comps)
|
|
|
|
| 198 |
arch_comp, ipa_preset_list, lora_accordion, cn_accordion, anima_cn_accordion, diffsynth_cn_accordion, krea2_cn_accordion, ipa_accordion, sd3_ipa_accordion, flux1_ipa_accordion, style_accordion, embedding_accordion, conditioning_accordion,
|
| 199 |
ref_latent_accordion, hidream_o1_ref_accordion, prompt_comp, neg_prompt_comp, steps_comp, cfg_comp, sampler_comp, scheduler_comp,
|
| 200 |
pid_acc=pid_accordion, vae_acc=vae_accordion, joyai_ref_acc=joyai_ref_accordion, krea2_identity_edit_acc=krea2_identity_edit_accordion, krea2_reference_edit_acc=krea2_reference_edit_accordion, qwen_image_edit_acc=qwen_image_edit_accordion, ref_img_acc=ref_img_accordion, sensenova_ref_acc=sensenova_ref_accordion,
|
| 201 |
+
task_type=task_type, type_comp=type_comp, qwen_pe_acc=qwen_pe_accordion, ming_pe_acc=ming_pe_accordion
|
| 202 |
)
|
| 203 |
if type_comp:
|
| 204 |
inputs2.append(type_comp)
|
|
|
|
| 293 |
f'qwen_image_edit_accordion_{prefix}': 'qwen_image_edit',
|
| 294 |
f'reference_image_accordion_{prefix}': 'reference_image',
|
| 295 |
f'qwen_image_2_1_prompt_enhancer_accordion_{prefix}': 'qwen_image_2_1_prompt_enhancer',
|
| 296 |
+
f'ming_image_prompt_enhancer_accordion_{prefix}': 'ming_image_prompt_enhancer',
|
| 297 |
f'pid_accordion_{prefix}': 'pid',
|
| 298 |
f'vae_accordion_{prefix}': 'vae',
|
| 299 |
}
|
|
|
|
| 333 |
'qwen_image_edit_accordion_imagegen',
|
| 334 |
'reference_image_accordion_imagegen',
|
| 335 |
'qwen_image_2_1_prompt_enhancer_accordion_imagegen',
|
| 336 |
+
'ming_image_prompt_enhancer_accordion_imagegen',
|
| 337 |
'pid_accordion_imagegen', 'vae_accordion_imagegen',
|
| 338 |
]
|
| 339 |
for key in accordion_keys:
|
ui/events/run_handlers.py
CHANGED
|
@@ -44,6 +44,15 @@ def create_run_event(prefix: str, task_type: str = None, ui_components: dict = N
|
|
| 44 |
if ui_components.get(f'qwen_image_2_1_prompt_enhancer_max_length_{prefix}'):
|
| 45 |
run_inputs_map['qwen_image_2_1_prompt_enhancer_max_length'] = ui_components[f'qwen_image_2_1_prompt_enhancer_max_length_{prefix}']
|
| 46 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 47 |
# Generic / Txt2Img inputs
|
| 48 |
if ui_components.get(f'width_{prefix}') or ui_components.get(f'{prefix}_width'):
|
| 49 |
run_inputs_map['width'] = ui_components.get(f'width_{prefix}') or ui_components.get(f'{prefix}_width')
|
|
|
|
| 44 |
if ui_components.get(f'qwen_image_2_1_prompt_enhancer_max_length_{prefix}'):
|
| 45 |
run_inputs_map['qwen_image_2_1_prompt_enhancer_max_length'] = ui_components[f'qwen_image_2_1_prompt_enhancer_max_length_{prefix}']
|
| 46 |
|
| 47 |
+
if ui_components.get(f'ming_image_prompt_enhancer_enable_{prefix}'):
|
| 48 |
+
run_inputs_map['ming_image_prompt_enhancer_enable'] = ui_components[f'ming_image_prompt_enhancer_enable_{prefix}']
|
| 49 |
+
if ui_components.get(f'ming_image_prompt_enhancer_thinking_{prefix}'):
|
| 50 |
+
run_inputs_map['ming_image_prompt_enhancer_thinking'] = ui_components[f'ming_image_prompt_enhancer_thinking_{prefix}']
|
| 51 |
+
if ui_components.get(f'ming_image_prompt_enhancer_max_length_{prefix}'):
|
| 52 |
+
run_inputs_map['ming_image_prompt_enhancer_max_length'] = ui_components[f'ming_image_prompt_enhancer_max_length_{prefix}']
|
| 53 |
+
if ui_components.get(f'ming_image_prompt_enhancer_system_prompt_{prefix}'):
|
| 54 |
+
run_inputs_map['ming_image_prompt_enhancer_system_prompt'] = ui_components[f'ming_image_prompt_enhancer_system_prompt_{prefix}']
|
| 55 |
+
|
| 56 |
# Generic / Txt2Img inputs
|
| 57 |
if ui_components.get(f'width_{prefix}') or ui_components.get(f'{prefix}_width'):
|
| 58 |
run_inputs_map['width'] = ui_components.get(f'width_{prefix}') or ui_components.get(f'{prefix}_width')
|
ui/imagegen_ui.py
CHANGED
|
@@ -9,7 +9,7 @@ from .shared.ui_components import (
|
|
| 9 |
create_model_architecture_filter_ui, create_category_filter_ui,
|
| 10 |
create_sd3_ipadapter_ui, create_flux1_ipadapter_ui, create_style_ui,
|
| 11 |
create_reference_latent_ui, create_hidream_o1_reference_ui, create_sensenova_reference_ui, create_joyai_reference_ui, create_reference_image_ui, create_krea2_identity_edit_ui, create_krea2_reference_edit_ui, create_qwen_image_edit_ui,
|
| 12 |
-
create_pid_ui, create_qwen_image_2_1_prompt_enhancer_ui
|
| 13 |
)
|
| 14 |
|
| 15 |
default_vals = MODEL_DEFAULTS_CONFIG.get('Default', {})
|
|
@@ -191,6 +191,7 @@ def create_ui():
|
|
| 191 |
components.update(create_qwen_image_edit_ui(prefix))
|
| 192 |
components.update(create_reference_image_ui(prefix))
|
| 193 |
components.update(create_qwen_image_2_1_prompt_enhancer_ui(prefix))
|
|
|
|
| 194 |
components.update(create_vae_override_ui(prefix))
|
| 195 |
components.update(create_pid_ui(prefix))
|
| 196 |
components[f'accordion_wrapper_{prefix}'] = accordion_wrapper
|
|
|
|
| 9 |
create_model_architecture_filter_ui, create_category_filter_ui,
|
| 10 |
create_sd3_ipadapter_ui, create_flux1_ipadapter_ui, create_style_ui,
|
| 11 |
create_reference_latent_ui, create_hidream_o1_reference_ui, create_sensenova_reference_ui, create_joyai_reference_ui, create_reference_image_ui, create_krea2_identity_edit_ui, create_krea2_reference_edit_ui, create_qwen_image_edit_ui,
|
| 12 |
+
create_pid_ui, create_qwen_image_2_1_prompt_enhancer_ui, create_ming_image_prompt_enhancer_ui
|
| 13 |
)
|
| 14 |
|
| 15 |
default_vals = MODEL_DEFAULTS_CONFIG.get('Default', {})
|
|
|
|
| 191 |
components.update(create_qwen_image_edit_ui(prefix))
|
| 192 |
components.update(create_reference_image_ui(prefix))
|
| 193 |
components.update(create_qwen_image_2_1_prompt_enhancer_ui(prefix))
|
| 194 |
+
components.update(create_ming_image_prompt_enhancer_ui(prefix))
|
| 195 |
components.update(create_vae_override_ui(prefix))
|
| 196 |
components.update(create_pid_ui(prefix))
|
| 197 |
components[f'accordion_wrapper_{prefix}'] = accordion_wrapper
|
ui/shared/ui_components.py
CHANGED
|
@@ -900,14 +900,14 @@ def create_qwen_image_2_1_prompt_enhancer_ui(prefix: str):
|
|
| 900 |
scale=2
|
| 901 |
)
|
| 902 |
components[key('qwen_image_2_1_prompt_enhancer_thinking')] = gr.Checkbox(
|
| 903 |
-
label="Enable Thinking (
|
| 904 |
value=False,
|
| 905 |
interactive=True,
|
| 906 |
scale=2
|
| 907 |
)
|
| 908 |
components[key('qwen_image_2_1_prompt_enhancer_max_length')] = gr.Number(
|
| 909 |
label="Max Length",
|
| 910 |
-
value=
|
| 911 |
precision=0,
|
| 912 |
minimum=16,
|
| 913 |
step=16,
|
|
@@ -915,4 +915,56 @@ def create_qwen_image_2_1_prompt_enhancer_ui(prefix: str):
|
|
| 915 |
scale=1
|
| 916 |
)
|
| 917 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 918 |
return components
|
|
|
|
| 900 |
scale=2
|
| 901 |
)
|
| 902 |
components[key('qwen_image_2_1_prompt_enhancer_thinking')] = gr.Checkbox(
|
| 903 |
+
label="Enable Thinking (Requires longer ZeroGPU Duration)",
|
| 904 |
value=False,
|
| 905 |
interactive=True,
|
| 906 |
scale=2
|
| 907 |
)
|
| 908 |
components[key('qwen_image_2_1_prompt_enhancer_max_length')] = gr.Number(
|
| 909 |
label="Max Length",
|
| 910 |
+
value=4096,
|
| 911 |
precision=0,
|
| 912 |
minimum=16,
|
| 913 |
step=16,
|
|
|
|
| 915 |
scale=1
|
| 916 |
)
|
| 917 |
|
| 918 |
+
return components
|
| 919 |
+
|
| 920 |
+
|
| 921 |
+
DEFAULT_MING_IMAGE_SYSTEM_PROMPT = (
|
| 922 |
+
"You are a senior visual designer and image-prompt engineer. Expand the user's request into one precise, high-resolution Figma-style caption. Return only one JSON object. "
|
| 923 |
+
"Use exactly two top-level keys. `canvas_settings` contains exactly `aspect_ratio`, `ambient_lighting`, and `image_style`. `layers` lists visible groups from background to topmost overlay. Every layer contains exactly `description`, `coordinates`, `hierarchy_and_relation`, and `color_specs`; `color_specs` is an array of hex colors. "
|
| 924 |
+
'`coordinates` MUST be one string, never an object or array, in exactly this form: `"cx: 0.500, cy: 0.500, w: 1.000, h: 1.000"`. Values are normalized; each bbox encloses its complete owned object and stays inside the canvas. '
|
| 925 |
+
"A layer is one selectable visible semantic group: background, full person, coherent object, panel, card, row, or text block. Prefer the fewest groups that preserve the layout. Keep people and objects intact. Never create invisible parents, guides, placeholders, empty layers, duplicate summaries, or multiple owners for one element. "
|
| 926 |
+
"Preserve every user-supplied rendered string character-for-character and as one contiguous string. Unless multiple visible copies are requested, it must occur exactly once across all `description` fields and zero times in `hierarchy_and_relation`. Quote it only where describing its visible rendering; refer to the related subject elsewhere with unquoted semantic wording. Enumerate intended copy, invent extra copy sparingly, and never hide content behind \"other text\", \"remaining labels\", or \"etc.\" "
|
| 927 |
+
"Describe concrete composition, typography, materials, texture, lighting, pose, and camera treatment without literary filler. Use `hierarchy_and_relation` only for ownership, alignment, containment, stacking, and occlusion. "
|
| 928 |
+
"Infer structured layouts first. Use one complete layer per card and state its row and column. A compact secondary table may be one layer only if every header and cell is listed; otherwise use a visible shared frame when present, one complete header, and one complete layer per body row, binding values to columns and stating blanks. Enumerate sequences, schedules, spans, gaps, and vacant tracks in visual order. Do not mistake ordinary alignment for a table. "
|
| 929 |
+
"Silently verify schema, string coordinates, Z-order, exact-text counts, geometry, bbox validity, and completeness."
|
| 930 |
+
)
|
| 931 |
+
|
| 932 |
+
|
| 933 |
+
def create_ming_image_prompt_enhancer_ui(prefix: str):
|
| 934 |
+
components = {}
|
| 935 |
+
key = lambda name: f"{name}_{prefix}"
|
| 936 |
+
|
| 937 |
+
with gr.Accordion("Ming-Image Prompt Enhancer Settings", open=False, visible=('ming_image_prompt_enhancer' in default_enabled_chains)) as enhancer_accordion:
|
| 938 |
+
components[key('ming_image_prompt_enhancer_accordion')] = enhancer_accordion
|
| 939 |
+
gr.Markdown("💡 **Tip:** Leverages [Qwen3.8-27B](https://huggingface.co/Qwen/Qwen3.8-27B) guided by the official [System Prompt](https://github.com/inclusionAI/Ming-Image#text-to-image-prompt-rewriting) to automatically enhance prompts into structured layouts based on input context.")
|
| 940 |
+
with gr.Row():
|
| 941 |
+
components[key('ming_image_prompt_enhancer_enable')] = gr.Checkbox(
|
| 942 |
+
label="Enable Ming-Image Prompt Enhancer",
|
| 943 |
+
value=False,
|
| 944 |
+
interactive=True,
|
| 945 |
+
scale=2
|
| 946 |
+
)
|
| 947 |
+
components[key('ming_image_prompt_enhancer_thinking')] = gr.Checkbox(
|
| 948 |
+
label="Enable Thinking (Requires longer ZeroGPU Duration)",
|
| 949 |
+
value=False,
|
| 950 |
+
interactive=True,
|
| 951 |
+
scale=2
|
| 952 |
+
)
|
| 953 |
+
components[key('ming_image_prompt_enhancer_max_length')] = gr.Number(
|
| 954 |
+
label="Max Length",
|
| 955 |
+
value=4096,
|
| 956 |
+
precision=0,
|
| 957 |
+
minimum=16,
|
| 958 |
+
step=16,
|
| 959 |
+
interactive=True,
|
| 960 |
+
scale=1
|
| 961 |
+
)
|
| 962 |
+
components[key('ming_image_prompt_enhancer_system_prompt')] = gr.Textbox(
|
| 963 |
+
label="System Prompt",
|
| 964 |
+
value=DEFAULT_MING_IMAGE_SYSTEM_PROMPT,
|
| 965 |
+
lines=4,
|
| 966 |
+
max_lines=12,
|
| 967 |
+
interactive=True
|
| 968 |
+
)
|
| 969 |
+
|
| 970 |
return components
|
yaml/chain_features.yaml
CHANGED
|
@@ -545,5 +545,30 @@ qwen_image_2_1_prompt_enhancer:
|
|
| 545 |
description: "Enable thinking (reasoning CoT). Thought text inside <think> tags will be automatically stripped before passing prompt to downstream text encoding."
|
| 546 |
max_length:
|
| 547 |
type: integer
|
| 548 |
-
default:
|
| 549 |
-
description: "Maximum generation length for prompt enhancement."
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 545 |
description: "Enable thinking (reasoning CoT). Thought text inside <think> tags will be automatically stripped before passing prompt to downstream text encoding."
|
| 546 |
max_length:
|
| 547 |
type: integer
|
| 548 |
+
default: 4096
|
| 549 |
+
description: "Maximum generation length for prompt enhancement (default 4096)."
|
| 550 |
+
|
| 551 |
+
ming_image_prompt_enhancer:
|
| 552 |
+
chains: ming_image_prompt_enhancer
|
| 553 |
+
display_name: "Ming-Image Prompt Enhancer Settings"
|
| 554 |
+
description: "Enhance prompts using Ming-Image Prompt Enhancer (Qwen3.8-27B) guided by official Figma-style rewriter System Prompt, with optional reasoning CoT thinking mode."
|
| 555 |
+
max_count: 1
|
| 556 |
+
usage_guideline: "Set enable to true to activate Ming-Image prompt enhancement. Supports single/multi-reference images, thinking mode, and custom system prompt."
|
| 557 |
+
parameters_schema:
|
| 558 |
+
type: object
|
| 559 |
+
properties:
|
| 560 |
+
enable:
|
| 561 |
+
type: boolean
|
| 562 |
+
default: false
|
| 563 |
+
description: "Enable Ming-Image Prompt Enhancer."
|
| 564 |
+
thinking:
|
| 565 |
+
type: boolean
|
| 566 |
+
default: false
|
| 567 |
+
description: "Enable thinking (reasoning CoT)."
|
| 568 |
+
max_length:
|
| 569 |
+
type: integer
|
| 570 |
+
default: 4096
|
| 571 |
+
description: "Maximum generation length for prompt enhancement (default 4096)."
|
| 572 |
+
system_prompt:
|
| 573 |
+
type: string
|
| 574 |
+
description: "System prompt for Qwen3.8-27B prompt enhancer (defaults to official Ming-Image Figma-style rewriter prompt)."
|
yaml/constants.yaml
CHANGED
|
@@ -103,6 +103,14 @@ RESOLUTION_MAP:
|
|
| 103 |
"3:4 (Classic Portrait)": [896, 1152]
|
| 104 |
"3:2 (Photography)": [1216, 832]
|
| 105 |
"2:3 (Photography Portrait)": [832, 1216]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 106 |
qwen-image-2.1:
|
| 107 |
"1:1 (Square)": [1024, 1024]
|
| 108 |
"16:9 (Landscape)": [1344, 768]
|
|
|
|
| 103 |
"3:4 (Classic Portrait)": [896, 1152]
|
| 104 |
"3:2 (Photography)": [1216, 832]
|
| 105 |
"2:3 (Photography Portrait)": [832, 1216]
|
| 106 |
+
ming-image:
|
| 107 |
+
"1:1 (Square)": [1024, 1024]
|
| 108 |
+
"16:9 (Landscape)": [1344, 768]
|
| 109 |
+
"9:16 (Portrait)": [768, 1344]
|
| 110 |
+
"4:3 (Classic)": [1152, 896]
|
| 111 |
+
"3:4 (Classic Portrait)": [896, 1152]
|
| 112 |
+
"3:2 (Photography)": [1216, 832]
|
| 113 |
+
"2:3 (Photography Portrait)": [832, 1216]
|
| 114 |
qwen-image-2.1:
|
| 115 |
"1:1 (Square)": [1024, 1024]
|
| 116 |
"16:9 (Landscape)": [1344, 768]
|
yaml/file_list.yaml
CHANGED
|
@@ -642,6 +642,11 @@ file:
|
|
| 642 |
source: "hf"
|
| 643 |
repo_id: "Comfy-Org/Qwen-Image-2.1"
|
| 644 |
repository_file_path: "diffusion_models/qwen_image_2.1_int8_convrot.safetensors"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 645 |
# Qwen-Image
|
| 646 |
- filename: "qwen_image_2512_fp8_e4m3fn_scaled_comfyui_4steps_v1.0.safetensors"
|
| 647 |
source: "hf"
|
|
@@ -920,6 +925,16 @@ file:
|
|
| 920 |
source: "hf"
|
| 921 |
repo_id: "Comfy-Org/Qwen-Image-2.1"
|
| 922 |
repository_file_path: "text_encoders/qwen3.5_9b_qwen_image_2.1_pe_t2i.int8_convrot.safetensors"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 923 |
# PixelDiT
|
| 924 |
- filename: "gemma_2_2b_it_elm_fp8_scaled.safetensors"
|
| 925 |
source: "hf"
|
|
@@ -1047,4 +1062,9 @@ file:
|
|
| 1047 |
- filename: "mage_flow_vae_bf16.safetensors"
|
| 1048 |
source: "hf"
|
| 1049 |
repo_id: "Comfy-Org/Mage-Flow"
|
| 1050 |
-
repository_file_path: "vae/mage_flow_vae_bf16.safetensors"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 642 |
source: "hf"
|
| 643 |
repo_id: "Comfy-Org/Qwen-Image-2.1"
|
| 644 |
repository_file_path: "diffusion_models/qwen_image_2.1_int8_convrot.safetensors"
|
| 645 |
+
# Ming-Image
|
| 646 |
+
- filename: "ming_image_0.1_design_int8_convrot.safetensors"
|
| 647 |
+
source: "hf"
|
| 648 |
+
repo_id: "Comfy-Org/Ming-Image"
|
| 649 |
+
repository_file_path: "diffusion_models/ming_image_0.1_design_int8_convrot.safetensors"
|
| 650 |
# Qwen-Image
|
| 651 |
- filename: "qwen_image_2512_fp8_e4m3fn_scaled_comfyui_4steps_v1.0.safetensors"
|
| 652 |
source: "hf"
|
|
|
|
| 925 |
source: "hf"
|
| 926 |
repo_id: "Comfy-Org/Qwen-Image-2.1"
|
| 927 |
repository_file_path: "text_encoders/qwen3.5_9b_qwen_image_2.1_pe_t2i.int8_convrot.safetensors"
|
| 928 |
+
# Ming-Image
|
| 929 |
+
- filename: "ming_image_0.1_ling_mini_2.0_w4a8.safetensors"
|
| 930 |
+
source: "hf"
|
| 931 |
+
repo_id: "Comfy-Org/Ming-Image"
|
| 932 |
+
repository_file_path: "text_encoders/ming_image_0.1_ling_mini_2.0_w4a8.safetensors"
|
| 933 |
+
# Qwen3.8-27B (Ming-Image Prompt Enhancer)
|
| 934 |
+
- filename: "qwen3.8_27b_w4a8.safetensors"
|
| 935 |
+
source: "hf"
|
| 936 |
+
repo_id: "Comfy-Org/Qwen3.8-27B"
|
| 937 |
+
repository_file_path: "text_encoders/qwen3.8_27b_w4a8.safetensors"
|
| 938 |
# PixelDiT
|
| 939 |
- filename: "gemma_2_2b_it_elm_fp8_scaled.safetensors"
|
| 940 |
source: "hf"
|
|
|
|
| 1062 |
- filename: "mage_flow_vae_bf16.safetensors"
|
| 1063 |
source: "hf"
|
| 1064 |
repo_id: "Comfy-Org/Mage-Flow"
|
| 1065 |
+
repository_file_path: "vae/mage_flow_vae_bf16.safetensors"
|
| 1066 |
+
# Ming-Image
|
| 1067 |
+
- filename: "ming_image_vae_bf16.safetensors"
|
| 1068 |
+
source: "hf"
|
| 1069 |
+
repo_id: "Comfy-Org/Ming-Image"
|
| 1070 |
+
repository_file_path: "vae/ming_image_vae_bf16.safetensors"
|
yaml/model_architecture_features.yaml
CHANGED
|
@@ -70,6 +70,12 @@ z-image:
|
|
| 70 |
- vae
|
| 71 |
- pid
|
| 72 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 73 |
qwen-image-2.1:
|
| 74 |
enabled_chains:
|
| 75 |
- lora
|
|
|
|
| 70 |
- vae
|
| 71 |
- pid
|
| 72 |
|
| 73 |
+
ming-image:
|
| 74 |
+
enabled_chains:
|
| 75 |
+
- lora
|
| 76 |
+
- reference_image
|
| 77 |
+
- ming_image_prompt_enhancer
|
| 78 |
+
|
| 79 |
qwen-image-2.1:
|
| 80 |
enabled_chains:
|
| 81 |
- lora
|
yaml/model_architectures.yaml
CHANGED
|
@@ -1,6 +1,7 @@
|
|
| 1 |
architecture_order:
|
| 2 |
- "Krea-2"
|
| 3 |
- "Qwen-Image-2.1"
|
|
|
|
| 4 |
- "SenseNova-U1.5"
|
| 5 |
- "Mage-Flow"
|
| 6 |
- "JoyAI-Image"
|
|
@@ -62,6 +63,8 @@ architectures:
|
|
| 62 |
"Z-Image":
|
| 63 |
model_type: "z-image"
|
| 64 |
controlnet_key: "Z-Image"
|
|
|
|
|
|
|
| 65 |
"Qwen-Image-2.1":
|
| 66 |
model_type: "qwen-image-2.1"
|
| 67 |
controlnet_key: "Qwen-Image"
|
|
|
|
| 1 |
architecture_order:
|
| 2 |
- "Krea-2"
|
| 3 |
- "Qwen-Image-2.1"
|
| 4 |
+
- "Ming-Image"
|
| 5 |
- "SenseNova-U1.5"
|
| 6 |
- "Mage-Flow"
|
| 7 |
- "JoyAI-Image"
|
|
|
|
| 63 |
"Z-Image":
|
| 64 |
model_type: "z-image"
|
| 65 |
controlnet_key: "Z-Image"
|
| 66 |
+
"Ming-Image":
|
| 67 |
+
model_type: "ming-image"
|
| 68 |
"Qwen-Image-2.1":
|
| 69 |
model_type: "qwen-image-2.1"
|
| 70 |
controlnet_key: "Qwen-Image"
|
yaml/model_defaults.yaml
CHANGED
|
@@ -131,6 +131,13 @@ Z-Image:
|
|
| 131 |
steps: 9
|
| 132 |
cfg: 1.0
|
| 133 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 134 |
Qwen-Image-2.1:
|
| 135 |
_defaults:
|
| 136 |
steps: 25
|
|
|
|
| 131 |
steps: 9
|
| 132 |
cfg: 1.0
|
| 133 |
|
| 134 |
+
Ming-Image:
|
| 135 |
+
_defaults:
|
| 136 |
+
steps: 12
|
| 137 |
+
cfg: 1.0
|
| 138 |
+
sampler_name: "euler"
|
| 139 |
+
scheduler: "simple"
|
| 140 |
+
|
| 141 |
Qwen-Image-2.1:
|
| 142 |
_defaults:
|
| 143 |
steps: 25
|
yaml/model_list.yaml
CHANGED
|
@@ -20,6 +20,14 @@ Checkpoint:
|
|
| 20 |
unet: "qwen_image_2.1_int8_convrot.safetensors"
|
| 21 |
vae: "qwen_image_2.1_vae_bf16.safetensors"
|
| 22 |
clip: "qwen3vl_8b_nvfp4.safetensors"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 23 |
SenseNova-U1.5:
|
| 24 |
latent_type: hidream_o1_latent
|
| 25 |
models:
|
|
|
|
| 20 |
unet: "qwen_image_2.1_int8_convrot.safetensors"
|
| 21 |
vae: "qwen_image_2.1_vae_bf16.safetensors"
|
| 22 |
clip: "qwen3vl_8b_nvfp4.safetensors"
|
| 23 |
+
Ming-Image:
|
| 24 |
+
latent_type: latent
|
| 25 |
+
models:
|
| 26 |
+
- display_name: "Ming-Image-0.1-Design"
|
| 27 |
+
components:
|
| 28 |
+
unet: "ming_image_0.1_design_int8_convrot.safetensors"
|
| 29 |
+
clip: "ming_image_0.1_ling_mini_2.0_w4a8.safetensors"
|
| 30 |
+
vae: "ming_image_vae_bf16.safetensors"
|
| 31 |
SenseNova-U1.5:
|
| 32 |
latent_type: hidream_o1_latent
|
| 33 |
models:
|
yaml/task_features.yaml
CHANGED
|
@@ -26,6 +26,7 @@ txt2img:
|
|
| 26 |
- qwen_image_edit
|
| 27 |
- reference_image
|
| 28 |
- qwen_image_2_1_prompt_enhancer
|
|
|
|
| 29 |
- pid
|
| 30 |
- vae
|
| 31 |
|
|
@@ -52,6 +53,7 @@ img2img:
|
|
| 52 |
- qwen_image_edit
|
| 53 |
- reference_image
|
| 54 |
- qwen_image_2_1_prompt_enhancer
|
|
|
|
| 55 |
- vae
|
| 56 |
|
| 57 |
inpaint:
|
|
@@ -77,6 +79,7 @@ inpaint:
|
|
| 77 |
- qwen_image_edit
|
| 78 |
- reference_image
|
| 79 |
- qwen_image_2_1_prompt_enhancer
|
|
|
|
| 80 |
- vae
|
| 81 |
|
| 82 |
outpaint:
|
|
@@ -102,6 +105,7 @@ outpaint:
|
|
| 102 |
- qwen_image_edit
|
| 103 |
- reference_image
|
| 104 |
- qwen_image_2_1_prompt_enhancer
|
|
|
|
| 105 |
- vae
|
| 106 |
|
| 107 |
hires_fix:
|
|
@@ -127,4 +131,5 @@ hires_fix:
|
|
| 127 |
- qwen_image_edit
|
| 128 |
- reference_image
|
| 129 |
- qwen_image_2_1_prompt_enhancer
|
|
|
|
| 130 |
- vae
|
|
|
|
| 26 |
- qwen_image_edit
|
| 27 |
- reference_image
|
| 28 |
- qwen_image_2_1_prompt_enhancer
|
| 29 |
+
- ming_image_prompt_enhancer
|
| 30 |
- pid
|
| 31 |
- vae
|
| 32 |
|
|
|
|
| 53 |
- qwen_image_edit
|
| 54 |
- reference_image
|
| 55 |
- qwen_image_2_1_prompt_enhancer
|
| 56 |
+
- ming_image_prompt_enhancer
|
| 57 |
- vae
|
| 58 |
|
| 59 |
inpaint:
|
|
|
|
| 79 |
- qwen_image_edit
|
| 80 |
- reference_image
|
| 81 |
- qwen_image_2_1_prompt_enhancer
|
| 82 |
+
- ming_image_prompt_enhancer
|
| 83 |
- vae
|
| 84 |
|
| 85 |
outpaint:
|
|
|
|
| 105 |
- qwen_image_edit
|
| 106 |
- reference_image
|
| 107 |
- qwen_image_2_1_prompt_enhancer
|
| 108 |
+
- ming_image_prompt_enhancer
|
| 109 |
- vae
|
| 110 |
|
| 111 |
hires_fix:
|
|
|
|
| 131 |
- qwen_image_edit
|
| 132 |
- reference_image
|
| 133 |
- qwen_image_2_1_prompt_enhancer
|
| 134 |
+
- ming_image_prompt_enhancer
|
| 135 |
- vae
|