diff --git a/server/operators/flux_local/manifest.json b/server/operators/flux_local/manifest.json index 1cb32c5..5ccebdd 100644 --- a/server/operators/flux_local/manifest.json +++ b/server/operators/flux_local/manifest.json @@ -2,23 +2,97 @@ "id": "flux_local", "name": "FLUX (local, MLX)", "category": "generate", - "description": "Prompt → image entirely on this Mac via mflux 0.18 (MLX). flux2-klein-4b = Apache/UNGATED, works out of the box, better than FLUX.1-schnell (~15GB first download). schnell-4bit = ungated community quant. schnell/dev/krea-dev/klein-9b are HF-gated (license accept + HF token in Settings). No API cost, no cloud. Elo context: klein ~1083-1119, FLUX.1-dev ~1027, nano-banana ~1154.", + "description": "Prompt -> image entirely on this Mac via mflux (MLX). Covers FLUX.1, FLUX.2 klein, Z-Image Turbo and Qwen-Image, with LoRA support. flux2-klein-4b/schnell-4bit/z-image-turbo are ungated. Z-Image Turbo is the pick for anatomy (correct hands at baseline). Cannot load SD/SDXL LoRAs - those need ComfyUI (comfyui_sd).", "accepts": [], - "produces": ["image"], + "produces": [ + "image" + ], "resources": "gpu", "entry": "run.py", "python": "venvs/mflux/bin/python", "params_schema": { "type": "object", "properties": { - "prompt": {"type": "string", "default": "", "description": "What to generate"}, - "model": {"type": "string", "enum": ["flux2-klein-4b", "flux2-klein-9b", "schnell", "schnell-4bit", "dev", "krea-dev"], "default": "flux2-klein-4b", "description": "klein-4b + schnell-4bit are ungated; the rest need the HF token"}, - "steps": {"type": "integer", "default": 4, "minimum": 1, "maximum": 50, "description": "klein/schnell: 2-4; dev/krea-dev: 20-28"}, - "width": {"type": "integer", "default": 1024, "minimum": 256, "maximum": 2048}, - "height": {"type": "integer", "default": 1024, "minimum": 256, "maximum": 2048}, - "seed": {"type": "integer", "default": 42, "description": "Fixed seed = reproducible A/B tests"}, - "quantize": {"type": "string", "enum": ["none", "8", "4"], "default": "8", "description": "FLUX.1 models only: 8-bit ≈ full quality at half memory"}, - "guidance": {"type": "number", "default": 3.5, "minimum": 0, "maximum": 10, "description": "dev/krea-dev only"} + "prompt": { + "type": "string", + "default": "", + "description": "What to generate" + }, + "model": { + "type": "string", + "enum": [ + "flux2-klein-4b", + "flux2-klein-9b", + "schnell", + "schnell-4bit", + "dev", + "krea-dev", + "z-image-turbo", + "z-image", + "qwen" + ], + "default": "flux2-klein-4b", + "description": "flux2-klein-4b + schnell-4bit + z-image-turbo are UNGATED (work out of the box). schnell/dev/krea-dev/klein-9b need an HF license accept + token. z-image-turbo = Alibaba Z-Image (DiT): best anatomy/hands, few-step (6-10), 4-bit ~6GB. qwen = Qwen-Image. NOTE these are DIFFERENT ARCHITECTURES \u2014 a LoRA only works on its own." + }, + "steps": { + "type": "integer", + "default": 4, + "minimum": 1, + "maximum": 50, + "description": "klein/schnell: 2-4; dev/krea-dev: 20-28" + }, + "width": { + "type": "integer", + "default": 1024, + "minimum": 256, + "maximum": 2048 + }, + "height": { + "type": "integer", + "default": 1024, + "minimum": 256, + "maximum": 2048 + }, + "seed": { + "type": "integer", + "default": 42, + "description": "Fixed seed = reproducible A/B tests" + }, + "quantize": { + "type": "string", + "enum": [ + "none", + "8", + "4" + ], + "default": "8", + "description": "FLUX.1: 4 or 8. Z-Image/Qwen: 3-8 (4 recommended, ~6GB). Ignored for klein." + }, + "guidance": { + "type": "number", + "default": 3.5, + "minimum": 0, + "maximum": 10, + "description": "dev/krea-dev only" + }, + "lora": { + "type": [ + "string", + "array" + ], + "default": "", + "description": "LoRA file(s): full path, or a bare name resolved against MB_LORA_DIRS (localmodels/Lora, civit-lib/{sd15,sdxl}/Lora). Accepts 'name=0.7' shorthand. MUST match the model's architecture \u2014 a mismatched LoRA is a SILENT no-op, not an error." + }, + "lora_weight": { + "type": [ + "number", + "array" + ], + "default": 0.8, + "minimum": 0, + "maximum": 2, + "description": "LoRA strength, scalar or per-LoRA list. 0.6-1.0 is the useful range." + } } } -} +} \ No newline at end of file diff --git a/server/operators/flux_local/manifest.json.bak-20260724-164905 b/server/operators/flux_local/manifest.json.bak-20260724-164905 new file mode 100644 index 0000000..1cb32c5 --- /dev/null +++ b/server/operators/flux_local/manifest.json.bak-20260724-164905 @@ -0,0 +1,24 @@ +{ + "id": "flux_local", + "name": "FLUX (local, MLX)", + "category": "generate", + "description": "Prompt → image entirely on this Mac via mflux 0.18 (MLX). flux2-klein-4b = Apache/UNGATED, works out of the box, better than FLUX.1-schnell (~15GB first download). schnell-4bit = ungated community quant. schnell/dev/krea-dev/klein-9b are HF-gated (license accept + HF token in Settings). No API cost, no cloud. Elo context: klein ~1083-1119, FLUX.1-dev ~1027, nano-banana ~1154.", + "accepts": [], + "produces": ["image"], + "resources": "gpu", + "entry": "run.py", + "python": "venvs/mflux/bin/python", + "params_schema": { + "type": "object", + "properties": { + "prompt": {"type": "string", "default": "", "description": "What to generate"}, + "model": {"type": "string", "enum": ["flux2-klein-4b", "flux2-klein-9b", "schnell", "schnell-4bit", "dev", "krea-dev"], "default": "flux2-klein-4b", "description": "klein-4b + schnell-4bit are ungated; the rest need the HF token"}, + "steps": {"type": "integer", "default": 4, "minimum": 1, "maximum": 50, "description": "klein/schnell: 2-4; dev/krea-dev: 20-28"}, + "width": {"type": "integer", "default": 1024, "minimum": 256, "maximum": 2048}, + "height": {"type": "integer", "default": 1024, "minimum": 256, "maximum": 2048}, + "seed": {"type": "integer", "default": 42, "description": "Fixed seed = reproducible A/B tests"}, + "quantize": {"type": "string", "enum": ["none", "8", "4"], "default": "8", "description": "FLUX.1 models only: 8-bit ≈ full quality at half memory"}, + "guidance": {"type": "number", "default": 3.5, "minimum": 0, "maximum": 10, "description": "dev/krea-dev only"} + } + } +} diff --git a/server/operators/flux_local/run.py b/server/operators/flux_local/run.py index b5337ef..9dd3833 100644 --- a/server/operators/flux_local/run.py +++ b/server/operators/flux_local/run.py @@ -1,6 +1,7 @@ import argparse import json import os +import shlex import subprocess import sys import time @@ -23,11 +24,30 @@ outdir = Path(a.outdir) out_png = outdir / f"flux_{model.replace('/', '_')}_s{p.get('seed', 42)}.png" bindir = Path(sys.executable).parent +# Each family needs a different mflux CLI, and they do NOT accept each other's flags. +# z-image/qwen are separate architectures from FLUX despite living in the same venv. if model.startswith("flux2-"): cli = bindir / "mflux-generate-flux2" model_args = ["--model", model] quant_args = [] # klein is small; skip quantization flags guidance_args = [] # distilled klein: guidance fixed at 1.0 +elif model in ("z-image", "z-image-turbo"): + # Alibaba Z-Image (DiT). filipstrand's 4-bit MLX build runs in ~6GB, which is what + # makes it viable on the 16-32GB boxes. Turbo is few-step: 6-10, not 25. + cli = bindir / ("mflux-generate-z-image-turbo" if model.endswith("turbo") + else "mflux-generate-z-image") + model_args = ["--base-model", model] + if model == "z-image-turbo": + model_args += ["-m", p.get("hf_repo", "filipstrand/Z-Image-Turbo-mflux-4bit")] + quant = str(p.get("quantize", "4")) + quant_args = ["-q", quant] if quant in ("3", "4", "5", "6", "8") else [] + guidance_args = [] +elif model == "qwen": + cli = bindir / "mflux-generate-qwen" + model_args = ["--base-model", "qwen"] + quant = str(p.get("quantize", "4")) + quant_args = ["-q", quant] if quant in ("3", "4", "5", "6", "8") else [] + guidance_args = [] elif model == "schnell-4bit": cli = bindir / "mflux-generate" # ungated community pre-quantized weights (~10GB) — no HF license wall @@ -44,9 +64,66 @@ else: if model in ("dev", "krea-dev") else []) if not cli.exists(): - print(f"ERROR: mflux not installed ({cli} missing). Run scripts/install_mflux.sh") + print(f"ERROR: mflux CLI not installed ({cli} missing). Run scripts/install_mflux.sh") sys.exit(1) +# ---- LoRA ------------------------------------------------------------------- +# THE RULE: a LoRA only works on its own base architecture. Loading an SD1.5 LoRA on +# FLUX, or a FLUX.1 LoRA on FLUX.2, is a SILENT no-op — mflux does not error, the image +# just comes back unchanged and you blame the LoRA. So we fail loudly on a missing file +# and log exactly what was applied. +# +# `lora` accepts a path, a bare filename resolved against LORA_DIRS, or a list. Weights +# come from `lora_weight` (scalar or list, default 0.8). +lora_dirs = [Path(d).expanduser() for d in + os.environ.get("MB_LORA_DIRS", + "~/Documents/localmodels/Lora:~/Documents/civit-lib/sd15/Lora:" + "~/Documents/civit-lib/sdxl/Lora:~/Documents/loras").split(":")] + + +def resolve_lora(name): + q = Path(name).expanduser() + if q.is_file(): + return q + for d in lora_dirs: + c = d / name + if c.is_file(): + return c + if not name.endswith(".safetensors"): + c = d / f"{name}.safetensors" + if c.is_file(): + return c + return None + + +lora_args = [] +raw = p.get("lora") or p.get("loras") or [] +if isinstance(raw, str): + raw = [raw] if raw.strip() else [] +if raw: + weights = p.get("lora_weight", p.get("lora_scales", 0.8)) + if not isinstance(weights, list): + weights = [weights] * len(raw) + paths, scales = [], [] + for i, nm in enumerate(raw): + # allow "name=0.7" shorthand + if isinstance(nm, str) and "=" in nm and not Path(nm).exists(): + nm, _, w = nm.rpartition("=") + try: + weights[i] = float(w) + except ValueError: + pass + hit = resolve_lora(nm) + if not hit: + print(f"ERROR: LoRA not found: {nm}") + print(f" searched: {', '.join(str(d) for d in lora_dirs)}") + sys.exit(1) + paths.append(str(hit)) + scales.append(str(weights[i] if i < len(weights) else 0.8)) + lora_args = ["--lora-paths", *paths, "--lora-scales", *scales] + for pth, sc in zip(paths, scales): + print(f"[lora] {Path(pth).name} @ {sc}", flush=True) + cmd = [str(cli), *model_args, "--prompt", p["prompt"], "--steps", str(steps), @@ -54,9 +131,9 @@ cmd = [str(cli), *model_args, "--height", str(p.get("height", 1024)), "--seed", str(p.get("seed", 42)), "--output", str(out_png), - *quant_args, *guidance_args] + *quant_args, *guidance_args, *lora_args] -print("+", " ".join(cmd), flush=True) +print("+", " ".join(shlex.quote(c) for c in cmd), flush=True) print("(first run per model downloads weights from HuggingFace)", flush=True) env = os.environ.copy() # hf_xet's chunked downloader intermittently fails ("Unable to parse string as @@ -66,8 +143,8 @@ t0 = time.time() res = subprocess.run(cmd, stdout=sys.stdout, stderr=subprocess.STDOUT, env=env) if res.returncode != 0: print("HINT: gated-repo/auth errors mean this model needs an HF license accept " - "+ token (Settings → HuggingFace token). flux2-klein-4b and schnell-4bit " - "are ungated and need nothing.") + "+ token (Settings → HuggingFace token). flux2-klein-4b, schnell-4bit and " + "z-image-turbo are ungated and need nothing.") sys.exit(res.returncode) elapsed = round(time.time() - t0, 1) @@ -81,5 +158,6 @@ if not out_png.exists(): (outdir / "result.json").write_text(json.dumps({"outputs": [ {"path": out_png.name, "meta": {"tool": "mflux", "model": model, "steps": steps, - "seed": p.get("seed", 42), "seconds": elapsed}}]})) + "seed": p.get("seed", 42), "seconds": elapsed, + "loras": [Path(x).name for x in lora_args[1:1 + len(raw)]] if raw else []}}]})) print(f"done in {elapsed}s: {out_png.name}", flush=True) diff --git a/server/operators/flux_local/run.py.bak-20260724-164905 b/server/operators/flux_local/run.py.bak-20260724-164905 new file mode 100644 index 0000000..b5337ef --- /dev/null +++ b/server/operators/flux_local/run.py.bak-20260724-164905 @@ -0,0 +1,85 @@ +import argparse +import json +import os +import subprocess +import sys +import time +from pathlib import Path + +ap = argparse.ArgumentParser() +ap.add_argument("--input", action="append", default=[]) +ap.add_argument("--outdir", required=True) +ap.add_argument("--params", default="{}") +a = ap.parse_args() +p = json.loads(a.params) + +if not p.get("prompt"): + print("ERROR: prompt is required") + sys.exit(1) + +model = p.get("model", "flux2-klein-4b") +steps = int(p.get("steps", 4)) +outdir = Path(a.outdir) +out_png = outdir / f"flux_{model.replace('/', '_')}_s{p.get('seed', 42)}.png" +bindir = Path(sys.executable).parent + +if model.startswith("flux2-"): + cli = bindir / "mflux-generate-flux2" + model_args = ["--model", model] + quant_args = [] # klein is small; skip quantization flags + guidance_args = [] # distilled klein: guidance fixed at 1.0 +elif model == "schnell-4bit": + cli = bindir / "mflux-generate" + # ungated community pre-quantized weights (~10GB) — no HF license wall + model_args = ["--model", "dhairyashil/FLUX.1-schnell-mflux-4bit", + "--base-model", "schnell"] + quant_args = [] + guidance_args = [] +else: + cli = bindir / "mflux-generate" + model_args = ["--model", model] + quant = str(p.get("quantize", "8")) + quant_args = ["--quantize", quant] if quant in ("4", "8") else [] + guidance_args = (["--guidance", str(p.get("guidance", 3.5))] + if model in ("dev", "krea-dev") else []) + +if not cli.exists(): + print(f"ERROR: mflux not installed ({cli} missing). Run scripts/install_mflux.sh") + sys.exit(1) + +cmd = [str(cli), *model_args, + "--prompt", p["prompt"], + "--steps", str(steps), + "--width", str(p.get("width", 1024)), + "--height", str(p.get("height", 1024)), + "--seed", str(p.get("seed", 42)), + "--output", str(out_png), + *quant_args, *guidance_args] + +print("+", " ".join(cmd), flush=True) +print("(first run per model downloads weights from HuggingFace)", flush=True) +env = os.environ.copy() +# hf_xet's chunked downloader intermittently fails ("Unable to parse string as +# hex hash value"); the plain HTTP path is reliable. +env["HF_HUB_DISABLE_XET"] = "1" +t0 = time.time() +res = subprocess.run(cmd, stdout=sys.stdout, stderr=subprocess.STDOUT, env=env) +if res.returncode != 0: + print("HINT: gated-repo/auth errors mean this model needs an HF license accept " + "+ token (Settings → HuggingFace token). flux2-klein-4b and schnell-4bit " + "are ungated and need nothing.") + sys.exit(res.returncode) +elapsed = round(time.time() - t0, 1) + +if not out_png.exists(): + pngs = sorted(outdir.glob("*.png")) + if not pngs: + print("ERROR: no image produced") + sys.exit(1) + out_png = pngs[-1] + +(outdir / "result.json").write_text(json.dumps({"outputs": [ + {"path": out_png.name, + "meta": {"tool": "mflux", "model": model, "steps": steps, + "seed": p.get("seed", 42), "seconds": elapsed}}]})) +print(f"done in {elapsed}s: {out_png.name}", flush=True)