File size: 6,924 Bytes
18c1466
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
#!/usr/bin/env bash
# Paired pipeline: Ling-3.0-flash-VL prompt enhancement -> Ming-Image text-to-image.
#
#   Stage 1  pe_ling.py  caption -> validated structured JSON prompt
#                        (system prompt: assets/t2i_rewriter_system_prompt.txt)
#   Stage 2  infer.py    --task text-to-image --prompt <json file> -> PNG
#                        (infer.py reads --prompt as a file when the path exists)
#
# Artifacts land in --output-dir: enhanced_prompt.json (overwritten per run)
# plus the PNG(s) infer.py writes (image_00.png for text-to-image).
# Fails loudly at every stage (set -Eeuo pipefail + ERR trap + stage checks).
set -Eeuo pipefail
trap 'printf "generate_paired: FAILED at line %d (exit %d)\n" "$LINENO" "$?" >&2' ERR

SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd)"
PYTHON="${PYTHON:-python3}"

# Local llama-server seat serving Ling-3.0-flash-VL on the target box.
DEFAULT_BASE_URL="http://127.0.0.1:8090/v1"
DEFAULT_PE_MODEL="ling-3.0-flash-vl-mtp-halo-STRIX_LEAN"

usage() {
  cat <<'EOF'
Usage: generate_paired.sh --model DIR_OR_REPO CAPTION [options] [-- EXTRA_INFER_ARGS...]

Enhances CAPTION with Ling-3.0-flash-VL (pe_ling.py), validates the structured
JSON rewrite, then renders it with infer.py --task text-to-image.

Required:
  CAPTION                free-form design caption (positional)
  --model DIR_OR_REPO    Ming checkpoint directory or HF repo id
                         (may also be set via the MING_MODEL environment variable)

Passthrough to infer.py (all optional; infer.py defaults in parentheses):
  --resolution N         resolution bucket, 1024 or 2048 for text-to-image;
                         other positive values snap to the nearest bucket (2048)
  --seed N               generation seed (42)
  --steps N              diffusion steps (12)
  --                     everything after this is passed to infer.py verbatim
                         (e.g. -- --validate-only --dtype float16)

Prompt-enhancement endpoint:
  --base-url URL         OpenAI-compatible base URL (http://127.0.0.1:8090/v1)
  --pe-model ID          chat model id served there
                         (ling-3.0-flash-vl-mtp-halo-STRIX_LEAN)
  LITELLM_API_KEY env    exported key is sent as a Bearer token (for a gated
                         OpenAI-compatible gateway such as LiteLLM)

Other:
  --output-dir DIR       artifact directory (outputs/paired)
  -h, --help             this help

Examples:
  ./generate_paired.sh --model /models/Ming-Image-0.1-Design \
      "espresso machine product poster, warm morning light" --resolution 2048

  LITELLM_API_KEY=sk-... ./generate_paired.sh \
      --base-url http://<gateway-host>:4000/v1 --pe-model <gateway-model-name> \
      --model /models/Ming-Image-0.1-Design "a caption" --seed 7
EOF
}

die() {
  printf 'generate_paired: %s\n' "$*" >&2
  exit 1
}

model="${MING_MODEL:-}"
base_url="$DEFAULT_BASE_URL"
pe_model="$DEFAULT_PE_MODEL"
output_dir="outputs/paired"
resolution=""
seed=""
steps=""
caption=""
extra_infer_args=()

while [[ $# -gt 0 ]]; do
  case "$1" in
    --model)      [[ $# -ge 2 ]] || die "--model requires a value";      model="$2";      shift 2 ;;
    --base-url)   [[ $# -ge 2 ]] || die "--base-url requires a value";   base_url="$2";   shift 2 ;;
    --pe-model)   [[ $# -ge 2 ]] || die "--pe-model requires a value";   pe_model="$2";   shift 2 ;;
    --output-dir) [[ $# -ge 2 ]] || die "--output-dir requires a value"; output_dir="$2"; shift 2 ;;
    --resolution) [[ $# -ge 2 ]] || die "--resolution requires a value"; resolution="$2"; shift 2 ;;
    --seed)       [[ $# -ge 2 ]] || die "--seed requires a value";       seed="$2";       shift 2 ;;
    --steps)      [[ $# -ge 2 ]] || die "--steps requires a value";      steps="$2";      shift 2 ;;
    -h|--help)    usage; exit 0 ;;
    --)           shift; extra_infer_args+=("$@"); break ;;
    -*)           usage >&2; die "unknown option: $1" ;;
    *)
      if [[ -n "$caption" ]]; then
        usage >&2
        die "unexpected extra argument: $1 (CAPTION was already given)"
      fi
      caption="$1"
      shift
      ;;
  esac
done

[[ -n "$caption" ]] || { usage >&2; die "CAPTION is required"; }
[[ -n "$model" ]]   || { usage >&2; die "--model DIR_OR_REPO is required (or set MING_MODEL)"; }
if [[ -n "$resolution" && ! "$resolution" =~ ^[0-9]+$ ]]; then
  die "--resolution must be a positive integer, got: $resolution"
fi
if [[ -n "$seed" && ! "$seed" =~ ^-?[0-9]+$ ]]; then
  die "--seed must be an integer, got: $seed"
fi
if [[ -n "$steps" && ! "$steps" =~ ^[0-9]+$ ]]; then
  die "--steps must be a positive integer, got: $steps"
fi
[[ -f "$SCRIPT_DIR/pe_ling.py" ]] || die "missing stage-1 script: $SCRIPT_DIR/pe_ling.py"
[[ -f "$SCRIPT_DIR/infer.py" ]]   || die "missing stage-2 script: $SCRIPT_DIR/infer.py"
command -v "$PYTHON" >/dev/null 2>&1 || die "python interpreter not found: $PYTHON (override with PYTHON=...)"

mkdir -p -- "$output_dir" || die "cannot create output directory: $output_dir"
prompt_json="$output_dir/enhanced_prompt.json"

printf '== stage 1/2: prompt enhancement (pe_ling.py, model %s @ %s)\n' "$pe_model" "$base_url" >&2
"$PYTHON" "$SCRIPT_DIR/pe_ling.py" "$caption" \
  --out "$prompt_json" \
  --base-url "$base_url" \
  --model "$pe_model"
[[ -s "$prompt_json" ]] || die "prompt enhancement produced no prompt file: $prompt_json"

printf '== stage 2/2: Ming-Image text-to-image (infer.py, model %s)\n' "$model" >&2
infer_args=(
  --model "$model"
  --task text-to-image
  --prompt "$prompt_json"
  --output-dir "$output_dir"
)
if [[ -n "$resolution" ]]; then infer_args+=(--resolution "$resolution"); fi
if [[ -n "$seed" ]];       then infer_args+=(--seed "$seed"); fi
if [[ -n "$steps" ]];      then infer_args+=(--steps "$steps"); fi
if [[ ${#extra_infer_args[@]} -gt 0 ]]; then infer_args+=("${extra_infer_args[@]}"); fi
validate_only=0
for arg in ${extra_infer_args[@]+"${extra_infer_args[@]}"}; do
  if [[ "$arg" == "--validate-only" ]]; then validate_only=1; fi
done
"$PYTHON" "$SCRIPT_DIR/infer.py" "${infer_args[@]}"

if [[ "$validate_only" -eq 1 ]]; then
  printf 'generate_paired: --validate-only dry run, no PNG expected; enhanced prompt: %s\n' \
    "$prompt_json" >&2
  exit 0
fi

# infer.py exits non-zero on failure (set -e above); additionally verify the
# promised PNG artifacts actually exist so a silent no-write still fails
# loudly. -newer pins the check to THIS run: stage 2 always writes its PNG
# after stage 1 wrote enhanced_prompt.json, so stale PNGs do not satisfy it.
pngs=()
while IFS= read -r png; do
  pngs+=("$png")
done < <(find "$output_dir" -maxdepth 1 -name '*.png' -type f -newer "$prompt_json" | sort)
if [[ ${#pngs[@]} -eq 0 ]]; then
  die "infer.py exited 0 but wrote no PNG under $output_dir in this run"
fi
printf 'generate_paired: enhanced prompt: %s\n' "$prompt_json" >&2
printf 'generate_paired: %d PNG(s):\n' "${#pngs[@]}" >&2
printf '%s\n' "${pngs[@]}"