-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathstart.sh
More file actions
executable file
·507 lines (467 loc) · 19.7 KB
/
Copy pathstart.sh
File metadata and controls
executable file
·507 lines (467 loc) · 19.7 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
#!/usr/bin/env bash
set -euo pipefail
cd "$(dirname "$0")"
detect_gpu() {
case "${GPU:-auto}" in
nvidia|amd|cpu) echo "$GPU"; return ;;
esac
if { command -v nvidia-smi >/dev/null 2>&1 && nvidia-smi -L >/dev/null 2>&1; } \
|| [ -e /proc/driver/nvidia/version ]; then
echo nvidia
elif [ -e /dev/kfd ] && compgen -G "/dev/dri/renderD*" >/dev/null; then
echo amd
else
echo cpu
fi
}
# Sort by VRAM so the biggest card ends up as device 0. it has to
# be UUIDs and NOT indices, nvidia-smi counts cards in slot order
# while CUDA sorts them fastest first, so index 1 means a
# different card to each of them.
nvidia_visible() {
nvidia-smi --query-gpu=memory.total,uuid --format=csv,noheader,nounits 2>/dev/null \
| sort -t, -k1 -nr | cut -d, -f2 | tr -d ' \r' | paste -sd, - || true
}
nvidia_count() {
nvidia-smi --query-gpu=uuid --format=csv,noheader 2>/dev/null | grep -c . || true
}
# VRAM on the biggest card, in MiB. php has no GPU device of its
# own so this is the ONLY way it finds out, see default_num_ctx()
# in api/lib/providers/context.php.
nvidia_vram_mb() {
nvidia-smi --query-gpu=memory.total --format=csv,noheader,nounits 2>/dev/null \
| sort -nr | head -n1 | tr -d ' \r' || true
}
amd_vram_mb() {
rocm-smi --showmeminfo vram --csv 2>/dev/null \
| awk -F, 'NR>1 { gsub(/[^0-9]/,"",$2); if ($2 != "") print int($2/1048576) }' \
| sort -nr | head -n1 || true
}
amd_visible() {
rocm-smi --showmeminfo vram --csv 2>/dev/null \
| awk -F, 'NR>1 { gsub(/[^0-9]/,"",$1); gsub(/[^0-9]/,"",$2); if ($2 != "") print $2","$1 }' \
| sort -t, -k1 -nr | cut -d, -f2 | paste -sd, - || true
}
# the card the MTP tune was measured on, as one string. MTP is
# multi-token prediction, a small drafter model guesses the next
# few tokens (how many = the draft depth) and the chat model
# checks them all in one pass. string is the vendor, then every
# GPU's name and how much VRAM it has. sorted biggest card first,
# so moving cards between slots is not a change, only a real
# swap is.
#
# the AMD half takes VRAM and nothing else. rocm-smi moves its
# product-name columns around between versions, and a name read
# out of the wrong column would make every boot look like a new
# card. missing a swap between two cards of the same size is the
# cheaper mistake. prints nothing when neither tool is here, an
# empty string is how the callers know we could not tell.
gpu_signature() {
if command -v nvidia-smi >/dev/null 2>&1; then
nvidia-smi --query-gpu=name,memory.total --format=csv,noheader,nounits 2>/dev/null \
| sed -e 's/\r//' -e 's/[[:space:]]*,[[:space:]]*/:/' \
| sort -t: -k2 -nr \
| awk 'NF { s = s (s ? "," : "") $0 } END { if (s) print "nvidia:" s }'
elif command -v rocm-smi >/dev/null 2>&1; then
rocm-smi --showmeminfo vram --csv 2>/dev/null \
| awk -F, 'NR>1 { gsub(/[^0-9]/,"",$2); if ($2 != "") print int($2/1048576) }' \
| sort -nr \
| awk 'NF { s = s (s ? "," : "") $0 } END { if (s) print "amd:" s }'
fi
}
# the model servers sit behind compose profiles, `ollama` runs
# the Ollama one and `llamacpp` the llama.cpp one. we MERGE with
# whatever the shell already set, like
# COMPOSE_PROFILES=prod ./start.sh, and whatever .env set. never
# replace it. then add what AI_PROVIDER implies on top. that way
# an old .env from before providers existed still boots ollama.
env_get() { sed -n "s/^$1=//p" .env 2>/dev/null | tail -n1 || true; }
add_profile() {
case ",${profiles}," in *,"$1",*) ;; *) profiles="${profiles:+$profiles,}$1" ;; esac
}
profiles="${COMPOSE_PROFILES:-}"
file_profiles="$(env_get COMPOSE_PROFILES)"
for p in ${file_profiles//,/ }; do add_profile "$p"; done
provider="$(env_get AI_PROVIDER)"; provider="${provider:-ollama}"
llamacpp_url="$(env_get LLAMACPP_URL)"
case "$provider" in
llamacpp)
# We run llama-server ourselves unless you pointed us at your own.
case "${llamacpp_url:-http://llamacpp:8080}" in
http://llamacpp:8080) add_profile llamacpp ;;
esac
llamacpp_model_file="$(env_get LLAMACPP_MODEL_FILE)"
if [ -n "$llamacpp_model_file" ]; then
export LLAMACPP_MODEL_FILE="$llamacpp_model_file"
llamacpp_models_dir="$(env_get LLAMACPP_MODELS_DIR)"
export LLAMACPP_MODELS_DIR="${llamacpp_models_dir:-./models}"
if [ ! -f "$LLAMACPP_MODELS_DIR/$LLAMACPP_MODEL_FILE" ]; then
echo "error: LLAMACPP_MODEL_FILE not found: $LLAMACPP_MODELS_DIR/$LLAMACPP_MODEL_FILE" >&2
exit 1
fi
llamacpp_alias="$(env_get LLAMACPP_MODEL_ALIAS)"
[ -z "$llamacpp_alias" ] || export LLAMACPP_MODEL_ALIAS="$llamacpp_alias"
fi
llamacpp_mtp="$(env_get LLAMACPP_MTP)"
if [ -n "$llamacpp_mtp" ]; then
export LLAMACPP_MTP="$llamacpp_mtp"
llamacpp_mtp_n_max="$(env_get LLAMACPP_MTP_N_MAX)"
[ -z "$llamacpp_mtp_n_max" ] || export LLAMACPP_MTP_N_MAX="$llamacpp_mtp_n_max"
fi
;;
openrouter) : ;;
*)
# We run ollama ourselves unless you pointed us at your own.
case "$(env_get OLLAMA_URL)" in
''|http://ollama:11434) add_profile ollama ;;
esac
;;
esac
voice="${VOICE:-$(env_get VOICE)}"
case "$(printf '%s' "${voice:-on}" | tr '[:upper:]' '[:lower:]')" in
off|0|false|no) ;;
*) add_profile voice ;;
esac
karaoke="${KARAOKE:-$(env_get KARAOKE)}"
case "$(printf '%s' "$karaoke" | tr '[:upper:]' '[:lower:]')" in
on|1|true|yes) add_profile karaoke ;;
esac
export COMPOSE_PROFILES="$profiles"
echo "AI provider: $provider${profiles:+ (compose profiles: $profiles)}"
case ",${profiles}," in
*,karaoke,*) ;;
*) echo "karaoke: off (set KARAOKE=on in .env to build its sidecar)" ;;
esac
# Compose publishes on BIND_ADDR, loopback unless somebody
# changed it. Say which it is, out loud, every start. "it's only
# on my machine" is the kind of thing people believe long after
# it stopped being true.
bind_addr="${BIND_ADDR:-$(env_get BIND_ADDR)}"
bind_addr="${bind_addr:-127.0.0.1}"
export BIND_ADDR="$bind_addr"
case "$bind_addr" in
127.0.0.1|localhost|::1) echo "listening on: $bind_addr (this machine only)" ;;
*) echo "listening on: $bind_addr - anything that can reach this box can open Jun" ;;
esac
# install.sh and install.ps1 both write this on first run, but
# cp .env.example .env && ./start.sh never went through either,
# and that box has open signup until someone notices. same rule
# as the installers, only when the line is MISSING. an empty
# OMEGA_REGISTRATION_KEY= is the operator saying "off".
if [ -f .env ] && ! grep -qE '^OMEGA_REGISTRATION_KEY=' .env; then
reg_key="$(openssl rand -hex 16 2>/dev/null || head -c16 /dev/urandom | od -An -tx1 | tr -d ' \n')"
printf 'OMEGA_REGISTRATION_KEY=%s\n' "$reg_key" >> .env
echo "registration key: $reg_key (written to .env, the first account skips it, everyone after needs it)"
fi
# same shape for the header php shows the tts/karaoke sidecars.
# an empty SIDECAR_SECRET= is honoured too, the sidecar then runs
# on its Host allowlist alone and warns about it at startup.
if [ -f .env ] && ! grep -qE '^SIDECAR_SECRET=' .env; then
printf 'SIDECAR_SECRET=%s\n' "$(openssl rand -hex 32 2>/dev/null || head -c32 /dev/urandom | od -An -tx1 | tr -d ' \n')" >> .env
fi
# nginx and php both refuse a Host they don't know (444 and 421),
# so opening the phone at https://192.168.1.42 needs that exact
# address in the allowlist. the containers can't work it out
# themselves, all they see is the docker bridge, so we read the
# host's own private v4 addresses here and hand them down. only
# when BIND_ADDR is off loopback: on the default install nothing
# outside this box can connect anyway, so widening the list buys
# nothing. 10.*, 172.16-31.* and 192.168.* only, and never the
# docker bridges, or we'd be naming addresses that aren't ours to
# answer for. DHCP moves these, so a new lease means a restart.
# OMEGA_EXTRA_HOSTS stays for anything we can't guess: an mDNS
# name, a tailscale address, whatever the proxy calls you.
lan_hosts() {
command -v ip >/dev/null 2>&1 || return 0
ip -4 -o addr show scope global up 2>/dev/null | awk '
$2 ~ /^(docker|br-|veth|virbr)/ { next }
{ split($4, a, "/")
if (a[1] ~ /^10\./ || a[1] ~ /^192\.168\./ || a[1] ~ /^172\.(1[6-9]|2[0-9]|3[01])\./) print a[1] }'
}
case "$bind_addr" in
127.0.0.1|127.*|localhost|::1) ;;
*)
extra_hosts="${OMEGA_EXTRA_HOSTS:-$(env_get OMEGA_EXTRA_HOSTS)}"
detected="$(lan_hosts | tr '\n' ' ')"
# commas out FIRST. php takes either, nginx's server_name only
# takes spaces and would happily register a host called
# "jun.local,".
extra_hosts="$(printf '%s %s' "$extra_hosts" "$detected" | tr ',' ' ' | tr -s ' ' | sed 's/^ //; s/ $//')"
export OMEGA_EXTRA_HOSTS="$extra_hosts"
[ -z "$detected" ] || echo "reachable as: $(printf '%s' "$detected" | sed 's/ $//; s/[^ ]*/https:\/\/&/g')"
;;
esac
tls_mode="${TLS_MODE:-$(env_get TLS_MODE)}"
tls_mode_normalized="$(printf '%s' "${tls_mode:-off}" | tr '[:upper:]' '[:lower:]')"
case "$tls_mode_normalized" in
off|"") ;;
*) case "$bind_addr" in
127.0.0.1|localhost|::1)
echo "note: TLS_MODE=$tls_mode but we only listen on $bind_addr, so Let's Encrypt can't reach the challenge. set BIND_ADDR=0.0.0.0 in .env." ;;
esac ;;
esac
export TLS_MODE="${tls_mode:-off}"
tts_device="${TTS_DEVICE:-$(env_get TTS_DEVICE)}"
export TTS_DEVICE="${tts_device:-cpu}"
# The GPU overlays build the karaoke sidecar with a CUDA or ROCm
# torch. when separation is set to run on the CPU we pin the CPU
# index instead, so we don't pull down a multi-GB wheel for
# hardware nobody asked to use.
sep_device="${SEP_DEVICE:-$(env_get SEP_DEVICE)}"
sep_device="${sep_device:-auto}"
karaoke_torch_index="${KARAOKE_TORCH_INDEX:-$(env_get KARAOKE_TORCH_INDEX)}"
if [ "$(printf '%s' "$sep_device" | tr '[:upper:]' '[:lower:]')" = cpu ] \
&& [ -z "$karaoke_torch_index" ]; then
karaoke_torch_index=https://download.pytorch.org/whl/cpu
fi
export SEP_DEVICE="$sep_device"
[ -z "$karaoke_torch_index" ] || export KARAOKE_TORCH_INDEX="$karaoke_torch_index"
gpu="$(detect_gpu)"
files=(-f docker-compose.yml)
[ -z "${LLAMACPP_MODEL_FILE:-}" ] || files+=(-f docker-compose.llamacpp-local.yml)
[ -z "${LLAMACPP_MTP:-}" ] || files+=(-f docker-compose.llamacpp-mtp.yml)
case "$gpu" in
nvidia)
files+=(-f docker-compose.nvidia.yml)
if ! command -v nvidia-smi >/dev/null 2>&1; then
echo "warning: NVIDIA selected but nvidia-smi not found - you also need" >&2
echo " nvidia-container-toolkit installed, or run GPU=cpu ./start.sh" >&2
fi
ngpus="$(nvidia_count)"
if [ "${ngpus:-0}" -gt 0 ]; then
export NVIDIA_GPU_COUNT="$ngpus"
fi
probed_vram_mb="$(nvidia_vram_mb)"
;;
amd)
files+=(-f docker-compose.amd.yml)
# the container has to be in the host groups that own the GPU
# device nodes, or it can't open them.
vgid="$(getent group video | cut -d: -f3 || true)"
rgid="$(stat -c '%g' /dev/dri/renderD* 2>/dev/null | head -n1 || true)"
[ -n "$rgid" ] || rgid="$(getent group render | cut -d: -f3 || true)"
export VIDEO_GID="${vgid:-44}"
export RENDER_GID="${rgid:-105}"
probed_vram_mb="$(amd_vram_mb)"
;;
esac
# A hand-set value always wins: probing reports the whole card,
# which is wrong when something else on the machine permanently
# owns part of it.
vram_mb="${OMEGA_GPU_VRAM_MB:-$(env_get OMEGA_GPU_VRAM_MB)}"
vram_mb="${vram_mb:-${probed_vram_mb:-}}"
[ -z "$vram_mb" ] || export OMEGA_GPU_VRAM_MB="$vram_mb"
devices="${GPU_DEVICES:-$(env_get GPU_DEVICES)}"
case "$devices" in
""|auto)
case "$gpu" in
nvidia) devices="$(nvidia_visible)" ;;
amd) devices="$(amd_visible)" ;;
*) devices="" ;;
esac
;;
all) devices="" ;;
esac
if [ -n "$devices" ]; then
case "$gpu" in
nvidia) export CUDA_VISIBLE_DEVICES="$devices" ;;
amd) export HIP_VISIBLE_DEVICES="$devices" ROCR_VISIBLE_DEVICES="$devices" GGML_VK_VISIBLE_DEVICES="$devices" ;;
esac
fi
tp="${TENSOR_PARALLEL:-$(env_get TENSOR_PARALLEL)}"
case "$(printf '%s' "$tp" | tr '[:upper:]' '[:lower:]')" in
on|1|true|yes) tp=on ;;
*) tp=off ;;
esac
if [ "$tp" = on ]; then
export OLLAMA_SCHED_SPREAD="${OLLAMA_SCHED_SPREAD:-1}"
if [ "$gpu" = nvidia ]; then
export LLAMA_ARG_SPLIT_MODE="${LLAMA_ARG_SPLIT_MODE:-row}"
fi
fi
echo "GPU detected: $gpu"
if [ "$gpu" = amd ]; then
echo " video gid=$VIDEO_GID, render gid=$RENDER_GID${HSA_OVERRIDE_GFX_VERSION:+, HSA_OVERRIDE_GFX_VERSION=$HSA_OVERRIDE_GFX_VERSION}"
fi
if [ -n "$devices" ]; then
echo " devices (largest VRAM first): $devices"
fi
if [ "$tp" = on ]; then
echo " tensor parallelism: on"
fi
if [ -n "${OMEGA_GPU_VRAM_MB:-}" ]; then
echo " vram: ${OMEGA_GPU_VRAM_MB} MiB"
fi
# ollama picks the layer split at load time and then it's pinned
# (see default_num_ctx() and the keep_alive=-1 pin in
# api/lib/providers/). so a model that loads while the karaoke
# sidecar's CUDA torch is still initialising stays mostly on the
# CPU. that's ~1000x slower on prefill, the pass that reads the
# prompt in. karaoke waits until the model server answers.
wait_for_ollama() {
local i status
for i in $(seq 1 90); do
status="$(docker inspect -f '{{.State.Health.Status}}' omega-ollama 2>/dev/null || true)"
[ "$status" = healthy ] && return 0
sleep 2
done
echo "warning: omega-ollama did not report healthy; starting karaoke anyway" >&2
}
# "is she even on the right card" is the first thing anyone asks
# after an install, and the only thing that actually knows is
# ollama's own startup log. a box with an iGPU next to a real one
# is where this bites: we hand down a device list sorted biggest
# VRAM first, ollama picks from it, and nothing has ever said out
# loud which one it took. so say it. waits up to 20s for the
# line, then gives up without a word. the model server pulling
# an 8 GB fine-tune on first boot is not an error.
report_gpu_placement() {
local i line
[ "$gpu" != cpu ] || return 0
case ",${profiles}," in *,ollama,*) ;; *) return 0 ;; esac
for i in $(seq 1 10); do
line="$(docker logs omega-ollama 2>&1 | grep 'inference compute' || true)"
[ -n "$line" ] && break
sleep 2
done
[ -n "$line" ] || return 0
echo "ollama is running on:"
printf '%s\n' "$line" | awk '{
lib = ""; nm = ""; tot = "";
if (match($0, /library=[^ ]+/)) lib = substr($0, RSTART + 8, RLENGTH - 8);
# 0.32 moved the card name to description= and left name=CUDA0
# unquoted. older builds only have name="...". take either.
if (match($0, /description="[^"]*"/)) nm = substr($0, RSTART + 13, RLENGTH - 14);
else if (match($0, /name="[^"]*"/)) nm = substr($0, RSTART + 6, RLENGTH - 7);
if (match($0, /total="[^"]*"/)) tot = substr($0, RSTART + 7, RLENGTH - 8);
# a CPU-only line is description=cpu, unquoted, so nm stays empty
if (lib == "cpu") print " CPU. no GPU at all, ollama gave up on the card";
else if (nm != "") printf " %s (%s%s)\n", nm, lib, (tot != "" ? ", " tot : "");
}'
if printf '%s\n' "$line" | grep -q 'library=cpu'; then
echo " nvidia-smi working != CUDA working. a stale /etc/cdi/nvidia.yaml hands the" >&2
echo " container the wrong nvidia-uvm major (it moves between boots) and cuInit dies" >&2
echo " with 999. regen it: sudo nvidia-ctk cdi generate --output=/etc/cdi/nvidia.yaml" >&2
return 0
fi
echo " wrong card? pin it with GPU_DEVICES= in .env, biggest-VRAM-first is only our guess."
}
# The draft depth in .env is a measurement, and it only describes
# the card it was measured on. Swap the GPU and the number in
# there is about hardware that left the building, so hold the
# stamp the tuner wrote against what is in the box now and measure
# again when the two don't match.
mtp_recheck() {
local want tuned sig drafter server ready i
# On the llamacpp side the tuner restarts the stack through
# ./start.sh, and that is us. without this we start a tune inside
# a tune, forever.
[ -z "${MTP_AUTOTUNE_RUNNING:-}" ] || return 0
want="${MTP_AUTOTUNE:-$(env_get MTP_AUTOTUNE)}"
case "$(printf '%s' "$want" | tr '[:upper:]' '[:lower:]')" in
off|0|false|no) return 0 ;;
esac
# No stamp means no tune ever finished on this box, so there is
# nothing to compare and nothing to nag about.
tuned="$(env_get MTP_TUNED_GPU)"
[ -n "$tuned" ] || return 0
sig="$(gpu_signature)"
[ -n "$sig" ] || return 0
[ "$sig" != "$tuned" ] || return 0
case "$provider" in
llamacpp) drafter="$(env_get LLAMACPP_MTP)"; server=llamacpp ;;
ollama|'') drafter="$(env_get OLLAMA_MTP)"; server=ollama ;;
*) return 0 ;;
esac
# MTP is off, so there is no depth to measure. the tuner would
# only die on it and we would come back here every single boot.
[ -n "$drafter" ] || return 0
# Point us at your own model server and none of this is ours to
# tune, there is no omega- container to wait on and the tuner
# wants one anyway. Both waits below would just run out their
# clocks, every boot, and tell you nothing.
case ",${profiles}," in *,"$server",*) ;; *) return 0 ;; esac
echo ""
echo "the GPU changed since MTP was tuned:"
echo " tuned on: $tuned"
echo " here now: $sig"
echo "the draft depth in .env was measured on a card that is not in this box any"
echo "more, so we are measuring it again. a few minutes on ollama, considerably"
echo "longer on llamacpp where every depth needs a llama-server restart."
echo "set MTP_AUTOTUNE=off in .env to skip this."
# ollama's healthcheck is /api/tags, and that answers the moment
# the server is up, long before ollama-entrypoint.sh has finished
# pulling the chat model and the drafter. The tuner needs the
# drafter blob on disk or it dies with "could not find the
# drafter blob", so wait untill ollama admits it has one.
ready=
case "$provider" in
llamacpp)
for i in $(seq 1 60); do
if [ "$(docker inspect -f '{{.State.Health.Status}}' omega-llamacpp 2>/dev/null || true)" = healthy ]; then
ready=1; break
fi
sleep 5
done
;;
*)
wait_for_ollama
for i in $(seq 1 60); do
if docker exec omega-ollama ollama show --modelfile "$drafter" >/dev/null 2>&1; then
ready=1; break
fi
sleep 5
done
;;
esac
# Leave the stamp stale on purpose. it is the only thing that
# makes the next boot try again, and a pull that is still running
# now is probably done by then.
if [ -z "$ready" ]; then
echo "warning: the drafter is not here yet, so the re-tune is skipped." >&2
echo " run ./mtp-autotune.sh by hand once it has finished pulling." >&2
return 0
fi
if ! MTP_AUTOTUNE_RUNNING=1 ./mtp-autotune.sh; then
echo "warning: autotune did not finish, run ./mtp-autotune.sh by hand" >&2
fi
}
staged_up() {
case ",${profiles}," in
*,karaoke,*) ;;
*) return 1 ;;
esac
case ",${profiles}," in
*,ollama,*) ;;
*) return 1 ;;
esac
[ "$#" -eq 0 ]
}
# No exec here. it replaces the shell, so nothing after the
# compose call gets to run, mtp_recheck included. set -e still
# takes a compose failure out on the spot, with compose's own
# exit code.
up_and_check() {
set -x; docker compose "${files[@]}" up -d --build "$@"
{ set +x; } 2>/dev/null
report_gpu_placement
mtp_recheck
}
# a bare first word is a lifecycle subcommand. anything else (a
# flag like --build, or service names) goes straight through to
# `up -d`.
case "${1:-up}" in
stop|down) shift; set -x; exec docker compose "${files[@]}" down "$@" ;;
restart) shift; docker compose "${files[@]}" down
up_and_check "$@" ;;
status|ps) shift; set -x; exec docker compose "${files[@]}" ps "$@" ;;
logs) shift; set -x; exec docker compose "${files[@]}" logs -f "$@" ;;
*) if staged_up "$@"; then
set -x
docker compose "${files[@]}" up -d --build --scale karaoke=0
{ set +x; } 2>/dev/null
wait_for_ollama
set -x
fi
up_and_check "$@" ;;
esac