-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathdocker-compose.yml
More file actions
311 lines (301 loc) · 10.1 KB
/
Copy pathdocker-compose.yml
File metadata and controls
311 lines (301 loc) · 10.1 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
name: omega
services:
nginx:
build:
context: .
dockerfile: docker/nginx.Dockerfile
container_name: omega-nginx
restart: unless-stopped
mem_limit: "${NGINX_MEMORY_LIMIT:-256m}"
cpus: "${NGINX_CPU_LIMIT:-1.0}"
pids_limit: 128
# Loopback by default. A bare "80:80" is reachable from every
# machine on the coffee shop wifi, and there is no login wall in
# front of Jun worth that. BIND_ADDR=0.0.0.0 opens it up, which
# is also what a public TLS_MODE needs so Let's Encrypt can reach
# the challenge.
ports:
- "${BIND_ADDR:-127.0.0.1}:80:80"
- "${BIND_ADDR:-127.0.0.1}:443:443"
environment:
DOMAIN: "${DOMAIN:-localhost}"
TLS_MODE: "${TLS_MODE:-off}"
OMEGA_EXTRA_HOSTS: "${OMEGA_EXTRA_HOSTS:-}"
volumes:
- letsencrypt:/etc/letsencrypt:ro
- certbot_webroot:/var/www/certbot:ro
- selfsigned:/etc/nginx/selfsigned
depends_on:
php:
condition: service_healthy
healthcheck:
test: ["CMD", "curl", "-fsS", "http://127.0.0.1/health"]
interval: 10s
timeout: 3s
retries: 5
start_period: 10s
security_opt:
- "no-new-privileges:true"
cap_drop: [ALL]
cap_add: [CHOWN, DAC_OVERRIDE, NET_BIND_SERVICE, SETGID, SETUID]
networks: [omega]
php:
build:
context: .
dockerfile: docker/php.Dockerfile
container_name: omega-php
restart: unless-stopped
mem_limit: "${PHP_MEMORY_LIMIT:-1g}"
cpus: "${PHP_CPU_LIMIT:-2.0}"
pids_limit: 256
# NO read_only here, unlike ollama and llamacpp. sync-webapp.sh
# pushes webapp/ in with docker cp and the daemon refuses that
# against a read-only rootfs ("container rootfs is marked
# read-only"), which kills the entire dev loop. php still runs as
# www-data with cap_drop ALL and no-new-privileges.
tmpfs:
- /tmp:size=128m,mode=1777
expose:
- "9000"
environment:
AI_PROVIDER: "${AI_PROVIDER:-ollama}"
OLLAMA_URL: "${OLLAMA_URL:-http://ollama:11434}"
OLLAMA_MODELS_TO_PULL: "${OLLAMA_MODELS_TO_PULL:-hf.co/efficiencyx/Jun-LoRA-E2B-GGUF:Q4_K_M}"
TITLE_MODEL: "${TITLE_MODEL-hf.co/efficiencyx/Titlewen-GGUF:F16}"
OLLAMA_MTP: "${OLLAMA_MTP:-}"
OLLAMA_MTP_MODEL: "${OLLAMA_MTP_MODEL:-jun-mtp}"
OPENROUTER_API_KEY: "${OPENROUTER_API_KEY:-}"
OPENROUTER_MODEL: "${OPENROUTER_MODEL:-}"
LLAMACPP_URL: "${LLAMACPP_URL:-http://llamacpp:8080}"
OMEGA_NUM_CTX: "${OMEGA_NUM_CTX:-}"
OMEGA_TURN_TIMEOUT_S: "${OMEGA_TURN_TIMEOUT_S:-}"
OMEGA_STREAM_IDLE_S: "${OMEGA_STREAM_IDLE_S:-}"
OMEGA_DEV_KEY: "${OMEGA_DEV_KEY:-}"
OMEGA_REGISTRATION_KEY: "${OMEGA_REGISTRATION_KEY:-}"
OMEGA_ALLOWED_HOSTS: "${DOMAIN:-localhost},localhost,127.0.0.1,::1,${OMEGA_EXTRA_HOSTS:-}"
OMEGA_ALLOWED_ORIGINS: "${OMEGA_ALLOWED_ORIGINS:-}"
TRUST_PROXY: "${TRUST_PROXY:-}"
# start.sh reads this on the host, php has no GPU device of
# its own
OMEGA_GPU_VRAM_MB: "${OMEGA_GPU_VRAM_MB:-}"
LLAMACPP_TOOLS: "${LLAMACPP_TOOLS:-}"
FLEE_BANS: "${FLEE_BANS:-on}"
FREE_ROAM: "${FREE_ROAM:-off}"
# KOKORO_URL is the old name, kept so existing .env files
# still work
TTS_URL: "${TTS_URL:-}"
KOKORO_URL: "${KOKORO_URL:-http://tts:8001}"
KARAOKE_URL: "${KARAOKE_URL:-http://karaoke:8001}"
SIDECAR_SECRET: "${SIDECAR_SECRET:-}"
volumes:
- omega_state:/var/lib/omega
- ./tools:/var/www/omega/tools
# no depends_on for the model server or TTS, ON PURPOSE. they
# come and go with their profiles, so they're soft dependencies
healthcheck:
test: ["CMD", "cgi-fcgi", "-bind", "-connect", "127.0.0.1:9000"]
interval: 10s
timeout: 3s
retries: 5
start_period: 15s
security_opt:
- "no-new-privileges:true"
cap_drop: [ALL]
networks: [omega]
ollama:
build:
context: .
dockerfile: docker/ollama.Dockerfile
network: host
container_name: omega-ollama
profiles: [ollama]
restart: unless-stopped
mem_limit: "${OLLAMA_MEMORY_LIMIT:-16g}"
cpus: "${OLLAMA_CPU_LIMIT:-8.0}"
pids_limit: 512
read_only: true
tmpfs:
- /tmp:size=256m,mode=1777
environment:
OLLAMA_MODELS_TO_PULL: "${OLLAMA_MODELS_TO_PULL:-hf.co/efficiencyx/Jun-LoRA-E2B-GGUF:Q4_K_M}"
TITLE_MODEL: "${TITLE_MODEL-hf.co/efficiencyx/Titlewen-GGUF:F16}"
OLLAMA_MTP: "${OLLAMA_MTP:-}"
OLLAMA_MTP_N_MAX: "${OLLAMA_MTP_N_MAX:-4}"
OLLAMA_MTP_MODEL: "${OLLAMA_MTP_MODEL:-jun-mtp}"
OMEGA_NUM_CTX: "${OMEGA_NUM_CTX:-}"
# cap how many requests and models Ollama holds at once. each
# one keeps its own KV cache, the memory of the prompt it
# already read, so they add up fast.
OLLAMA_NUM_PARALLEL: "${OLLAMA_NUM_PARALLEL:-1}"
OLLAMA_MAX_LOADED_MODELS: "${OLLAMA_MAX_LOADED_MODELS:-3}"
OLLAMA_KEEP_ALIVE: "${OLLAMA_KEEP_ALIVE:-5m}"
# q8_0 stores that cache smaller. it needs flash attention,
# without it we get f16.
OLLAMA_FLASH_ATTENTION: "${OLLAMA_FLASH_ATTENTION:-1}"
OLLAMA_KV_CACHE_TYPE: "${OLLAMA_KV_CACHE_TYPE:-q8_0}"
volumes:
- ollama_data:/root/.ollama
healthcheck:
test: ["CMD", "curl", "-fsS", "http://127.0.0.1:11434/api/tags"]
interval: 10s
timeout: 3s
retries: 30
start_period: 30s
security_opt:
- "no-new-privileges:true"
cap_drop: [ALL]
networks: [omega]
llamacpp:
image: ghcr.io/ggml-org/llama.cpp:server@sha256:190813e82f33a82f506e66826f367004a3159f8b8139b11d07566437aecdac93
container_name: omega-llamacpp
profiles: [llamacpp]
restart: unless-stopped
mem_limit: "${LLAMACPP_MEMORY_LIMIT:-16g}"
cpus: "${LLAMACPP_CPU_LIMIT:-8.0}"
pids_limit: 512
read_only: true
tmpfs:
- /tmp:size=256m,mode=1777
environment:
LLAMA_ARG_HF_REPO: "${LLAMACPP_MODEL_HF:-efficiencyx/Jun-LoRA-E2B-GGUF:Q4_K_M}"
LLAMA_ARG_CTX_SIZE: "16384"
LLAMA_ARG_JINJA: "1" # enables template processing for tool calls
LLAMA_ARG_HOST: "0.0.0.0"
LLAMA_ARG_PORT: "8080"
volumes:
- llamacpp_cache:/root/.cache/llama.cpp
healthcheck:
test: ["CMD", "curl", "-fsS", "http://127.0.0.1:8080/health"]
interval: 10s
timeout: 3s
retries: 90
start_period: 60s
security_opt:
- "no-new-privileges:true"
cap_drop: [ALL]
networks: [omega]
tts:
build:
context: .
dockerfile: docker/tts.Dockerfile
network: host
args:
TORCH_INDEX: "${TTS_TORCH_INDEX:-https://download.pytorch.org/whl/cpu}"
container_name: omega-tts
restart: unless-stopped
profiles: [voice]
mem_limit: "${TTS_MEMORY_LIMIT:-8g}"
cpus: "${TTS_CPU_LIMIT:-4.0}"
pids_limit: 256
read_only: true
tmpfs:
- /tmp:size=512m,mode=1777
environment:
SIDECAR_SECRET: "${SIDECAR_SECRET:-}"
TTS_DEVICE: "${TTS_DEVICE:-cpu}"
STT_MODEL: "${STT_MODEL:-base}"
STT_LANG: "${STT_LANG-}"
STT_DEVICE: "${STT_DEVICE:-cpu}"
STT_MAX_DURATION_S: "${STT_MAX_DURATION_S:-120}"
STT_MAX_CONCURRENT: "${STT_MAX_CONCURRENT:-1}"
TTS_MAX_CONCURRENT: "${TTS_MAX_CONCURRENT:-2}"
TTS_MAX_QUEUE: "${TTS_MAX_QUEUE:-8}"
volumes:
- tts_cache:/home/omega/.cache
healthcheck:
test: ["CMD", "curl", "-fsS", "http://127.0.0.1:8001/health"]
interval: 10s
timeout: 5s
retries: 30
start_period: 300s
security_opt:
- "no-new-privileges:true"
cap_drop: [ALL]
cap_add: [CHOWN, DAC_OVERRIDE, SETGID, SETUID]
networks: [omega]
# the same server.py as `tts`, built with demucs (splits a song
# into vocals and backing) and, on the GPU overlays, a CUDA or
# ROCm torch. it sits behind a profile because the image is
# several GB and only karaoke wants it. with the profile off
# api/karaoke.php just can't reach the host, and the webapp
# greys the karaoke button out.
karaoke:
build:
context: .
dockerfile: docker/karaoke.Dockerfile
network: host
args:
TORCH_INDEX: "${KARAOKE_TORCH_INDEX:-https://download.pytorch.org/whl/cpu}"
container_name: omega-karaoke
profiles: [karaoke]
restart: unless-stopped
mem_limit: "${KARAOKE_MEMORY_LIMIT:-12g}"
cpus: "${KARAOKE_CPU_LIMIT:-6.0}"
pids_limit: 384
read_only: true
tmpfs:
- /tmp:size=1g,mode=1777
environment:
SIDECAR_SECRET: "${SIDECAR_SECRET:-}"
SEP_DEVICE: "${SEP_DEVICE:-auto}"
STT_MODEL: "${STT_MODEL:-base}"
STT_LANG: "${STT_LANG-}"
STT_DEVICE: "${STT_DEVICE:-cpu}"
SEP_MAX_DURATION_S: "${SEP_MAX_DURATION_S:-900}"
SEP_MAX_CONCURRENT: "${SEP_MAX_CONCURRENT:-1}"
SEP_MAX_JOBS: "${SEP_MAX_JOBS:-4}"
STT_MAX_CONCURRENT: "${STT_MAX_CONCURRENT:-1}"
volumes:
- karaoke_cache:/home/omega/.cache
healthcheck:
test: ["CMD", "curl", "-fsS", "http://127.0.0.1:8001/health"]
interval: 10s
timeout: 5s
retries: 30
start_period: 300s
security_opt:
- "no-new-privileges:true"
cap_drop: [ALL]
cap_add: [CHOWN, DAC_OVERRIDE, SETGID, SETUID]
networks: [omega]
certbot:
build:
context: .
dockerfile: docker/certbot.Dockerfile
container_name: omega-certbot
profiles: [prod]
restart: unless-stopped
mem_limit: "${CERTBOT_MEMORY_LIMIT:-512m}"
cpus: "${CERTBOT_CPU_LIMIT:-1.0}"
pids_limit: 128
environment:
DOMAIN: "${DOMAIN:-localhost}"
EMAIL: "${EMAIL:-admin@localhost}"
volumes:
- letsencrypt:/etc/letsencrypt
- certbot_webroot:/var/www/certbot
# the challenge is answered by nginx on :80, so asking before it's
# up is a guaranteed failed validation
depends_on:
nginx:
condition: service_healthy
security_opt:
- "no-new-privileges:true"
cap_drop: [ALL]
networks: [omega]
volumes:
ollama_data:
llamacpp_cache:
tts_cache:
# Pinned to the old volume name so an install that already exists
# keeps the weights it downloaded, 300MB and up, instead of
# fetching them again.
name: omega_kokoro_cache
karaoke_cache:
omega_state:
letsencrypt:
certbot_webroot:
selfsigned:
networks:
omega:
name: omega