Repository navigation
Expand file tree
/
Copy pathdocker-compose.yml
More file actions
404 lines (388 loc) · 13.9 KB
/
Copy pathdocker-compose.yml
File metadata and controls
404 lines (388 loc) · 13.9 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
# ML4T 3rd Edition — Docker Compose Configuration
#
# QUICK START (pre-built images from Docker Hub):
# docker compose pull ml4t # Pull pre-built image (~12 GB amd64, ~3 GB arm64)
# docker compose up ml4t # Start Jupyter Lab at http://localhost:8888
#
# To build locally instead:
# docker compose build ml4t # Build from Dockerfile (slow, ~45 min)
#
# PLATFORM SUPPORT:
# - Linux x86_64: All services work, GPU optional
# - Windows x86_64: All services via Docker Desktop (WSL2), GPU via WSL2+nvidia-toolkit
# - macOS Intel: All services except GPU
# - Apple Silicon: ml4t + benchmark native arm64; ml4t-py312 amd64 only, runs under
# Rosetta emulation (see below)
#
# IMAGES ON DOCKER HUB (docker.io/ml4t/):
# ml4t:latest — All chapters, Python 3.14, PyTorch CUDA 12.8 (amd64+arm64)
# ml4t-py312:latest — Python 3.12: signatory, esig, gensim, tfcausalimpact (amd64 only)
# ml4t-benchmark:latest — Storage benchmarks: DuckDB, HDF5, DB clients (amd64+arm64)
#
# WHICH IMAGE DO I NEED?
# Most readers: ml4t only (covers Ch01-Ch27 + case studies)
# Ch05 NB01/03/07, Ch09 NB06/12, Ch10 NB01-03, Ch12 NB10, Ch14 NB06, Ch15 NB06:
# ml4t-py312 (amd64; on Apple Silicon read the committed .ipynb outputs, or run it
# under Rosetta with DOCKER_DEFAULT_PLATFORM=linux/amd64)
# Ch02 storage benchmarks: benchmark + database services
# Ch12 GBM GPU benchmark: rapids (requires NVIDIA GPU)
#
# See docs/installation.md for detailed setup instructions
x-common: &common
user: "${UID:-1000}:${GID:-1000}"
volumes:
- .:/app:rw
# Read-write so the in-container download workflow (data/download_all.py)
# can populate /data. ML4T_DATA_PATH=/data below points every downloader
# here; a read-only mount makes each write fail with "Read-only file system".
- ${ML4T_DATA_PATH:-./data}:/data:rw
# Every notebook that pins a HuggingFace checkpoint loads it with
# local_files_only=True, so a container run has to see the host's cache or the
# load raises LocalEntryNotFoundError. Read-write, because the notebooks that do
# not pin a checkpoint fetch it on first use (10_text_feature_engineering/09
# loads FinBERT through a bare from_pretrained), and a read-only mount turns
# that download into a failure. What a pinned load must not do is silently
# substitute a different checkpoint, and `revision=` with local_files_only is
# what prevents that, not the mount mode.
#
# The host directory has to exist before the first `docker compose` command.
# Docker creates a missing bind source as root, and the container runs as the
# invoking user, so an uncached download then fails on permissions despite the
# :rw. `mkdir -p ~/.cache/huggingface` once is the whole fix; any host that has
# run a HuggingFace library already has it.
- ${HF_HOME:-${HOME}/.cache/huggingface}:/home/ml4t/.cache/huggingface:rw
environment:
- ML4T_DATA_PATH=/data
- ML4T_PATH=/app
- TEST=${TEST:-0}
- NEO4J_URI=${NEO4J_URI:-bolt://neo4j:7687}
- NEO4J_USER=${NEO4J_USER:-neo4j}
- NEO4J_PASSWORD=${NEO4J_PASSWORD:-password}
- NUMBA_CACHE_DIR=/tmp/numba_cache
- MPLCONFIGDIR=/tmp/matplotlib
- POLARS_FMT_MAX_ROWS=20
- POLARS_FMT_STR_LEN=50
- VIRTUAL_ENV=/opt/ml4t
- UV_PROJECT_ENVIRONMENT=/opt/ml4t
- PATH=/opt/ml4t/bin:/usr/local/bin:/usr/bin:/bin:/usr/local/games:/usr/games
env_file:
- path: .env
required: false
working_dir: /app
stdin_open: true
tty: true
services:
# ============================================
# ML4T MAIN (All Chapters)
# Works on all platforms including Apple Silicon
# ============================================
ml4t:
<<: *common
image: ml4t/ml4t:latest
build:
context: .
dockerfile: envs/ml4t/Dockerfile
container_name: ml4t-review
ports:
- "127.0.0.1:8888:8888"
command: ["jupyter", "lab", "--ip=0.0.0.0", "--port=8888", "--no-browser", "--allow-root"]
# ============================================
# ML4T GPU (NVIDIA only — Linux/Windows x86)
# Same as ml4t but with GPU passthrough
# ============================================
ml4t-gpu:
<<: *common
image: ml4t/ml4t:latest
build:
context: .
dockerfile: envs/ml4t/Dockerfile
container_name: ml4t-review-gpu
ports:
- "127.0.0.1:8889:8888"
command: ["jupyter", "lab", "--ip=0.0.0.0", "--port=8888", "--no-browser", "--allow-root"]
profiles: ["gpu"]
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
# ============================================
# BENCHMARK (ARM64 Compatible)
# Storage benchmarks: Parquet, DuckDB, HDF5, database clients
# Excludes ArcticDB (no ARM64 wheels)
# Works on Apple Silicon
# ============================================
benchmark:
<<: *common
image: ml4t/ml4t-benchmark:latest
build:
context: .
dockerfile: envs/benchmark/Dockerfile
container_name: ml4t-benchmark
ports:
- "127.0.0.1:8887:8888"
profiles: ["benchmark"]
# Re-declares the x-common volumes because a service-level `volumes:` key
# replaces the anchor's rather than extending it.
volumes:
- .:/app:rw
- ${ML4T_DATA_PATH:-./data}:/data:rw
# kdb+/PyKX licence + q binary, mounted read-only from the host at RUNTIME.
# Never baked into the image: the licence is personal and must not ship in
# a published image. If you have no kdb+ licence these resolve to empty
# directories, PyKX finds no licence, and 21_storage_benchmark_database
# skips kdb+ and says so. Get a free personal edition from
# https://kx.com/kdb-personal-edition-download/ and unpack it to ~/.kx.
- ${KX_HOME:-${HOME}/.kx}:/home/ml4t/.kx:ro
- ${PYKX_HOME:-${HOME}/.pykx}:/home/ml4t/.pykx:ro
environment:
- ML4T_DATA_PATH=/data
- ML4T_PATH=/app
- TEST=${TEST:-0}
# q resolves its licence from QLIC, not from the file's presence on disk;
# without this q starts and immediately fails "no license loaded".
- QLIC=/home/ml4t/.kx
- QHOME=/home/ml4t/.kx/q
# Database connections (when running with database services)
- CLICKHOUSE_HOST=ml4t-clickhouse
- CLICKHOUSE_PORT=8123
- QUESTDB_HOST=ml4t-questdb
- QUESTDB_HTTP_PORT=9000
- TIMESCALE_HOST=ml4t-timescaledb
- TIMESCALE_PORT=5432
- TIMESCALE_PASSWORD=benchmark
- INFLUXDB_HOST=ml4t-influxdb
- INFLUXDB_PORT=8086
- INFLUXDB_ORG=ml4t
- INFLUXDB_TOKEN=benchmark-token
- INFLUXDB_BUCKET=market_data
- POSTGRES_HOST=ml4t-postgres
- POSTGRES_PORT=5432
- POSTGRES_USER=postgres
- POSTGRES_PASSWORD=benchmark
- POSTGRES_DB=ml4t
# ============================================
# BENCHMARK FULL (x86 Only)
# All benchmarks including ArcticDB
# WARNING: Does not work on Apple Silicon
# ============================================
benchmark-full:
<<: *common
build:
context: .
dockerfile: envs/benchmark/Dockerfile.full
# Force x86 build (required for ArcticDB)
platforms:
- linux/amd64
container_name: ml4t-benchmark-full
ports:
- "127.0.0.1:8887:8888"
profiles: ["benchmark-full"]
environment:
- ML4T_DATA_PATH=/data
- ML4T_PATH=/app
- TEST=${TEST:-0}
# Database connections
- CLICKHOUSE_HOST=ml4t-clickhouse
- CLICKHOUSE_PORT=8123
- QUESTDB_HOST=ml4t-questdb
- QUESTDB_HTTP_PORT=9000
- TIMESCALE_HOST=ml4t-timescaledb
- TIMESCALE_PORT=5432
- TIMESCALE_PASSWORD=benchmark
- INFLUXDB_HOST=ml4t-influxdb
- INFLUXDB_PORT=8086
- INFLUXDB_ORG=ml4t
- INFLUXDB_TOKEN=benchmark-token
- INFLUXDB_BUCKET=market_data
- POSTGRES_HOST=ml4t-postgres
- POSTGRES_PORT=5432
- POSTGRES_USER=postgres
- POSTGRES_PASSWORD=benchmark
- POSTGRES_DB=ml4t
# ============================================
# RAPIDS GBM BENCHMARK (NVIDIA GPU Required)
# Ch12 GBM library benchmark: RAPIDS cuML XGBoost, LightGBM CUDA,
# CatBoost GPU, plus CPU baselines. Single-purpose image.
# Start with: docker compose --profile rapids run --rm rapids python 12_gradient_boosting/02_gbm_comparison.py
# ============================================
rapids:
<<: *common
build:
context: .
dockerfile: envs/rapids/Dockerfile
platforms:
- linux/amd64
container_name: ml4t-rapids
profiles: ["rapids"]
user: root
entrypoint: []
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
# ============================================
# PY312 (Python 3.12, x86 Only)
# For packages without Python 3.14 support: gensim, signatory, esig,
# tfcausalimpact, plus notebooks hitting the Python 3.14 torch CUDA import bug.
# Covers: Ch05 NB01/03/07, Ch09 NB06/12, Ch10 NB01-03, Ch12 NB10, Ch14 NB06,
# Ch15 NB06. Ch21 deep hedging is NOT here: pfhedge is a main dependency and
# 0.22.0 works on Python 3.14, so that notebook runs in the default image.
# Start with: docker compose --profile py312 run --rm py312 python ...
# This service reserves no GPU, so it starts on macOS and on any machine
# without NVIDIA. Use the py312-gpu service below where a GPU is available.
# Apple Silicon: there is no native arm64 image (signatory/esig/gensim have no
# arm64 wheels). Either read the committed .ipynb outputs for the notebooks
# above, or run the amd64 image under Rosetta emulation (slow, no GPU):
# DOCKER_DEFAULT_PLATFORM=linux/amd64 docker compose --profile py312 run --rm py312 python ...
# ============================================
py312:
<<: *common
image: ml4t/ml4t-py312:latest
build:
context: .
dockerfile: envs/py312/Dockerfile
platforms:
- linux/amd64
container_name: ml4t-py312
profiles: ["py312"]
# Same image with an NVIDIA GPU attached, for the six GPU-tagged py312 notebooks
# (Ch05 NB01/03/07, Ch10 NB03, Ch12 NB10, Ch14 NB06), whether they train or run
# inference. Linux and Windows WSL2 only, exactly like ml4t-gpu.
# Start with: docker compose --profile py312-gpu run --rm py312-gpu python ...
py312-gpu:
<<: *common
image: ml4t/ml4t-py312:latest
build:
context: .
dockerfile: envs/py312/Dockerfile
platforms:
- linux/amd64
container_name: ml4t-py312-gpu
profiles: ["py312-gpu"]
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: 1
capabilities: [gpu]
# ============================================
# KNOWLEDGE GRAPH (Neo4j Community Edition)
# Start with: docker compose --profile kg up -d neo4j
# ============================================
neo4j:
image: neo4j:2026.02.2
container_name: ml4t-neo4j
profiles: ["kg"]
ports:
- "127.0.0.1:7474:7474"
- "127.0.0.1:7687:7687"
environment:
- NEO4J_AUTH=${NEO4J_AUTH:-neo4j/password}
volumes:
- neo4j_data:/data
- neo4j_logs:/logs
healthcheck:
test: ["CMD-SHELL", "cypher-shell -u neo4j -p password 'RETURN 1;'"]
interval: 10s
timeout: 10s
retries: 10
# ============================================
# Database Services (for benchmarks)
# Start with: docker compose --profile databases up -d
# ============================================
timescaledb:
image: timescale/timescaledb:latest-pg16
container_name: ml4t-timescaledb
environment:
- POSTGRES_PASSWORD=benchmark
- POSTGRES_DB=ml4t
- POSTGRES_USER=postgres
ports:
- "127.0.0.1:5437:5432"
profiles: ["benchmark", "benchmark-full", "databases"]
healthcheck:
test: ["CMD-SHELL", "pg_isready -U postgres"]
interval: 10s
timeout: 5s
retries: 5
postgres:
image: postgres:16
container_name: ml4t-postgres
environment:
- POSTGRES_PASSWORD=benchmark
- POSTGRES_DB=ml4t
- POSTGRES_USER=postgres
ports:
- "127.0.0.1:5436:5432"
profiles: ["benchmark", "benchmark-full", "databases"]
healthcheck:
test: ["CMD-SHELL", "pg_isready -U postgres"]
interval: 10s
timeout: 5s
retries: 5
clickhouse:
image: clickhouse/clickhouse-server:latest
container_name: ml4t-clickhouse
environment:
- CLICKHOUSE_DEFAULT_ACCESS_MANAGEMENT=1
- CLICKHOUSE_USER=default
- CLICKHOUSE_PASSWORD=
ports:
- "127.0.0.1:8123:8123"
profiles: ["benchmark", "benchmark-full", "databases"]
deploy:
resources:
limits:
memory: 4G
healthcheck:
test: ["CMD", "wget", "--spider", "-q", "http://localhost:8123/ping"]
interval: 10s
timeout: 5s
retries: 5
questdb:
image: questdb/questdb:latest
container_name: ml4t-questdb
ports:
- "127.0.0.1:9000:9000"
- "127.0.0.1:9009:9009"
profiles: ["benchmark", "benchmark-full", "databases"]
deploy:
resources:
limits:
memory: 2G
healthcheck:
test: ["CMD", "curl", "-f", "http://localhost:9000"]
interval: 10s
timeout: 5s
retries: 5
influxdb:
image: influxdb:2.7
container_name: ml4t-influxdb
environment:
- DOCKER_INFLUXDB_INIT_MODE=setup
- DOCKER_INFLUXDB_INIT_USERNAME=admin
- DOCKER_INFLUXDB_INIT_PASSWORD=benchmark123
- DOCKER_INFLUXDB_INIT_ORG=ml4t
- DOCKER_INFLUXDB_INIT_BUCKET=market_data
- DOCKER_INFLUXDB_INIT_ADMIN_TOKEN=benchmark-token
ports:
- "127.0.0.1:8086:8086"
profiles: ["benchmark", "benchmark-full", "databases"]
healthcheck:
test: ["CMD", "curl", "-f", "http://localhost:8086/health"]
interval: 10s
timeout: 5s
retries: 5
networks:
default:
name: ml4t-review-network
volumes:
neo4j_data:
neo4j_logs: