forked from LibreChat-AI/code-interpreter
-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathdocker-compose.scalable.yml
More file actions
256 lines (248 loc) · 10.5 KB
/
Copy pathdocker-compose.scalable.yml
File metadata and controls
256 lines (248 loc) · 10.5 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
# Scalable Architecture Stack (MicroVM)
#
# This demonstrates the split API + Worker architecture for horizontal scaling.
# The sandbox runs inside a libkrun microVM with its own guest kernel.
# In production, use Kubernetes with HPA for auto-scaling.
#
# Requires: /dev/kvm accessible in Docker (WSL2 with nested virt, or bare metal Linux)
#
# Usage: set CODEAPI_INTERNAL_SERVICE_TOKEN as documented below, then run:
# docker compose -f docker-compose.scalable.yml up --scale worker-sandbox=3 --scale api=2
#
# Architecture:
# ┌─────────────────────────────────────────────────────────────────────────┐
# │ Shared Infrastructure │
# │ ┌───────────┐ ┌───────────┐ ┌─────────────┐ ┌──────────────────┐ │
# │ │ Redis │ │ MinIO │ │ file_server │ │ tool_call_server │ │
# │ │ (queue) │ │ (files) │ │ (stateless) │ │ (stateless) │ │
# │ └───────────┘ └───────────┘ └─────────────┘ └──────────────────┘ │
# └─────────────────────────────────────────────────────────────────────────┘
# │
# ┌─────────────────────────┴─────────────────────────┐
# │ │
# ▼ ▼
# ┌─────────────────────┐ ┌─────────────────────────────┐
# │ API Service │ │ Worker-Sandbox Pods │
# │ (HTTP only) │ │ │
# │ ┌───────────────┐ │ │ ┌─────────────────────────┐ │
# │ │ api (N pods) │ │ ──submits jobs──▶ │ │ worker + sandbox (M) │ │
# │ └───────────────┘ │ │ │ worker + sandbox (M) │ │
# │ │ ◀──receives results─│ │ worker + sandbox (M) │ │
# └─────────────────────┘ │ └─────────────────────────┘ │
# └─────────────────────────────┘
#
# Scaling:
# - api: Scale based on HTTP traffic (requests/sec, latency)
# - worker-sandbox: Scale based on queue depth (jobs waiting)
# - file_server: Scale based on upload/download throughput
# - tool_call_server: Scale based on tool call volume
#
services:
# =============================================================================
# SHARED INFRASTRUCTURE (Scale independently)
# =============================================================================
# Redis - Shared job queue and state store
redis:
image: redis:7-alpine
container_name: redis
command: redis-server --requirepass localdev
ports:
- 6379:6379
healthcheck:
test: ["CMD", "redis-cli", "-a", "localdev", "ping"]
interval: 5s
timeout: 3s
retries: 3
# MinIO (S3-compatible storage) - Shared file storage
minio:
image: quay.io/minio/minio
container_name: minio
ports:
- 9000:9000
- 9001:9001
volumes:
- minio_data:/data
environment:
- MINIO_ROOT_USER=minioadmin
- MINIO_ROOT_PASSWORD=minioadmin
command: server /data --console-address ":9001"
healthcheck:
test: ["CMD", "curl", "-f", "http://localhost:9000/minio/health/live"]
interval: 5s
timeout: 3s
retries: 3
# File Server - Stateless, scales independently
file_server:
build:
context: .
dockerfile: service/Dockerfile
target: production
ports:
- 3000:3000
environment:
- MINIO_BUCKET=test-bucket
- MINIO_ENDPOINT=minio
- MINIO_PORT=9000
- MINIO_USE_SSL=false
- MINIO_ACCESS_KEY=minioadmin
- MINIO_SECRET_KEY=minioadmin
- FILE_SERVER_PORT=3000
- REDIS_HOST=redis
- REDIS_PORT=6379
- REDIS_PASSWORD=localdev
- CODEAPI_INTERNAL_SERVICE_TOKEN=${CODEAPI_INTERNAL_SERVICE_TOKEN:?CODEAPI_INTERNAL_SERVICE_TOKEN is required}
depends_on:
redis:
condition: service_healthy
minio:
condition: service_healthy
deploy:
replicas: 1 # Can scale independently
# Tool Call Server - Stateless, scales independently
tool_call_server:
build:
context: .
dockerfile: service/Dockerfile.tool-call-server
target: production
ports:
- 3033:3033
environment:
- TOOL_CALL_SERVER_PORT=3033
- TOOL_CALL_REQUEST_TIMEOUT=300000
- TOOL_CALL_SESSION_EXPIRY=600
- REDIS_HOST=redis
- REDIS_PORT=6379
- REDIS_PASSWORD=localdev
- CODEAPI_INTERNAL_SERVICE_TOKEN=${CODEAPI_INTERNAL_SERVICE_TOKEN:?CODEAPI_INTERNAL_SERVICE_TOKEN is required}
depends_on:
redis:
condition: service_healthy
deploy:
replicas: 1 # Can scale independently
# =============================================================================
# API SERVICE (HTTP handlers only, no workers)
# Scale based on: HTTP traffic, request latency
# =============================================================================
api:
build:
context: .
dockerfile: service/Dockerfile.api
target: production
ports:
- "3112-3119:3112" # Port range for multiple instances
environment:
- LOCAL_MODE=${LOCAL_MODE:-true}
- FILE_SERVER_URL=http://file_server:3000
- TOOL_CALL_SERVER_URL=http://tool_call_server:3033
- CODEAPI_INTERNAL_SERVICE_TOKEN=${CODEAPI_INTERNAL_SERVICE_TOKEN:?CODEAPI_INTERNAL_SERVICE_TOKEN is required}
- REDIS_HOST=redis
- REDIS_PORT=6379
- REDIS_PASSWORD=localdev
- PORT=3112
# Note: No SANDBOX_ENDPOINT - API doesn't talk to sandbox directly
# Note: No PYTHON_CONCURRENCY - API doesn't process jobs
depends_on:
redis:
condition: service_healthy
file_server:
condition: service_started
tool_call_server:
condition: service_started
deploy:
replicas: 2 # Scale based on HTTP traffic
healthcheck:
test: ["CMD", "curl", "-f", "http://localhost:3112/v1/health"]
interval: 10s
timeout: 5s
retries: 3
# =============================================================================
# WORKER-SANDBOX PODS (Worker + MicroVM Sandbox as a unit)
# Scale based on: Queue depth, job wait time
# Each pod runs its own microVM — PYTHON_CONCURRENCY is per-sandbox
# =============================================================================
worker-sandbox:
build:
context: .
dockerfile: docker/Dockerfile.worker-sandbox
target: worker-sandbox-${KVM_ENABLED:-true}
privileged: false
devices:
- ${KVM_DEVICE_PATH:-/dev/kvm}:/dev/kvm
environment:
- KVM_ENABLED=${KVM_ENABLED:-true}
- LOCAL_MODE=${LOCAL_MODE:-true}
# Worker config
- SANDBOX_ENDPOINT=http://localhost:2000/api/v2
- FILE_SERVER_URL=http://file_server:3000
- TOOL_CALL_SERVER_URL=http://tool_call_server:3033
- CODEAPI_INTERNAL_SERVICE_TOKEN=${CODEAPI_INTERNAL_SERVICE_TOKEN:?CODEAPI_INTERNAL_SERVICE_TOKEN is required}
- REDIS_HOST=redis
- REDIS_PORT=6379
- REDIS_PASSWORD=localdev
- PYTHON_CONCURRENCY=1
- OTHER_CONCURRENCY=8
- JOB_WINDOW=1000
- JOB_TIMEOUT=300000
- WORKER_HEALTH_PORT=3113
# Launcher config (microVM)
- LAUNCHER_VCPUS=2
- LAUNCHER_RAM_MIB=2048
- LAUNCHER_LOG_LEVEL=3
# Sandbox config (passed through to microVM guest)
- SANDBOX_LOG_LEVEL=INFO
- SANDBOX_PACKAGES_DIRECTORY=/pkgs
- SANDBOX_MAX_PROCESS_COUNT=100
- SANDBOX_MAX_CONCURRENT_JOBS=8
- SANDBOX_PER_JOB_UIDS=true
- SANDBOX_JOB_UID_BASE=200000
- SANDBOX_JOB_GID_BASE=200000
- SANDBOX_WORKSPACE_REAPER_MAX_AGE_SECONDS=3600
- SANDBOX_RUN_CPU_TIME=60000
- SANDBOX_RUN_TIMEOUT=300000
- SANDBOX_OUTPUT_MAX_SIZE=65536
- SANDBOX_DISABLE_NETWORKING=true
- SANDBOX_ALLOWED_LOCAL_NETWORK_PORT=3033
- SANDBOX_FORWARD_TARGET=tool_call_server:3033
- LAUNCHER_PACKAGES_HOST=/disabled-host-packages
volumes:
- ${SANDBOX_PACKAGES_PATH:-./data/pkgs}:/host-packages:ro
depends_on:
redis:
condition: service_healthy
file_server:
condition: service_started
tool_call_server:
condition: service_started
deploy:
replicas: 3 # Scale based on queue depth
healthcheck:
test: ["CMD", "curl", "-f", "http://localhost:3113/health"]
interval: 10s
timeout: 5s
retries: 3
volumes:
minio_data:
# =============================================================================
# USAGE NOTES
# =============================================================================
#
# Generate the shared internal credential once in the current shell:
# export CODEAPI_INTERNAL_SERVICE_TOKEN="$(openssl rand -hex 32)"
#
# Start with default replicas:
# docker compose -f docker-compose.scalable.yml up
#
# Scale API pods (high HTTP traffic):
# docker compose -f docker-compose.scalable.yml up --scale api=5
#
# Scale worker-sandbox pods (high execution demand):
# docker compose -f docker-compose.scalable.yml up --scale worker-sandbox=10
#
# Scale both:
# docker compose -f docker-compose.scalable.yml up --scale api=5 --scale worker-sandbox=10
#
# In Kubernetes:
# - Use HorizontalPodAutoscaler for api based on CPU/requests
# - Use HorizontalPodAutoscaler for worker-sandbox based on queue depth (custom metric)
# - Use K8s services for file_server and tool_call_server with multiple replicas
#