-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathdocker-compose.test.yml
More file actions
49 lines (47 loc) · 1.81 KB
/
Copy pathdocker-compose.test.yml
File metadata and controls
49 lines (47 loc) · 1.81 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
# inferhost v0.5+ test stack.
# Use: ./run.sh docker-build, then docker-smoke / docker-test / docker-shell.
# Requires NVIDIA Container Toolkit on the host for --gpus all.
services:
inferhost:
build:
context: .
dockerfile: Dockerfile
image: inferhost-test:v0.5
container_name: inferhost-test
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
environment:
NVIDIA_VISIBLE_DEVICES: all
NVIDIA_DRIVER_CAPABILITIES: compute,utility
INFERHOST_GATEWAY_PORT: "9001"
# llama-swap stays loopback (v0.5 default); inside the container we
# don't need to reach it externally — only LiteLLM is exposed.
INFERHOST_SWAP_PORT: "9090"
# Force the prebuilt llama-server target if needed (uncomment for
# CPU-only host builds). The default uses the binaries.py auto-detect.
# INFERHOST_LLAMACPP_BACKEND: cuda
ports:
# Published on a non-default host port: the host's own inferhost
# daemons (llama-swap :9090, litellm :9001, tts :9096) may already be
# bound to the defaults, and `docker compose up` would fail to start
# if we tried to reuse them. Container-internal port stays 9001
# (INFERHOST_GATEWAY_PORT below) — docker-functional talks to it via
# `docker compose exec`, which never touches this host mapping.
- "19001:9001" # LiteLLM gateway (host-side only; avoids host's live :9001)
volumes:
- hf-cache:/inferhost/hf-cache
- inferhost-config:/inferhost/config
- inferhost-data:/inferhost/data
stdin_open: true
tty: true
# Keep the container alive so `docker compose run` / `exec` are cheap.
command: ["sleep", "infinity"]
volumes:
hf-cache:
inferhost-config:
inferhost-data: