Repository navigation
Expand file tree
/
Copy pathcompose.http.yaml
More file actions
99 lines (98 loc) · 5.36 KB
/
Copy pathcompose.http.yaml
File metadata and controls
99 lines (98 loc) · 5.36 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
# Serve Laya over HTTP from the same image, so a Jev client can point its `baseUrl` at it:
#
# docker compose -f compose.yaml -f compose.http.yaml up --build laya-serve
# curl -s localhost:8000/health
# curl -s localhost:8000/v1/systemone -H 'content-type: application/json' \
# --data @examples/docker/request.json
#
# For NVIDIA, append -f compose.cuda.yaml before "up":
#
# docker compose -f compose.yaml -f compose.http.yaml -f compose.cuda.yaml up --build laya-serve
#
# This file is additive. `laya` in compose.yaml still runs the one-shot SDK quickstart and
# still publishes no ports, so `docker compose run --rm laya` behaves exactly as before.
# Weights share the same named cache volume as that service.
#
# Set LAYA_API_KEY (or LAYA_API_KEY_FILE) to require `Authorization: Bearer <key>`.
# LAYA_PRELOAD=1 builds every checkpoint at startup so the first request does not pay for
# it; the default here is 0 because it makes the container download the whole family on
# first boot, which is a surprise in a quickstart.
services:
laya-serve:
build:
context: .
args:
TORCH_INDEX: "${LAYA_TORCH_INDEX:-cpu}"
TORCH_VERSION: "${LAYA_TORCH_VERSION:-2.14.0}"
# Overrides the image's one-shot quickstart CMD; the ENTRYPOINT that loads *_FILE
# secrets still runs first.
command: ["laya-serve"]
ports:
# Host address, host port, container port. The port comes from LAYA_PORT, which
# `laya-serve` itself reads, so the two sides cannot drift apart. The host side
# stays on loopback unless LAYA_BIND_ADDRESS says otherwise, because the API has
# no authentication until LAYA_API_KEY is set.
- "${LAYA_BIND_ADDRESS:-127.0.0.1}:${LAYA_PORT:-8000}:${LAYA_PORT:-8000}"
environment:
LAYA_HOST: "${LAYA_HOST:-0.0.0.0}"
LAYA_PORT: "${LAYA_PORT:-8000}"
LAYA_DEVICE: "${LAYA_DEVICE:-cpu}"
# FastAPI's external URL prefix behind a reverse proxy; forward stripped paths.
LAYA_ROOT_PATH: "${LAYA_ROOT_PATH:-}"
# Autocast dtype the runtime reads when building the model; empty keeps the checkpoint's own
# `amp_dtype`. Set LAYA_CUDA_AMP=fp16 to serve fp16 on a GPU host.
LAYA_CUDA_AMP: "${LAYA_CUDA_AMP:-}"
LAYA_CPU_AMP: "${LAYA_CPU_AMP:-}"
LAYA_PRELOAD: "${LAYA_PRELOAD:-0}"
LAYA_MODELS: "${LAYA_MODELS:-}"
LAYA_THREADS: "${LAYA_THREADS:-${OMP_NUM_THREADS:-4}}"
LAYA_AUTO_TASK: "${LAYA_AUTO_TASK:-0}"
# The checkpoint a state with no language evidence falls back to. Empty keeps Router's own
# default (english); set `multilingual` for mostly-non-English traffic. An unresolvable
# name stops the container at startup rather than serving a configuration nobody asked for.
LAYA_DEFAULT_MODEL: "${LAYA_DEFAULT_MODEL:-}"
# Empty keeps Router's own default (2). Set 3 when LAYA_AUTO_TASK is on: that adds a
# third checkpoint routing can choose on demand, and a cap below it reloads one per switch.
LAYA_MAX_LOADED: "${LAYA_MAX_LOADED:-}"
# Requests admitted at once; one that arrives with every slot taken is refused with 503
# instead of queued, which is what bounds the bodies a burst holds in memory. Empty keeps
# the server's own default (16), so this line changes nothing until it is set.
LAYA_MAX_CONCURRENT: "${LAYA_MAX_CONCURRENT:-}"
# Ceiling on the `max_len` / `head_max_len` a request may ask for; a larger value comes back
# as 422. The SDK integrations quote 8192 because that is this service's default, so the
# default here stays empty: an operator who sets nothing gets the number the docs print.
LAYA_MAX_TOKEN_BUDGET: "${LAYA_MAX_TOKEN_BUDGET:-}"
LAYA_LOG_LEVEL: "${LAYA_LOG_LEVEL:-info}"
# The supply-chain pair: LAYA_REVISION says which Hub commit a checkpoint may load from (a
# commit, branch, tag, or `reviewed` for the SHAs in laya/revisions.py) and
# LAYA_SHA256_DIGESTS verifies the artifact that arrives. An empty revision is "not asked
# for", so unlike a name the server resolves, nothing is validated at startup here.
LAYA_REVISION: "${LAYA_REVISION:-}"
LAYA_SHA256_DIGESTS: "${LAYA_SHA256_DIGESTS:-}"
LAYA_API_KEY: "${LAYA_API_KEY:-}"
LAYA_API_KEY_FILE: "${LAYA_API_KEY_FILE:-}"
HF_HUB_OFFLINE: "${HF_HUB_OFFLINE:-0}"
HF_TOKEN: "${HF_TOKEN:-}"
HF_TOKEN_FILE: "${HF_TOKEN_FILE:-}"
OMP_NUM_THREADS: "${OMP_NUM_THREADS:-4}"
volumes:
- model-cache:/home/laya/.cache/huggingface
# Optional token files, readable by UID 10001. Never put a secret value in the
# environment; these paths are read once at startup and the *_FILE variable is
# then removed before the server execs.
# - ${HF_TOKEN_PATH:?Set HF_TOKEN_PATH}:/run/secrets/hf_token:ro
# - ${LAYA_API_KEY_PATH:?Set LAYA_API_KEY_PATH}:/run/secrets/laya_api_key:ro
init: true
restart: unless-stopped
# The server preloads before it binds, so /health answers once the preloaded
# checkpoints are ready. `up --wait` therefore waits for a usable server.
healthcheck:
test: ["CMD", "python", "-c", "import os, urllib.request; urllib.request.urlopen('http://127.0.0.1:' + os.environ['LAYA_PORT'] + '/health', timeout=4)"]
interval: 30s
timeout: 5s
retries: 3
start_period: 10m
start_interval: 2s
volumes:
model-cache:
name: "${LAYA_CACHE_VOLUME:-${COMPOSE_PROJECT_NAME}_model-cache}"