mirror of
https://github.com/NandhaKishorM/laya.git
synced 2026-09-28 07:52:57 +08:00
laya/agent.py selects the forward's dtype from LAYA_CUDA_AMP and LAYA_CPU_AMP, and no compose file that sets LAYA_DEVICE passed either name, so the published deployment always served the checkpoint's amp_dtype whatever the operator exported. README measures the difference as decisions, not rounding: bf16 flips 3 of 864 argmaxes on the parity_fast set where fp16 flips none, at the same latency. Both GPU services get the pair, because an override on `laya` never reaches `laya-serve`. An empty value means "the checkpoint's own amp_dtype", which is what these images shipped before, so the passthrough default changes nothing until it is set. The MPS row gate is deliberately left out: no container here can select MPS.
30 lines
1.1 KiB
YAML
30 lines
1.1 KiB
YAML
services:
|
|
laya:
|
|
build:
|
|
context: .
|
|
args:
|
|
TORCH_INDEX: "${LAYA_TORCH_INDEX:-cpu}"
|
|
TORCH_VERSION: "${LAYA_TORCH_VERSION:-2.14.0}"
|
|
environment:
|
|
LAYA_DEVICE: "${LAYA_DEVICE:-cpu}"
|
|
# The autocast dtype the runtime picks when it builds the model on the selected device
|
|
# (`laya/agent.py`). An empty value means "use the checkpoint's own `amp_dtype`", which is
|
|
# what these images have always served. README's threshold section measures what fp16 vs
|
|
# bf16 decides differently.
|
|
LAYA_CUDA_AMP: "${LAYA_CUDA_AMP:-}"
|
|
LAYA_CPU_AMP: "${LAYA_CPU_AMP:-}"
|
|
LAYA_MODEL: "${LAYA_MODEL:-auto}"
|
|
LAYA_MODEL_PATH: "${LAYA_MODEL_PATH:-}"
|
|
LAYA_REQUEST_FILE: "${LAYA_REQUEST_FILE:-/opt/laya/examples/request.json}"
|
|
HF_HUB_OFFLINE: "${HF_HUB_OFFLINE:-0}"
|
|
HF_TOKEN: "${HF_TOKEN:-}"
|
|
HF_TOKEN_FILE: "${HF_TOKEN_FILE:-}"
|
|
OMP_NUM_THREADS: "${OMP_NUM_THREADS:-4}"
|
|
volumes:
|
|
- model-cache:/home/laya/.cache/huggingface
|
|
init: true
|
|
|
|
volumes:
|
|
model-cache:
|
|
name: "${LAYA_CACHE_VOLUME:-${COMPOSE_PROJECT_NAME}_model-cache}"
|