mirror of
https://github.com/NandhaKishorM/laya.git
synced 2026-09-28 07:52:57 +08:00
* feat: self-hostable server + Nix flake (Jev-compatible /v1/systemone) Laya's predict() output is already schema-compatible with TypeSafe Jev's decision API, so this adds only the HTTP surface a Jev client needs to talk to a self-hosted Laya: - laya/serve.py: FastAPI app exposing POST /v1/systemone (+ /health), calling Router.predict, with optional LAYA_API_KEY bearer auth. Config via env (LAYA_HOST/PORT/DEVICE/PRELOAD/MODELS/AUTO_TASK). Heavy imports are deferred so `import laya.serve` stays GPU- and web-dep-free. - laya-serve console script + laya[serve] extra (fastapi, uvicorn). - flake.nix: overlay building laya from torch-bin (prebuilt CUDA), a laya-serve runner, a dev shell, and nixosModules.default. - nix/laya-serve.nix: hardened DynamicUser systemd service (services.laya-serve) with CUDA device access and a HF weight cache. - tests/test_serve.py: shim tests (injected router, real FastAPI TestClient). Verified end-to-end on an RTX 3090: all three checkpoints preload on CUDA, English/Hindi requests route correctly and return Jev-shaped payloads, 6/6 tests pass. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_019NnoUnPhBMoS4S7eYWHYfe * fix(nix): build the service package from the host pkgs, not an overlay Hosts that inject `pkgs` via specialArgs (read-only nixpkgs) ignore module-level `nixpkgs.overlays`, so `pkgs.laya-serve` was missing and the NixOS toplevel failed to evaluate. Factor the package into nix/package.nix and have services.laya-serve.package default to `(pkgs.callPackage ./package.nix {}).laya-serve`, built from the host's own pkgs. The flake overlay/packages now reuse the same file. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_019NnoUnPhBMoS4S7eYWHYfe * feat(serve): LAYA_THREADS knob + services.laya-serve.threads for CPU tuning Cap torch intra-op threads for CPU inference. serve.py reads LAYA_THREADS and calls torch.set_num_threads (torch imported only when a limit is set); the NixOS module exposes `threads` and wires it to LAYA_THREADS + OMP_NUM_THREADS. Motivated by a thread sweep on a 32-core/64-thread EPYC: oversubscribing the logical core count is a ~4x latency regression, and single-request latency is often best below the physical core count (e.g. 8-16), while batched throughput peaks at the physical count. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_019NnoUnPhBMoS4S7eYWHYfe * chore(nix): commit flake.lock pinning nixpkgs to ccad53c Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_019NnoUnPhBMoS4S7eYWHYfe --------- Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
66 lines
2.3 KiB
Nix
66 lines
2.3 KiB
Nix
{
|
|
description = "laya + laya-serve: a self-hosted, TypeSafe Jev-compatible System-1 decision server";
|
|
|
|
inputs = {
|
|
# Pinned to the same rev missionctrl-infra uses, so hq's binary cache
|
|
# (cache.nixos.org + missionctrl.cachix.org) hits instead of rebuilding torch.
|
|
nixpkgs.url = "github:NixOS/nixpkgs/ccad53cd79cf4cf3bc338805d007d68565e75bda";
|
|
flake-utils.url = "github:numtide/flake-utils";
|
|
};
|
|
|
|
outputs = { self, nixpkgs, flake-utils }:
|
|
let
|
|
# Overlay exposing `laya` + `laya-serve`, built from ./nix/package.nix.
|
|
overlay = final: prev:
|
|
let p = final.callPackage ./nix/package.nix { };
|
|
in { inherit (p) laya laya-serve; };
|
|
in
|
|
{
|
|
overlays.default = overlay;
|
|
|
|
# The module builds its package from the host's own `pkgs` (see
|
|
# nix/laya-serve.nix), so it works even on hosts that inject `pkgs` via
|
|
# specialArgs and ignore module-level `nixpkgs.overlays` — no overlay
|
|
# required here.
|
|
nixosModules.default = ./nix/laya-serve.nix;
|
|
nixosModules.laya-serve = ./nix/laya-serve.nix;
|
|
}
|
|
// flake-utils.lib.eachDefaultSystem (system:
|
|
let
|
|
pkgs = import nixpkgs {
|
|
inherit system;
|
|
config.allowUnfree = true; # torch-bin bundles CUDA (unfree)
|
|
overlays = [ overlay ];
|
|
};
|
|
in
|
|
{
|
|
packages = {
|
|
default = pkgs.laya-serve;
|
|
laya-serve = pkgs.laya-serve;
|
|
laya = pkgs.laya;
|
|
};
|
|
|
|
devShells.default = pkgs.mkShell {
|
|
packages = [
|
|
(pkgs.python3.withPackages (ps: [
|
|
ps.torch-bin
|
|
ps.transformers
|
|
ps.safetensors
|
|
ps.huggingface-hub
|
|
ps.numpy
|
|
ps.fastapi
|
|
ps.uvicorn
|
|
ps.pytest
|
|
ps.httpx # fastapi TestClient
|
|
]))
|
|
];
|
|
shellHook = ''
|
|
export PYTHONPATH="$PWD:$PYTHONPATH"
|
|
# torch-bin's CUDA needs the host NVIDIA userspace driver.
|
|
export LD_LIBRARY_PATH="/run/opengl-driver/lib''${LD_LIBRARY_PATH:+:$LD_LIBRARY_PATH}"
|
|
echo "laya dev shell — python $(python --version 2>&1 | cut -d' ' -f2), torch $(python -c 'import torch; print(torch.__version__)' 2>/dev/null)"
|
|
'';
|
|
};
|
|
});
|
|
}
|