mirror of
https://github.com/NandhaKishorM/laya.git
synced 2026-09-28 16:02:56 +08:00
fix(server): keep exception text out of responses, log it instead # Conflicts: # .github/workflows/ci.yml # examples/server.py
392 lines
17 KiB
YAML
392 lines
17 KiB
YAML
name: CI
|
|
|
|
on:
|
|
pull_request:
|
|
push:
|
|
branches: [main]
|
|
|
|
permissions:
|
|
contents: read
|
|
|
|
concurrency:
|
|
group: ci-${{ github.event_name == 'pull_request' && github.ref || github.sha }}
|
|
# Cancel outdated pull request runs. On main, use a per-commit group so each push
|
|
# gets its own run and started runs finish without cancelling or being cancelled by adjacent merges.
|
|
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
|
|
|
|
env:
|
|
# transformers probes for TensorFlow at import; when TF is present its abseil runtime can
|
|
# deadlock model construction. Laya is torch-only.
|
|
USE_TF: "0"
|
|
USE_TORCH: "1"
|
|
TOKENIZERS_PARALLELISM: "false"
|
|
|
|
jobs:
|
|
typescript:
|
|
name: TypeScript (node${{ matrix.node }})
|
|
runs-on: ubuntu-latest
|
|
timeout-minutes: 10
|
|
defaults:
|
|
run:
|
|
working-directory: laya-ts
|
|
strategy:
|
|
fail-fast: false
|
|
matrix:
|
|
node: ["20", "22"]
|
|
steps:
|
|
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
|
|
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
|
|
with:
|
|
node-version: ${{ matrix.node }}
|
|
cache: npm
|
|
cache-dependency-path: laya-ts/package-lock.json
|
|
- run: npm ci
|
|
- name: Unit tests
|
|
run: npm test -- --run
|
|
- name: Packed package end-to-end test
|
|
run: npm run test:package
|
|
|
|
feishu-benchmark:
|
|
name: Chinese benchmark audit (no models or API)
|
|
runs-on: ubuntu-latest
|
|
timeout-minutes: 5
|
|
steps:
|
|
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
|
|
- uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
|
|
with:
|
|
python-version: "3.11"
|
|
- run: python research/benchmarks/feishu_zh/audit.py
|
|
- run: python -m unittest discover -s research/benchmarks/feishu_zh/tests -v
|
|
# This job installs nothing, so running the audit here is what proves it needs
|
|
# nothing but the standard library. Its test suite imports numpy and runs in `test`.
|
|
- run: python research/benchmarks/zh_short_commands/audit.py
|
|
|
|
test:
|
|
name: tests (py${{ matrix.python }})
|
|
runs-on: ubuntu-latest
|
|
timeout-minutes: 15
|
|
strategy:
|
|
fail-fast: false
|
|
matrix:
|
|
# Keep this in lockstep with pyproject classifiers; packaging tests enforce that.
|
|
python: ["3.10", "3.11", "3.12", "3.13"]
|
|
steps:
|
|
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
|
|
- uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
|
|
with:
|
|
python-version: ${{ matrix.python }}
|
|
cache: pip
|
|
|
|
- name: Install
|
|
run: |
|
|
python -m pip install --upgrade pip
|
|
# CPU-only torch keeps the job under a minute instead of pulling CUDA wheels
|
|
pip install torch --index-url https://download.pytorch.org/whl/cpu
|
|
# `mcp` so laya[mcp] resolution and the laya.mcp layer are both exercised, and
|
|
# `serve` because laya.serve and examples/server.py both import fastapi -- without
|
|
# it every test that touches them skips itself silently and passes.
|
|
# `langchain` for the same reason: without langchain-core the runnables fall back
|
|
# to their plain-object path, so tests/test_langchain.py never checks the real
|
|
# Runnable behaviour (batch delegation, pydantic field validation) it is there for.
|
|
pip install -e ".[mcp,serve,langchain]" httpx
|
|
# The pytest-style suites need the runner; nothing else in this job uses it.
|
|
pip install pytest
|
|
|
|
- name: Import check
|
|
run: python -c "import laya; print(laya.__version__); print(sorted(laya.__all__))"
|
|
|
|
- name: Routing and language-detection tests
|
|
run: |
|
|
python tests/test_router.py
|
|
python tests/test_criteria.py
|
|
python tests/test_batch.py
|
|
python tests/test_predict_long.py
|
|
python tests/test_attention_dynamic_shapes.py
|
|
python tests/test_hooks.py
|
|
python tests/test_hooks_api.py
|
|
python tests/test_structured.py
|
|
python tests/test_structured_api.py
|
|
python tests/test_structured_docs.py
|
|
python tests/test_onnx_lang_parity.py
|
|
python tests/test_onnx_batch.py
|
|
python tests/test_onnx_long.py
|
|
python tests/test_onnx_sort.py
|
|
python tests/test_download.py
|
|
python tests/test_shortlist.py
|
|
python tests/test_decision_model.py
|
|
python tests/test_head_checkpointing.py
|
|
python tests/test_packaging.py
|
|
python tests/test_compose_env.py
|
|
python tests/test_audit_scope.py
|
|
python tests/test_doc_tables.py
|
|
python tests/test_env_docs.py
|
|
python tests/test_tokenizer_cache.py
|
|
python tests/test_tokenizer_concurrency.py
|
|
python tests/test_question_token_reuse.py
|
|
python tests/test_lazy_import.py
|
|
python tests/test_runtime_fixes.py
|
|
python tests/test_option_collapse.py
|
|
python tests/test_lang_guess.py
|
|
python tests/test_identifier_complexity.py
|
|
python tests/test_lang_stats.py
|
|
python tests/test_calibration_persistence.py
|
|
python tests/test_context_manager.py
|
|
python tests/test_criteria_normalization.py
|
|
python tests/test_docker_entrypoint.py
|
|
python tests/test_empty_questions.py
|
|
python tests/test_router_memory.py
|
|
python tests/test_shortlist_cosine.py
|
|
python tests/test_temperature_loading.py
|
|
python tests/test_confidence.py
|
|
python tests/test_cli.py
|
|
python tests/test_mcp.py
|
|
python tests/test_langchain.py
|
|
python tests/test_portability.py
|
|
python tests/test_training.py
|
|
python tests/test_example_server_limits.py
|
|
python tests/test_blank_lang_routing.py
|
|
python tests/test_export_onnx_safety.py
|
|
python tests/test_load_errors.py
|
|
python tests/test_predict_batch.py
|
|
python tests/test_revision_pinning.py
|
|
python tests/test_example_server_errors.py
|
|
python tests/test_crewai.py
|
|
python tests/test_llamaindex.py
|
|
|
|
- name: LangChain integration runs on the real base class
|
|
run: |
|
|
# The install above is what makes tests/test_langchain.py run on real Runnables, and
|
|
# nothing proves it took: with langchain-core missing, the eight checks inside that
|
|
# file's `if _RUNNABLE_AVAILABLE:` block vanish instead of failing, and it still
|
|
# prints FAIL: 0 -- measured here: PASS: 87 with the extra, PASS: 80 without, exit 0
|
|
# both ways. This step asserts the install took, then checks what only a real install
|
|
# can check.
|
|
python -c "from laya.integrations.langchain import RunnableSerializable; \
|
|
assert RunnableSerializable is not object, \
|
|
'langchain-core is missing: laya.integrations.langchain fell back to a plain object, ' \
|
|
'so the LangChain suites in this job test a shape no install produces'"
|
|
python tests/test_langchain_real_mode.py
|
|
|
|
- name: Email cleaning tests
|
|
run: python tests/test_email.py
|
|
# Full dependency tree here, so the stub pipeline and the metric pins that need numpy
|
|
# run instead of skipping. The audit itself runs in `feishu-benchmark`, where nothing
|
|
# is installed.
|
|
- name: Chinese short-command benchmark tests
|
|
run: python -m unittest discover -s research/benchmarks/zh_short_commands/tests -v
|
|
|
|
# These are pytest suites, and `python tests/<name>.py` does not run them: it defines
|
|
# the test functions and exits 0, having asserted nothing. They were counted by the suite
|
|
# list above without ever executing, which is how the HTTP surface came to have no coverage
|
|
# in CI at all (#374).
|
|
#
|
|
# test_onnx.py and test_fast.py are deliberately not here: both are skip-guarded on an extra
|
|
# this job does not install, so they would pass trivially, and their real execution belongs
|
|
# in a lane that installs `onnx` or has a GPU.
|
|
- name: Pytest suites (serve, batch, system-one, audit regressions, truncation, compile)
|
|
run: |
|
|
python -m pytest tests/test_serve.py \
|
|
tests/test_router_batch.py \
|
|
tests/test_predict_batch.py \
|
|
tests/test_system_one_lang.py \
|
|
tests/test_audit_regressions.py \
|
|
tests/test_truncation_direction.py \
|
|
tests/test_compile.py \
|
|
-q
|
|
|
|
# The `test` job above deliberately stays free of `onnx`: this is the lane that
|
|
# installs it, so the exporter's quantization suite runs for real instead of
|
|
# skip-guarding into a trivial pass. Weight-free -- it builds tiny synthetic
|
|
# graphs, no checkpoint download.
|
|
onnx-export:
|
|
name: onnx export (quantization, weight-free)
|
|
runs-on: ubuntu-latest
|
|
timeout-minutes: 10
|
|
steps:
|
|
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
|
|
- uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
|
|
with:
|
|
python-version: "3.11"
|
|
cache: pip
|
|
|
|
- name: Install
|
|
run: |
|
|
python -m pip install --upgrade pip
|
|
pip install torch --index-url https://download.pytorch.org/whl/cpu
|
|
pip install -e .
|
|
pip install onnx onnxruntime
|
|
|
|
- name: Quantized-export tests
|
|
run: python tests/test_onnx_quantize.py
|
|
|
|
# The eval harness needs numpy and laya only, so it stays a separate, weight-free job: metric
|
|
# math, dataset parsing, the CLI and the API guard. The real-checkpoint gate is the scheduled
|
|
# `evals` workflow, which needs weights and network.
|
|
evals:
|
|
name: evals (metric math, dataset, API guard)
|
|
runs-on: ubuntu-latest
|
|
timeout-minutes: 20
|
|
steps:
|
|
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
|
|
- uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
|
|
with:
|
|
python-version: "3.11"
|
|
cache: pip
|
|
|
|
- name: Install
|
|
run: |
|
|
python -m pip install --upgrade pip
|
|
pip install torch --index-url https://download.pytorch.org/whl/cpu
|
|
pip install -e . pytest
|
|
|
|
- name: Eval harness tests
|
|
run: python -m pytest tests/test_evals.py -q
|
|
|
|
- name: Eval API guard
|
|
run: python tests/test_evals_api.py
|
|
|
|
# The ONNX runner suite fakes the agent, so it stays weight-free and needs no
|
|
# onnxruntime: the module only imports numpy, and ONNXAgent's constructor is
|
|
# never called.
|
|
- name: Eval ONNX runner tests
|
|
run: python tests/test_evals_onnx.py
|
|
|
|
- name: Fixture dataset validates
|
|
run: laya-evals validate research/evals/fixture.jsonl
|
|
|
|
# Every workflow in this repo ran on ubuntu-latest only, so a test that fails off Linux was
|
|
# invisible until a user hit it -- which is what happened to tests/test_download.py (#140).
|
|
#
|
|
# A separate job rather than an extra axis on `test`, deliberately: it leaves that job's name,
|
|
# and therefore the `tests (pyX.Y)` required checks, exactly as they are. It runs one Python
|
|
# because the variable that was untested is the OS, not the interpreter.
|
|
test-windows:
|
|
name: tests (windows)
|
|
runs-on: windows-latest
|
|
timeout-minutes: 20
|
|
steps:
|
|
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
|
|
- uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
|
|
with:
|
|
python-version: "3.11"
|
|
cache: pip
|
|
|
|
# The +cpu index used on Linux is Linux/macOS only. On Windows the default PyPI wheel is
|
|
# already CPU-only, so installing torch plainly is both correct and what a user gets.
|
|
# The serve extra and pytest are here for the same reason as the Linux job: the HTTP
|
|
# surface is otherwise skipped and the pytest suites otherwise do not run at all.
|
|
- name: Install
|
|
run: |
|
|
python -m pip install --upgrade pip
|
|
pip install torch
|
|
pip install -e ".[serve]"
|
|
pip install pytest
|
|
shell: bash
|
|
|
|
# bash on this runner so the quoting rules match the Linux job; the default shell on a
|
|
# Windows runner is PowerShell, where `python -c "...; ..."` is not the same command.
|
|
- name: Import check
|
|
run: python -c "import laya; print(laya.__version__); print(sorted(laya.__all__))"
|
|
shell: bash
|
|
|
|
# Same suites as the Linux job. Guarded with -f so a suite added by another open branch
|
|
# cannot make this job red on a tree that does not have it yet.
|
|
#
|
|
# test_docker_entrypoint is skipped here on purpose. It drives docker/entrypoint.py, which
|
|
# only ever runs inside a Linux container, through `os.execvp`. On Windows exec* rebuilds
|
|
# the command line from argv without re-applying the quoting `subprocess` used to build it,
|
|
# so its `-c "<script>"` argument arrives split at the first space and the child dies with
|
|
# "SyntaxError: invalid syntax" before printing. The environment passthrough the suite
|
|
# exists to check is unaffected, and switching the entrypoint to `subprocess.run` makes it
|
|
# pass -- a change to a container script, so it belongs in its own PR, not this one.
|
|
- name: Routing and language-detection tests
|
|
run: |
|
|
for t in test_router test_criteria test_batch test_predict_long \
|
|
test_structured test_structured_api test_onnx_lang_parity \
|
|
test_download test_shortlist test_decision_model \
|
|
test_head_checkpointing test_packaging test_doc_tables \
|
|
test_tokenizer_cache test_tokenizer_concurrency test_lazy_import \
|
|
test_runtime_fixes test_lang_guess test_identifier_complexity \
|
|
test_lang_stats test_calibration_persistence test_context_manager \
|
|
test_criteria_normalization test_docker_entrypoint \
|
|
test_empty_questions test_router_memory test_shortlist_cosine \
|
|
test_temperature_loading test_confidence test_cli test_mcp \
|
|
test_langchain test_portability test_training \
|
|
test_example_server_limits test_blank_lang_routing \
|
|
test_export_onnx_safety test_load_errors test_revision_pinning; do
|
|
if [ "$t" = "test_docker_entrypoint" ]; then
|
|
echo "skipping $t: docker/entrypoint.py is Linux-container only"
|
|
continue
|
|
fi
|
|
if [ -f "tests/$t.py" ]; then python "tests/$t.py"; fi
|
|
done
|
|
shell: bash
|
|
|
|
- name: Prediction hook tests
|
|
run: |
|
|
python tests/test_hooks.py
|
|
python tests/test_hooks_api.py
|
|
shell: bash
|
|
|
|
- name: Question token reuse tests
|
|
run: python tests/test_question_token_reuse.py
|
|
shell: bash
|
|
|
|
# Same pytest suites as the Linux job. They are named explicitly rather than added to
|
|
# the loop above, because the loop runs each file as a script and that is exactly what does
|
|
# not execute a pytest suite (#374).
|
|
- name: Pytest suites (serve, batch, system-one, audit regressions, truncation, compile)
|
|
run: |
|
|
python -m pytest tests/test_serve.py \
|
|
tests/test_router_batch.py \
|
|
tests/test_predict_batch.py \
|
|
tests/test_system_one_lang.py \
|
|
tests/test_audit_regressions.py \
|
|
tests/test_truncation_direction.py \
|
|
tests/test_compile.py \
|
|
-q
|
|
shell: bash
|
|
|
|
- name: Email cleaning tests
|
|
run: python tests/test_email.py
|
|
shell: bash
|
|
|
|
lint:
|
|
name: lint
|
|
runs-on: ubuntu-latest
|
|
timeout-minutes: 5
|
|
steps:
|
|
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
|
|
- uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
|
|
with:
|
|
python-version: "3.11"
|
|
- run: pip install ruff
|
|
- name: Ruff
|
|
run: ruff check laya/ --select=E9,F63,F7,F82,F401,F811 --line-length=120
|
|
- name: Byte-compile every module
|
|
run: python -m compileall -q laya/ tests/
|
|
|
|
build:
|
|
name: package
|
|
runs-on: ubuntu-latest
|
|
timeout-minutes: 10
|
|
steps:
|
|
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
|
|
- uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
|
|
with:
|
|
python-version: "3.11"
|
|
- run: pip install build twine
|
|
- run: python -m build
|
|
- name: Validate metadata
|
|
run: python -m twine check dist/*
|
|
- name: Version must match the tag on a release
|
|
if: startsWith(github.ref, 'refs/tags/v')
|
|
run: |
|
|
pkg=$(python -c "import re;print(re.search(r'version = \"(.*)\"', open('pyproject.toml').read()).group(1))")
|
|
tag="${GITHUB_REF_NAME#v}"
|
|
test "$pkg" = "$tag" || { echo "pyproject $pkg != tag $tag"; exit 1; }
|
|
- uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2
|
|
with:
|
|
name: dist
|
|
path: dist/
|