Files
Nandakishor 6e19780df8 Merge pull request #625 from modusensus/fix/example-server-error-text
fix(server): keep exception text out of responses, log it instead

# Conflicts:
#	.github/workflows/ci.yml
#	examples/server.py
2026-09-28 00:54:22 +05:30

392 lines
17 KiB
YAML

name: CI
on:
pull_request:
push:
branches: [main]
permissions:
contents: read
concurrency:
group: ci-${{ github.event_name == 'pull_request' && github.ref || github.sha }}
# Cancel outdated pull request runs. On main, use a per-commit group so each push
# gets its own run and started runs finish without cancelling or being cancelled by adjacent merges.
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
env:
# transformers probes for TensorFlow at import; when TF is present its abseil runtime can
# deadlock model construction. Laya is torch-only.
USE_TF: "0"
USE_TORCH: "1"
TOKENIZERS_PARALLELISM: "false"
jobs:
typescript:
name: TypeScript (node${{ matrix.node }})
runs-on: ubuntu-latest
timeout-minutes: 10
defaults:
run:
working-directory: laya-ts
strategy:
fail-fast: false
matrix:
node: ["20", "22"]
steps:
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
with:
node-version: ${{ matrix.node }}
cache: npm
cache-dependency-path: laya-ts/package-lock.json
- run: npm ci
- name: Unit tests
run: npm test -- --run
- name: Packed package end-to-end test
run: npm run test:package
feishu-benchmark:
name: Chinese benchmark audit (no models or API)
runs-on: ubuntu-latest
timeout-minutes: 5
steps:
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
- uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
with:
python-version: "3.11"
- run: python research/benchmarks/feishu_zh/audit.py
- run: python -m unittest discover -s research/benchmarks/feishu_zh/tests -v
# This job installs nothing, so running the audit here is what proves it needs
# nothing but the standard library. Its test suite imports numpy and runs in `test`.
- run: python research/benchmarks/zh_short_commands/audit.py
test:
name: tests (py${{ matrix.python }})
runs-on: ubuntu-latest
timeout-minutes: 15
strategy:
fail-fast: false
matrix:
# Keep this in lockstep with pyproject classifiers; packaging tests enforce that.
python: ["3.10", "3.11", "3.12", "3.13"]
steps:
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
- uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
with:
python-version: ${{ matrix.python }}
cache: pip
- name: Install
run: |
python -m pip install --upgrade pip
# CPU-only torch keeps the job under a minute instead of pulling CUDA wheels
pip install torch --index-url https://download.pytorch.org/whl/cpu
# `mcp` so laya[mcp] resolution and the laya.mcp layer are both exercised, and
# `serve` because laya.serve and examples/server.py both import fastapi -- without
# it every test that touches them skips itself silently and passes.
# `langchain` for the same reason: without langchain-core the runnables fall back
# to their plain-object path, so tests/test_langchain.py never checks the real
# Runnable behaviour (batch delegation, pydantic field validation) it is there for.
pip install -e ".[mcp,serve,langchain]" httpx
# The pytest-style suites need the runner; nothing else in this job uses it.
pip install pytest
- name: Import check
run: python -c "import laya; print(laya.__version__); print(sorted(laya.__all__))"
- name: Routing and language-detection tests
run: |
python tests/test_router.py
python tests/test_criteria.py
python tests/test_batch.py
python tests/test_predict_long.py
python tests/test_attention_dynamic_shapes.py
python tests/test_hooks.py
python tests/test_hooks_api.py
python tests/test_structured.py
python tests/test_structured_api.py
python tests/test_structured_docs.py
python tests/test_onnx_lang_parity.py
python tests/test_onnx_batch.py
python tests/test_onnx_long.py
python tests/test_onnx_sort.py
python tests/test_download.py
python tests/test_shortlist.py
python tests/test_decision_model.py
python tests/test_head_checkpointing.py
python tests/test_packaging.py
python tests/test_compose_env.py
python tests/test_audit_scope.py
python tests/test_doc_tables.py
python tests/test_env_docs.py
python tests/test_tokenizer_cache.py
python tests/test_tokenizer_concurrency.py
python tests/test_question_token_reuse.py
python tests/test_lazy_import.py
python tests/test_runtime_fixes.py
python tests/test_option_collapse.py
python tests/test_lang_guess.py
python tests/test_identifier_complexity.py
python tests/test_lang_stats.py
python tests/test_calibration_persistence.py
python tests/test_context_manager.py
python tests/test_criteria_normalization.py
python tests/test_docker_entrypoint.py
python tests/test_empty_questions.py
python tests/test_router_memory.py
python tests/test_shortlist_cosine.py
python tests/test_temperature_loading.py
python tests/test_confidence.py
python tests/test_cli.py
python tests/test_mcp.py
python tests/test_langchain.py
python tests/test_portability.py
python tests/test_training.py
python tests/test_example_server_limits.py
python tests/test_blank_lang_routing.py
python tests/test_export_onnx_safety.py
python tests/test_load_errors.py
python tests/test_predict_batch.py
python tests/test_revision_pinning.py
python tests/test_example_server_errors.py
python tests/test_crewai.py
python tests/test_llamaindex.py
- name: LangChain integration runs on the real base class
run: |
# The install above is what makes tests/test_langchain.py run on real Runnables, and
# nothing proves it took: with langchain-core missing, the eight checks inside that
# file's `if _RUNNABLE_AVAILABLE:` block vanish instead of failing, and it still
# prints FAIL: 0 -- measured here: PASS: 87 with the extra, PASS: 80 without, exit 0
# both ways. This step asserts the install took, then checks what only a real install
# can check.
python -c "from laya.integrations.langchain import RunnableSerializable; \
assert RunnableSerializable is not object, \
'langchain-core is missing: laya.integrations.langchain fell back to a plain object, ' \
'so the LangChain suites in this job test a shape no install produces'"
python tests/test_langchain_real_mode.py
- name: Email cleaning tests
run: python tests/test_email.py
# Full dependency tree here, so the stub pipeline and the metric pins that need numpy
# run instead of skipping. The audit itself runs in `feishu-benchmark`, where nothing
# is installed.
- name: Chinese short-command benchmark tests
run: python -m unittest discover -s research/benchmarks/zh_short_commands/tests -v
# These are pytest suites, and `python tests/<name>.py` does not run them: it defines
# the test functions and exits 0, having asserted nothing. They were counted by the suite
# list above without ever executing, which is how the HTTP surface came to have no coverage
# in CI at all (#374).
#
# test_onnx.py and test_fast.py are deliberately not here: both are skip-guarded on an extra
# this job does not install, so they would pass trivially, and their real execution belongs
# in a lane that installs `onnx` or has a GPU.
- name: Pytest suites (serve, batch, system-one, audit regressions, truncation, compile)
run: |
python -m pytest tests/test_serve.py \
tests/test_router_batch.py \
tests/test_predict_batch.py \
tests/test_system_one_lang.py \
tests/test_audit_regressions.py \
tests/test_truncation_direction.py \
tests/test_compile.py \
-q
# The `test` job above deliberately stays free of `onnx`: this is the lane that
# installs it, so the exporter's quantization suite runs for real instead of
# skip-guarding into a trivial pass. Weight-free -- it builds tiny synthetic
# graphs, no checkpoint download.
onnx-export:
name: onnx export (quantization, weight-free)
runs-on: ubuntu-latest
timeout-minutes: 10
steps:
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
- uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
with:
python-version: "3.11"
cache: pip
- name: Install
run: |
python -m pip install --upgrade pip
pip install torch --index-url https://download.pytorch.org/whl/cpu
pip install -e .
pip install onnx onnxruntime
- name: Quantized-export tests
run: python tests/test_onnx_quantize.py
# The eval harness needs numpy and laya only, so it stays a separate, weight-free job: metric
# math, dataset parsing, the CLI and the API guard. The real-checkpoint gate is the scheduled
# `evals` workflow, which needs weights and network.
evals:
name: evals (metric math, dataset, API guard)
runs-on: ubuntu-latest
timeout-minutes: 20
steps:
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
- uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
with:
python-version: "3.11"
cache: pip
- name: Install
run: |
python -m pip install --upgrade pip
pip install torch --index-url https://download.pytorch.org/whl/cpu
pip install -e . pytest
- name: Eval harness tests
run: python -m pytest tests/test_evals.py -q
- name: Eval API guard
run: python tests/test_evals_api.py
# The ONNX runner suite fakes the agent, so it stays weight-free and needs no
# onnxruntime: the module only imports numpy, and ONNXAgent's constructor is
# never called.
- name: Eval ONNX runner tests
run: python tests/test_evals_onnx.py
- name: Fixture dataset validates
run: laya-evals validate research/evals/fixture.jsonl
# Every workflow in this repo ran on ubuntu-latest only, so a test that fails off Linux was
# invisible until a user hit it -- which is what happened to tests/test_download.py (#140).
#
# A separate job rather than an extra axis on `test`, deliberately: it leaves that job's name,
# and therefore the `tests (pyX.Y)` required checks, exactly as they are. It runs one Python
# because the variable that was untested is the OS, not the interpreter.
test-windows:
name: tests (windows)
runs-on: windows-latest
timeout-minutes: 20
steps:
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
- uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
with:
python-version: "3.11"
cache: pip
# The +cpu index used on Linux is Linux/macOS only. On Windows the default PyPI wheel is
# already CPU-only, so installing torch plainly is both correct and what a user gets.
# The serve extra and pytest are here for the same reason as the Linux job: the HTTP
# surface is otherwise skipped and the pytest suites otherwise do not run at all.
- name: Install
run: |
python -m pip install --upgrade pip
pip install torch
pip install -e ".[serve]"
pip install pytest
shell: bash
# bash on this runner so the quoting rules match the Linux job; the default shell on a
# Windows runner is PowerShell, where `python -c "...; ..."` is not the same command.
- name: Import check
run: python -c "import laya; print(laya.__version__); print(sorted(laya.__all__))"
shell: bash
# Same suites as the Linux job. Guarded with -f so a suite added by another open branch
# cannot make this job red on a tree that does not have it yet.
#
# test_docker_entrypoint is skipped here on purpose. It drives docker/entrypoint.py, which
# only ever runs inside a Linux container, through `os.execvp`. On Windows exec* rebuilds
# the command line from argv without re-applying the quoting `subprocess` used to build it,
# so its `-c "<script>"` argument arrives split at the first space and the child dies with
# "SyntaxError: invalid syntax" before printing. The environment passthrough the suite
# exists to check is unaffected, and switching the entrypoint to `subprocess.run` makes it
# pass -- a change to a container script, so it belongs in its own PR, not this one.
- name: Routing and language-detection tests
run: |
for t in test_router test_criteria test_batch test_predict_long \
test_structured test_structured_api test_onnx_lang_parity \
test_download test_shortlist test_decision_model \
test_head_checkpointing test_packaging test_doc_tables \
test_tokenizer_cache test_tokenizer_concurrency test_lazy_import \
test_runtime_fixes test_lang_guess test_identifier_complexity \
test_lang_stats test_calibration_persistence test_context_manager \
test_criteria_normalization test_docker_entrypoint \
test_empty_questions test_router_memory test_shortlist_cosine \
test_temperature_loading test_confidence test_cli test_mcp \
test_langchain test_portability test_training \
test_example_server_limits test_blank_lang_routing \
test_export_onnx_safety test_load_errors test_revision_pinning; do
if [ "$t" = "test_docker_entrypoint" ]; then
echo "skipping $t: docker/entrypoint.py is Linux-container only"
continue
fi
if [ -f "tests/$t.py" ]; then python "tests/$t.py"; fi
done
shell: bash
- name: Prediction hook tests
run: |
python tests/test_hooks.py
python tests/test_hooks_api.py
shell: bash
- name: Question token reuse tests
run: python tests/test_question_token_reuse.py
shell: bash
# Same pytest suites as the Linux job. They are named explicitly rather than added to
# the loop above, because the loop runs each file as a script and that is exactly what does
# not execute a pytest suite (#374).
- name: Pytest suites (serve, batch, system-one, audit regressions, truncation, compile)
run: |
python -m pytest tests/test_serve.py \
tests/test_router_batch.py \
tests/test_predict_batch.py \
tests/test_system_one_lang.py \
tests/test_audit_regressions.py \
tests/test_truncation_direction.py \
tests/test_compile.py \
-q
shell: bash
- name: Email cleaning tests
run: python tests/test_email.py
shell: bash
lint:
name: lint
runs-on: ubuntu-latest
timeout-minutes: 5
steps:
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
- uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
with:
python-version: "3.11"
- run: pip install ruff
- name: Ruff
run: ruff check laya/ --select=E9,F63,F7,F82,F401,F811 --line-length=120
- name: Byte-compile every module
run: python -m compileall -q laya/ tests/
build:
name: package
runs-on: ubuntu-latest
timeout-minutes: 10
steps:
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
- uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
with:
python-version: "3.11"
- run: pip install build twine
- run: python -m build
- name: Validate metadata
run: python -m twine check dist/*
- name: Version must match the tag on a release
if: startsWith(github.ref, 'refs/tags/v')
run: |
pkg=$(python -c "import re;print(re.search(r'version = \"(.*)\"', open('pyproject.toml').read()).group(1))")
tag="${GITHUB_REF_NAME#v}"
test "$pkg" = "$tag" || { echo "pyproject $pkg != tag $tag"; exit 1; }
- uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2
with:
name: dist
path: dist/