mirror of
https://github.com/sandbaseai/sandbase-harness.git
synced 2026-09-28 06:03:00 +08:00
* chore(deps): bump ai v7, @ai-sdk/* v4, zod 4, vite 8, plugin-react 6 Supersedes the six dependabot PRs for these packages. The lockfile is rewritten wholesale: the old lock pinned vite 5 / plugin-react 4 as root ranges, and since those two are mutual peers npm could not move them incrementally (ERESOLVE). Regenerating resolves cleanly with no overrides. Required migrations, all driven by upstream renames or removals: - LanguageModelV1 -> LanguageModel (the v7 type union admits V2/V3/V4 only) - streamText: maxSteps -> stopWhen(stepCountIs), maxTokens -> maxOutputTokens, toolCallStreaming removed (now unconditional) - usage: promptTokens/completionTokens -> inputTokens/outputTokens - step.reasoning is an array; use step.reasoningText for the text - stream parts: textDelta -> text, step-finish -> finish-step - zod 4 requires an explicit key schema for z.record() Three breakages that neither typecheck nor the suite would have caught: - @ai-sdk/openai v2 switched the bare `openai(id)` call to the Responses API, which Ollama/vLLM/DeepSeek/minimax do not implement. Pinned to `.chat(id)` so every OpenAI-compatible provider stays on /chat/completions, with a test asserting the request URL. - Tools carry their schema in `inputSchema`, not `parameters`. A missing inputSchema is not rejected: asSchema(undefined) yields an empty object schema, so every tool would have reached the model stripped of its arguments. Converted at the single toAiTool boundary; the internal `parameters` convention is unchanged. - Conversation history rebuilt by events-to-messages used v4 part shapes (tool-call.args, tool-result.result, image.mimeType). Any second turn after a tool call would fail message validation. reasoning_effort now ships as a call-time provider option, since the provider dropped the constructor settings argument that used to carry it. Added tests that assert it reaches the request body, and that it is absent when unset -- the previous test only checked the config object, so a silently dropped value would still have passed. MCP talks to @modelcontextprotocol/sdk (already a dependency) directly, because the AI SDK removed its experimental_createMCPClient wrapper. The reconnect/backoff/degradation logic and the mcp_<server>_<tool> namespacing are untouched. Behavior change worth noting: an unusable tool call (malformed JSON, unknown tool) is now reported back to the model instead of throwing, so the loop recovers and the turn ends idle rather than failed. The call is still never announced or executed; two status assertions were updated to match. typecheck (src/tests/console), build, package:check and smoke:release all pass. Remaining suite failures are pre-existing platform gaps unrelated to these packages (tar CLI, /bin/sh, POSIX path assertions, symlink EPERM). TypeScript 7 (PR #100) is intentionally left for a separate commit. * chore(build): migrate tsup to tsdown, bump typescript to v7
110 lines
4.1 KiB
TypeScript
110 lines
4.1 KiB
TypeScript
/**
|
|
* Integration test: model/provider errors surface as a failed turn (not a
|
|
* silent idle). Covers the fullStream 'error' part handling in DefaultStrategy.
|
|
*/
|
|
|
|
import { describe, it, expect, beforeEach, afterEach } from 'vitest';
|
|
import { join } from 'node:path';
|
|
import { mkdtempSync, rmSync } from 'node:fs';
|
|
import { tmpdir } from 'node:os';
|
|
import { Database } from '@/core/db/database.js';
|
|
import { SessionManager } from '@/core/session/session-manager.js';
|
|
import { DefaultSessionExecutor } from '@/core/session/executor.js';
|
|
import { DefaultStrategy } from '@/strategy/default-strategy.js';
|
|
import { ModelRegistry } from '@/model/registry.js';
|
|
import { LocalSandboxProvider } from '@/sandbox/local-provider.js';
|
|
import type { LanguageModel } from 'ai';
|
|
|
|
/** A model whose stream throws — simulates a provider/auth/network failure. */
|
|
function throwingModel(): LanguageModel {
|
|
return {
|
|
specificationVersion: 'v4',
|
|
provider: 'test',
|
|
modelId: 'boom',
|
|
supportedUrls: {},
|
|
async doGenerate() {
|
|
throw new Error('401 unauthorized');
|
|
},
|
|
async doStream() {
|
|
throw new Error('401 unauthorized');
|
|
},
|
|
} as unknown as LanguageModel;
|
|
}
|
|
|
|
describe('Model error → failed turn', () => {
|
|
let db: Database;
|
|
let manager: SessionManager;
|
|
let tmpDir: string;
|
|
|
|
beforeEach(() => {
|
|
tmpDir = mkdtempSync(join(tmpdir(), 'ma-strerr-'));
|
|
db = new Database(join(tmpDir, 'test.db'));
|
|
db.runMigrations();
|
|
db.exec(`INSERT INTO environments (id, name, config) VALUES ('env_default', 'local', '{}')`);
|
|
db.exec(`INSERT INTO agents (id, name, definition) VALUES ('agent_b', 'b', '{}')`);
|
|
|
|
manager = new SessionManager(db);
|
|
const modelRegistry = new ModelRegistry();
|
|
(modelRegistry as any).createModel = () => throwingModel();
|
|
const executor = new DefaultSessionExecutor({
|
|
agents: [{ name: 'b', model: 'm', system: 'p' }],
|
|
modelRegistry,
|
|
sandboxProvider: new LocalSandboxProvider(tmpDir),
|
|
strategy: new DefaultStrategy(),
|
|
eventLogger: manager.getEventLogger(),
|
|
});
|
|
manager.setExecutor(executor);
|
|
});
|
|
|
|
afterEach(() => {
|
|
db.close();
|
|
rmSync(tmpDir, { recursive: true, force: true });
|
|
});
|
|
|
|
it('transitions to failed and records a session.error (not silent idle)', async () => {
|
|
const session = manager.create({ agent: 'agent_b' });
|
|
await manager.sendEvent(session.id, {
|
|
type: 'user.message',
|
|
content: [{ type: 'text', text: 'hi' }],
|
|
} as any);
|
|
|
|
// Wait for the turn to fail
|
|
await new Promise((r) => setTimeout(r, 200));
|
|
|
|
const status = manager.get(session.id)!.status;
|
|
expect(status).toBe('failed');
|
|
|
|
const events = manager.getEventLogger().getEvents(session.id);
|
|
const errEvent = events.find((e) => e.type === 'session.error');
|
|
expect(errEvent).toBeDefined();
|
|
});
|
|
|
|
it('accepts a new message on a failed session and resumes the turn', async () => {
|
|
const session = manager.create({ agent: 'agent_b' });
|
|
await manager.sendEvent(session.id, {
|
|
type: 'user.message',
|
|
content: [{ type: 'text', text: 'hi' }],
|
|
} as any);
|
|
await new Promise((r) => setTimeout(r, 200));
|
|
expect(manager.get(session.id)!.status).toBe('failed');
|
|
|
|
// A failed session is recoverable: sending another message must be
|
|
// accepted (not rejected as terminal) and must re-run the turn.
|
|
const result = await manager.sendEvent(session.id, {
|
|
type: 'user.message',
|
|
content: [{ type: 'text', text: 'retry please' }],
|
|
} as any);
|
|
expect(result.accepted).toBe(true);
|
|
|
|
// The resume runs a fresh turn. Its status_running event proves the
|
|
// failed → running transition was allowed. (This model always throws, so
|
|
// it lands back in failed — the point is the turn was re-entered.)
|
|
await new Promise((r) => setTimeout(r, 200));
|
|
const events = manager.getEventLogger().getEvents(session.id);
|
|
const runningEvents = events.filter((e) => e.type === 'session.status_running');
|
|
expect(runningEvents.length).toBe(2);
|
|
const userMessages = events.filter((e) => e.type === 'user.message');
|
|
expect(userMessages.length).toBe(2);
|
|
});
|
|
});
|