fix(ai): expose evaluation confidence (#51930)

This commit is contained in:
James Long
2026-09-28 15:24:21 -04:00
committed by GitHub
parent 6cf442b545
commit 46e53e3f2b
6 changed files with 22 additions and 9 deletions
+3 -2
View File
@@ -129,8 +129,9 @@ VercelAIGateway.configure().experimental.evaluation("typesafe-ai/jev")
OpenRouter reads `OPENROUTER_API_KEY`. Vercel reads `AI_GATEWAY_API_KEY`, then `VERCEL_OIDC_TOKEN`.
The common API uses `boolean`; System One routes lower it to native `noul`.
Choice and score confidence plus score legends remain available in provider metadata, and the
provider's rounded probabilities are returned unchanged.
Choice and score answers include `confidence` when the provider returns it, such as
`response.answers.department.confidence`. Score legends remain available in provider metadata, and
the provider's rounded probabilities are returned unchanged.
## Alibaba Cloud Model Studio
@@ -63,6 +63,7 @@ export const ChoiceAnswer = Schema.Struct({
type: Schema.Literal("choice"),
choice: Schema.String,
probabilities: Schema.optional(Schema.Record(Schema.String, Probability)),
confidence: Schema.optional(Probability),
})
export type ChoiceAnswer = Schema.Schema.Type<typeof ChoiceAnswer>
@@ -70,6 +71,7 @@ export const ScoreAnswer = Schema.Struct({
type: Schema.Literal("score"),
score: Schema.Number,
probabilities: Schema.optional(Schema.Record(Schema.String, Probability)),
confidence: Schema.optional(Probability),
})
export type ScoreAnswer = Schema.Schema.Type<typeof ScoreAnswer>
@@ -92,6 +94,7 @@ export type AnswerFor<Question extends EvaluationQuestion> = Question extends {
readonly type: "choice"
readonly choice: Extract<keyof Criteria, string>
readonly probabilities?: Readonly<Record<Extract<keyof Criteria, string>, number>>
readonly confidence?: number
}
: Question extends { readonly type: "score" }
? ScoreAnswer
+10 -5
View File
@@ -142,32 +142,37 @@ export const model = <Options extends EvaluationOptions = EvaluationOptions>(cfg
Effect.mapError((cause) => fail("System One returned an invalid response", cause, text)),
)
const confidence: Record<string, number> = {}
const legend: Record<string, Record<string, Schema.Json>> = {}
const answers = Object.fromEntries(
Object.entries(data.answers).map(([id, answer]): [string, EvaluationAnswer] => {
if (answer.type === "noul") return [id, { type: "boolean", probability: answer.noul }]
if (answer.type === "choice") {
if (answer.confidence !== undefined) confidence[id] = answer.confidence
return [
id,
{
type: "choice",
choice: answer.choice,
probabilities: answer.probabilities,
...(answer.confidence === undefined ? {} : { confidence: answer.confidence }),
},
]
}
if (answer.confidence !== undefined) confidence[id] = answer.confidence
if (answer.legend !== undefined) legend[id] = answer.legend
return [id, { type: "score", score: answer.score, probabilities: answer.probabilities }]
return [
id,
{
type: "score",
score: answer.score,
probabilities: answer.probabilities,
...(answer.confidence === undefined ? {} : { confidence: answer.confidence }),
},
]
}),
)
const meta = {
...(data.id === undefined ? {} : { responseId: data.id }),
...(data.provider === undefined ? {} : { provider: data.provider }),
...data.provider_metadata?.[cfg.providerMetadataKey],
...(Object.keys(confidence).length === 0 ? {} : { confidence }),
...(Object.keys(legend).length === 0 ? {} : { legend }),
}
return new EvaluationResponse({
+2 -1
View File
@@ -39,17 +39,18 @@ describe("experimental Evaluation", () => {
type: "choice",
choice: "billing",
probabilities: { billing: 0.9, technical: 0.1 },
confidence: 0.8,
})
expect(response.answers.urgency).toEqual({
type: "score",
score: 1.2,
probabilities: { "0": 0, "1": 0.8, "2": 0.2 },
confidence: 0.6,
})
expect(response.answers.refund).toEqual({ type: "boolean", probability: 0.97 })
expect(response.usage?.totalTokens).toBe(36)
expect(response.providerMetadata).toEqual({
typesafe: {
confidence: { department: 0.8, urgency: 0.6 },
legend: { urgency: { "0": "Can wait", "1": "Needs attention", "2": "Blocking" } },
},
})
+2
View File
@@ -26,8 +26,10 @@ const request = Evaluation.request({
const result = EvaluationClient.evaluate(request)
type Result = Success<typeof result>
type Choice = Assert<Equal<Result["answers"]["topic"]["choice"], "billing" | "support">>
type Confidence = Assert<Equal<Result["answers"]["topic"]["confidence"], number | undefined>>
type ClientRequirements = Assert<Equal<Requirements<typeof result>, Service>>
void (true satisfies Choice)
void (true satisfies Confidence)
void (true satisfies ClientRequirements)
Effect.gen(function* () {
@@ -92,5 +92,6 @@ const assertEvaluation = <Options extends EvaluationOptions>(
expect(response.answers.refund.probability).toBeGreaterThan(0.5)
expect(response.usage?.inputTokens).toBeGreaterThan(0)
expect(response.usage?.outputTokens).toBeGreaterThan(0)
expect(response.providerMetadata?.[metadataKey]?.confidence).toBeDefined()
expect(response.answers.department.confidence).toBeGreaterThan(0)
expect(response.answers.urgency.confidence).toBeGreaterThan(0)
})