mirror of
https://github.com/anomalyco/opencode.git
synced 2026-10-03 02:31:52 +08:00
fix(ai): expose evaluation confidence (#51930)
This commit is contained in:
@@ -129,8 +129,9 @@ VercelAIGateway.configure().experimental.evaluation("typesafe-ai/jev")
|
||||
|
||||
OpenRouter reads `OPENROUTER_API_KEY`. Vercel reads `AI_GATEWAY_API_KEY`, then `VERCEL_OIDC_TOKEN`.
|
||||
The common API uses `boolean`; System One routes lower it to native `noul`.
|
||||
Choice and score confidence plus score legends remain available in provider metadata, and the
|
||||
provider's rounded probabilities are returned unchanged.
|
||||
Choice and score answers include `confidence` when the provider returns it, such as
|
||||
`response.answers.department.confidence`. Score legends remain available in provider metadata, and
|
||||
the provider's rounded probabilities are returned unchanged.
|
||||
|
||||
## Alibaba Cloud Model Studio
|
||||
|
||||
|
||||
@@ -63,6 +63,7 @@ export const ChoiceAnswer = Schema.Struct({
|
||||
type: Schema.Literal("choice"),
|
||||
choice: Schema.String,
|
||||
probabilities: Schema.optional(Schema.Record(Schema.String, Probability)),
|
||||
confidence: Schema.optional(Probability),
|
||||
})
|
||||
export type ChoiceAnswer = Schema.Schema.Type<typeof ChoiceAnswer>
|
||||
|
||||
@@ -70,6 +71,7 @@ export const ScoreAnswer = Schema.Struct({
|
||||
type: Schema.Literal("score"),
|
||||
score: Schema.Number,
|
||||
probabilities: Schema.optional(Schema.Record(Schema.String, Probability)),
|
||||
confidence: Schema.optional(Probability),
|
||||
})
|
||||
export type ScoreAnswer = Schema.Schema.Type<typeof ScoreAnswer>
|
||||
|
||||
@@ -92,6 +94,7 @@ export type AnswerFor<Question extends EvaluationQuestion> = Question extends {
|
||||
readonly type: "choice"
|
||||
readonly choice: Extract<keyof Criteria, string>
|
||||
readonly probabilities?: Readonly<Record<Extract<keyof Criteria, string>, number>>
|
||||
readonly confidence?: number
|
||||
}
|
||||
: Question extends { readonly type: "score" }
|
||||
? ScoreAnswer
|
||||
|
||||
@@ -142,32 +142,37 @@ export const model = <Options extends EvaluationOptions = EvaluationOptions>(cfg
|
||||
Effect.mapError((cause) => fail("System One returned an invalid response", cause, text)),
|
||||
)
|
||||
|
||||
const confidence: Record<string, number> = {}
|
||||
const legend: Record<string, Record<string, Schema.Json>> = {}
|
||||
const answers = Object.fromEntries(
|
||||
Object.entries(data.answers).map(([id, answer]): [string, EvaluationAnswer] => {
|
||||
if (answer.type === "noul") return [id, { type: "boolean", probability: answer.noul }]
|
||||
if (answer.type === "choice") {
|
||||
if (answer.confidence !== undefined) confidence[id] = answer.confidence
|
||||
return [
|
||||
id,
|
||||
{
|
||||
type: "choice",
|
||||
choice: answer.choice,
|
||||
probabilities: answer.probabilities,
|
||||
...(answer.confidence === undefined ? {} : { confidence: answer.confidence }),
|
||||
},
|
||||
]
|
||||
}
|
||||
if (answer.confidence !== undefined) confidence[id] = answer.confidence
|
||||
if (answer.legend !== undefined) legend[id] = answer.legend
|
||||
return [id, { type: "score", score: answer.score, probabilities: answer.probabilities }]
|
||||
return [
|
||||
id,
|
||||
{
|
||||
type: "score",
|
||||
score: answer.score,
|
||||
probabilities: answer.probabilities,
|
||||
...(answer.confidence === undefined ? {} : { confidence: answer.confidence }),
|
||||
},
|
||||
]
|
||||
}),
|
||||
)
|
||||
const meta = {
|
||||
...(data.id === undefined ? {} : { responseId: data.id }),
|
||||
...(data.provider === undefined ? {} : { provider: data.provider }),
|
||||
...data.provider_metadata?.[cfg.providerMetadataKey],
|
||||
...(Object.keys(confidence).length === 0 ? {} : { confidence }),
|
||||
...(Object.keys(legend).length === 0 ? {} : { legend }),
|
||||
}
|
||||
return new EvaluationResponse({
|
||||
|
||||
@@ -39,17 +39,18 @@ describe("experimental Evaluation", () => {
|
||||
type: "choice",
|
||||
choice: "billing",
|
||||
probabilities: { billing: 0.9, technical: 0.1 },
|
||||
confidence: 0.8,
|
||||
})
|
||||
expect(response.answers.urgency).toEqual({
|
||||
type: "score",
|
||||
score: 1.2,
|
||||
probabilities: { "0": 0, "1": 0.8, "2": 0.2 },
|
||||
confidence: 0.6,
|
||||
})
|
||||
expect(response.answers.refund).toEqual({ type: "boolean", probability: 0.97 })
|
||||
expect(response.usage?.totalTokens).toBe(36)
|
||||
expect(response.providerMetadata).toEqual({
|
||||
typesafe: {
|
||||
confidence: { department: 0.8, urgency: 0.6 },
|
||||
legend: { urgency: { "0": "Can wait", "1": "Needs attention", "2": "Blocking" } },
|
||||
},
|
||||
})
|
||||
|
||||
@@ -26,8 +26,10 @@ const request = Evaluation.request({
|
||||
const result = EvaluationClient.evaluate(request)
|
||||
type Result = Success<typeof result>
|
||||
type Choice = Assert<Equal<Result["answers"]["topic"]["choice"], "billing" | "support">>
|
||||
type Confidence = Assert<Equal<Result["answers"]["topic"]["confidence"], number | undefined>>
|
||||
type ClientRequirements = Assert<Equal<Requirements<typeof result>, Service>>
|
||||
void (true satisfies Choice)
|
||||
void (true satisfies Confidence)
|
||||
void (true satisfies ClientRequirements)
|
||||
|
||||
Effect.gen(function* () {
|
||||
|
||||
@@ -92,5 +92,6 @@ const assertEvaluation = <Options extends EvaluationOptions>(
|
||||
expect(response.answers.refund.probability).toBeGreaterThan(0.5)
|
||||
expect(response.usage?.inputTokens).toBeGreaterThan(0)
|
||||
expect(response.usage?.outputTokens).toBeGreaterThan(0)
|
||||
expect(response.providerMetadata?.[metadataKey]?.confidence).toBeDefined()
|
||||
expect(response.answers.department.confidence).toBeGreaterThan(0)
|
||||
expect(response.answers.urgency.confidence).toBeGreaterThan(0)
|
||||
})
|
||||
|
||||
Reference in New Issue
Block a user