feat(agent): report the calibrated confidence alongside the existing one (#126)

`confidence` carries two definitions in one response payload:

    choice / score  ->  1 - H(p) / log(k)          (normalized entropy)
    noul            ->  max(p_true, 1 - p_true)    == max(p)

They are not on the same scale. The same two-option distribution comes back as
0.90 from a `noul` and 0.53 from an equivalent two-option `choice`. Every shipped
preset mixes the types in a single call -- `moderation_questions()` is four `noul`
and one `score` -- and the README's "Automated Confidence Gating" section gates all
of them on one threshold, so at P(true)=0.90 the same evidence is automated as a
noul and escalated as a choice.

Only max(p) has the property that section relies on. Both benchmark harnesses take
`conf = max(probs)` before calling `ece_score` (research/scripts/bench_local.py,
build_benchmark_nb.py), so every ECE figure in the README describes max(p), and
temperature scaling fits the same quantity. Normalized entropy carries no such
guarantee: on perfectly calibrated synthetic data, ECE on max(p) is 0.0111 while
ECE on normalized entropy is 0.3682.

This is additive. `confidence` is untouched on every question type, so nothing a
caller gates on today moves. A new `answer_confidence` reports max(p) on all three
types, so a caller can gate across types on one number now, and `answer_confidence`
is exported next to `confidence_from_probs` with each docstring saying what it does
and does not promise.

It also leaves a clean path for changing `confidence` itself in a deliberate
release: callers migrate to `answer_confidence` first, and the later switch is then
a no-op for them.

tests/test_confidence.py covers it without loading weights: max(p) behaviour, the
new field agreeing with the shipped noul value across the range, the threshold
disagreement between the two scales, and an end-to-end calibration check showing
ECE near zero on max(p) and far from it on entropy.

Co-authored-by: NandhaKishorM <nandakishor@convaiinnovations.com>
This commit is contained in:
君莫
2026-09-24 08:47:36 +05:30
committed by GitHub
co-authored by NandhaKishorM
parent 5877415e3c
commit 50df9bb3da
6 changed files with 162 additions and 1 deletions
+1
View File
@@ -104,6 +104,7 @@ jobs:
python tests/test_router_memory.py
python tests/test_shortlist_cosine.py
python tests/test_temperature_loading.py
python tests/test_confidence.py
python tests/test_cli.py
python tests/test_mcp.py
python tests/test_langchain.py
+2
View File
@@ -26,6 +26,7 @@ _LAZY_ATTRS = {
"proper_reward": (".common", "proper_reward"),
"td_lambda_targets": (".common", "td_lambda_targets"),
"ece_score": (".common", "ece_score"),
"answer_confidence": (".common", "answer_confidence"),
"confidence_from_probs": (".common", "confidence_from_probs"),
"render_options": (".common", "render_options"),
"QTYPES": (".common", "QTYPES"),
@@ -80,6 +81,7 @@ __all__ = [
"proper_reward",
"td_lambda_targets",
"ece_score",
"answer_confidence",
"confidence_from_probs",
"render_options",
"QTYPES",
+11
View File
@@ -20,6 +20,7 @@ from .common import (
build_sequence,
clamp_temperature,
collate_items,
answer_confidence,
confidence_from_probs,
_resolve_noul_labels,
render_options,
@@ -651,6 +652,12 @@ class Agent(HookRegistry):
p = np.exp(z - z.max())
p = p / p.sum()
# `confidence` means one thing for `noul` (max(p)) and another for `choice` and
# `score` (normalized entropy), and only the first is the quantity temperature
# scaling fits and ECE measures. Rather than change one underneath existing
# callers, report both: `answer_confidence` is the calibrated one, on every
# question type, so a caller can gate across types on a single number.
ans_conf = round(answer_confidence(p, k), 4)
ext = {"act_probability": round(float(act[r, 0]), 4)}
if q["t"] == "choice":
@@ -660,6 +667,7 @@ class Agent(HookRegistry):
"choice": keys[int(p.argmax())],
"probabilities": {kk: round(float(v), 4) for kk, v in zip(keys, p)},
"confidence": round(confidence_from_probs(p, k), 4),
"answer_confidence": ans_conf,
"action": ext,
}
elif q["t"] == "score":
@@ -670,6 +678,7 @@ class Agent(HookRegistry):
"legend": {str(i): c for i, c in enumerate(q["crit"])},
"probabilities": {str(i): round(float(v), 4) for i, v in enumerate(p)},
"confidence": round(confidence_from_probs(p, k), 4),
"answer_confidence": ans_conf,
"action": ext,
}
else:
@@ -677,6 +686,8 @@ class Agent(HookRegistry):
"type": "noul",
"noul": round(float(p[1]), 4),
"confidence": round(max(float(p[1]), 1.0 - float(p[1])), 4),
# identical here: over two options max(p_true, 1 - p_true) is max(p)
"answer_confidence": ans_conf,
"action": ext,
}
return answers
+21 -1
View File
@@ -270,8 +270,28 @@ def ece_score(conf: np.ndarray, correct: np.ndarray, bins: int = 15) -> float:
return float(e)
def answer_confidence(p: np.ndarray, k: int) -> float:
"""Probability mass on the answer being reported: max(p).
This is the quantity temperature scaling fits, and the quantity every calibration figure in
this repository is computed on -- both benchmark harnesses take `conf = max(probs)` before
calling `ece_score`. It is therefore the one confidence with the property the README's
gating section relies on: of the answers returned at confidence c, about c of them are right.
`confidence_from_probs` below reports a different quantity on a different scale and carries
no such guarantee, so the two must not be compared against the same threshold.
"""
if k < 1:
return 1.0
return float(np.clip(np.max(p[:k]), 0.0, 1.0))
def confidence_from_probs(p: np.ndarray, k: int) -> float:
"""Normalized Shannon entropy confidence: 1 - H(p) / log(k)."""
"""Normalized Shannon entropy confidence: 1 - H(p) / log(k).
How concentrated the whole distribution is. Useful, but not calibrated: it is not what
temperature scaling fits and not what the reported ECE measures. See `answer_confidence`.
"""
if k < 2:
return 1.0
p = p[:k]
+5
View File
@@ -244,11 +244,15 @@ with patch.object(_agent, "confidence_from_probs", wraps=_agent.confidence_from_
check("decode/mixed answers retain calibrated values", decoded, {
"pick": {"type": "choice", "choice": "right",
"probabilities": {"left": 0.25, "right": 0.75}, "confidence": 0.1887,
"answer_confidence": 0.75,
"action": {"act_probability": 0.125}},
"level": {"type": "score", "score": 0.75, "legend": {"0": "low", "1": "high"},
"probabilities": {"0": 0.25, "1": 0.75}, "confidence": 0.1887,
"answer_confidence": 0.75,
"action": {"act_probability": 0.25}},
"flag": {"type": "noul", "noul": 0.2, "confidence": 0.8,
# max(p) over two options is max(p_true, 1 - p_true): the same number
"answer_confidence": 0.8,
"action": {"act_probability": 0.75}},
})
check("decode/entropy only computed for choice and score", entropy.call_count, 2)
@@ -261,6 +265,7 @@ with patch.object(_agent, "confidence_from_probs", wraps=_agent.confidence_from_
check("decode/noul confidence at p=%s" % true_probability, result["flag"], {
"type": "noul", "noul": true_probability,
"confidence": max(true_probability, 1.0 - true_probability),
"answer_confidence": max(true_probability, 1.0 - true_probability),
"action": {"act_probability": 0.75},
})
check("decode/noul-only answers skip entropy", entropy.call_count, 0)
+122
View File
@@ -0,0 +1,122 @@
"""`confidence` carries two definitions, and only one of them is the calibrated quantity.
choice / score -> confidence_from_probs(p, k) == 1 - H(p) / log(k)
noul -> max(p_true, 1 - p_true) == max(p)
They are not on the same scale. The same two-option distribution comes back as 0.90 from a
`noul` and 0.53 from an equivalent two-option `choice`, and every shipped preset mixes the two
types in one call -- `moderation_questions()` is four `noul` and one `score`. The README's
"Automated Confidence Gating" section gates all of them on a single threshold.
Both benchmark harnesses in research/scripts take `conf = max(probs)` before calling
`ece_score`, so every ECE figure in the README describes max(p) and not normalized entropy.
`answer_confidence` reports that quantity on every question type, additively: `confidence` is
untouched, so nothing a caller gates on today moves.
No weights are loaded: the confidence helpers are pure.
"""
import math
import os
import sys
import numpy as np
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
from laya.common import answer_confidence, confidence_from_probs, ece_score # noqa: E402
PASS, FAIL = [], []
def check(name, got, want):
if got == want:
PASS.append(name)
else:
FAIL.append("%s: got %r, want %r" % (name, got, want))
def check_true(name, cond, detail=""):
if cond:
PASS.append(name)
else:
FAIL.append("%s %s" % (name, detail))
def close(a, b, tol=1e-9):
return abs(a - b) <= tol
# --------------------------------------------------------------- answer_confidence is max(p)
for probs in ([0.5, 0.5], [0.1, 0.9], [0.7, 0.2, 0.1], [0.25] * 4, [1.0, 0.0, 0.0]):
p = np.array(probs)
check_true("answer/max of %s" % (probs,), close(answer_confidence(p, len(probs)), max(probs)))
check("answer/only the first k entries count", answer_confidence(np.array([0.4, 0.6, 0.99]), 2), 0.6)
check("answer/k=1 is certain", answer_confidence(np.array([1.0]), 1), 1.0)
check("answer/k=0 does not crash", answer_confidence(np.array([]), 0), 1.0)
# --------------------------------------------------------------- it matches noul's definition
# `noul` reports max(p_true, 1 - p_true), which over two options is exactly max(p). So for a
# noul answer the new field equals the existing one, and the two agree by construction.
for p_true in (0.0, 0.05, 0.3, 0.5, 0.62, 0.9, 1.0):
p = np.array([1.0 - p_true, p_true])
check_true("noul/answer_confidence equals the shipped noul confidence at p=%.2f" % p_true,
close(answer_confidence(p, 2), max(p_true, 1.0 - p_true)))
# --------------------------------------------------------------- the two scales really differ
# Same distribution, two question types. 0.85 is the threshold the README's gating section uses.
disagree = []
for p_true in (0.60, 0.70, 0.80, 0.85, 0.90, 0.95):
p = np.array([1.0 - p_true, p_true])
if (answer_confidence(p, 2) >= 0.85) != (confidence_from_probs(p, 2) >= 0.85):
disagree.append(p_true)
check_true("scales/the two disagree across the documented threshold", disagree == [0.85, 0.90, 0.95],
"disagreed at %s" % (disagree,))
check_true("scales/entropy sits far below max(p) on the same distribution",
confidence_from_probs(np.array([0.1, 0.9]), 2) < 0.55 < answer_confidence(np.array([0.1, 0.9]), 2))
# max(p) over two options has a floor of 0.5; entropy reads 0.0 for the same coin flip.
check("scales/a coin flip is 0.5 on max(p)", answer_confidence(np.array([0.5, 0.5]), 2), 0.5)
check("scales/and 0.0 on entropy", confidence_from_probs(np.array([0.5, 0.5]), 2), 0.0)
# --------------------------------------------------------------- only one of them is calibrated
# Perfectly calibrated predictions: an answer reported at top probability c is right exactly c
# of the time. ECE on max(p) must be near zero. ECE on normalized entropy must not be, which is
# why it cannot be compared against a probability threshold.
rng = np.random.default_rng(0)
tops, ents, correct = [], [], []
for c in np.linspace(0.30, 0.99, 24):
rest = (1.0 - c) / 2.0
p = np.array([c, rest, rest])
for _ in range(400):
tops.append(answer_confidence(p, 3))
ents.append(confidence_from_probs(p, 3))
correct.append(1.0 if rng.random() < c else 0.0)
tops, ents, correct = np.array(tops), np.array(ents), np.array(correct)
ece_top, ece_ent = ece_score(tops, correct), ece_score(ents, correct)
check_true("calibration/max(p) is calibrated on calibrated data (ECE %.4f)" % ece_top,
ece_top < 0.03, "ECE %.4f" % ece_top)
check_true("calibration/entropy is not (ECE %.4f)" % ece_ent, ece_ent > 0.20, "ECE %.4f" % ece_ent)
check_true("calibration/entropy is worse by a wide margin", ece_ent > 5 * ece_top,
"top %.4f vs entropy %.4f" % (ece_top, ece_ent))
# --------------------------------------------------------------- the entropy helper is untouched
for probs, k in (([0.1, 0.9], 2), ([0.25] * 4, 4), ([0.7, 0.2, 0.1], 3)):
p = np.array(probs)
ent = -(p * np.log(np.clip(p, 1e-12, 1.0))).sum()
check_true("entropy/formula for k=%d" % k,
close(confidence_from_probs(p, k), float(np.clip(1 - ent / math.log(k), 0.0, 1.0))))
# --------------------------------------------------------------- exported
import laya # noqa: E402
check_true("export/answer_confidence is importable from laya", hasattr(laya, "answer_confidence"))
check_true("export/answer_confidence is in __all__", "answer_confidence" in laya.__all__)
check_true("export/answer_confidence is in dir()", "answer_confidence" in dir(laya))
print("\n%d passed, %d failed" % (len(PASS), len(FAIL)))
for f in FAIL:
print(" FAIL", f)
if not FAIL:
print("all confidence tests passed")
sys.exit(1 if FAIL else 0)