mirror of
https://github.com/NandhaKishorM/laya.git
synced 2026-09-28 07:52:57 +08:00
feat(agent): report the calibrated confidence alongside the existing one (#126)
`confidence` carries two definitions in one response payload:
choice / score -> 1 - H(p) / log(k) (normalized entropy)
noul -> max(p_true, 1 - p_true) == max(p)
They are not on the same scale. The same two-option distribution comes back as
0.90 from a `noul` and 0.53 from an equivalent two-option `choice`. Every shipped
preset mixes the types in a single call -- `moderation_questions()` is four `noul`
and one `score` -- and the README's "Automated Confidence Gating" section gates all
of them on one threshold, so at P(true)=0.90 the same evidence is automated as a
noul and escalated as a choice.
Only max(p) has the property that section relies on. Both benchmark harnesses take
`conf = max(probs)` before calling `ece_score` (research/scripts/bench_local.py,
build_benchmark_nb.py), so every ECE figure in the README describes max(p), and
temperature scaling fits the same quantity. Normalized entropy carries no such
guarantee: on perfectly calibrated synthetic data, ECE on max(p) is 0.0111 while
ECE on normalized entropy is 0.3682.
This is additive. `confidence` is untouched on every question type, so nothing a
caller gates on today moves. A new `answer_confidence` reports max(p) on all three
types, so a caller can gate across types on one number now, and `answer_confidence`
is exported next to `confidence_from_probs` with each docstring saying what it does
and does not promise.
It also leaves a clean path for changing `confidence` itself in a deliberate
release: callers migrate to `answer_confidence` first, and the later switch is then
a no-op for them.
tests/test_confidence.py covers it without loading weights: max(p) behaviour, the
new field agreeing with the shipped noul value across the range, the threshold
disagreement between the two scales, and an end-to-end calibration check showing
ECE near zero on max(p) and far from it on entropy.
Co-authored-by: NandhaKishorM <nandakishor@convaiinnovations.com>
This commit is contained in:
@@ -104,6 +104,7 @@ jobs:
|
||||
python tests/test_router_memory.py
|
||||
python tests/test_shortlist_cosine.py
|
||||
python tests/test_temperature_loading.py
|
||||
python tests/test_confidence.py
|
||||
python tests/test_cli.py
|
||||
python tests/test_mcp.py
|
||||
python tests/test_langchain.py
|
||||
|
||||
@@ -26,6 +26,7 @@ _LAZY_ATTRS = {
|
||||
"proper_reward": (".common", "proper_reward"),
|
||||
"td_lambda_targets": (".common", "td_lambda_targets"),
|
||||
"ece_score": (".common", "ece_score"),
|
||||
"answer_confidence": (".common", "answer_confidence"),
|
||||
"confidence_from_probs": (".common", "confidence_from_probs"),
|
||||
"render_options": (".common", "render_options"),
|
||||
"QTYPES": (".common", "QTYPES"),
|
||||
@@ -80,6 +81,7 @@ __all__ = [
|
||||
"proper_reward",
|
||||
"td_lambda_targets",
|
||||
"ece_score",
|
||||
"answer_confidence",
|
||||
"confidence_from_probs",
|
||||
"render_options",
|
||||
"QTYPES",
|
||||
|
||||
@@ -20,6 +20,7 @@ from .common import (
|
||||
build_sequence,
|
||||
clamp_temperature,
|
||||
collate_items,
|
||||
answer_confidence,
|
||||
confidence_from_probs,
|
||||
_resolve_noul_labels,
|
||||
render_options,
|
||||
@@ -651,6 +652,12 @@ class Agent(HookRegistry):
|
||||
p = np.exp(z - z.max())
|
||||
p = p / p.sum()
|
||||
|
||||
# `confidence` means one thing for `noul` (max(p)) and another for `choice` and
|
||||
# `score` (normalized entropy), and only the first is the quantity temperature
|
||||
# scaling fits and ECE measures. Rather than change one underneath existing
|
||||
# callers, report both: `answer_confidence` is the calibrated one, on every
|
||||
# question type, so a caller can gate across types on a single number.
|
||||
ans_conf = round(answer_confidence(p, k), 4)
|
||||
ext = {"act_probability": round(float(act[r, 0]), 4)}
|
||||
|
||||
if q["t"] == "choice":
|
||||
@@ -660,6 +667,7 @@ class Agent(HookRegistry):
|
||||
"choice": keys[int(p.argmax())],
|
||||
"probabilities": {kk: round(float(v), 4) for kk, v in zip(keys, p)},
|
||||
"confidence": round(confidence_from_probs(p, k), 4),
|
||||
"answer_confidence": ans_conf,
|
||||
"action": ext,
|
||||
}
|
||||
elif q["t"] == "score":
|
||||
@@ -670,6 +678,7 @@ class Agent(HookRegistry):
|
||||
"legend": {str(i): c for i, c in enumerate(q["crit"])},
|
||||
"probabilities": {str(i): round(float(v), 4) for i, v in enumerate(p)},
|
||||
"confidence": round(confidence_from_probs(p, k), 4),
|
||||
"answer_confidence": ans_conf,
|
||||
"action": ext,
|
||||
}
|
||||
else:
|
||||
@@ -677,6 +686,8 @@ class Agent(HookRegistry):
|
||||
"type": "noul",
|
||||
"noul": round(float(p[1]), 4),
|
||||
"confidence": round(max(float(p[1]), 1.0 - float(p[1])), 4),
|
||||
# identical here: over two options max(p_true, 1 - p_true) is max(p)
|
||||
"answer_confidence": ans_conf,
|
||||
"action": ext,
|
||||
}
|
||||
return answers
|
||||
|
||||
+21
-1
@@ -270,8 +270,28 @@ def ece_score(conf: np.ndarray, correct: np.ndarray, bins: int = 15) -> float:
|
||||
return float(e)
|
||||
|
||||
|
||||
def answer_confidence(p: np.ndarray, k: int) -> float:
|
||||
"""Probability mass on the answer being reported: max(p).
|
||||
|
||||
This is the quantity temperature scaling fits, and the quantity every calibration figure in
|
||||
this repository is computed on -- both benchmark harnesses take `conf = max(probs)` before
|
||||
calling `ece_score`. It is therefore the one confidence with the property the README's
|
||||
gating section relies on: of the answers returned at confidence c, about c of them are right.
|
||||
|
||||
`confidence_from_probs` below reports a different quantity on a different scale and carries
|
||||
no such guarantee, so the two must not be compared against the same threshold.
|
||||
"""
|
||||
if k < 1:
|
||||
return 1.0
|
||||
return float(np.clip(np.max(p[:k]), 0.0, 1.0))
|
||||
|
||||
|
||||
def confidence_from_probs(p: np.ndarray, k: int) -> float:
|
||||
"""Normalized Shannon entropy confidence: 1 - H(p) / log(k)."""
|
||||
"""Normalized Shannon entropy confidence: 1 - H(p) / log(k).
|
||||
|
||||
How concentrated the whole distribution is. Useful, but not calibrated: it is not what
|
||||
temperature scaling fits and not what the reported ECE measures. See `answer_confidence`.
|
||||
"""
|
||||
if k < 2:
|
||||
return 1.0
|
||||
p = p[:k]
|
||||
|
||||
@@ -244,11 +244,15 @@ with patch.object(_agent, "confidence_from_probs", wraps=_agent.confidence_from_
|
||||
check("decode/mixed answers retain calibrated values", decoded, {
|
||||
"pick": {"type": "choice", "choice": "right",
|
||||
"probabilities": {"left": 0.25, "right": 0.75}, "confidence": 0.1887,
|
||||
"answer_confidence": 0.75,
|
||||
"action": {"act_probability": 0.125}},
|
||||
"level": {"type": "score", "score": 0.75, "legend": {"0": "low", "1": "high"},
|
||||
"probabilities": {"0": 0.25, "1": 0.75}, "confidence": 0.1887,
|
||||
"answer_confidence": 0.75,
|
||||
"action": {"act_probability": 0.25}},
|
||||
"flag": {"type": "noul", "noul": 0.2, "confidence": 0.8,
|
||||
# max(p) over two options is max(p_true, 1 - p_true): the same number
|
||||
"answer_confidence": 0.8,
|
||||
"action": {"act_probability": 0.75}},
|
||||
})
|
||||
check("decode/entropy only computed for choice and score", entropy.call_count, 2)
|
||||
@@ -261,6 +265,7 @@ with patch.object(_agent, "confidence_from_probs", wraps=_agent.confidence_from_
|
||||
check("decode/noul confidence at p=%s" % true_probability, result["flag"], {
|
||||
"type": "noul", "noul": true_probability,
|
||||
"confidence": max(true_probability, 1.0 - true_probability),
|
||||
"answer_confidence": max(true_probability, 1.0 - true_probability),
|
||||
"action": {"act_probability": 0.75},
|
||||
})
|
||||
check("decode/noul-only answers skip entropy", entropy.call_count, 0)
|
||||
|
||||
@@ -0,0 +1,122 @@
|
||||
"""`confidence` carries two definitions, and only one of them is the calibrated quantity.
|
||||
|
||||
choice / score -> confidence_from_probs(p, k) == 1 - H(p) / log(k)
|
||||
noul -> max(p_true, 1 - p_true) == max(p)
|
||||
|
||||
They are not on the same scale. The same two-option distribution comes back as 0.90 from a
|
||||
`noul` and 0.53 from an equivalent two-option `choice`, and every shipped preset mixes the two
|
||||
types in one call -- `moderation_questions()` is four `noul` and one `score`. The README's
|
||||
"Automated Confidence Gating" section gates all of them on a single threshold.
|
||||
|
||||
Both benchmark harnesses in research/scripts take `conf = max(probs)` before calling
|
||||
`ece_score`, so every ECE figure in the README describes max(p) and not normalized entropy.
|
||||
`answer_confidence` reports that quantity on every question type, additively: `confidence` is
|
||||
untouched, so nothing a caller gates on today moves.
|
||||
|
||||
No weights are loaded: the confidence helpers are pure.
|
||||
"""
|
||||
import math
|
||||
import os
|
||||
import sys
|
||||
|
||||
import numpy as np
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
from laya.common import answer_confidence, confidence_from_probs, ece_score # noqa: E402
|
||||
|
||||
PASS, FAIL = [], []
|
||||
|
||||
|
||||
def check(name, got, want):
|
||||
if got == want:
|
||||
PASS.append(name)
|
||||
else:
|
||||
FAIL.append("%s: got %r, want %r" % (name, got, want))
|
||||
|
||||
|
||||
def check_true(name, cond, detail=""):
|
||||
if cond:
|
||||
PASS.append(name)
|
||||
else:
|
||||
FAIL.append("%s %s" % (name, detail))
|
||||
|
||||
|
||||
def close(a, b, tol=1e-9):
|
||||
return abs(a - b) <= tol
|
||||
|
||||
|
||||
# --------------------------------------------------------------- answer_confidence is max(p)
|
||||
for probs in ([0.5, 0.5], [0.1, 0.9], [0.7, 0.2, 0.1], [0.25] * 4, [1.0, 0.0, 0.0]):
|
||||
p = np.array(probs)
|
||||
check_true("answer/max of %s" % (probs,), close(answer_confidence(p, len(probs)), max(probs)))
|
||||
|
||||
check("answer/only the first k entries count", answer_confidence(np.array([0.4, 0.6, 0.99]), 2), 0.6)
|
||||
check("answer/k=1 is certain", answer_confidence(np.array([1.0]), 1), 1.0)
|
||||
check("answer/k=0 does not crash", answer_confidence(np.array([]), 0), 1.0)
|
||||
|
||||
# --------------------------------------------------------------- it matches noul's definition
|
||||
# `noul` reports max(p_true, 1 - p_true), which over two options is exactly max(p). So for a
|
||||
# noul answer the new field equals the existing one, and the two agree by construction.
|
||||
for p_true in (0.0, 0.05, 0.3, 0.5, 0.62, 0.9, 1.0):
|
||||
p = np.array([1.0 - p_true, p_true])
|
||||
check_true("noul/answer_confidence equals the shipped noul confidence at p=%.2f" % p_true,
|
||||
close(answer_confidence(p, 2), max(p_true, 1.0 - p_true)))
|
||||
|
||||
# --------------------------------------------------------------- the two scales really differ
|
||||
# Same distribution, two question types. 0.85 is the threshold the README's gating section uses.
|
||||
disagree = []
|
||||
for p_true in (0.60, 0.70, 0.80, 0.85, 0.90, 0.95):
|
||||
p = np.array([1.0 - p_true, p_true])
|
||||
if (answer_confidence(p, 2) >= 0.85) != (confidence_from_probs(p, 2) >= 0.85):
|
||||
disagree.append(p_true)
|
||||
check_true("scales/the two disagree across the documented threshold", disagree == [0.85, 0.90, 0.95],
|
||||
"disagreed at %s" % (disagree,))
|
||||
check_true("scales/entropy sits far below max(p) on the same distribution",
|
||||
confidence_from_probs(np.array([0.1, 0.9]), 2) < 0.55 < answer_confidence(np.array([0.1, 0.9]), 2))
|
||||
# max(p) over two options has a floor of 0.5; entropy reads 0.0 for the same coin flip.
|
||||
check("scales/a coin flip is 0.5 on max(p)", answer_confidence(np.array([0.5, 0.5]), 2), 0.5)
|
||||
check("scales/and 0.0 on entropy", confidence_from_probs(np.array([0.5, 0.5]), 2), 0.0)
|
||||
|
||||
# --------------------------------------------------------------- only one of them is calibrated
|
||||
# Perfectly calibrated predictions: an answer reported at top probability c is right exactly c
|
||||
# of the time. ECE on max(p) must be near zero. ECE on normalized entropy must not be, which is
|
||||
# why it cannot be compared against a probability threshold.
|
||||
rng = np.random.default_rng(0)
|
||||
tops, ents, correct = [], [], []
|
||||
for c in np.linspace(0.30, 0.99, 24):
|
||||
rest = (1.0 - c) / 2.0
|
||||
p = np.array([c, rest, rest])
|
||||
for _ in range(400):
|
||||
tops.append(answer_confidence(p, 3))
|
||||
ents.append(confidence_from_probs(p, 3))
|
||||
correct.append(1.0 if rng.random() < c else 0.0)
|
||||
|
||||
tops, ents, correct = np.array(tops), np.array(ents), np.array(correct)
|
||||
ece_top, ece_ent = ece_score(tops, correct), ece_score(ents, correct)
|
||||
check_true("calibration/max(p) is calibrated on calibrated data (ECE %.4f)" % ece_top,
|
||||
ece_top < 0.03, "ECE %.4f" % ece_top)
|
||||
check_true("calibration/entropy is not (ECE %.4f)" % ece_ent, ece_ent > 0.20, "ECE %.4f" % ece_ent)
|
||||
check_true("calibration/entropy is worse by a wide margin", ece_ent > 5 * ece_top,
|
||||
"top %.4f vs entropy %.4f" % (ece_top, ece_ent))
|
||||
|
||||
# --------------------------------------------------------------- the entropy helper is untouched
|
||||
for probs, k in (([0.1, 0.9], 2), ([0.25] * 4, 4), ([0.7, 0.2, 0.1], 3)):
|
||||
p = np.array(probs)
|
||||
ent = -(p * np.log(np.clip(p, 1e-12, 1.0))).sum()
|
||||
check_true("entropy/formula for k=%d" % k,
|
||||
close(confidence_from_probs(p, k), float(np.clip(1 - ent / math.log(k), 0.0, 1.0))))
|
||||
|
||||
# --------------------------------------------------------------- exported
|
||||
import laya # noqa: E402
|
||||
|
||||
check_true("export/answer_confidence is importable from laya", hasattr(laya, "answer_confidence"))
|
||||
check_true("export/answer_confidence is in __all__", "answer_confidence" in laya.__all__)
|
||||
check_true("export/answer_confidence is in dir()", "answer_confidence" in dir(laya))
|
||||
|
||||
print("\n%d passed, %d failed" % (len(PASS), len(FAIL)))
|
||||
for f in FAIL:
|
||||
print(" FAIL", f)
|
||||
if not FAIL:
|
||||
print("all confidence tests passed")
|
||||
sys.exit(1 if FAIL else 0)
|
||||
Reference in New Issue
Block a user