Darwin-397B-ZTC / ztc /usage.py
SeaWolf-AI's picture
ZTC usage example
6f06b4d verified
Raw History Blame Contribute Delete
2.14 kB
# Zero-Token Confidence (ZTC) — usage
#
# ZTC reads the model's own internal state ONCE, before generation, and returns
# the probability that the answer the model is about to produce will be correct.
# No extra tokens are generated. No second model is required.
#
# probe file : ztc/ztc_probe_darwin397b.npz (45 KB)
# input : final-layer hidden state of the last prompt token (4096-dim)
# output : score, and a calibrated probability in [0, 1]
#
# Reported performance on this model (PubMedQA, 539 items, 146 incorrect):
# self-reported confidence AUROC 0.7646
# ZTC AUROC 0.8801 (permutation null z = 13.31)
import numpy as np
import torch
from transformers import AutoModel, AutoTokenizer
MODEL = "FINAL-Bench/Darwin-397B-ZTC"
PROBE = "ztc/ztc_probe_darwin397b.npz"
class ZTC:
def __init__(self, path=PROBE):
z = np.load(path)
self.w = z["w"].astype(np.float32)
self.mu = z["mu"].astype(np.float32)
self.sd = z["sd"].astype(np.float32)
self.s_mean = float(z["s_mean"]); self.s_std = float(z["s_std"])
self.A = float(z["cal_A"]); self.B = float(z["cal_B"])
def score(self, hidden):
"""hidden: (4096,) or (batch, 4096) final-layer state of the last prompt token."""
h = np.asarray(hidden, dtype=np.float32)
s = ((h - self.mu) / self.sd) @ self.w
p = 1.0 / (1.0 + np.exp(-(self.A * (s - self.s_mean) / self.s_std + self.B)))
return s, p
# --- one forward pass, zero generated tokens -------------------------------
tok = AutoTokenizer.from_pretrained(MODEL)
model = AutoModel.from_pretrained(MODEL, dtype=torch.bfloat16, device_map="auto").eval()
ztc = ZTC()
prompt = "Question: ...\nAnswer:"
b = tok(prompt, return_tensors="pt").to(next(model.parameters()).device)
with torch.no_grad():
h = model(**b).last_hidden_state[0, -1].float().cpu().numpy()
s, p = ztc.score(h)
print("ZTC score %.3f -> P(correct) = %.3f" % (s, p))
# Gate the action, not the answer:
# if p < THRESHOLD: do not call the tool / escalate / answer "I don't know"
# else: generate as usual