botp
/

Solomon / src /solomon /service.py
orz99's picture ArcherHume's picture
Duplicate from DoccyHealth/Solomon
1d2de8a
Raw History Blame Contribute Delete
39.5 kB
"""Solomon decision layer: the Solomon serving contract over the pinned readout chain.
A new layer over the unchanged, hash-pinned serving chain:
PackageEvidenceService (solomon/service_packages) -> ConfidenceHeadsService (solomon/serving) -> HeadsService
(solomon/service_heads) -> solomon/service_answers -> the answer contract readout -> engine (solomon/engine).
None of those files is edited. The layer does three things the chain cannot:
1. Engine proxy (`_EngineProxy`), passed to the chain instead of the raw engine:
- records the real letter logits and hidden state of every answer branch (so the layer reads P(yes) / listed
probabilities itself, rather than trusting the pinned chain's own four-state reading);
- in 'two_letter' mode rewrites each four-state Noul prompt (BOOLEAN_TASK / LABEL header) to the registered
two-letter A=yes / B=no header and passes the semantic head key explicitly (the pinned
`semantic_head_key` routes by prompt prefix and would raise on the new header). The 2 real logits are padded
with a finite -1e4 to the 4 letters the pinned readout expects; padding never reaches a Solomon response;
- answers the S-form sufficiency block synthetically ("A", zero tokens, no model call): the sufficiency branch is
removed from Solomon (the design note). Only listed-option probabilities are read.
- returns reliability None from `predict_correctness`: the old four-state correctness head is never applied.
2. The serving stack is bound to the stack that was measured (`RuntimeBinding`): every key of the binding's
`runtime` must equal the live engine identity, or the service refuses to load. That is the only safety
property this layer enforces, and it is enforced before a single question is asked.
3. The request/response shapes of the serving contract (see `parse_questions` and solomon/api.py).
v1.1 changes (Solomon v1.1):
* The entity answer type is REMOVED from the public API. An entity question is a yes/no question with the
candidate substituted in: ask one yes/no question per candidate, or send the candidates as labels for a
multi-label answer. An old-style request (candidate_kind 'entity', or a {candidate} placeholder) is refused with
a message naming both replacements (ENTITY_REMOVED).
* One merged yes/no head. Yes/no and multi-label branches read the same trained head (boolean/state4) through the
head routing table (`HEAD_ROUTING`, or the binding's `head_routing`); temperatures and reporting stay per answer
type (a solomon-readout-temperature-v3 artifact may map one head_key to several types).
* Evidence: the trained relevance head is the default selector at evidence='support' when one is configured
(solomon/evidence_selector.py, method 'trained_relevance_head'); word overlap is the labelled fallback
('lexical_overlap_fallback'). Every response names the method that produced its spans.
* All per-type temperatures are applied, single choice included, from the bound calibration.
v1.1 semantics -- THE MODEL ALWAYS ANSWERS. There is no abstention, no threshold, no policy freeze and no fitted
correctness head anywhere on this path. Every question of every answer type is answered with a probability, and
entity / multi-label questions return one probability per candidate, which is the unit they are accurate at.
The probabilities ORDER well and their MAGNITUDE overstates reliability at the top of the range (see
ORDERING_DISCLOSURE). That is disclosed, not gated on, and no response field may read as a certified error rate.
Readout modes (named by the selection file, see `load_selection`):
four_collapsed the four-state prompts; P(yes) = softmax over the 4 letters [A]; choice renormalised over the
listed options (reserved R slots dropped). This is the shipped readout.
two_letter two-letter prompts (A = yes, B = no); P(yes) = softmax over 2 letters [A]; choice S (listed).
"""
from solomon.binding import (BINDING_SCHEMA, CONTRACT, DEFAULT_BINDING, DESIGN_FREEZE, MODES, # noqa: F401
SELECTION, RuntimeBinding, TwoLetterPrompts, UnboundServing, default_design,
load_binding, load_prompts, load_selection)
from solomon.calibration import (ROOT, SERVED_TASKS, TEMPERATURE, TEMPERATURE_SCHEMA, TEMPERATURE_SCHEMAS, # noqa: F401
TEMPERATURE_TASKS, NoTemperature, ReadoutTemperature, _all_keys,
_structural_keys, digest, load_temperature)
import copy
import hashlib
import importlib
import json
import math
from pathlib import Path
import numpy as np
from solomon.engine_contract import BOOLEAN_TASK, LABEL, SUFFICIENCY
from solomon.heads import semantic_head_key
from solomon import semantics as sem
from solomon import evidence_selector as ES
PAD = -1e4
MAX_CANDIDATES = 64
# Head routing: served head_key -> the trained head that answers it. The v1.1 heads file carries one merged yes/no head;
# a route is only taken when the routed-from head is absent from the engine (a v1 ten-head file routes nothing).
HEAD_ROUTING = {'multilabel/state4': 'boolean/state4', 'entity/state4': 'boolean/state4'}
ENTITY_REMOVED = ('the entity answer type was removed in Solomon v1.1. Ask one yes/no question per candidate with the '
'candidate written into the question (e.g. {"type": "noul", "instructions": "Is Rookwood Ltd an approved '
'grower?"}), or send the candidates as labels for a multi-label answer ({"type": "noul", "instructions": '
'"Which growers are approved?", "candidates": [...], "candidate_kind": "label"}, no {candidate} '
'placeholder).')
# Response keys that must never appear in a Solomon answer (the design note: no reserved labels).
RESERVED_KEYS = ('not_stated', 'conflicting', 'none_of_listed', 'sufficiency', 'evidence_sufficiency')
# Response keys that would reintroduce abstention or imply a certified error rate. Nothing may emit these (v1.1).
FORBIDDEN_KEYS = ('abstain', 'abstain_reason', 'abstention', 'confidence', 'confidence_source', 'policy_status',
'threshold', 'certified_error_rate', 'error_rate_guarantee', 'calibration_guarantee',
'guaranteed_accuracy', 'certified_accuracy')
ORDERING_DISCLOSURE = (
'ordering_score is the readout\'s own top-class probability, read off the UNSCALED softmax (temperature 1.0 for '
'every answer type in v1.1: per-type temperatures fitted on development documents did not improve held-out '
'calibration, so none is applied). It has no fitted decision behind it and it certifies nothing. IT MEANS TWO '
'DIFFERENT THINGS depending on how many units the question has. SINGLE-UNIT (a yes/no (boolean) question, a single '
'or ordered choice, and EVERY per-candidate value inside a multi-label answer): the score is exactly the readout '
'probability. On held-out real test documents its expected calibration error is about 0.02 for per-candidate '
'multi-label values, 0.01-0.02 for single choice, and 0.05-0.07 for yes/no and ordered questions (cells of roughly '
'100-150 questions), and at the low end (stated 5-20%) it under-calls real yeses: an approximate magnitude, most '
'useful for ordering. MULTI-UNIT, i.e. the rolled-up question-level '
'score for a multi-label question with more than one candidate: it is the PRODUCT of the per-unit probabilities, '
'which assumes the units are independent. That assumption has never been validated as a joint probability. It '
'orders such questions well and it is a HEURISTIC ORDERING, not a calibrated probability -- read the per-candidate '
'numbers if you need a magnitude. All of this is measured behaviour on AI-labelled evaluation documents, NOT a '
'certified error rate, and nothing here is guaranteed. Every question is answered; nothing abstains.')
# ------------------------------------------------------------------ engine proxy
class _EngineProxy:
"""See module docstring. Phases: 'idle' (ignore), 'answer' (record rows), 'evidence' (record fresh-prefill segments)."""
def __init__(self, engine, mode, prompts=None):
if mode not in MODES:
raise ValueError('unknown readout mode')
if mode == 'two_letter' and prompts is None:
raise ValueError('two_letter mode requires two-letter prompts')
self._engine, self._mode, self._prompts = engine, mode, prompts
self._task = None
self.phase = 'idle'
self.rows = []
self.segments = []
self.routing = {} # served head_key -> trained head_key (set by `service` from the binding / HEAD_ROUTING)
self.tap_layers = () # layers the evidence head reads on the branch (mid-layer variants only)
def __getattr__(self, name):
return getattr(self._engine, name)
def reset(self):
self.phase = 'idle'; self.rows = []; self.segments = []
def task_context(self, task):
from contextlib import contextmanager
@contextmanager
def context():
previous = self._task; self._task = task
try:
with self._engine.task_context(task):
yield self
finally:
self._task = previous
return context()
def prefill(self, *args, **kwargs):
state = self._engine.prefill(*args, **kwargs)
if self.phase == 'evidence':
self.segments.append([])
return state
def predict_correctness(self, task, branches):
return {'reliability': None}
def estimate(self, state, block, n, stage='fast', max_new_tokens=512):
if block.startswith(SUFFICIENCY):
return {'input_tokens': 0, 'generated_tokens': 0, 'branches': 1, 'full_prompt_tokens': 0,
'reused_prefix_tokens': state.get('prefix_tokens', 0) if isinstance(state, dict) else 0}
if self._mode == 'two_letter':
swapped = self._prompts.rewrite(self._task, block)
if swapped is not None:
block, n = swapped
return self._engine.estimate(state, block, n, stage=stage, max_new_tokens=max_new_tokens)
def ask(self, state, block, n, execution='cached', head_key=None, **kwargs):
if block.startswith(SUFFICIENCY):
# Removed from Solomon: constant "sufficient", no model call, zero tokens.
logits = np.r_[0., PAD, PAD][:n]
p = np.exp(logits - logits.max()); p /= p.sum()
return {'letter_logits': logits, 'probabilities': p, 'prediction': 0, 'hidden': None, 'head_key': None,
'mass': None, 'top_is_letter': None, 'mass_available': False, 'execution': execution, 'fallback': 'solomon-removed',
'prompt_tokens': 0, 'branch_tokens': 0, 'input_tokens': 0, 'reused_prefix_tokens': 0, 'generation_tokens': 0,
'synthetic': True, 'fingerprint': self._engine.identity.get('fingerprint')}
width = n
if self.routing and head_key is None:
try:
natural = semantic_head_key(self._task, block)
except ValueError:
natural = None
if natural in self.routing:
head_key = self.routing[natural]
if self._mode == 'two_letter':
swapped = self._prompts.rewrite(self._task, block)
if swapped is not None:
head_key = head_key or semantic_head_key(self._task, block)
block, width = swapped
if self.tap_layers and self.phase == 'answer':
kwargs = {**kwargs, 'tap_layers': tuple(self.tap_layers)}
row = (self._engine.ask(state, block, width, execution=execution, head_key=head_key, **kwargs) if head_key is not None
else self._engine.ask(state, block, width, execution=execution, **kwargs))
logits = np.asarray(row['letter_logits'], np.float64)
if logits.shape != (width,) or not np.isfinite(logits).all():
raise ValueError('engine returned invalid letter logits')
record = {'block': block, 'task': self._task, 'letter_logits': logits.copy(), 'hidden': row.get('hidden'),
'head_key': row.get('head_key', head_key), 'input_tokens': int(row.get('input_tokens', row.get('branch_tokens', 0))),
'execution': execution, 'taps': row.get('taps'), 'state': state}
if self.phase == 'answer':
self.rows.append(record)
elif self.phase == 'evidence':
if not self.segments:
self.segments.append([])
self.segments[-1].append(record)
if width == n:
return row
padded = np.r_[logits, np.full(n - width, PAD)]
p = np.exp(padded - padded.max()); p /= p.sum()
return {**row, 'letter_logits': padded, 'probabilities': p, 'prediction': int(p.argmax())}
class _PhasedAnswers:
"""Wraps the ConfidenceHeadsService so the proxy knows which branches belong to the answer."""
def __init__(self, answers, proxy):
self._answers, self._proxy = answers, proxy
def __getattr__(self, name):
return getattr(self._answers, name)
def ask(self, *args, **kwargs):
self._proxy.phase = 'answer'
try:
return self._answers.ask(*args, **kwargs)
finally:
self._proxy.phase = 'evidence'
# ------------------------------------------------------------------ identity binding and the ordering score
def ordering_score(distributions):
"""The readout's own top-class probability, multiplied over the units of a question.
Read ORDERING_DISCLOSURE before using this number. The short version, and the reason this is not called
`probability`:
* ONE unit -> the score IS the tempered readout probability; the held-out panel validates it as a calibrated magnitude.
This covers yes/no, single and ordered choice, and every PER-CANDIDATE value that entity and
multi-label answers report in `candidates` / `candidate_ordering_scores`.
* MANY units -> the score is a product across units, which assumes they are independent. That assumption has
never been validated as a joint probability. It orders well; it is not a calibrated joint.
Only the rolled-up question-level number for a multi-candidate entity or multi-label question
is affected -- the per-candidate numbers beside it are single-unit and calibrated.
Naming it `probability` would be right for the single-unit case and would overclaim for the multi-unit one,
which is the wrong direction to be wrong in on a serving path with no abstention behind it.
Delegates the per-unit statement to solomon.reliability so there is one implementation of "what the readout says
about itself"; the qualification side uses the same function. No fitted parameters beyond the temperature."""
from solomon import reliability as R
units = []
for d in distributions:
d = np.asarray(d, float)
units.append({'kind': 'choice', 'logits': np.log(np.maximum(d, 1e-300))})
return R.question_reliability(units, 1.0)
def forbidden_keys(value):
"""Every FORBIDDEN_KEYS / RESERVED_KEYS name anywhere in a response. Must always be empty (v1.1)."""
return _all_keys(value) & set(FORBIDDEN_KEYS + RESERVED_KEYS)
# ------------------------------------------------------------------ request parsing
def _text(value, name):
if not isinstance(value, str) or not value.strip():
raise ValueError(name + ' must be a nonempty string')
return value
def parse_questions(questions):
"""Questions {id: spec} -> ordered list of normalised specs.
noul: {"type": "noul", "instructions": str} -> task boolean
{"type": "noul", "instructions": str, "candidates": [...]} -> task multilabel (one Noul per label)
("candidate_kind": "label" is accepted and is the only kind. v1.1 removed the entity type: a request with
candidate_kind "entity" or a {candidate} placeholder is refused with ENTITY_REMOVED.)
choice: {"type": "choice", "instructions": str, "options": [str, ...] | {key: text}, "ordered": bool}
('criteria' is accepted as a compatibility alias of 'options')
score: {"type": "score", "instructions": str, "levels": [str, ...]} (alias 'criteria'); an ordered choice keyed
"0".."K-1" with a legend and score = sum_i i * p_i.
"""
if not isinstance(questions, dict) or not questions:
raise ValueError('questions must be a nonempty mapping of id -> question')
out = []
for qid, spec in questions.items():
if not isinstance(qid, str) or not qid:
raise ValueError('question ids must be nonempty strings')
if isinstance(spec, str):
spec = {'type': 'noul', 'instructions': spec}
if not isinstance(spec, dict):
raise ValueError(f'question {qid}: spec must be an object')
kind = str(spec.get('type', 'noul')).lower()
instructions = _text(spec.get('instructions', spec.get('question')), f'question {qid}: instructions')
if kind == 'noul':
if 'candidates' not in spec:
out.append({'id': qid, 'type': 'noul', 'task': 'boolean', 'request': {'question': instructions}})
continue
candidates = spec['candidates']
if (not isinstance(candidates, list) or not 1 <= len(candidates) <= MAX_CANDIDATES or len(set(candidates)) != len(candidates)
or any(not isinstance(c, str) or not c.strip() for c in candidates)):
raise ValueError(f'question {qid}: candidates must be 1 to {MAX_CANDIDATES} distinct nonempty strings')
ckind = spec.get('candidate_kind', 'label')
if ckind == 'entity' or '{candidate}' in instructions or '{entity}' in instructions:
raise ValueError(f'question {qid}: ' + ENTITY_REMOVED)
if ckind != 'label':
raise ValueError(f'question {qid}: candidate_kind must be label (the entity type was removed in v1.1)')
request = {'question': instructions, 'labels': list(candidates)}
task = 'multilabel'
out.append({'id': qid, 'type': 'noul', 'task': task, 'candidates': list(candidates), 'request': request})
elif kind in ('choice', 'score'):
raw = spec.get('levels', spec.get('options', spec.get('criteria'))) if kind == 'score' else spec.get('options', spec.get('criteria'))
if isinstance(raw, dict):
keys, texts = [str(k) for k in raw], [raw[k] if isinstance(raw[k], str) and raw[k].strip() else str(k) for k in raw]
elif isinstance(raw, list):
texts = list(raw); keys = [str(i) for i in range(len(raw))] if kind == 'score' else list(raw)
else:
raise ValueError(f'question {qid}: options must be a list or a mapping')
if not 2 <= len(texts) <= 8 or any(not isinstance(t, str) or not t.strip() for t in texts) or len(set(texts)) != len(texts):
raise ValueError(f'question {qid}: provide 2 to 8 distinct nonempty options')
ordered = kind == 'score' or bool(spec.get('ordered', False))
if not isinstance(spec.get('ordered', False), bool):
raise ValueError(f'question {qid}: ordered must be Boolean')
out.append({'id': qid, 'type': kind, 'task': 'ordered' if ordered else 'single', 'keys': keys, 'texts': texts,
'request': {'question': instructions, 'options': texts}})
else:
raise ValueError(f'question {qid}: type must be noul, choice or score')
return out
def _state_parts(state):
if isinstance(state, str):
return [{'text': state}]
if isinstance(state, list):
return state
if isinstance(state, dict):
return [{'text': json.dumps(state, ensure_ascii=False, indent=2, sort_keys=False)}]
raise ValueError('state must be text, an object, or a list of document parts')
# ------------------------------------------------------------------ the layer
class SolomonService:
def __init__(self, api, proxy, *, mode, design, binding, prompts=None, selection=None, temperature=None):
self.api, self.proxy, self.mode, self.design = api, proxy, mode, copy.deepcopy(design)
self.binding, self.prompts, self.selection = binding, prompts, selection
# An explicit temperature= overrides the binding's (development and tests only); otherwise the bound one.
self.calibration = load_temperature(temperature) if temperature is not None else binding.calibration
if self.design.get('single_choice') not in ('R', 'S') or self.design.get('ordered') not in ('R', 'S'):
raise ValueError('Solomon serves the R or S listwise readout only')
@property
def runtime_identity(self):
return self.api.runtime_identity
def evidence_selector(self):
selector = getattr(self.api, 'selector', None)
if selector is not None and hasattr(selector, 'describe'):
return selector.describe()
return {'method': 'external_selector' if selector is not None else ES.FALLBACK_METHOD,
'faithfulness_established': False}
def temperature(self, task, head_keys=()):
"""The readout temperature actually applied to this task's logits, verified against the served head keys."""
return self.calibration.temperature(task, head_keys)
def health(self):
return {'contract': CONTRACT, 'readout': self.mode, 'question_types': ['noul', 'choice', 'score'],
'answer_types': list(SERVED_TASKS), 'removed_answer_types': {'entity': ENTITY_REMOVED},
'noul_candidate_kinds': ['label'], 'evidence_levels': ['none', 'support', 'sufficiency', 'removal'],
'evidence_selector': self.evidence_selector(), 'head_routing': dict(self.proxy.routing),
'answer_policy': 'always_answers',
'ordering_score_semantics': ORDERING_DISCLOSURE,
'calibration': self.calibration.describe(),
'design': self.design, 'binding': self.binding.describe(),
'prompts_sha256': self.prompts.sha256 if self.prompts is not None else None,
'selection_sha256': (self.selection or {}).get('sha256'), 'runtime_identity': copy.deepcopy(self.runtime_identity)}
def create(self, state):
result = self.api.create(_state_parts(state))
return {'state_id': result['state_id'], 'contract': CONTRACT, 'document_sha256': result.get('document_sha256')}
def decide(self, *, questions, state=None, state_id=None, population=None, evidence='support', evidence_max_calls=64,
evidence_detail=False, budget=None):
specs = parse_questions(questions)
if (state is None) == (state_id is None):
raise ValueError('provide exactly one of state or state_id')
if evidence not in ('none', 'support', 'sufficiency', 'removal'):
raise ValueError('invalid evidence level')
if population is not None and (not isinstance(population, str) or not population):
raise ValueError('population must be a nonempty string')
if type(evidence_detail) is not bool:
raise ValueError('evidence_detail must be Boolean')
key = state_id if state_id is not None else self.create(state)['state_id']
answers, usage = {}, {'branches': 0, 'input_tokens': 0, 'document_prefill_tokens': 0, 'evidence_calls': 0}
for spec in specs:
answer, cost = self._one(key, spec, population=population, evidence=evidence, evidence_max_calls=evidence_max_calls,
evidence_detail=evidence_detail, budget=budget)
answers[spec['id']] = answer
for k in usage:
usage[k] += cost.get(k, 0)
out = {'contract': CONTRACT, 'readout': self.mode, 'state_id': key, 'answers': answers, 'usage': usage,
'answer_policy': 'always_answers', 'ordering_score_semantics': ORDERING_DISCLOSURE,
'binding': self.binding.describe(), 'runtime_fingerprint': self.runtime_identity.get('fingerprint')}
bad = forbidden_keys(out) # structural: a response can never carry abstention or a certified rate
if bad:
raise ValueError('response carries forbidden fields: ' + ', '.join(sorted(bad)))
return out
# -------------------------------------------------------------- one question
def _budget(self, spec, budget):
if budget is not None:
return budget
units = len(spec.get('candidates', [])) or 1
return {'branches': units + 1, 'input_tokens': 4096 * (units + 1), 'generated_tokens': 0}
def _one(self, key, spec, *, population, evidence, evidence_max_calls, evidence_detail, budget):
task = spec['task']
with self.api.answers.lock:
self.proxy.reset()
try:
result = self.api.ask(key, task, evidence=evidence, evidence_max_calls=evidence_max_calls,
budget=self._budget(spec, budget), population=population, question_id=spec['id'], **spec['request'])
rows, segments = self.proxy.rows, self.proxy.segments
state = self.api.answers._warm(key)
finally:
self.proxy.phase = 'idle'
cost = {'branches': len(rows), 'input_tokens': sum(r['input_tokens'] for r in rows),
'document_prefill_tokens': (result.get('document_prefill') or {}).get('tokens', 0) or 0}
out = {'type': spec['type']}
if result.get('candidate') is None:
# Not an abstention: the caller's own branch/token budget stopped the chain before it produced anything.
stop = (result.get('compute') or {}).get('stop_reason', 'no_answer')
out.update(self._empty(spec), error=stop, ordering_score=None, evidence=[], evidence_status='not_run')
return out, cost
T = self.temperature(task, [r.get('head_key') for r in rows])
units = self._units(spec, rows, result['candidate'])
shown = self._distributions(spec, units, T)
out.update(self._present(spec, shown))
out['ordering_score'] = ordering_score(shown)
body = result.get('evidence')
out['temperature'] = float(T)
ev = self._evidence(spec, body, segments, out, T, evidence_detail)
cost['evidence_calls'] = ev.pop('_calls', 0)
out.update(ev)
return out, cost
def _units(self, spec, rows, candidate):
"""Answer-branch logits per unit, cross-checked against the pinned chain's own candidate."""
task = spec['task']
n_units = len(spec.get('candidates', [])) or 1
if len(rows) != n_units:
raise ValueError(f'expected {n_units} answer branches, captured {len(rows)} (only R/S single-order designs are served)')
logits = [r['letter_logits'] for r in rows]
if task in ('boolean', 'entity', 'multilabel'):
width = 2 if self.mode == 'two_letter' else 4
if any(len(x) != width for x in logits):
raise ValueError('Noul branch width does not match the readout mode')
pinned = [candidate['probabilities']] if task == 'boolean' else candidate['probabilities']
for x, p in zip(logits, pinned):
if abs(sem.p_yes(x) - float(p[0])) > 1e-6:
raise ValueError('captured branch does not match the pinned candidate')
else:
n = len(spec['texts'])
p = np.asarray(candidate['probabilities'], float)
if len(logits[0]) < n or np.max(np.abs(sem.listed_probs(logits[0], n) - p[:n] / p[:n].sum())) > 1e-6:
raise ValueError('captured choice branch does not match the pinned candidate')
return logits
def _distributions(self, spec, units, T):
if spec['task'] in ('boolean', 'entity', 'multilabel'):
return [np.array([sem.p_yes(x, T), 1 - sem.p_yes(x, T)]) for x in units]
return [sem.listed_probs(units[0], len(spec['texts']), T)]
@staticmethod
def _empty(spec):
if spec['type'] == 'noul':
return {'candidates': None} if 'candidates' in spec else {'noul': None}
return {'probabilities': None, 'answer': None}
@staticmethod
def _present(spec, dists):
if spec['type'] == 'noul':
if 'candidates' in spec:
# Entity and multi-label are accurate per candidate, so the probability is reported per candidate
# and so is the ordering score. There is no single whole-question number for these types.
return {'candidate_kind': 'label',
'candidates': {c: float(d[0]) for c, d in zip(spec['candidates'], dists)},
'candidate_ordering_scores': {c: float(max(d[0], 1 - d[0])) for c, d in zip(spec['candidates'], dists)}}
return {'noul': float(dists[0][0])}
p = dists[0]; keys = spec['keys']
out = {'probabilities': {k: float(v) for k, v in zip(keys, p)}, 'answer': keys[int(np.argmax(p))]}
out['choice'] = out['answer'] # compatibility alias of 'answer'
if spec['type'] == 'score':
out['score'] = float(np.dot(np.arange(len(p)), p))
out['legend'] = dict(zip(keys, spec['texts']))
elif spec['task'] == 'ordered':
out['ordered'] = True
return out
def _evidence(self, spec, body, segments, answer, T, detail):
if body is None:
return {'evidence': [], 'evidence_status': 'not_requested'}
status = body.get('status')
out = {'evidence_status': status, 'evidence': [], '_calls': int(body.get('calls', 0))}
if 'method' in body:
# Which selector produced the spans: 'trained_relevance_head', or the labelled 'lexical_overlap_fallback'
# (with the reason it ran). Neither establishes faithfulness.
out['evidence_method'] = body['method']
out['evidence_faithfulness_established'] = False
reason = getattr(getattr(self.api, 'selector', None), 'last_reason', None)
if body['method'] == ES.FALLBACK_METHOD and reason:
out['evidence_fallback_reason'] = reason
ranking = getattr(getattr(self.api, 'selector', None), 'last_ranking', None) if body.get('method') == ES.RANKED_METHOD else None
if status in ('found', 'no_support_found'):
out['evidence'] = [{k: r[k] for k in ('start', 'end', 'text', 'page', 'path') if k in r} for r in body.get('references', [])]
if 'candidates' in spec and body.get('package_format') == 'per_unit' and len(body.get('packages', [])) == len(spec['candidates']):
out['candidate_evidence'] = {c: [{k: s[k] for k in ('start', 'end', 'text', 'pages') if k in s}
for s in p['spans'] if s.get('role') == 'evidence']
for c, p in zip(spec['candidates'], body['packages'])}
if ranking is not None:
# v1.1 ranked pointers: the top sentences per answer unit IN SCORE ORDER, each with its relevance score.
# A unit whose answer is 'not stated' carries none (evidence_suppressed). Scores rank; they certify nothing.
ranked = [[{k: s[k] for k in ('start', 'end', 'text', 'score', 'rank')} for s in u['ranked']] for u in ranking]
suppressed = [bool(u['suppressed']) for u in ranking]
if 'candidates' in spec and len(ranked) == len(spec['candidates']):
out['candidate_evidence'] = dict(zip(spec['candidates'], ranked))
out['candidate_evidence_suppressed'] = dict(zip(spec['candidates'], suppressed))
elif len(ranked) == 1:
out['evidence'] = ranked[0]
out['evidence_suppressed'] = suppressed[0]
out['evidence_semantics'] = ES.RANKED_SEMANTICS
verification = self._verification(spec, body, segments, answer, T)
if verification:
out['evidence_verification'] = verification
if detail:
out['evidence_detail'] = {k: v for k, v in body.items() if k not in ('evidence_only', 'evidence_removed', 'union_package')}
return out
def _verification(self, spec, body, segments, answer, T):
"""Evidence-only / evidence-removed re-asks in Solomon terms (from the captured fresh-prefill segments)."""
modes = [m for m in ('evidence_only', 'evidence_removed') if m in body]
if not modes:
return None
per_unit = 'evidence_only' in body and 'per_unit' in body['evidence_only']
calls = (len(body['evidence_only']['per_unit']) if per_unit else int('evidence_only' in body)) + int('evidence_removed' in body)
if len(segments) < calls:
raise ValueError('evidence re-ask branches were not captured')
used = segments[len(segments) - calls:]
groups = {}
if 'evidence_only' in body:
k = len(body['evidence_only']['per_unit']) if per_unit else 1
groups['evidence_only'] = [r for seg in used[:k] for r in seg]
if 'evidence_removed' in body:
groups['evidence_removed'] = used[-1]
out = {}
for mode, rows in groups.items():
n_units = len(spec.get('candidates', [])) or 1
if len(rows) != n_units:
raise ValueError(f'{mode}: expected {n_units} branches, captured {len(rows)}')
dists = self._distributions(spec, [r['letter_logits'] for r in rows], T)
shown = self._present(spec, dists)
shown.pop('choice', None)
shown.pop('candidate_ordering_scores', None)
out[mode] = {**shown, 'agrees_with_full': self._decision(spec, shown) == self._decision(spec, answer)}
return out
@staticmethod
def _decision(spec, shown):
if spec['type'] == 'noul':
if 'candidates' in spec:
return {c: p >= .5 for c, p in shown['candidates'].items()}
return shown['noul'] >= .5
return shown['answer']
def head_routing(binding_routing, engine):
"""The served routing table: the binding's when it names one, else HEAD_ROUTING restricted to heads the engine
lacks. Every target must be a head the engine actually loaded (checked when the engine exposes its heads)."""
heads = getattr(engine, 'answer_heads', None)
if binding_routing is not None:
routing = dict(binding_routing)
elif heads is None:
routing = {}
else:
routing = {k: v for k, v in HEAD_ROUTING.items() if k not in heads}
if heads is not None:
missing = sorted({v for v in routing.values() if v not in heads})
if missing:
raise ValueError('head routing points at heads the engine did not load: ' + ', '.join(missing))
return routing
def evidence_selector(proxy, evidence_head=None, evidence_thresholds=None, device='cuda'):
"""(prefix, thresholds path) -> the trained selector; neither -> the labelled lexical fallback."""
if (evidence_head is None) != (evidence_thresholds is None):
raise ValueError('an evidence head needs its per-task thresholds (and thresholds need a head)')
if evidence_head is None:
return ES.LexicalFallback('no evidence head configured')
selector = ES.load_selector(proxy, _resolve(evidence_head), _resolve(evidence_thresholds), device=device)
# Recorded in the runtime identity BEFORE the binding check, so a binding issued for this head pins it.
identity = getattr(proxy._engine, 'identity', None)
if isinstance(identity, dict):
identity.update(evidence_head_sha256=selector.head_sha256, evidence_thresholds_sha256=selector.thresholds_sha256)
proxy.tap_layers = selector.tap_layers
if selector.tap_layers:
proxy._engine.prefix_layers = tuple(selector.tap_layers)
return selector
def _resolve(path):
path = Path(path)
return path if path.is_absolute() else ROOT / path
def service(store, engine, *, selection=SELECTION, mode=None, design=None, prompts=None, binding=None, temperature=None,
selector=None, page_selector=None, page_maps=None, size_cap=None, package_format='per_unit',
evidence_head=None, evidence_thresholds=None):
"""Build the pinned chain around a proxied engine and wrap it in the Solomon layer.
mode/design/prompts/binding override selection.json (for tests and development); binding may be a path, a
binding object, or None (selection 'binding' key, else DEFAULT_BINDING, else UnboundServing). A binding that
is present but does not match the live runtime identity raises here, before any question is asked.
"""
from solomon.serving import ConfidenceHeadsService, FreshDecider
from solomon.service_packages import PackageEvidenceService, SIZE_CAP
chosen = None
if mode is None:
chosen = load_selection(selection)
mode = chosen['mode']
design = design or chosen['design']
prompts = prompts if prompts is not None else chosen['prompts']
binding = binding if binding is not None else chosen['binding']
head_config = chosen.get('evidence_head') or {}
beside = lambda v: (str(Path(chosen['dir']) / v) if isinstance(v, str) and not Path(v).is_absolute()
and chosen.get('dir') and (Path(chosen['dir']) / (v + '.json' if not v.endswith('.json') else v)).exists() else v)
evidence_head = evidence_head or beside(head_config.get('prefix'))
evidence_thresholds = evidence_thresholds or beside(head_config.get('thresholds'))
if isinstance(binding, str) and not Path(binding).is_absolute() and (Path(chosen['dir']) / binding).exists():
binding = Path(chosen['dir']) / binding # the selection names a binding beside itself
if mode not in MODES:
raise ValueError('unknown readout mode')
design = copy.deepcopy(design or default_design(mode))
prompt_set = load_prompts(prompts) if mode == 'two_letter' else None
proxy = _EngineProxy(engine, mode, prompt_set)
if selector is None:
selector = evidence_selector(proxy, evidence_head, evidence_thresholds)
answers = ConfidenceHeadsService(store, proxy, design)
api = PackageEvidenceService(_PhasedAnswers(answers, proxy), FreshDecider(proxy, design, Path(store) / '.evidence-readout'),
selector, page_selector, page_maps=page_maps, size_cap=size_cap or SIZE_CAP, package_format=package_format)
runtime = answers.runtime_identity
loaded = load_binding(binding, mode=mode, runtime=runtime, design=design, prompts_sha256=prompt_set.sha256 if prompt_set else None)
proxy.routing = head_routing(getattr(loaded, 'head_routing', None), engine)
return SolomonService(api, proxy, mode=mode, design=design, binding=loaded, prompts=prompt_set, selection=chosen,
temperature=temperature)