"""Solomon decision layer: the Solomon serving contract over the pinned readout chain. A new layer over the unchanged, hash-pinned serving chain: PackageEvidenceService (solomon/service_packages) -> ConfidenceHeadsService (solomon/serving) -> HeadsService (solomon/service_heads) -> solomon/service_answers -> the answer contract readout -> engine (solomon/engine). None of those files is edited. The layer does three things the chain cannot: 1. Engine proxy (`_EngineProxy`), passed to the chain instead of the raw engine: - records the real letter logits and hidden state of every answer branch (so the layer reads P(yes) / listed probabilities itself, rather than trusting the pinned chain's own four-state reading); - in 'two_letter' mode rewrites each four-state Noul prompt (BOOLEAN_TASK / LABEL header) to the registered two-letter A=yes / B=no header and passes the semantic head key explicitly (the pinned `semantic_head_key` routes by prompt prefix and would raise on the new header). The 2 real logits are padded with a finite -1e4 to the 4 letters the pinned readout expects; padding never reaches a Solomon response; - answers the S-form sufficiency block synthetically ("A", zero tokens, no model call): the sufficiency branch is removed from Solomon (the design note). Only listed-option probabilities are read. - returns reliability None from `predict_correctness`: the old four-state correctness head is never applied. 2. The serving stack is bound to the stack that was measured (`RuntimeBinding`): every key of the binding's `runtime` must equal the live engine identity, or the service refuses to load. That is the only safety property this layer enforces, and it is enforced before a single question is asked. 3. The request/response shapes of the serving contract (see `parse_questions` and solomon/api.py). v1.1 changes (Solomon v1.1): * The entity answer type is REMOVED from the public API. An entity question is a yes/no question with the candidate substituted in: ask one yes/no question per candidate, or send the candidates as labels for a multi-label answer. An old-style request (candidate_kind 'entity', or a {candidate} placeholder) is refused with a message naming both replacements (ENTITY_REMOVED). * One merged yes/no head. Yes/no and multi-label branches read the same trained head (boolean/state4) through the head routing table (`HEAD_ROUTING`, or the binding's `head_routing`); temperatures and reporting stay per answer type (a solomon-readout-temperature-v3 artifact may map one head_key to several types). * Evidence: the trained relevance head is the default selector at evidence='support' when one is configured (solomon/evidence_selector.py, method 'trained_relevance_head'); word overlap is the labelled fallback ('lexical_overlap_fallback'). Every response names the method that produced its spans. * All per-type temperatures are applied, single choice included, from the bound calibration. v1.1 semantics -- THE MODEL ALWAYS ANSWERS. There is no abstention, no threshold, no policy freeze and no fitted correctness head anywhere on this path. Every question of every answer type is answered with a probability, and entity / multi-label questions return one probability per candidate, which is the unit they are accurate at. The probabilities ORDER well and their MAGNITUDE overstates reliability at the top of the range (see ORDERING_DISCLOSURE). That is disclosed, not gated on, and no response field may read as a certified error rate. Readout modes (named by the selection file, see `load_selection`): four_collapsed the four-state prompts; P(yes) = softmax over the 4 letters [A]; choice renormalised over the listed options (reserved R slots dropped). This is the shipped readout. two_letter two-letter prompts (A = yes, B = no); P(yes) = softmax over 2 letters [A]; choice S (listed). """ from solomon.binding import (BINDING_SCHEMA, CONTRACT, DEFAULT_BINDING, DESIGN_FREEZE, MODES, # noqa: F401 SELECTION, RuntimeBinding, TwoLetterPrompts, UnboundServing, default_design, load_binding, load_prompts, load_selection) from solomon.calibration import (ROOT, SERVED_TASKS, TEMPERATURE, TEMPERATURE_SCHEMA, TEMPERATURE_SCHEMAS, # noqa: F401 TEMPERATURE_TASKS, NoTemperature, ReadoutTemperature, _all_keys, _structural_keys, digest, load_temperature) import copy import hashlib import importlib import json import math from pathlib import Path import numpy as np from solomon.engine_contract import BOOLEAN_TASK, LABEL, SUFFICIENCY from solomon.heads import semantic_head_key from solomon import semantics as sem from solomon import evidence_selector as ES PAD = -1e4 MAX_CANDIDATES = 64 # Head routing: served head_key -> the trained head that answers it. The v1.1 heads file carries one merged yes/no head; # a route is only taken when the routed-from head is absent from the engine (a v1 ten-head file routes nothing). HEAD_ROUTING = {'multilabel/state4': 'boolean/state4', 'entity/state4': 'boolean/state4'} ENTITY_REMOVED = ('the entity answer type was removed in Solomon v1.1. Ask one yes/no question per candidate with the ' 'candidate written into the question (e.g. {"type": "noul", "instructions": "Is Rookwood Ltd an approved ' 'grower?"}), or send the candidates as labels for a multi-label answer ({"type": "noul", "instructions": ' '"Which growers are approved?", "candidates": [...], "candidate_kind": "label"}, no {candidate} ' 'placeholder).') # Response keys that must never appear in a Solomon answer (the design note: no reserved labels). RESERVED_KEYS = ('not_stated', 'conflicting', 'none_of_listed', 'sufficiency', 'evidence_sufficiency') # Response keys that would reintroduce abstention or imply a certified error rate. Nothing may emit these (v1.1). FORBIDDEN_KEYS = ('abstain', 'abstain_reason', 'abstention', 'confidence', 'confidence_source', 'policy_status', 'threshold', 'certified_error_rate', 'error_rate_guarantee', 'calibration_guarantee', 'guaranteed_accuracy', 'certified_accuracy') ORDERING_DISCLOSURE = ( 'ordering_score is the readout\'s own top-class probability, read off the UNSCALED softmax (temperature 1.0 for ' 'every answer type in v1.1: per-type temperatures fitted on development documents did not improve held-out ' 'calibration, so none is applied). It has no fitted decision behind it and it certifies nothing. IT MEANS TWO ' 'DIFFERENT THINGS depending on how many units the question has. SINGLE-UNIT (a yes/no (boolean) question, a single ' 'or ordered choice, and EVERY per-candidate value inside a multi-label answer): the score is exactly the readout ' 'probability. On held-out real test documents its expected calibration error is about 0.02 for per-candidate ' 'multi-label values, 0.01-0.02 for single choice, and 0.05-0.07 for yes/no and ordered questions (cells of roughly ' '100-150 questions), and at the low end (stated 5-20%) it under-calls real yeses: an approximate magnitude, most ' 'useful for ordering. MULTI-UNIT, i.e. the rolled-up question-level ' 'score for a multi-label question with more than one candidate: it is the PRODUCT of the per-unit probabilities, ' 'which assumes the units are independent. That assumption has never been validated as a joint probability. It ' 'orders such questions well and it is a HEURISTIC ORDERING, not a calibrated probability -- read the per-candidate ' 'numbers if you need a magnitude. All of this is measured behaviour on AI-labelled evaluation documents, NOT a ' 'certified error rate, and nothing here is guaranteed. Every question is answered; nothing abstains.') # ------------------------------------------------------------------ engine proxy class _EngineProxy: """See module docstring. Phases: 'idle' (ignore), 'answer' (record rows), 'evidence' (record fresh-prefill segments).""" def __init__(self, engine, mode, prompts=None): if mode not in MODES: raise ValueError('unknown readout mode') if mode == 'two_letter' and prompts is None: raise ValueError('two_letter mode requires two-letter prompts') self._engine, self._mode, self._prompts = engine, mode, prompts self._task = None self.phase = 'idle' self.rows = [] self.segments = [] self.routing = {} # served head_key -> trained head_key (set by `service` from the binding / HEAD_ROUTING) self.tap_layers = () # layers the evidence head reads on the branch (mid-layer variants only) def __getattr__(self, name): return getattr(self._engine, name) def reset(self): self.phase = 'idle'; self.rows = []; self.segments = [] def task_context(self, task): from contextlib import contextmanager @contextmanager def context(): previous = self._task; self._task = task try: with self._engine.task_context(task): yield self finally: self._task = previous return context() def prefill(self, *args, **kwargs): state = self._engine.prefill(*args, **kwargs) if self.phase == 'evidence': self.segments.append([]) return state def predict_correctness(self, task, branches): return {'reliability': None} def estimate(self, state, block, n, stage='fast', max_new_tokens=512): if block.startswith(SUFFICIENCY): return {'input_tokens': 0, 'generated_tokens': 0, 'branches': 1, 'full_prompt_tokens': 0, 'reused_prefix_tokens': state.get('prefix_tokens', 0) if isinstance(state, dict) else 0} if self._mode == 'two_letter': swapped = self._prompts.rewrite(self._task, block) if swapped is not None: block, n = swapped return self._engine.estimate(state, block, n, stage=stage, max_new_tokens=max_new_tokens) def ask(self, state, block, n, execution='cached', head_key=None, **kwargs): if block.startswith(SUFFICIENCY): # Removed from Solomon: constant "sufficient", no model call, zero tokens. logits = np.r_[0., PAD, PAD][:n] p = np.exp(logits - logits.max()); p /= p.sum() return {'letter_logits': logits, 'probabilities': p, 'prediction': 0, 'hidden': None, 'head_key': None, 'mass': None, 'top_is_letter': None, 'mass_available': False, 'execution': execution, 'fallback': 'solomon-removed', 'prompt_tokens': 0, 'branch_tokens': 0, 'input_tokens': 0, 'reused_prefix_tokens': 0, 'generation_tokens': 0, 'synthetic': True, 'fingerprint': self._engine.identity.get('fingerprint')} width = n if self.routing and head_key is None: try: natural = semantic_head_key(self._task, block) except ValueError: natural = None if natural in self.routing: head_key = self.routing[natural] if self._mode == 'two_letter': swapped = self._prompts.rewrite(self._task, block) if swapped is not None: head_key = head_key or semantic_head_key(self._task, block) block, width = swapped if self.tap_layers and self.phase == 'answer': kwargs = {**kwargs, 'tap_layers': tuple(self.tap_layers)} row = (self._engine.ask(state, block, width, execution=execution, head_key=head_key, **kwargs) if head_key is not None else self._engine.ask(state, block, width, execution=execution, **kwargs)) logits = np.asarray(row['letter_logits'], np.float64) if logits.shape != (width,) or not np.isfinite(logits).all(): raise ValueError('engine returned invalid letter logits') record = {'block': block, 'task': self._task, 'letter_logits': logits.copy(), 'hidden': row.get('hidden'), 'head_key': row.get('head_key', head_key), 'input_tokens': int(row.get('input_tokens', row.get('branch_tokens', 0))), 'execution': execution, 'taps': row.get('taps'), 'state': state} if self.phase == 'answer': self.rows.append(record) elif self.phase == 'evidence': if not self.segments: self.segments.append([]) self.segments[-1].append(record) if width == n: return row padded = np.r_[logits, np.full(n - width, PAD)] p = np.exp(padded - padded.max()); p /= p.sum() return {**row, 'letter_logits': padded, 'probabilities': p, 'prediction': int(p.argmax())} class _PhasedAnswers: """Wraps the ConfidenceHeadsService so the proxy knows which branches belong to the answer.""" def __init__(self, answers, proxy): self._answers, self._proxy = answers, proxy def __getattr__(self, name): return getattr(self._answers, name) def ask(self, *args, **kwargs): self._proxy.phase = 'answer' try: return self._answers.ask(*args, **kwargs) finally: self._proxy.phase = 'evidence' # ------------------------------------------------------------------ identity binding and the ordering score def ordering_score(distributions): """The readout's own top-class probability, multiplied over the units of a question. Read ORDERING_DISCLOSURE before using this number. The short version, and the reason this is not called `probability`: * ONE unit -> the score IS the tempered readout probability; the held-out panel validates it as a calibrated magnitude. This covers yes/no, single and ordered choice, and every PER-CANDIDATE value that entity and multi-label answers report in `candidates` / `candidate_ordering_scores`. * MANY units -> the score is a product across units, which assumes they are independent. That assumption has never been validated as a joint probability. It orders well; it is not a calibrated joint. Only the rolled-up question-level number for a multi-candidate entity or multi-label question is affected -- the per-candidate numbers beside it are single-unit and calibrated. Naming it `probability` would be right for the single-unit case and would overclaim for the multi-unit one, which is the wrong direction to be wrong in on a serving path with no abstention behind it. Delegates the per-unit statement to solomon.reliability so there is one implementation of "what the readout says about itself"; the qualification side uses the same function. No fitted parameters beyond the temperature.""" from solomon import reliability as R units = [] for d in distributions: d = np.asarray(d, float) units.append({'kind': 'choice', 'logits': np.log(np.maximum(d, 1e-300))}) return R.question_reliability(units, 1.0) def forbidden_keys(value): """Every FORBIDDEN_KEYS / RESERVED_KEYS name anywhere in a response. Must always be empty (v1.1).""" return _all_keys(value) & set(FORBIDDEN_KEYS + RESERVED_KEYS) # ------------------------------------------------------------------ request parsing def _text(value, name): if not isinstance(value, str) or not value.strip(): raise ValueError(name + ' must be a nonempty string') return value def parse_questions(questions): """Questions {id: spec} -> ordered list of normalised specs. noul: {"type": "noul", "instructions": str} -> task boolean {"type": "noul", "instructions": str, "candidates": [...]} -> task multilabel (one Noul per label) ("candidate_kind": "label" is accepted and is the only kind. v1.1 removed the entity type: a request with candidate_kind "entity" or a {candidate} placeholder is refused with ENTITY_REMOVED.) choice: {"type": "choice", "instructions": str, "options": [str, ...] | {key: text}, "ordered": bool} ('criteria' is accepted as a compatibility alias of 'options') score: {"type": "score", "instructions": str, "levels": [str, ...]} (alias 'criteria'); an ordered choice keyed "0".."K-1" with a legend and score = sum_i i * p_i. """ if not isinstance(questions, dict) or not questions: raise ValueError('questions must be a nonempty mapping of id -> question') out = [] for qid, spec in questions.items(): if not isinstance(qid, str) or not qid: raise ValueError('question ids must be nonempty strings') if isinstance(spec, str): spec = {'type': 'noul', 'instructions': spec} if not isinstance(spec, dict): raise ValueError(f'question {qid}: spec must be an object') kind = str(spec.get('type', 'noul')).lower() instructions = _text(spec.get('instructions', spec.get('question')), f'question {qid}: instructions') if kind == 'noul': if 'candidates' not in spec: out.append({'id': qid, 'type': 'noul', 'task': 'boolean', 'request': {'question': instructions}}) continue candidates = spec['candidates'] if (not isinstance(candidates, list) or not 1 <= len(candidates) <= MAX_CANDIDATES or len(set(candidates)) != len(candidates) or any(not isinstance(c, str) or not c.strip() for c in candidates)): raise ValueError(f'question {qid}: candidates must be 1 to {MAX_CANDIDATES} distinct nonempty strings') ckind = spec.get('candidate_kind', 'label') if ckind == 'entity' or '{candidate}' in instructions or '{entity}' in instructions: raise ValueError(f'question {qid}: ' + ENTITY_REMOVED) if ckind != 'label': raise ValueError(f'question {qid}: candidate_kind must be label (the entity type was removed in v1.1)') request = {'question': instructions, 'labels': list(candidates)} task = 'multilabel' out.append({'id': qid, 'type': 'noul', 'task': task, 'candidates': list(candidates), 'request': request}) elif kind in ('choice', 'score'): raw = spec.get('levels', spec.get('options', spec.get('criteria'))) if kind == 'score' else spec.get('options', spec.get('criteria')) if isinstance(raw, dict): keys, texts = [str(k) for k in raw], [raw[k] if isinstance(raw[k], str) and raw[k].strip() else str(k) for k in raw] elif isinstance(raw, list): texts = list(raw); keys = [str(i) for i in range(len(raw))] if kind == 'score' else list(raw) else: raise ValueError(f'question {qid}: options must be a list or a mapping') if not 2 <= len(texts) <= 8 or any(not isinstance(t, str) or not t.strip() for t in texts) or len(set(texts)) != len(texts): raise ValueError(f'question {qid}: provide 2 to 8 distinct nonempty options') ordered = kind == 'score' or bool(spec.get('ordered', False)) if not isinstance(spec.get('ordered', False), bool): raise ValueError(f'question {qid}: ordered must be Boolean') out.append({'id': qid, 'type': kind, 'task': 'ordered' if ordered else 'single', 'keys': keys, 'texts': texts, 'request': {'question': instructions, 'options': texts}}) else: raise ValueError(f'question {qid}: type must be noul, choice or score') return out def _state_parts(state): if isinstance(state, str): return [{'text': state}] if isinstance(state, list): return state if isinstance(state, dict): return [{'text': json.dumps(state, ensure_ascii=False, indent=2, sort_keys=False)}] raise ValueError('state must be text, an object, or a list of document parts') # ------------------------------------------------------------------ the layer class SolomonService: def __init__(self, api, proxy, *, mode, design, binding, prompts=None, selection=None, temperature=None): self.api, self.proxy, self.mode, self.design = api, proxy, mode, copy.deepcopy(design) self.binding, self.prompts, self.selection = binding, prompts, selection # An explicit temperature= overrides the binding's (development and tests only); otherwise the bound one. self.calibration = load_temperature(temperature) if temperature is not None else binding.calibration if self.design.get('single_choice') not in ('R', 'S') or self.design.get('ordered') not in ('R', 'S'): raise ValueError('Solomon serves the R or S listwise readout only') @property def runtime_identity(self): return self.api.runtime_identity def evidence_selector(self): selector = getattr(self.api, 'selector', None) if selector is not None and hasattr(selector, 'describe'): return selector.describe() return {'method': 'external_selector' if selector is not None else ES.FALLBACK_METHOD, 'faithfulness_established': False} def temperature(self, task, head_keys=()): """The readout temperature actually applied to this task's logits, verified against the served head keys.""" return self.calibration.temperature(task, head_keys) def health(self): return {'contract': CONTRACT, 'readout': self.mode, 'question_types': ['noul', 'choice', 'score'], 'answer_types': list(SERVED_TASKS), 'removed_answer_types': {'entity': ENTITY_REMOVED}, 'noul_candidate_kinds': ['label'], 'evidence_levels': ['none', 'support', 'sufficiency', 'removal'], 'evidence_selector': self.evidence_selector(), 'head_routing': dict(self.proxy.routing), 'answer_policy': 'always_answers', 'ordering_score_semantics': ORDERING_DISCLOSURE, 'calibration': self.calibration.describe(), 'design': self.design, 'binding': self.binding.describe(), 'prompts_sha256': self.prompts.sha256 if self.prompts is not None else None, 'selection_sha256': (self.selection or {}).get('sha256'), 'runtime_identity': copy.deepcopy(self.runtime_identity)} def create(self, state): result = self.api.create(_state_parts(state)) return {'state_id': result['state_id'], 'contract': CONTRACT, 'document_sha256': result.get('document_sha256')} def decide(self, *, questions, state=None, state_id=None, population=None, evidence='support', evidence_max_calls=64, evidence_detail=False, budget=None): specs = parse_questions(questions) if (state is None) == (state_id is None): raise ValueError('provide exactly one of state or state_id') if evidence not in ('none', 'support', 'sufficiency', 'removal'): raise ValueError('invalid evidence level') if population is not None and (not isinstance(population, str) or not population): raise ValueError('population must be a nonempty string') if type(evidence_detail) is not bool: raise ValueError('evidence_detail must be Boolean') key = state_id if state_id is not None else self.create(state)['state_id'] answers, usage = {}, {'branches': 0, 'input_tokens': 0, 'document_prefill_tokens': 0, 'evidence_calls': 0} for spec in specs: answer, cost = self._one(key, spec, population=population, evidence=evidence, evidence_max_calls=evidence_max_calls, evidence_detail=evidence_detail, budget=budget) answers[spec['id']] = answer for k in usage: usage[k] += cost.get(k, 0) out = {'contract': CONTRACT, 'readout': self.mode, 'state_id': key, 'answers': answers, 'usage': usage, 'answer_policy': 'always_answers', 'ordering_score_semantics': ORDERING_DISCLOSURE, 'binding': self.binding.describe(), 'runtime_fingerprint': self.runtime_identity.get('fingerprint')} bad = forbidden_keys(out) # structural: a response can never carry abstention or a certified rate if bad: raise ValueError('response carries forbidden fields: ' + ', '.join(sorted(bad))) return out # -------------------------------------------------------------- one question def _budget(self, spec, budget): if budget is not None: return budget units = len(spec.get('candidates', [])) or 1 return {'branches': units + 1, 'input_tokens': 4096 * (units + 1), 'generated_tokens': 0} def _one(self, key, spec, *, population, evidence, evidence_max_calls, evidence_detail, budget): task = spec['task'] with self.api.answers.lock: self.proxy.reset() try: result = self.api.ask(key, task, evidence=evidence, evidence_max_calls=evidence_max_calls, budget=self._budget(spec, budget), population=population, question_id=spec['id'], **spec['request']) rows, segments = self.proxy.rows, self.proxy.segments state = self.api.answers._warm(key) finally: self.proxy.phase = 'idle' cost = {'branches': len(rows), 'input_tokens': sum(r['input_tokens'] for r in rows), 'document_prefill_tokens': (result.get('document_prefill') or {}).get('tokens', 0) or 0} out = {'type': spec['type']} if result.get('candidate') is None: # Not an abstention: the caller's own branch/token budget stopped the chain before it produced anything. stop = (result.get('compute') or {}).get('stop_reason', 'no_answer') out.update(self._empty(spec), error=stop, ordering_score=None, evidence=[], evidence_status='not_run') return out, cost T = self.temperature(task, [r.get('head_key') for r in rows]) units = self._units(spec, rows, result['candidate']) shown = self._distributions(spec, units, T) out.update(self._present(spec, shown)) out['ordering_score'] = ordering_score(shown) body = result.get('evidence') out['temperature'] = float(T) ev = self._evidence(spec, body, segments, out, T, evidence_detail) cost['evidence_calls'] = ev.pop('_calls', 0) out.update(ev) return out, cost def _units(self, spec, rows, candidate): """Answer-branch logits per unit, cross-checked against the pinned chain's own candidate.""" task = spec['task'] n_units = len(spec.get('candidates', [])) or 1 if len(rows) != n_units: raise ValueError(f'expected {n_units} answer branches, captured {len(rows)} (only R/S single-order designs are served)') logits = [r['letter_logits'] for r in rows] if task in ('boolean', 'entity', 'multilabel'): width = 2 if self.mode == 'two_letter' else 4 if any(len(x) != width for x in logits): raise ValueError('Noul branch width does not match the readout mode') pinned = [candidate['probabilities']] if task == 'boolean' else candidate['probabilities'] for x, p in zip(logits, pinned): if abs(sem.p_yes(x) - float(p[0])) > 1e-6: raise ValueError('captured branch does not match the pinned candidate') else: n = len(spec['texts']) p = np.asarray(candidate['probabilities'], float) if len(logits[0]) < n or np.max(np.abs(sem.listed_probs(logits[0], n) - p[:n] / p[:n].sum())) > 1e-6: raise ValueError('captured choice branch does not match the pinned candidate') return logits def _distributions(self, spec, units, T): if spec['task'] in ('boolean', 'entity', 'multilabel'): return [np.array([sem.p_yes(x, T), 1 - sem.p_yes(x, T)]) for x in units] return [sem.listed_probs(units[0], len(spec['texts']), T)] @staticmethod def _empty(spec): if spec['type'] == 'noul': return {'candidates': None} if 'candidates' in spec else {'noul': None} return {'probabilities': None, 'answer': None} @staticmethod def _present(spec, dists): if spec['type'] == 'noul': if 'candidates' in spec: # Entity and multi-label are accurate per candidate, so the probability is reported per candidate # and so is the ordering score. There is no single whole-question number for these types. return {'candidate_kind': 'label', 'candidates': {c: float(d[0]) for c, d in zip(spec['candidates'], dists)}, 'candidate_ordering_scores': {c: float(max(d[0], 1 - d[0])) for c, d in zip(spec['candidates'], dists)}} return {'noul': float(dists[0][0])} p = dists[0]; keys = spec['keys'] out = {'probabilities': {k: float(v) for k, v in zip(keys, p)}, 'answer': keys[int(np.argmax(p))]} out['choice'] = out['answer'] # compatibility alias of 'answer' if spec['type'] == 'score': out['score'] = float(np.dot(np.arange(len(p)), p)) out['legend'] = dict(zip(keys, spec['texts'])) elif spec['task'] == 'ordered': out['ordered'] = True return out def _evidence(self, spec, body, segments, answer, T, detail): if body is None: return {'evidence': [], 'evidence_status': 'not_requested'} status = body.get('status') out = {'evidence_status': status, 'evidence': [], '_calls': int(body.get('calls', 0))} if 'method' in body: # Which selector produced the spans: 'trained_relevance_head', or the labelled 'lexical_overlap_fallback' # (with the reason it ran). Neither establishes faithfulness. out['evidence_method'] = body['method'] out['evidence_faithfulness_established'] = False reason = getattr(getattr(self.api, 'selector', None), 'last_reason', None) if body['method'] == ES.FALLBACK_METHOD and reason: out['evidence_fallback_reason'] = reason ranking = getattr(getattr(self.api, 'selector', None), 'last_ranking', None) if body.get('method') == ES.RANKED_METHOD else None if status in ('found', 'no_support_found'): out['evidence'] = [{k: r[k] for k in ('start', 'end', 'text', 'page', 'path') if k in r} for r in body.get('references', [])] if 'candidates' in spec and body.get('package_format') == 'per_unit' and len(body.get('packages', [])) == len(spec['candidates']): out['candidate_evidence'] = {c: [{k: s[k] for k in ('start', 'end', 'text', 'pages') if k in s} for s in p['spans'] if s.get('role') == 'evidence'] for c, p in zip(spec['candidates'], body['packages'])} if ranking is not None: # v1.1 ranked pointers: the top sentences per answer unit IN SCORE ORDER, each with its relevance score. # A unit whose answer is 'not stated' carries none (evidence_suppressed). Scores rank; they certify nothing. ranked = [[{k: s[k] for k in ('start', 'end', 'text', 'score', 'rank')} for s in u['ranked']] for u in ranking] suppressed = [bool(u['suppressed']) for u in ranking] if 'candidates' in spec and len(ranked) == len(spec['candidates']): out['candidate_evidence'] = dict(zip(spec['candidates'], ranked)) out['candidate_evidence_suppressed'] = dict(zip(spec['candidates'], suppressed)) elif len(ranked) == 1: out['evidence'] = ranked[0] out['evidence_suppressed'] = suppressed[0] out['evidence_semantics'] = ES.RANKED_SEMANTICS verification = self._verification(spec, body, segments, answer, T) if verification: out['evidence_verification'] = verification if detail: out['evidence_detail'] = {k: v for k, v in body.items() if k not in ('evidence_only', 'evidence_removed', 'union_package')} return out def _verification(self, spec, body, segments, answer, T): """Evidence-only / evidence-removed re-asks in Solomon terms (from the captured fresh-prefill segments).""" modes = [m for m in ('evidence_only', 'evidence_removed') if m in body] if not modes: return None per_unit = 'evidence_only' in body and 'per_unit' in body['evidence_only'] calls = (len(body['evidence_only']['per_unit']) if per_unit else int('evidence_only' in body)) + int('evidence_removed' in body) if len(segments) < calls: raise ValueError('evidence re-ask branches were not captured') used = segments[len(segments) - calls:] groups = {} if 'evidence_only' in body: k = len(body['evidence_only']['per_unit']) if per_unit else 1 groups['evidence_only'] = [r for seg in used[:k] for r in seg] if 'evidence_removed' in body: groups['evidence_removed'] = used[-1] out = {} for mode, rows in groups.items(): n_units = len(spec.get('candidates', [])) or 1 if len(rows) != n_units: raise ValueError(f'{mode}: expected {n_units} branches, captured {len(rows)}') dists = self._distributions(spec, [r['letter_logits'] for r in rows], T) shown = self._present(spec, dists) shown.pop('choice', None) shown.pop('candidate_ordering_scores', None) out[mode] = {**shown, 'agrees_with_full': self._decision(spec, shown) == self._decision(spec, answer)} return out @staticmethod def _decision(spec, shown): if spec['type'] == 'noul': if 'candidates' in spec: return {c: p >= .5 for c, p in shown['candidates'].items()} return shown['noul'] >= .5 return shown['answer'] def head_routing(binding_routing, engine): """The served routing table: the binding's when it names one, else HEAD_ROUTING restricted to heads the engine lacks. Every target must be a head the engine actually loaded (checked when the engine exposes its heads).""" heads = getattr(engine, 'answer_heads', None) if binding_routing is not None: routing = dict(binding_routing) elif heads is None: routing = {} else: routing = {k: v for k, v in HEAD_ROUTING.items() if k not in heads} if heads is not None: missing = sorted({v for v in routing.values() if v not in heads}) if missing: raise ValueError('head routing points at heads the engine did not load: ' + ', '.join(missing)) return routing def evidence_selector(proxy, evidence_head=None, evidence_thresholds=None, device='cuda'): """(prefix, thresholds path) -> the trained selector; neither -> the labelled lexical fallback.""" if (evidence_head is None) != (evidence_thresholds is None): raise ValueError('an evidence head needs its per-task thresholds (and thresholds need a head)') if evidence_head is None: return ES.LexicalFallback('no evidence head configured') selector = ES.load_selector(proxy, _resolve(evidence_head), _resolve(evidence_thresholds), device=device) # Recorded in the runtime identity BEFORE the binding check, so a binding issued for this head pins it. identity = getattr(proxy._engine, 'identity', None) if isinstance(identity, dict): identity.update(evidence_head_sha256=selector.head_sha256, evidence_thresholds_sha256=selector.thresholds_sha256) proxy.tap_layers = selector.tap_layers if selector.tap_layers: proxy._engine.prefix_layers = tuple(selector.tap_layers) return selector def _resolve(path): path = Path(path) return path if path.is_absolute() else ROOT / path def service(store, engine, *, selection=SELECTION, mode=None, design=None, prompts=None, binding=None, temperature=None, selector=None, page_selector=None, page_maps=None, size_cap=None, package_format='per_unit', evidence_head=None, evidence_thresholds=None): """Build the pinned chain around a proxied engine and wrap it in the Solomon layer. mode/design/prompts/binding override selection.json (for tests and development); binding may be a path, a binding object, or None (selection 'binding' key, else DEFAULT_BINDING, else UnboundServing). A binding that is present but does not match the live runtime identity raises here, before any question is asked. """ from solomon.serving import ConfidenceHeadsService, FreshDecider from solomon.service_packages import PackageEvidenceService, SIZE_CAP chosen = None if mode is None: chosen = load_selection(selection) mode = chosen['mode'] design = design or chosen['design'] prompts = prompts if prompts is not None else chosen['prompts'] binding = binding if binding is not None else chosen['binding'] head_config = chosen.get('evidence_head') or {} beside = lambda v: (str(Path(chosen['dir']) / v) if isinstance(v, str) and not Path(v).is_absolute() and chosen.get('dir') and (Path(chosen['dir']) / (v + '.json' if not v.endswith('.json') else v)).exists() else v) evidence_head = evidence_head or beside(head_config.get('prefix')) evidence_thresholds = evidence_thresholds or beside(head_config.get('thresholds')) if isinstance(binding, str) and not Path(binding).is_absolute() and (Path(chosen['dir']) / binding).exists(): binding = Path(chosen['dir']) / binding # the selection names a binding beside itself if mode not in MODES: raise ValueError('unknown readout mode') design = copy.deepcopy(design or default_design(mode)) prompt_set = load_prompts(prompts) if mode == 'two_letter' else None proxy = _EngineProxy(engine, mode, prompt_set) if selector is None: selector = evidence_selector(proxy, evidence_head, evidence_thresholds) answers = ConfidenceHeadsService(store, proxy, design) api = PackageEvidenceService(_PhasedAnswers(answers, proxy), FreshDecider(proxy, design, Path(store) / '.evidence-readout'), selector, page_selector, page_maps=page_maps, size_cap=size_cap or SIZE_CAP, package_format=package_format) runtime = answers.runtime_identity loaded = load_binding(binding, mode=mode, runtime=runtime, design=design, prompts_sha256=prompt_set.sha256 if prompt_set else None) proxy.routing = head_routing(getattr(loaded, 'head_routing', None), engine) return SolomonService(api, proxy, mode=mode, design=design, binding=loaded, prompts=prompt_set, selection=chosen, temperature=temperature)