"""Deterministic spaced sample for literal-span and data-format diagnostics.""" import ast,collections,csv,gzip,json,re,time from pathlib import Path from evaluate_model import ROOT,dump def main(): csv.field_size_limit(64*1024*1024);stats=collections.Counter();examples=[];lengths=[];start=time.time() with gzip.open(ROOT/'final_gliner_file.csv.gz','rt',newline='') as stream: for index,row in enumerate(csv.DictReader(stream),1): if index%512!=137:continue stats['sampled_rows']+=1 try: try:item=json.loads(row['gliner_ner']) except json.JSONDecodeError:item=ast.literal_eval(row['gliner_ner']) text=item['text'];entities=item['entities'];schema=item.get('entity_descriptions',{}) stats['valid_sampled_rows']+=1;lengths.append(len(text.split())) stats['schema_types']+=len(schema) stats['schema_types_without_positive_mentions']+=sum(not entities.get(k) for k in schema) stats['empty_gold_rows']+=not any(entities.values()) for label,values in entities.items(): if not isinstance(values,list):stats['nonlist_values']+=1;continue for value in values: stats['mentions']+=1 if not isinstance(value,str):stats['nonstring_mentions']+=1;continue if not value:stats['empty_string_mentions']+=1;continue stats['mentions_over8_whitespace_words']+=len(value.split())>8 stats['mentions_over8_regex_units']+=len(re.findall(r'\w+(?:[-_]\w+)*|\S',value))>8 n=text.count(value) if n==0: stats['mentions_not_literal_substrings']+=1 if value.casefold() in text.casefold():stats['case_only_mismatch']+=1 if len(examples)<30:examples.append({'row':index,'label':label,'mention':value,'text':text}) elif n>1:stats['mentions_with_multiple_literal_occurrences']+=1 except Exception:stats['parse_errors']+=1 summary={'method':'Every 512th CSV record, offset 137; spaced sample rather than a random or domain-stratified sample.', 'counts':dict(stats),'seconds':time.time()-start, 'text_word_lengths':{'min':min(lengths),'median':sorted(lengths)[len(lengths)//2],'max':max(lengths)}} dump(ROOT/'investigation/training-span-quality.json',summary) # Training excerpts are private local diagnostics, excluded from publication. dump(ROOT/'investigation/private-training-mismatch-examples.json',examples) print(json.dumps(summary,indent=2),flush=True) if __name__=='__main__':main()