gliner2-small / evaluation /scripts /sample_training_quality.py
Siddharth63's picture
Document audited training labels, context settings and reproducible NER evaluation
99bfb2e verified
Raw History Blame Contribute Delete
2.81 kB
"""Deterministic spaced sample for literal-span and data-format diagnostics."""
import ast,collections,csv,gzip,json,re,time
from pathlib import Path
from evaluate_model import ROOT,dump
def main():
csv.field_size_limit(64*1024*1024);stats=collections.Counter();examples=[];lengths=[];start=time.time()
with gzip.open(ROOT/'final_gliner_file.csv.gz','rt',newline='') as stream:
for index,row in enumerate(csv.DictReader(stream),1):
if index%512!=137:continue
stats['sampled_rows']+=1
try:
try:item=json.loads(row['gliner_ner'])
except json.JSONDecodeError:item=ast.literal_eval(row['gliner_ner'])
text=item['text'];entities=item['entities'];schema=item.get('entity_descriptions',{})
stats['valid_sampled_rows']+=1;lengths.append(len(text.split()))
stats['schema_types']+=len(schema)
stats['schema_types_without_positive_mentions']+=sum(not entities.get(k) for k in schema)
stats['empty_gold_rows']+=not any(entities.values())
for label,values in entities.items():
if not isinstance(values,list):stats['nonlist_values']+=1;continue
for value in values:
stats['mentions']+=1
if not isinstance(value,str):stats['nonstring_mentions']+=1;continue
if not value:stats['empty_string_mentions']+=1;continue
stats['mentions_over8_whitespace_words']+=len(value.split())>8
stats['mentions_over8_regex_units']+=len(re.findall(r'\w+(?:[-_]\w+)*|\S',value))>8
n=text.count(value)
if n==0:
stats['mentions_not_literal_substrings']+=1
if value.casefold() in text.casefold():stats['case_only_mismatch']+=1
if len(examples)<30:examples.append({'row':index,'label':label,'mention':value,'text':text})
elif n>1:stats['mentions_with_multiple_literal_occurrences']+=1
except Exception:stats['parse_errors']+=1
summary={'method':'Every 512th CSV record, offset 137; spaced sample rather than a random or domain-stratified sample.',
'counts':dict(stats),'seconds':time.time()-start,
'text_word_lengths':{'min':min(lengths),'median':sorted(lengths)[len(lengths)//2],'max':max(lengths)}}
dump(ROOT/'investigation/training-span-quality.json',summary)
# Training excerpts are private local diagnostics, excluded from publication.
dump(ROOT/'investigation/private-training-mismatch-examples.json',examples)
print(json.dumps(summary,indent=2),flush=True)
if __name__=='__main__':main()