Instructions to use Siddharth63/gliner2-small with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- GLiNER2
How to use Siddharth63/gliner2-small with GLiNER2:
from gliner2 import AutoExtractor extractor = AutoExtractor.from_pretrained("Siddharth63/gliner2-small") # Extract entities text = "Apple CEO Tim Cook announced iPhone 15 in Cupertino yesterday." result = extractor.extract_entities(text, ["company", "person", "product", "location"]) print(result) - Notebooks
- Google Colab
- Kaggle
Download evaluation/scripts/sample_training_quality.py from Siddharth63/gliner2-small: direct link, hf CLI and curl.
- Browser
- Download file 2.81 kB
-
https://huggingface.co/Siddharth63/gliner2-small/resolve/main/evaluation/scripts/sample_training_quality.py
- Command line
-
hf download hf://Siddharth63/gliner2-small/evaluation/scripts/sample_training_quality.py
-
curl -L -o sample_training_quality.py https://huggingface.co/Siddharth63/gliner2-small/resolve/main/evaluation/scripts/sample_training_quality.py
2.81 kB
| """Deterministic spaced sample for literal-span and data-format diagnostics.""" | |
| import ast,collections,csv,gzip,json,re,time | |
| from pathlib import Path | |
| from evaluate_model import ROOT,dump | |
| def main(): | |
| csv.field_size_limit(64*1024*1024);stats=collections.Counter();examples=[];lengths=[];start=time.time() | |
| with gzip.open(ROOT/'final_gliner_file.csv.gz','rt',newline='') as stream: | |
| for index,row in enumerate(csv.DictReader(stream),1): | |
| if index%512!=137:continue | |
| stats['sampled_rows']+=1 | |
| try: | |
| try:item=json.loads(row['gliner_ner']) | |
| except json.JSONDecodeError:item=ast.literal_eval(row['gliner_ner']) | |
| text=item['text'];entities=item['entities'];schema=item.get('entity_descriptions',{}) | |
| stats['valid_sampled_rows']+=1;lengths.append(len(text.split())) | |
| stats['schema_types']+=len(schema) | |
| stats['schema_types_without_positive_mentions']+=sum(not entities.get(k) for k in schema) | |
| stats['empty_gold_rows']+=not any(entities.values()) | |
| for label,values in entities.items(): | |
| if not isinstance(values,list):stats['nonlist_values']+=1;continue | |
| for value in values: | |
| stats['mentions']+=1 | |
| if not isinstance(value,str):stats['nonstring_mentions']+=1;continue | |
| if not value:stats['empty_string_mentions']+=1;continue | |
| stats['mentions_over8_whitespace_words']+=len(value.split())>8 | |
| stats['mentions_over8_regex_units']+=len(re.findall(r'\w+(?:[-_]\w+)*|\S',value))>8 | |
| n=text.count(value) | |
| if n==0: | |
| stats['mentions_not_literal_substrings']+=1 | |
| if value.casefold() in text.casefold():stats['case_only_mismatch']+=1 | |
| if len(examples)<30:examples.append({'row':index,'label':label,'mention':value,'text':text}) | |
| elif n>1:stats['mentions_with_multiple_literal_occurrences']+=1 | |
| except Exception:stats['parse_errors']+=1 | |
| summary={'method':'Every 512th CSV record, offset 137; spaced sample rather than a random or domain-stratified sample.', | |
| 'counts':dict(stats),'seconds':time.time()-start, | |
| 'text_word_lengths':{'min':min(lengths),'median':sorted(lengths)[len(lengths)//2],'max':max(lengths)}} | |
| dump(ROOT/'investigation/training-span-quality.json',summary) | |
| # Training excerpts are private local diagnostics, excluded from publication. | |
| dump(ROOT/'investigation/private-training-mismatch-examples.json',examples) | |
| print(json.dumps(summary,indent=2),flush=True) | |
| if __name__=='__main__':main() | |