"""LIBE: sentence-to-language-description similarity search.""" import csv import logging import os import pickle import re from pathlib import Path # Avoid CUDA-runtime probing in the parent process before ZeroGPU forks. os.environ.setdefault('PYTORCH_NVML_BASED_CUDA_CHECK', '1') from importlib.metadata import version import spaces # Import before torch / sentence_transformers for ZeroGPU. import torch import gradio as gr from sentence_transformers import SentenceTransformer, util ROOT = Path(__file__).resolve().parent MODEL_ID = os.environ.get('MODEL_ID', '').strip() MODEL_REVISION = os.environ.get('MODEL_REVISION') or None MAX_CHARS = 5000 logging.basicConfig(level=logging.INFO) def select_device(zero_gpu_flag, cuda_available): is_zero = zero_gpu_flag.strip().lower() in {'1', 'true', 'yes', 'on', 't'} return is_zero, 'cuda' if is_zero or cuda_available else 'cpu' IS_ZERO_GPU = os.environ.get('SPACES_ZERO_GPU', '').strip().lower() in { '1', 'true', 'yes', 'on', 't' } # Short-circuit: never query CUDA availability in the ZeroGPU parent. DEVICE = 'cuda' if IS_ZERO_GPU else ('cuda' if torch.cuda.is_available() else 'cpu') logging.info('LIBE runtime: torch=%s cuda_build=%s spaces=%s gradio=%s', torch.__version__, torch.version.cuda, version('spaces'), version('gradio')) logging.info('LIBE hardware: %s', 'ZeroGPU' if IS_ZERO_GPU else DEVICE) def gpu_duration(text, cached_embeddings): return 120 if cached_embeddings is None else 30 @spaces.GPU(duration=gpu_duration) def search_on_gpu(text, cached_embeddings): logging.info('LIBE GPU worker entered') with torch.inference_mode(): if cached_embeddings is None: logging.info('Encoding language descriptions inside GPU worker') corpus = model.encode( descriptions, convert_to_tensor=True, normalize_embeddings=True, batch_size=16, show_progress_bar=True, device=DEVICE, ) else: corpus = cached_embeddings.to(DEVICE) query = model.encode( [text], convert_to_tensor=True, normalize_embeddings=True, show_progress_bar=False, device=DEVICE, ) hits = util.semantic_search(query, corpus, top_k=top_k)[0] # Return the initial embeddings to the parent process explicitly. # Worker-global cache mutations alone do not persist reliably on ZeroGPU. new_cache = corpus.cpu() if cached_embeddings is None else None return hits, new_cache def load_descriptions(tsv_path, pickle_path): # Only load the maintainer's trusted pickle; never accept user-uploaded pickles. with open(pickle_path, 'rb') as f: selected_raw = pickle.load(f)['glotlid-corpus'] selected = {str(lang).split('_')[0] for lang in selected_raw} descriptions = {} with open(tsv_path, encoding='utf-8') as f: for row in csv.reader(f, delimiter='\t'): if len(row) < 2 or row[0] not in selected: continue desc = row[1].replace('\xa0', ' ') desc = re.sub(r'\[\d+\]', '', desc) desc = re.sub(r'\s+', ' ', desc).strip() if desc: descriptions[row[0]] = desc missing = selected - descriptions.keys() logging.info('Matched %d languages; %d missing or empty descriptions.', len(descriptions), len(missing)) if missing: logging.warning('Missing examples: %s', sorted(missing)[:20]) if not descriptions: raise ValueError('No languages matched. Check the TSV and glotlid-corpus IDs.') labels = sorted(descriptions) return labels, [descriptions[label] for label in labels] if not MODEL_ID: raise ValueError('Set MODEL_ID in Space Settings → Variables to your trained model repository ID.') labels, descriptions = load_descriptions( ROOT / 'data/WorldsLangs.tsv', ROOT / 'data/all_stats.pkl' ) # Load the trained LIBE checkpoint, not the unmodified base LaBSE model. model = SentenceTransformer( MODEL_ID, device='cpu' if IS_ZERO_GPU else DEVICE, revision=MODEL_REVISION, token=os.environ.get('HF_TOKEN') or None, ) # No model inference in the parent process before the ZeroGPU worker starts. corpus_embeddings = None logging.info('LIBE startup: no description encoding; deferred to first request') top_k = min(5, len(labels)) # Place model on CUDA outside the decorated function; ZeroGPU emulates this # allocation until a request receives a real GPU. No GPU inference at startup. model.to(DEVICE) def create_demo(): def identify(text): global corpus_embeddings text = (text or '').strip() if not text: raise gr.Error('Please enter some text.') if len(text) > MAX_CHARS: raise gr.Error(f'Please use at most {MAX_CHARS:,} characters.') hits, new_cache = search_on_gpu(text, corpus_embeddings) if new_cache is not None: corpus_embeddings = new_cache best = hits[0] i = best['corpus_id'] rows = [ [rank, labels[h['corpus_id']], round(float(h['score']), 4), descriptions[h['corpus_id']][:200]] for rank, h in enumerate(hits, 1) ] return labels[i], round(float(best['score']), 4), descriptions[i], rows demo = gr.Interface( fn=identify, inputs=gr.Textbox(label='Text to identify', lines=4, placeholder='Type or paste a sentence…', max_length=MAX_CHARS), outputs=[ gr.Textbox(label='Best-matching language', interactive=False), gr.Number(label='Cosine similarity — not a probability', precision=4), gr.Textbox(label='Language description', lines=5, interactive=False), gr.Dataframe(headers=['Rank', 'Language', 'Cosine similarity', 'Description excerpt'], datatype=['number', 'str', 'number', 'str'], label='Top candidates', interactive=False), ], title='LIBE · Language identification', description=( "**LIBE** identifies languages by comparing text embeddings with embeddings " "of natural-language descriptions. This demo uses a **LaBSE-based bi-encoder " "trained on 10 million sentences from GlotLID-C**. " f"It ranks {len(labels)} candidate languages by cosine similarity." ), article=( "## About LIBE\n\n" "Language identification is a key component of multilingual NLP pipelines and " "corpus construction. Identification errors can create spurious text–language " "associations and harm downstream model quality, particularly for low-resource " "languages and closely related languages or dialects.\n\n" "LIBE uses a bi-encoder initialized from **LaBSE** to embed input sentences and " "natural-language descriptions of languages into a shared semantic space. " "Language identification is then performed through similarity search. This " "formulation supports data-efficient learning and adding candidate languages " "through their descriptions. The checkpoint used here was trained on " "**10 million GlotLID-C sentences**; its performance should be distinguished " "from the paper's separate zero- and few-shot adaptation experiments.\n\n" "### Research findings\n\n" "Across multiple benchmarks, the paper reports competitive performance with " "strong LID systems, with substantially less task-specific supervision in " "several adaptation settings. It also introduces an optional **cross-encoder " "re-ranking stage** to improve discrimination in ambiguous or low-resource cases. " "The hybrid approach achieves up to **90% F1 with as few as five labeled examples** " "in the reported challenging open-set settings. This result concerns the hybrid " "experimental setup, not a performance guarantee for this demo. " "**This demo uses the bi-encoder only, without cross-encoder re-ranking.**\n\n" "### Interpreting predictions\n\n" "The top candidates are ranked by **cosine similarity, not calibrated confidence " "probabilities**. The demo always returns the closest candidate; it does not " "automatically reject unsupported languages. Short, mixed-language, or ambiguous " "texts may be difficult to identify. Adding a language description does not " "guarantee reliable identification without evaluation. " f"Long inputs may be truncated at the model's limit of {model.max_seq_length} tokens.\n\n" "### Paper and resources\n\n" "Rocco Tripodi (2026). **Representation Learning Enables Language Identification " "in Zero- and Few-Shot Settings.** *Transactions of the Association for " "Computational Linguistics*. To appear.\n\n" "[Code and data](https://github.com/cafoscari-nlp/LIBE) · " "[LIBE10M](https://huggingface.co/cafoscari-nlp/LIBE)\n\n" "```bibtex\n" "@article{tripodi-2026-libe,\n" " author = {Tripodi, Rocco},\n" " title = {Representation Learning Enables Language Identification in Zero- and Few-Shot Settings},\n" " journal = {Transactions of the Association for Computational Linguistics},\n" " year = {2026},\n" " note = {To appear}\n" "}\n```" ), examples=[['Questa è una frase in italiano.'], ['This is a sentence in English.']], cache_examples=False, flagging_mode='never', submit_btn='Identify language', ) return demo.queue(default_concurrency_limit=1, max_size=20) if __name__ == '__main__': create_demo().launch( server_name='0.0.0.0', server_port=7860, prevent_thread_lock=False, )