Download app.py from roccot/LIBE: direct link, hf CLI and curl.
- Browser
- Download file 10.1 kB
-
https://huggingface.co/spaces/roccot/LIBE/resolve/main/app.py
- Command line
-
hf download hf://spaces/roccot/LIBE/app.py
-
curl -L -o app.py https://huggingface.co/spaces/roccot/LIBE/resolve/main/app.py
10.1 kB
| """LIBE: sentence-to-language-description similarity search.""" | |
| import csv | |
| import logging | |
| import os | |
| import pickle | |
| import re | |
| from pathlib import Path | |
| # Avoid CUDA-runtime probing in the parent process before ZeroGPU forks. | |
| os.environ.setdefault('PYTORCH_NVML_BASED_CUDA_CHECK', '1') | |
| from importlib.metadata import version | |
| import spaces # Import before torch / sentence_transformers for ZeroGPU. | |
| import torch | |
| import gradio as gr | |
| from sentence_transformers import SentenceTransformer, util | |
| ROOT = Path(__file__).resolve().parent | |
| MODEL_ID = os.environ.get('MODEL_ID', '').strip() | |
| MODEL_REVISION = os.environ.get('MODEL_REVISION') or None | |
| MAX_CHARS = 5000 | |
| logging.basicConfig(level=logging.INFO) | |
| def select_device(zero_gpu_flag, cuda_available): | |
| is_zero = zero_gpu_flag.strip().lower() in {'1', 'true', 'yes', 'on', 't'} | |
| return is_zero, 'cuda' if is_zero or cuda_available else 'cpu' | |
| IS_ZERO_GPU = os.environ.get('SPACES_ZERO_GPU', '').strip().lower() in { | |
| '1', 'true', 'yes', 'on', 't' | |
| } | |
| # Short-circuit: never query CUDA availability in the ZeroGPU parent. | |
| DEVICE = 'cuda' if IS_ZERO_GPU else ('cuda' if torch.cuda.is_available() else 'cpu') | |
| logging.info('LIBE runtime: torch=%s cuda_build=%s spaces=%s gradio=%s', | |
| torch.__version__, torch.version.cuda, version('spaces'), version('gradio')) | |
| logging.info('LIBE hardware: %s', 'ZeroGPU' if IS_ZERO_GPU else DEVICE) | |
| def gpu_duration(text, cached_embeddings): | |
| return 120 if cached_embeddings is None else 30 | |
| def search_on_gpu(text, cached_embeddings): | |
| logging.info('LIBE GPU worker entered') | |
| with torch.inference_mode(): | |
| if cached_embeddings is None: | |
| logging.info('Encoding language descriptions inside GPU worker') | |
| corpus = model.encode( | |
| descriptions, convert_to_tensor=True, normalize_embeddings=True, | |
| batch_size=16, show_progress_bar=True, device=DEVICE, | |
| ) | |
| else: | |
| corpus = cached_embeddings.to(DEVICE) | |
| query = model.encode( | |
| [text], convert_to_tensor=True, normalize_embeddings=True, | |
| show_progress_bar=False, device=DEVICE, | |
| ) | |
| hits = util.semantic_search(query, corpus, top_k=top_k)[0] | |
| # Return the initial embeddings to the parent process explicitly. | |
| # Worker-global cache mutations alone do not persist reliably on ZeroGPU. | |
| new_cache = corpus.cpu() if cached_embeddings is None else None | |
| return hits, new_cache | |
| def load_descriptions(tsv_path, pickle_path): | |
| # Only load the maintainer's trusted pickle; never accept user-uploaded pickles. | |
| with open(pickle_path, 'rb') as f: | |
| selected_raw = pickle.load(f)['glotlid-corpus'] | |
| selected = {str(lang).split('_')[0] for lang in selected_raw} | |
| descriptions = {} | |
| with open(tsv_path, encoding='utf-8') as f: | |
| for row in csv.reader(f, delimiter='\t'): | |
| if len(row) < 2 or row[0] not in selected: | |
| continue | |
| desc = row[1].replace('\xa0', ' ') | |
| desc = re.sub(r'\[\d+\]', '', desc) | |
| desc = re.sub(r'\s+', ' ', desc).strip() | |
| if desc: | |
| descriptions[row[0]] = desc | |
| missing = selected - descriptions.keys() | |
| logging.info('Matched %d languages; %d missing or empty descriptions.', | |
| len(descriptions), len(missing)) | |
| if missing: | |
| logging.warning('Missing examples: %s', sorted(missing)[:20]) | |
| if not descriptions: | |
| raise ValueError('No languages matched. Check the TSV and glotlid-corpus IDs.') | |
| labels = sorted(descriptions) | |
| return labels, [descriptions[label] for label in labels] | |
| if not MODEL_ID: | |
| raise ValueError('Set MODEL_ID in Space Settings → Variables to your trained model repository ID.') | |
| labels, descriptions = load_descriptions( | |
| ROOT / 'data/WorldsLangs.tsv', ROOT / 'data/all_stats.pkl' | |
| ) | |
| # Load the trained LIBE checkpoint, not the unmodified base LaBSE model. | |
| model = SentenceTransformer( | |
| MODEL_ID, device='cpu' if IS_ZERO_GPU else DEVICE, revision=MODEL_REVISION, | |
| token=os.environ.get('HF_TOKEN') or None, | |
| ) | |
| # No model inference in the parent process before the ZeroGPU worker starts. | |
| corpus_embeddings = None | |
| logging.info('LIBE startup: no description encoding; deferred to first request') | |
| top_k = min(5, len(labels)) | |
| # Place model on CUDA outside the decorated function; ZeroGPU emulates this | |
| # allocation until a request receives a real GPU. No GPU inference at startup. | |
| model.to(DEVICE) | |
| def create_demo(): | |
| def identify(text): | |
| global corpus_embeddings | |
| text = (text or '').strip() | |
| if not text: | |
| raise gr.Error('Please enter some text.') | |
| if len(text) > MAX_CHARS: | |
| raise gr.Error(f'Please use at most {MAX_CHARS:,} characters.') | |
| hits, new_cache = search_on_gpu(text, corpus_embeddings) | |
| if new_cache is not None: | |
| corpus_embeddings = new_cache | |
| best = hits[0] | |
| i = best['corpus_id'] | |
| rows = [ | |
| [rank, labels[h['corpus_id']], round(float(h['score']), 4), | |
| descriptions[h['corpus_id']][:200]] | |
| for rank, h in enumerate(hits, 1) | |
| ] | |
| return labels[i], round(float(best['score']), 4), descriptions[i], rows | |
| demo = gr.Interface( | |
| fn=identify, | |
| inputs=gr.Textbox(label='Text to identify', lines=4, | |
| placeholder='Type or paste a sentence…', max_length=MAX_CHARS), | |
| outputs=[ | |
| gr.Textbox(label='Best-matching language', interactive=False), | |
| gr.Number(label='Cosine similarity — not a probability', precision=4), | |
| gr.Textbox(label='Language description', lines=5, interactive=False), | |
| gr.Dataframe(headers=['Rank', 'Language', 'Cosine similarity', 'Description excerpt'], | |
| datatype=['number', 'str', 'number', 'str'], | |
| label='Top candidates', interactive=False), | |
| ], | |
| title='LIBE · Language identification', | |
| description=( | |
| "**LIBE** identifies languages by comparing text embeddings with embeddings " | |
| "of natural-language descriptions. This demo uses a **LaBSE-based bi-encoder " | |
| "trained on 10 million sentences from GlotLID-C**. " | |
| f"It ranks {len(labels)} candidate languages by cosine similarity." | |
| ), | |
| article=( | |
| "## About LIBE\n\n" | |
| "Language identification is a key component of multilingual NLP pipelines and " | |
| "corpus construction. Identification errors can create spurious text–language " | |
| "associations and harm downstream model quality, particularly for low-resource " | |
| "languages and closely related languages or dialects.\n\n" | |
| "LIBE uses a bi-encoder initialized from **LaBSE** to embed input sentences and " | |
| "natural-language descriptions of languages into a shared semantic space. " | |
| "Language identification is then performed through similarity search. This " | |
| "formulation supports data-efficient learning and adding candidate languages " | |
| "through their descriptions. The checkpoint used here was trained on " | |
| "**10 million GlotLID-C sentences**; its performance should be distinguished " | |
| "from the paper's separate zero- and few-shot adaptation experiments.\n\n" | |
| "### Research findings\n\n" | |
| "Across multiple benchmarks, the paper reports competitive performance with " | |
| "strong LID systems, with substantially less task-specific supervision in " | |
| "several adaptation settings. It also introduces an optional **cross-encoder " | |
| "re-ranking stage** to improve discrimination in ambiguous or low-resource cases. " | |
| "The hybrid approach achieves up to **90% F1 with as few as five labeled examples** " | |
| "in the reported challenging open-set settings. This result concerns the hybrid " | |
| "experimental setup, not a performance guarantee for this demo. " | |
| "**This demo uses the bi-encoder only, without cross-encoder re-ranking.**\n\n" | |
| "### Interpreting predictions\n\n" | |
| "The top candidates are ranked by **cosine similarity, not calibrated confidence " | |
| "probabilities**. The demo always returns the closest candidate; it does not " | |
| "automatically reject unsupported languages. Short, mixed-language, or ambiguous " | |
| "texts may be difficult to identify. Adding a language description does not " | |
| "guarantee reliable identification without evaluation. " | |
| f"Long inputs may be truncated at the model's limit of {model.max_seq_length} tokens.\n\n" | |
| "### Paper and resources\n\n" | |
| "Rocco Tripodi (2026). **Representation Learning Enables Language Identification " | |
| "in Zero- and Few-Shot Settings.** *Transactions of the Association for " | |
| "Computational Linguistics*. To appear.\n\n" | |
| "[Code and data](https://github.com/cafoscari-nlp/LIBE) · " | |
| "[LIBE10M](https://huggingface.co/cafoscari-nlp/LIBE)\n\n" | |
| "```bibtex\n" | |
| "@article{tripodi-2026-libe,\n" | |
| " author = {Tripodi, Rocco},\n" | |
| " title = {Representation Learning Enables Language Identification in Zero- and Few-Shot Settings},\n" | |
| " journal = {Transactions of the Association for Computational Linguistics},\n" | |
| " year = {2026},\n" | |
| " note = {To appear}\n" | |
| "}\n```" | |
| ), | |
| examples=[['Questa è una frase in italiano.'], ['This is a sentence in English.']], | |
| cache_examples=False, | |
| flagging_mode='never', | |
| submit_btn='Identify language', | |
| ) | |
| return demo.queue(default_concurrency_limit=1, max_size=20) | |
| if __name__ == '__main__': | |
| create_demo().launch( | |
| server_name='0.0.0.0', server_port=7860, prevent_thread_lock=False, | |
| ) | |