LIBE / app.py
rocco
Adding correct link to LIBE model
876dd1e
Raw History Blame Contribute Delete
10.1 kB
"""LIBE: sentence-to-language-description similarity search."""
import csv
import logging
import os
import pickle
import re
from pathlib import Path
# Avoid CUDA-runtime probing in the parent process before ZeroGPU forks.
os.environ.setdefault('PYTORCH_NVML_BASED_CUDA_CHECK', '1')
from importlib.metadata import version
import spaces # Import before torch / sentence_transformers for ZeroGPU.
import torch
import gradio as gr
from sentence_transformers import SentenceTransformer, util
ROOT = Path(__file__).resolve().parent
MODEL_ID = os.environ.get('MODEL_ID', '').strip()
MODEL_REVISION = os.environ.get('MODEL_REVISION') or None
MAX_CHARS = 5000
logging.basicConfig(level=logging.INFO)
def select_device(zero_gpu_flag, cuda_available):
is_zero = zero_gpu_flag.strip().lower() in {'1', 'true', 'yes', 'on', 't'}
return is_zero, 'cuda' if is_zero or cuda_available else 'cpu'
IS_ZERO_GPU = os.environ.get('SPACES_ZERO_GPU', '').strip().lower() in {
'1', 'true', 'yes', 'on', 't'
}
# Short-circuit: never query CUDA availability in the ZeroGPU parent.
DEVICE = 'cuda' if IS_ZERO_GPU else ('cuda' if torch.cuda.is_available() else 'cpu')
logging.info('LIBE runtime: torch=%s cuda_build=%s spaces=%s gradio=%s',
torch.__version__, torch.version.cuda, version('spaces'), version('gradio'))
logging.info('LIBE hardware: %s', 'ZeroGPU' if IS_ZERO_GPU else DEVICE)
def gpu_duration(text, cached_embeddings):
return 120 if cached_embeddings is None else 30
@spaces.GPU(duration=gpu_duration)
def search_on_gpu(text, cached_embeddings):
logging.info('LIBE GPU worker entered')
with torch.inference_mode():
if cached_embeddings is None:
logging.info('Encoding language descriptions inside GPU worker')
corpus = model.encode(
descriptions, convert_to_tensor=True, normalize_embeddings=True,
batch_size=16, show_progress_bar=True, device=DEVICE,
)
else:
corpus = cached_embeddings.to(DEVICE)
query = model.encode(
[text], convert_to_tensor=True, normalize_embeddings=True,
show_progress_bar=False, device=DEVICE,
)
hits = util.semantic_search(query, corpus, top_k=top_k)[0]
# Return the initial embeddings to the parent process explicitly.
# Worker-global cache mutations alone do not persist reliably on ZeroGPU.
new_cache = corpus.cpu() if cached_embeddings is None else None
return hits, new_cache
def load_descriptions(tsv_path, pickle_path):
# Only load the maintainer's trusted pickle; never accept user-uploaded pickles.
with open(pickle_path, 'rb') as f:
selected_raw = pickle.load(f)['glotlid-corpus']
selected = {str(lang).split('_')[0] for lang in selected_raw}
descriptions = {}
with open(tsv_path, encoding='utf-8') as f:
for row in csv.reader(f, delimiter='\t'):
if len(row) < 2 or row[0] not in selected:
continue
desc = row[1].replace('\xa0', ' ')
desc = re.sub(r'\[\d+\]', '', desc)
desc = re.sub(r'\s+', ' ', desc).strip()
if desc:
descriptions[row[0]] = desc
missing = selected - descriptions.keys()
logging.info('Matched %d languages; %d missing or empty descriptions.',
len(descriptions), len(missing))
if missing:
logging.warning('Missing examples: %s', sorted(missing)[:20])
if not descriptions:
raise ValueError('No languages matched. Check the TSV and glotlid-corpus IDs.')
labels = sorted(descriptions)
return labels, [descriptions[label] for label in labels]
if not MODEL_ID:
raise ValueError('Set MODEL_ID in Space Settings → Variables to your trained model repository ID.')
labels, descriptions = load_descriptions(
ROOT / 'data/WorldsLangs.tsv', ROOT / 'data/all_stats.pkl'
)
# Load the trained LIBE checkpoint, not the unmodified base LaBSE model.
model = SentenceTransformer(
MODEL_ID, device='cpu' if IS_ZERO_GPU else DEVICE, revision=MODEL_REVISION,
token=os.environ.get('HF_TOKEN') or None,
)
# No model inference in the parent process before the ZeroGPU worker starts.
corpus_embeddings = None
logging.info('LIBE startup: no description encoding; deferred to first request')
top_k = min(5, len(labels))
# Place model on CUDA outside the decorated function; ZeroGPU emulates this
# allocation until a request receives a real GPU. No GPU inference at startup.
model.to(DEVICE)
def create_demo():
def identify(text):
global corpus_embeddings
text = (text or '').strip()
if not text:
raise gr.Error('Please enter some text.')
if len(text) > MAX_CHARS:
raise gr.Error(f'Please use at most {MAX_CHARS:,} characters.')
hits, new_cache = search_on_gpu(text, corpus_embeddings)
if new_cache is not None:
corpus_embeddings = new_cache
best = hits[0]
i = best['corpus_id']
rows = [
[rank, labels[h['corpus_id']], round(float(h['score']), 4),
descriptions[h['corpus_id']][:200]]
for rank, h in enumerate(hits, 1)
]
return labels[i], round(float(best['score']), 4), descriptions[i], rows
demo = gr.Interface(
fn=identify,
inputs=gr.Textbox(label='Text to identify', lines=4,
placeholder='Type or paste a sentence…', max_length=MAX_CHARS),
outputs=[
gr.Textbox(label='Best-matching language', interactive=False),
gr.Number(label='Cosine similarity — not a probability', precision=4),
gr.Textbox(label='Language description', lines=5, interactive=False),
gr.Dataframe(headers=['Rank', 'Language', 'Cosine similarity', 'Description excerpt'],
datatype=['number', 'str', 'number', 'str'],
label='Top candidates', interactive=False),
],
title='LIBE · Language identification',
description=(
"**LIBE** identifies languages by comparing text embeddings with embeddings "
"of natural-language descriptions. This demo uses a **LaBSE-based bi-encoder "
"trained on 10 million sentences from GlotLID-C**. "
f"It ranks {len(labels)} candidate languages by cosine similarity."
),
article=(
"## About LIBE\n\n"
"Language identification is a key component of multilingual NLP pipelines and "
"corpus construction. Identification errors can create spurious text–language "
"associations and harm downstream model quality, particularly for low-resource "
"languages and closely related languages or dialects.\n\n"
"LIBE uses a bi-encoder initialized from **LaBSE** to embed input sentences and "
"natural-language descriptions of languages into a shared semantic space. "
"Language identification is then performed through similarity search. This "
"formulation supports data-efficient learning and adding candidate languages "
"through their descriptions. The checkpoint used here was trained on "
"**10 million GlotLID-C sentences**; its performance should be distinguished "
"from the paper's separate zero- and few-shot adaptation experiments.\n\n"
"### Research findings\n\n"
"Across multiple benchmarks, the paper reports competitive performance with "
"strong LID systems, with substantially less task-specific supervision in "
"several adaptation settings. It also introduces an optional **cross-encoder "
"re-ranking stage** to improve discrimination in ambiguous or low-resource cases. "
"The hybrid approach achieves up to **90% F1 with as few as five labeled examples** "
"in the reported challenging open-set settings. This result concerns the hybrid "
"experimental setup, not a performance guarantee for this demo. "
"**This demo uses the bi-encoder only, without cross-encoder re-ranking.**\n\n"
"### Interpreting predictions\n\n"
"The top candidates are ranked by **cosine similarity, not calibrated confidence "
"probabilities**. The demo always returns the closest candidate; it does not "
"automatically reject unsupported languages. Short, mixed-language, or ambiguous "
"texts may be difficult to identify. Adding a language description does not "
"guarantee reliable identification without evaluation. "
f"Long inputs may be truncated at the model's limit of {model.max_seq_length} tokens.\n\n"
"### Paper and resources\n\n"
"Rocco Tripodi (2026). **Representation Learning Enables Language Identification "
"in Zero- and Few-Shot Settings.** *Transactions of the Association for "
"Computational Linguistics*. To appear.\n\n"
"[Code and data](https://github.com/cafoscari-nlp/LIBE) · "
"[LIBE10M](https://huggingface.co/cafoscari-nlp/LIBE)\n\n"
"```bibtex\n"
"@article{tripodi-2026-libe,\n"
" author = {Tripodi, Rocco},\n"
" title = {Representation Learning Enables Language Identification in Zero- and Few-Shot Settings},\n"
" journal = {Transactions of the Association for Computational Linguistics},\n"
" year = {2026},\n"
" note = {To appear}\n"
"}\n```"
),
examples=[['Questa è una frase in italiano.'], ['This is a sentence in English.']],
cache_examples=False,
flagging_mode='never',
submit_btn='Identify language',
)
return demo.queue(default_concurrency_limit=1, max_size=20)
if __name__ == '__main__':
create_demo().launch(
server_name='0.0.0.0', server_port=7860, prevent_thread_lock=False,
)