Sentence Similarity
sentence-transformers
ONNX
Safetensors
Transformers.js
modernbert
feature-extraction
embeddings
retrieval
bge-m3
distillation
mmbert
text-embeddings-inference
Instructions to use Horizon-Labs/multilingual-embedding-base with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- sentence-transformers
How to use Horizon-Labs/multilingual-embedding-base with sentence-transformers:
from sentence_transformers import SentenceTransformer model = SentenceTransformer("Horizon-Labs/multilingual-embedding-base") sentences = [ "How tall is the Eiffel Tower?", "La tour Eiffel mesure 330 mètres.", "Der Eiffelturm ist das höchste Bauwerk von Paris.", "The Statue of Liberty is 93 metres tall.", "Ich esse gern Pizza." ] embeddings = model.encode(sentences) similarities = model.similarity(embeddings, embeddings) print(similarities.shape) # [5, 5] - Transformers.js
How to use Horizon-Labs/multilingual-embedding-base with Transformers.js:
// npm i @huggingface/transformers import { pipeline } from '@huggingface/transformers'; // Allocate pipeline const pipe = await pipeline('feature-extraction', 'Horizon-Labs/multilingual-embedding-base'); - Notebooks
- Google Colab
- Kaggle
Download code/rerank/gen_queries.py from Horizon-Labs/multilingual-embedding-base: direct link, hf CLI and curl.
- Browser
- Download file 2.93 kB
-
https://huggingface.co/Horizon-Labs/multilingual-embedding-base/resolve/main/code/rerank/gen_queries.py
- Command line
-
hf download hf://Horizon-Labs/multilingual-embedding-base/code/rerank/gen_queries.py
-
curl -L -o gen_queries.py https://huggingface.co/Horizon-Labs/multilingual-embedding-base/resolve/main/code/rerank/gen_queries.py
2.93 kB
| """Search queries for passages, Qwen3.8-27B via vLLM (Apache-2.0 outputs). | |
| IN=passages.parquet SHARD=i NSHARD=n OUT=dir python rerank/gen_queries.py -> $OUT/q_<shard>.jsonl {pid, query, qtype, qlang} | |
| Per passage: one natural-language question and one short keyword-style query, in the passage's language; for 15% of | |
| non-English passages one more query in English (cross-lingual search). Queries must be answerable by the passage but | |
| should not copy long phrases from it. | |
| """ | |
| import json, os, random, re, sys | |
| sys.path.insert(0, os.path.join(os.path.dirname(os.path.abspath(__file__)), "..", "sentiment")) | |
| def parse_json(t): | |
| t = re.sub(r"^```(?:json)?|```$", "", t.strip(), flags=re.M).strip() | |
| s, e = t.find("{"), t.rfind("}") | |
| try: | |
| return json.loads(t[s: e + 1]) | |
| except Exception: | |
| return None | |
| def prompt(text, xl): | |
| extra = ('\n- "english": a natural question in English that this passage answers (for cross-lingual search)' if xl else "") | |
| return f"""Passage: | |
| <<< | |
| {text} | |
| >>> | |
| Write search queries that a user might type into a search engine and for which this passage is a good result. | |
| - "question": a natural question in the same language as the passage, answered by the passage | |
| - "keywords": a short keyword-style query (2-6 words) in the same language as the passage{extra} | |
| Do not copy long phrases from the passage; use the words a searcher would use. Return only JSON with these keys.""" | |
| if __name__ == "__main__": | |
| import pandas as pd | |
| from vllm import LLM, SamplingParams | |
| from common import hard_exit | |
| df = pd.read_parquet(os.environ["IN"]); sh, ns = int(os.environ.get("SHARD", 0)), int(os.environ.get("NSHARD", 1)) | |
| df = df.iloc[sh::ns] | |
| rng = random.Random(11 + sh) | |
| xl = [l != "eng_Latn" and rng.random() < 0.15 for l in df.lang] | |
| llm = LLM(os.environ.get("MODEL", "Qwen/Qwen3.8-27B-FP8"), max_model_len=4096, gpu_memory_utilization=0.92, max_num_seqs=512, | |
| limit_mm_per_prompt={"image": 0, "video": 0}) | |
| outs = llm.chat([[{"role": "user", "content": prompt(t, x)}] for t, x in zip(df.text, xl)], | |
| SamplingParams(temperature=0.7, top_p=0.95, max_tokens=300, seed=5 + sh), chat_template_kwargs={"enable_thinking": False}) | |
| os.makedirs(os.environ["OUT"], exist_ok=True); n = 0 | |
| with open(f"{os.environ['OUT']}/q_{sh}.jsonl", "w") as f: | |
| for pid, lang, o in zip(df.pid, df.lang, outs): | |
| j = parse_json(o.outputs[0].text) | |
| if not isinstance(j, dict): | |
| continue | |
| for qt in ["question", "keywords", "english"]: | |
| q = j.get(qt) | |
| if isinstance(q, str) and 3 <= len(q.strip()) <= 300: | |
| f.write(json.dumps(dict(pid=int(pid), query=q.strip(), qtype=qt, qlang="eng_Latn" if qt == "english" else lang), ensure_ascii=False) + "\n"); n += 1 | |
| print("queries", n, "for", len(df), "passages", flush=True) | |
| hard_exit(0) | |