Text Generation
Transformers
PyTorch
Safetensors
GGUF
English
collision
conversational
rag
reasoning
deepseek-r1-style
system-2
agent
ollama
llama.cpp
fastapi
openai-compatible
slm
edge-ai
cpu-first
in-house-nlp
math
keyphrase-extraction
topic-classification
grammar-correction
reading-comprehension
sentiment-analysis
research
educational
custom_code
Instructions to use collision-10M/Collision-1B with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use collision-10M/Collision-1B with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-generation", model="collision-10M/Collision-1B", trust_remote_code=True)# pip install -U transformers accelerate # Load model directly from transformers import AutoModelForCausalLM model = AutoModelForCausalLM.from_pretrained("collision-10M/Collision-1B", trust_remote_code=True, device_map="auto") - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use collision-10M/Collision-1B with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "collision-10M/Collision-1B" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "collision-10M/Collision-1B", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker
docker model run hf.co/collision-10M/Collision-1B
- SGLang
How to use collision-10M/Collision-1B with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "collision-10M/Collision-1B" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "collision-10M/Collision-1B", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "collision-10M/Collision-1B" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "collision-10M/Collision-1B", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }' - Docker Model Runner
How to use collision-10M/Collision-1B with Docker Model Runner:
docker model run hf.co/collision-10M/Collision-1B
Download data/stats.py from collision-10M/Collision-1B: direct link, hf CLI and curl.
- Browser
- Download file 2.44 kB
-
https://huggingface.co/collision-10M/Collision-1B/resolve/main/data/stats.py
- Command line
-
hf download hf://collision-10M/Collision-1B/data/stats.py
-
curl -L -o stats.py https://huggingface.co/collision-10M/Collision-1B/resolve/main/data/stats.py
2.44 kB
| import os | |
| import json | |
| import re | |
| def get_latest_version_dir(datasets_dir="datasets"): | |
| if not os.path.exists(datasets_dir): | |
| return None | |
| versions = [] | |
| for d in os.listdir(datasets_dir): | |
| match = re.match(r'collision_dataset_v(\d+)', d) | |
| if match: | |
| versions.append((int(match.group(1)), d)) | |
| if not versions: | |
| return None | |
| # Sort by version number | |
| latest_name = sorted(versions, key=lambda x: x[0])[-1][1] | |
| return os.path.join(datasets_dir, latest_name) | |
| def main(): | |
| latest_dir = get_latest_version_dir() | |
| if not latest_dir: | |
| print("No prepared dataset versions found. Please run 'python -m data.prepare' first.") | |
| return | |
| meta_path = os.path.join(latest_dir, "metadata.json") | |
| if not os.path.exists(meta_path): | |
| print(f"Error: metadata.json missing in {latest_dir}") | |
| return | |
| with open(meta_path, "r", encoding="utf-8") as f: | |
| meta = json.load(f) | |
| # Calculate average document length | |
| raw_path = os.path.join("data", "raw") | |
| txt_files = meta.get("source_files", []) | |
| lengths = [] | |
| for f_name in txt_files: | |
| f_path = os.path.join(raw_path, f_name) | |
| if os.path.exists(f_path): | |
| lengths.append(os.path.getsize(f_path)) | |
| avg_len = sum(lengths) / len(lengths) if lengths else 0 | |
| print("## COLLISION DATASET\n") | |
| print(f"Dataset Version: {meta.get('dataset_version', 'N/A')}") | |
| print(f"Files: {', '.join(txt_files)}") | |
| print(f"Characters: {meta.get('cleaned_characters', 0):,}") | |
| print(f"Estimated tokens: {meta.get('token_count', 0):,}") | |
| print(f"Training tokens: {meta.get('train_tokens', 0):,}") | |
| print(f"Validation tokens: {meta.get('validation_tokens', 0):,}") | |
| print(f"Vocabulary: {meta.get('vocabulary_size', 0)}") | |
| print(f"Average doc length: {avg_len:,.1f} bytes") | |
| print() | |
| # Warn about dataset size | |
| token_count = meta.get("token_count", 0) | |
| if token_count < 100_000: | |
| print("WARNING: Dataset is suitable only for testing.") | |
| elif 100_000 <= token_count < 1_000_000: | |
| print("Dataset classification: Small training dataset.") | |
| elif 1_000_000 <= token_count < 10_000_000: | |
| print("Dataset classification: Useful small-model training dataset.") | |
| else: | |
| print("Dataset classification: Good for extended experiments.") | |
| if __name__ == "__main__": | |
| main() | |