Text Generation
Transformers
Safetensors
English
llama
gpt-u
tiny-lm
pretrained-from-scratch
text-generation-inference
Instructions to use DedeProGames/GPT-U-20M with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use DedeProGames/GPT-U-20M with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-generation", model="DedeProGames/GPT-U-20M")# pip install -U transformers accelerate # Load model directly from transformers import AutoTokenizer, AutoModelForCausalLM tokenizer = AutoTokenizer.from_pretrained("DedeProGames/GPT-U-20M") model = AutoModelForCausalLM.from_pretrained("DedeProGames/GPT-U-20M", device_map="auto") - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use DedeProGames/GPT-U-20M with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "DedeProGames/GPT-U-20M" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "DedeProGames/GPT-U-20M", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }'Use Docker
docker model run hf.co/DedeProGames/GPT-U-20M
- SGLang
How to use DedeProGames/GPT-U-20M with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "DedeProGames/GPT-U-20M" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "DedeProGames/GPT-U-20M", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "DedeProGames/GPT-U-20M" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "DedeProGames/GPT-U-20M", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }' - Docker Model Runner
How to use DedeProGames/GPT-U-20M with Docker Model Runner:
docker model run hf.co/DedeProGames/GPT-U-20M
Download scripts/plot.py from DedeProGames/GPT-U-20M: direct link, hf CLI and curl.
- Browser
- Download file 4.83 kB
-
https://huggingface.co/DedeProGames/GPT-U-20M/resolve/main/scripts/plot.py
- Command line
-
hf download hf://DedeProGames/GPT-U-20M/scripts/plot.py
-
curl -L -o plot.py https://huggingface.co/DedeProGames/GPT-U-20M/resolve/main/scripts/plot.py
4.83 kB
| """Loss curve for GPT-U-20M: training loss (logged value + 100-step mean) and validation loss per domain.""" | |
| import json | |
| from pathlib import Path | |
| import matplotlib | |
| matplotlib.use("Agg") | |
| import matplotlib.pyplot as plt | |
| SURFACE, INK, INK_2, MUTED, GRID, AXIS = "#fcfcfb", "#0b0b0b", "#52514e", "#898781", "#e1e0d9", "#c3c2b7" | |
| DOMAIN_COLORS = {"dclm": "#2a78d6", "edu": "#eb6834", "code": "#1baf7a"} # categorical slots 1-3, fixed order | |
| PHASES = ((500, "warmup ends"), (15_872, "decay starts")) | |
| def read_log(path: Path) -> tuple[list, list]: | |
| """Last record per step wins (a resumed run re-logs the steps after its checkpoint).""" | |
| train, evals = {}, {} | |
| for line in path.read_text(encoding="utf-8").splitlines(): | |
| rec = json.loads(line) | |
| if rec.get("event") == "eval": | |
| evals[rec["step"]] = rec | |
| elif "event" not in rec: | |
| train[rec["step"]] = rec["loss"] | |
| return sorted(train.items()), sorted(evals.items()) | |
| def _style(ax, title: str, ylabel: str) -> None: | |
| ax.set_facecolor(SURFACE) | |
| ax.set_title(title, loc="left", color=INK, fontsize=11) | |
| ax.set_xlabel("step (131,072 tokens each)", color=INK_2, fontsize=9) | |
| ax.set_ylabel(ylabel, color=INK_2, fontsize=9) | |
| ax.grid(axis="y", color=GRID, linewidth=0.5) | |
| ax.set_axisbelow(True) | |
| for side in ("top", "right"): | |
| ax.spines[side].set_visible(False) | |
| for side in ("left", "bottom"): | |
| ax.spines[side].set_color(AXIS) | |
| ax.tick_params(colors=MUTED, labelsize=8) | |
| def _phase_lines(ax, x_max: int) -> None: | |
| for step, label in PHASES: | |
| if step <= x_max: | |
| ax.axvline(step, color=AXIS, linewidth=0.6, linestyle=(0, (3, 3))) | |
| ax.annotate(label, (step, 1), xycoords=("data", "axes fraction"), xytext=(3, -10), | |
| textcoords="offset points", color=MUTED, fontsize=7) | |
| def plot_loss(log_path: Path, out_path: Path, total_steps: int) -> None: | |
| train, evals = read_log(Path(log_path)) | |
| fig, (ax_t, ax_v) = plt.subplots(1, 2, figsize=(12, 4.6), dpi=150, facecolor=SURFACE) | |
| _style(ax_t, "Training loss", "cross-entropy (nats/token)") | |
| _style(ax_v, "Validation loss by domain", "cross-entropy (nats/token)") | |
| x_max = max([s for s, _ in train] + [s for s, _ in evals] + [1]) | |
| if train: | |
| steps, loss = zip(*train) | |
| window = 10 # log points are 10 steps apart -> 100-step mean | |
| mean = [sum(loss[max(0, i - window + 1):i + 1]) / len(loss[max(0, i - window + 1):i + 1]) for i in range(len(loss))] | |
| ax_t.plot(steps, loss, color=AXIS, linewidth=0.6, label="logged loss") | |
| ax_t.plot(steps, mean, color=INK, linewidth=1.2, label="100-step mean") | |
| settled = [l for s, l in train if s >= min(300, steps[-1] // 5)] or list(loss) | |
| ax_t.set_ylim(min(loss) - 0.15, max(settled) + 0.3) | |
| ax_t.annotate(f"{mean[-1]:.3f}", (steps[-1], mean[-1]), xytext=(4, 0), textcoords="offset points", | |
| color=INK_2, fontsize=8, va="center") | |
| if steps[0] <= 10: | |
| ax_t.text(0.01, 0.02, f"step {steps[0]} loss {loss[0]:.2f} (off scale)", transform=ax_t.transAxes, | |
| ha="left", va="bottom", color=MUTED, fontsize=7) | |
| ax_t.legend(loc="upper right", bbox_to_anchor=(1, 0.92), frameon=False, fontsize=8, labelcolor=INK_2) | |
| if evals: | |
| ends = [] | |
| for domain, color in DOMAIN_COLORS.items(): | |
| pts = [(s, r[domain]["loss"]) for s, r in evals if domain in r] | |
| if pts: | |
| xs, ys = zip(*pts) | |
| ax_v.plot(xs, ys, color=color, linewidth=1.2, marker="o", markersize=4.5, | |
| markeredgecolor=SURFACE, markeredgewidth=1, label=domain) | |
| ends.append([ys[-1], domain, xs[-1], ys[-1]]) | |
| lo, hi = ax_v.get_ylim() | |
| gap = (hi - lo) * 0.06 # dodge direct labels that would collide | |
| ends.sort() | |
| for i in range(1, len(ends)): | |
| ends[i][0] = max(ends[i][0], ends[i - 1][0] + gap) | |
| for label_y, domain, x, last in ends: | |
| ax_v.annotate(f"{domain} {last:.3f}", (x, label_y), xytext=(6, 0), textcoords="offset points", | |
| color=INK_2, fontsize=8, va="center") | |
| ax_v.legend(loc="upper center", frameon=False, fontsize=8, labelcolor=INK_2) | |
| else: | |
| ax_v.text(0.5, 0.5, "no evaluation yet", transform=ax_v.transAxes, ha="center", color=MUTED) | |
| for ax in (ax_t, ax_v): | |
| ax.set_xlim(0, max(x_max * 1.08, 10)) | |
| _phase_lines(ax, x_max) | |
| fig.suptitle(f"GPT-U-20M · step {x_max:,} / {total_steps:,}", x=0.01, ha="left", color=INK_2, fontsize=9) | |
| fig.tight_layout() | |
| out_path.parent.mkdir(parents=True, exist_ok=True) | |
| tmp = out_path.with_suffix(".tmp.png") | |
| fig.savefig(tmp, facecolor=SURFACE) | |
| plt.close(fig) | |
| tmp.replace(out_path) | |