Text Generation
Transformers
Safetensors
English
k2_horizon
decision-model
jev
classification
calibration
pointer-head
conversational
custom_code
Instructions to use IFM/K2-Type-0.9B with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use IFM/K2-Type-0.9B with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-generation", model="IFM/K2-Type-0.9B", trust_remote_code=True) messages = [ {"role": "user", "content": "Who are you?"}, ] pipe(messages)# Load model directly from transformers import AutoModelForCausalLM model = AutoModelForCausalLM.from_pretrained("IFM/K2-Type-0.9B", trust_remote_code=True, device_map="auto") - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use IFM/K2-Type-0.9B with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "IFM/K2-Type-0.9B" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "IFM/K2-Type-0.9B", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker
docker model run hf.co/IFM/K2-Type-0.9B
- SGLang
How to use IFM/K2-Type-0.9B with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "IFM/K2-Type-0.9B" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "IFM/K2-Type-0.9B", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "IFM/K2-Type-0.9B" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "IFM/K2-Type-0.9B", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }' - Docker Model Runner
How to use IFM/K2-Type-0.9B with Docker Model Runner:
docker model run hf.co/IFM/K2-Type-0.9B
Download jev/encode.py from IFM/K2-Type-0.9B: direct link, hf CLI and curl.
- Browser
- Download file 6.78 kB
-
https://huggingface.co/IFM/K2-Type-0.9B/resolve/main/jev/encode.py
- Command line
-
hf download hf://IFM/K2-Type-0.9B/jev/encode.py
-
curl -L -o encode.py https://huggingface.co/IFM/K2-Type-0.9B/resolve/main/jev/encode.py
6.78 kB
| """Turn a jev record into one packed sequence: the state once, then every question after it. | |
| <state> state tokens | <q> instr <opt> option </opt> ... <decide> | <q> ... <decide> | ... | |
| Isolation: a question token may attend to the state and to earlier tokens of its own question, never to another | |
| question. Position ids restart at the end of the state for every question, so each question sees exactly the | |
| positions it would see if it were asked alone. | |
| Marker tokens are unused reserved ids of the K2 tokenizer, inserted by id (never through text tokenization). | |
| """ | |
| import json | |
| import torch | |
| MARKERS = {"state": "reserved_special_token_100", "q": "reserved_special_token_101", "opt": "reserved_special_token_102", | |
| "end_opt": "reserved_special_token_103", "decide": "reserved_special_token_104"} | |
| def render(v, indent=0): | |
| pad = " " * indent | |
| if isinstance(v, dict): | |
| return "\n".join(f"{pad}{k}:\n{render(x, indent + 1)}" if isinstance(x, (dict, list)) else f"{pad}{k}: {x}" | |
| for k, x in v.items()) | |
| if isinstance(v, list): | |
| return "\n".join(f"{pad}- {render(x, indent + 1).strip() if isinstance(x, (dict, list)) else x}" for x in v) | |
| return f"{pad}{v}" | |
| def options_and_target(q): | |
| """Option texts and the target distribution for one question.""" | |
| t = q["type"] | |
| if t == "choice": | |
| keys = list(q["criteria"]) | |
| opts = [f"{k}: {d}" if d else str(k) for k, d in q["criteria"].items()] | |
| if q.get("soft") is not None: | |
| target = [float(q["soft"][k]) for k in keys] | |
| else: | |
| target = [1.0 if k == q["label"] else 0.0 for k in keys] | |
| elif t == "score": | |
| opts = [str(x) for x in q["criteria"]] | |
| target = [float(x) for x in q["soft"]] if q.get("soft") is not None else \ | |
| [1.0 if i == q["label"] else 0.0 for i in range(len(opts))] | |
| else: | |
| crit = q.get("criteria") or {} # optional definitions of the two outcomes: {"false": ..., "true": ...} | |
| opts = [f"{k}: {render(crit[k]).strip()}" if crit.get(k) else k for k in ("false", "true")] | |
| p = float(q["soft"]) if q.get("soft") is not None else float(bool(q["label"])) | |
| target = [1.0 - p, p] | |
| s = sum(target) | |
| return opts, [x / s for x in target] | |
| def teacher_target(q): | |
| """The question's `teacher` field as a distribution in canonical option order (None if absent).""" | |
| t = q.get("teacher") | |
| if t is None: | |
| return None | |
| if q["type"] == "choice": | |
| return [float(t[k]) for k in q["criteria"]] | |
| if q["type"] == "score": | |
| return [float(x) for x in t] | |
| return [1.0 - float(t), float(t)] | |
| class Encoder: | |
| def __init__(self, tok, max_len=4096, max_state=3072, teacher_mix=None): | |
| """teacher_mix = alpha: target = alpha * gold + (1 - alpha) * teacher when a question has `teacher`.""" | |
| self.tok, self.max_len, self.max_state, self.teacher_mix = tok, max_len, max_state, teacher_mix | |
| vocab = tok.get_vocab() | |
| self.ids = {k: vocab[v] for k, v in MARKERS.items()} | |
| self.bos = tok.bos_token_id | |
| def text(self, s): | |
| return self.tok(s, add_special_tokens=False).input_ids | |
| def encode(self, rec, rng=None): | |
| """Returns a dict of python lists, or None when not even one question fits. | |
| With `rng`, choice and noul options are shown in a random order (targets permuted to match).""" | |
| state = self.text(render(rec["state"]))[: self.max_state] | |
| ids = [self.bos, self.ids["state"]] + state | |
| seg = [0] * len(ids) | |
| pos = list(range(len(ids))) | |
| S = len(ids) | |
| decide, opt_ends, targets, qtypes, qkeys, perms = [], [], [], [], [], [] | |
| for j, (k, q) in enumerate(rec["questions"].items()): | |
| opts, target = options_and_target(q) | |
| tt = teacher_target(q) if self.teacher_mix is not None else None | |
| if tt is not None: | |
| a = self.teacher_mix | |
| target = [a * g + (1 - a) * t for g, t in zip(target, tt)] | |
| order = list(range(len(opts))) | |
| if rng is not None and q["type"] != "score": | |
| rng.shuffle(order) | |
| opts, target = [opts[i] for i in order], [target[i] for i in order] | |
| instr = q["instructions"] if isinstance(q["instructions"], str) else render(q["instructions"]) # Kev uses dicts too | |
| qi = [self.ids["q"]] + self.text(instr) | |
| ends = [] | |
| for o in opts: | |
| qi += [self.ids["opt"]] + self.text(o)[:128] + [self.ids["end_opt"]] | |
| ends.append(len(qi) - 1) | |
| qi.append(self.ids["decide"]) | |
| if len(ids) + len(qi) > self.max_len: | |
| continue # skip questions that do not fit; the others still train | |
| base = len(ids) | |
| ids += qi | |
| seg += [j + 1] * len(qi) | |
| pos += list(range(S, S + len(qi))) | |
| decide.append(base + len(qi) - 1) | |
| opt_ends.append([base + e for e in ends]) | |
| targets.append(target) | |
| qtypes.append(q["type"]) | |
| qkeys.append(k) | |
| perms.append(order) # shown position -> canonical option index | |
| if not decide: | |
| return None | |
| return {"ids": ids, "seg": seg, "pos": pos, "decide": decide, "opt_ends": opt_ends, | |
| "targets": targets, "qtypes": qtypes, "qkeys": qkeys, "perms": perms} | |
| def collate(items, pad_id): | |
| """Right-pad a batch and build the [B, 1, L, L] boolean isolation mask (True = may attend).""" | |
| B, L = len(items), max(len(x["ids"]) for x in items) | |
| ids = torch.full((B, L), pad_id, dtype=torch.long) | |
| pos = torch.zeros((B, L), dtype=torch.long) | |
| seg = torch.full((B, L), -1, dtype=torch.long) | |
| for b, x in enumerate(items): | |
| n = len(x["ids"]) | |
| ids[b, :n] = torch.tensor(x["ids"]) | |
| pos[b, :n] = torch.tensor(x["pos"]) | |
| seg[b, :n] = torch.tensor(x["seg"]) | |
| causal = torch.ones(L, L, dtype=torch.bool).tril() | |
| sq, sk = seg[:, :, None], seg[:, None, :] | |
| mask = causal[None] & (sk >= 0) & (sq >= 0) & ((sk == 0) | (sk == sq)) | |
| mask |= torch.eye(L, dtype=torch.bool)[None] # padding rows attend to themselves only (avoids NaN softmax) | |
| # flat question index: (batch row, decide position, option end positions, target) | |
| qs = [(b, d, e, t, ty) for b, x in enumerate(items) | |
| for d, e, t, ty in zip(x["decide"], x["opt_ends"], x["targets"], x["qtypes"])] | |
| return {"input_ids": ids, "position_ids": pos, "mask": mask[:, None], "questions": qs} | |
| def load_jsonl(path, limit=0): | |
| out = [] | |
| with open(path) as fh: | |
| for i, line in enumerate(fh): | |
| if limit and i >= limit: | |
| break | |
| out.append(json.loads(line)) | |
| return out | |