Instructions to use HilaryTorn/rl-training-debug-artifacts with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- PEFT
How to use HilaryTorn/rl-training-debug-artifacts with PEFT:
Task type is invalid.
- Notebooks
- Google Colab
- Kaggle
Download code/prompts.py from HilaryTorn/rl-training-debug-artifacts: direct link, hf CLI and curl.
- Browser
- Download file 2.88 kB
-
https://huggingface.co/HilaryTorn/rl-training-debug-artifacts/resolve/main/code/prompts.py
- Command line
-
hf download hf://HilaryTorn/rl-training-debug-artifacts/code/prompts.py
-
curl -L -o prompts.py https://huggingface.co/HilaryTorn/rl-training-debug-artifacts/resolve/main/code/prompts.py
2.88 kB
| """Prompt formats shared by RL training, DPO pair generation, and reward eval.""" | |
| from __future__ import annotations | |
| CODING_PYTHON_IO_PROMPT = """You are solving a competitive programming problem. | |
| Write a complete Python 3 program that reads from standard input and writes to standard output. | |
| Return only the solution code. Do not include explanations. | |
| Problem: | |
| {problem}""" | |
| # Nemotron's normalized ``prompt`` field already contains this instruction | |
| # wrapper. Keeping it inside our own stricter code-only wrapper creates | |
| # conflicting instructions and makes Qwen3.5 emit long step-by-step comments | |
| # inside an unfinished code fence even when chat-template thinking is disabled. | |
| # Match the complete prefix exactly so ordinary problem statements are never | |
| # rewritten merely because they mention similar words. | |
| NEMOTRON_LEGACY_PROMPT_PREFIX = """You are a helpful and harmless assistant. You should think step-by-step before responding to the instruction below. | |
| Please use python programming language only. | |
| You must use ```python for just the final solution code block with the following format: | |
| ```python | |
| # Your code here | |
| ``` | |
| """ | |
| def normalize_coding_problem(problem: str) -> str: | |
| problem = problem.strip() | |
| if problem.startswith(NEMOTRON_LEGACY_PROMPT_PREFIX): | |
| problem = problem.removeprefix(NEMOTRON_LEGACY_PROMPT_PREFIX).strip() | |
| return problem | |
| def format_coding_prompt(problem: str, *, format_name: str = "python_io") -> str: | |
| problem = normalize_coding_problem(problem) | |
| if format_name == "raw": | |
| return problem | |
| if format_name == "python_io": | |
| return CODING_PYTHON_IO_PROMPT.format(problem=problem) | |
| raise ValueError(f"unknown coding prompt format {format_name!r}") | |
| def render_coding_prompt(tokenizer, problem: str, *, format_name: str = "python_io") -> str: | |
| """Render the exact prompt string the policy is trained/evaluated on. | |
| Chat-tuned models (all Qwen3.5 sizes) must see their chat template — feeding | |
| them raw text puts them in continuation mode and they ramble instead of | |
| answering. ``enable_thinking=False`` pins non-thinking mode across model | |
| sizes whose templates default differently; templates without that variable | |
| ignore it. Tokenizers without a chat template fall back to the raw format. | |
| Every consumer of coding prompts (GRPO rollouts, the per-checkpoint reward | |
| callback, DPO pair generation, prompt-length filtering) must go through this | |
| one function so the token stream is identical everywhere. | |
| """ | |
| user_text = format_coding_prompt(problem, format_name=format_name) | |
| if getattr(tokenizer, "chat_template", None): | |
| return tokenizer.apply_chat_template( | |
| [{"role": "user", "content": user_text}], | |
| tokenize=False, | |
| add_generation_prompt=True, | |
| enable_thinking=False, | |
| ) | |
| return user_text | |