File size: 2,880 Bytes
274951a
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
"""Prompt formats shared by RL training, DPO pair generation, and reward eval."""

from __future__ import annotations

CODING_PYTHON_IO_PROMPT = """You are solving a competitive programming problem.

Write a complete Python 3 program that reads from standard input and writes to standard output.
Return only the solution code. Do not include explanations.

Problem:
{problem}"""

# Nemotron's normalized ``prompt`` field already contains this instruction
# wrapper.  Keeping it inside our own stricter code-only wrapper creates
# conflicting instructions and makes Qwen3.5 emit long step-by-step comments
# inside an unfinished code fence even when chat-template thinking is disabled.
# Match the complete prefix exactly so ordinary problem statements are never
# rewritten merely because they mention similar words.
NEMOTRON_LEGACY_PROMPT_PREFIX = """You are a helpful and harmless assistant. You should think step-by-step before responding to the instruction below.

Please use python programming language only.

You must use ```python for just the final solution code block with the following format:
```python
# Your code here
```

"""


def normalize_coding_problem(problem: str) -> str:
    problem = problem.strip()
    if problem.startswith(NEMOTRON_LEGACY_PROMPT_PREFIX):
        problem = problem.removeprefix(NEMOTRON_LEGACY_PROMPT_PREFIX).strip()
    return problem


def format_coding_prompt(problem: str, *, format_name: str = "python_io") -> str:
    problem = normalize_coding_problem(problem)
    if format_name == "raw":
        return problem
    if format_name == "python_io":
        return CODING_PYTHON_IO_PROMPT.format(problem=problem)
    raise ValueError(f"unknown coding prompt format {format_name!r}")


def render_coding_prompt(tokenizer, problem: str, *, format_name: str = "python_io") -> str:
    """Render the exact prompt string the policy is trained/evaluated on.

    Chat-tuned models (all Qwen3.5 sizes) must see their chat template — feeding
    them raw text puts them in continuation mode and they ramble instead of
    answering. ``enable_thinking=False`` pins non-thinking mode across model
    sizes whose templates default differently; templates without that variable
    ignore it. Tokenizers without a chat template fall back to the raw format.

    Every consumer of coding prompts (GRPO rollouts, the per-checkpoint reward
    callback, DPO pair generation, prompt-length filtering) must go through this
    one function so the token stream is identical everywhere.
    """
    user_text = format_coding_prompt(problem, format_name=format_name)
    if getattr(tokenizer, "chat_template", None):
        return tokenizer.apply_chat_template(
            [{"role": "user", "content": user_text}],
            tokenize=False,
            add_generation_prompt=True,
            enable_thinking=False,
        )
    return user_text