File size: 5,935 Bytes
64b629a
 
 
 
 
 
 
39c0d86
64b629a
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
f4db68f
64b629a
f4db68f
 
 
 
 
 
64b629a
 
 
 
 
6deeb72
64b629a
6deeb72
64b629a
d8ef231
64b629a
 
 
 
 
 
39c0d86
64b629a
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
d8ef231
 
618f580
d8ef231
13c0c65
 
d8ef231
618f580
64b629a
6c4ff2f
d8ef231
6c4ff2f
64b629a
 
 
0b6774a
d8ef231
6c4ff2f
 
64b629a
 
 
 
 
 
 
 
 
 
f4db68f
64b629a
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
FROM ./Janus-35B-A3B.Q4_K_M.gguf

# Chat template β€” Qwen 3.6 ChatML in Ollama Go-template form, with the
# tool-calling blocks Ollama's capability detector looks for. Without a
# TEMPLATE that references .Tools and .ToolCalls, /api/chat and
# /v1/chat/completions reject any request carrying a `tools` array with
# `<model> does not support tools`. Same template as the 27B dense sibling
# (FoolDev/Thanatos-27B-HERETIC) β€” both share the Qwen 3.6 chat format.
TEMPLATE """{{- $lastUserIdx := -1 -}}
{{- range $idx, $msg := .Messages -}}
{{- if eq $msg.Role "user" }}{{ $lastUserIdx = $idx }}{{ end -}}
{{- end }}
{{- if or .System .Tools }}<|im_start|>system
{{ if .System }}{{ .System }}

{{ end }}
{{- if .Tools }}# Tools

You may call one or more functions to assist with the user query.

You are provided with function signatures within <tools></tools> XML tags:
<tools>
{{- range .Tools }}
{"type": "function", "function": {{ .Function }}}
{{- end }}
</tools>

For each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:
<tool_call>
{"name": <function-name>, "arguments": <args-json-object>}
</tool_call>
{{- end -}}<|im_end|>
{{ end }}
{{- range $i, $_ := .Messages }}
{{- $last := eq (len (slice $.Messages $i)) 1 -}}
{{- if eq .Role "user" }}<|im_start|>user
{{ .Content }}<|im_end|>
{{ else if eq .Role "assistant" }}<|im_start|>assistant
{{ if (and $.IsThinkSet (and .Thinking (or $last (gt $i $lastUserIdx)))) -}}
<think>{{ .Thinking }}</think>
{{ end -}}
{{ if .Content }}{{ .Content }}{{ end }}
{{- if .ToolCalls }}
{{- range .ToolCalls }}
<tool_call>
{"name": "{{ .Function.Name }}", "arguments": {{ .Function.Arguments }}}
</tool_call>
{{- end }}
{{- end }}{{ if not $last }}<|im_end|>
{{ end }}
{{- else if eq .Role "tool" }}<|im_start|>user
<tool_response>
{{ .Content }}
</tool_response><|im_end|>
{{ end }}
{{- if and (ne .Role "assistant") $last }}<|im_start|>assistant
{{ if and $.IsThinkSet (not $.Think) -}}
<think>

</think>

{{ else -}}
<think>
{{ end -}}
{{ end }}
{{- end }}"""

# Sampling tuned for reasoning + general use. See README "Recommended sampling"
# for creative/RP alternatives.
PARAMETER temperature 1.0
PARAMETER top_p 0.95
PARAMETER top_k 0
PARAMETER repeat_penalty 1.05
PARAMETER num_ctx 1010000

# Stop tokens. Without these, Ollama only honors <|im_end|> from the GGUF
# metadata; the model occasionally emits <|endoftext|> instead and Ollama
# keeps generating past it (synthesising a fake new user turn). Listing
# both β€” plus <|im_start|> as a belt-and-braces guard against the same
# loop β€” keeps responses cleanly terminated. Same fix the 27B sibling
# (FoolDev/Thanatos-27B-HERETIC) shipped in commit 6672746.
PARAMETER stop "<|im_end|>"
PARAMETER stop "<|endoftext|>"
PARAMETER stop "<|im_start|>"

SYSTEM """You are Janus, a precise and capable assistant for reasoning, writing, coding, and long-form dialogue.

Behavior rules:
- Answer the user's actual request directly.
- Be accurate, complete, and structured.
- Think before answering, but do not get stuck in repetitive loops or meta-commentary.
- If the request is ambiguous or incomplete, state what is missing and make the smallest reasonable assumption needed to continue.
- If the user wants creative writing, preserve tone, continuity, and character consistency.
- If the user wants analysis or technical help, prefer concrete steps, examples, and decisions over fluff.
- Finish with a usable answer, not just planning."""

# Hardware notes
# --------------
# This Q4_K_M is ~19 GB on disk. Real footprint at runtime:
#   weights mmap          ~19 GB
#   compute graph alloc   ~19 GB  (Ollama log: device.go:272 "total memory")
#   KV cache @ 262K ctx   ~16 GB  (q8_0, ~2 GB / 32K; so ~62 GB at the 1.01M default)
#   total minimum          ~100 GB at the 1010000 default / ~53 GB at 262144 native
#                                  (theoretical, extrapolated from ~2 GB/32K; the
#                                   default num_ctx is 1010000 β€” a ~1.01M ceiling ABOVE the
#                                   262144 native context. No YaRN rope-scaling is
#                                   baked in this GGUF, so output past ~262K degrades;
#                                   treat 1.01M as an advertised ceiling. Most
#                                   hosts must override num_ctx down, see below)
#
# Working configurations (rows assume num_ctx trimmed to a practical ~16-32K;
# at the 1.01M default the ~100 GB footprint fits only 128 GB+ hosts β€” override down
# everywhere else, see below):
#   βœ“ Single H100 80GB / A100 80GB              β€” full GPU offload
#   βœ“ RTX 5090 32GB / RTX 4090 24GB             β€” partial offload, ~15-25 tok/s
#   βœ“ Mac Studio M2/M3 Ultra 64GB+              β€” unified memory, ~20+ tok/s
#   βœ“ Linux box with 48GB+ RAM (CPU-only)       β€” ~3-6 tok/s
#   ⚠ ASUS ROG Flow Z13 (Ryzen AI Max+, 32GB)   β€” OOMs at the 1.01M default (and
#                                                 above ~32K); fits with num_ctx
#                                                 ≀ 4096 and num_batch ≀ 256 (verified)
#
# Measured data point (ASUS ROG Flow Z13 GZ302EA-RU004W, Ryzen AI Max+ 395 +
# Radeon 8060S iGPU, 32 GB unified, ROCm gfx1151, OLLAMA_FLASH_ATTENTION=1,
# OLLAMA_KV_CACHE_TYPE=q8_0, num_ctx 4096, num_batch 256):
#   Q4_K_M, 3-prompt mix β†’ 28.71 tok/s aggregate
#     (717 tokens / 25.0 s; 29.55 / 29.24 / 28.57 short/medium/long).
#   ~97% of layers offload to the iGPU via ROCm. Compute split per
#   `ollama ps` shows 3% CPU / 97% GPU at 4096 ctx.
#
# To run on a 32 GB unified-memory laptop, override these in your local
# Modelfile copy (or via `/set parameter` in the interactive `ollama run` REPL):
#   PARAMETER num_ctx 4096
#   PARAMETER num_batch 256
#
# If you have β‰₯48 GB RAM but want partial GPU offload, set:
#   PARAMETER num_gpu 24    # offload most layers (model has 40)