#!/usr/bin/env python3
"""Static guard for the Go ``template`` - and so for the Modelfile TEMPLATE, which
check_bridge_sync.py requires to match it.
Why this exists. Ollama decides at load time whether to run this Go template or the
Jinja template embedded in the GGUF, by comparing the capabilities each advertises.
The Go template's "thinking" capability is inferred solely from a ``.Thinking`` field
inside ``range .Messages``, wrapped in ````/```` text. On 2026-09-10 a
three-line edit removed that reference. Every render test still passed, but Ollama
silently switched to the embedded template, which returned HTTP 500 on
``reasoning_effort: "max"`` and, on Janus, returned a tool call twice. A follow-up restricted
the condition to ``$last`` and silently dropped a tool-call chain's reasoning. This
check fails on both.
Assertions (string-level; there is no Go toolchain in this repo's checks):
1. the ``$lastUserIdx`` loop is present, verbatim (Qwen's stock condition, the
README's edit for turning replay off, needs it);
2. the assistant branch opens with the thinking block under
``$.IsThinkSet``, which replays every earlier turn's reasoning, verbatim,
wrapping ``\n{{ .Thinking }}\n`` and a blank line. The condition
is deliberately not gated on ``.Thinking``: a turn that did no reasoning still
renders the empty ``\n\n`` block. NOTE this is a deliberate
divergence on the Ollama path, not parity: Qwen 3.6's embedded template - the
one that governs llama.cpp for this repo, since it ships no chat_template.jinja -
gates both of its show_think arms on ``reasoning_content|length > 0`` and so
renders NO block at all for a history turn that did no reasoning (README
"Reasoning replay" states this correctly). The reason for the bare condition is
replay consistency across a tool-call chain, not matching upstream;
3. ``\n{{ .Thinking }}\n`` sits inside the
``range $i, $_ := .Messages`` block, before the ``end`` that closes it;
4. the tool round trip Ollama looks for is intact (``.Tools``, ``.ToolCalls``,
``eq .Role "tool"``, ````), so the Go template keeps its tools
capability and keeps winning the tie.
4a. consecutive ``tool`` messages render as ONE ``<|im_start|>user`` turn
carrying one ```` block each - the ``$prevIsTool`` /
``$nextIsTool`` machinery - and an assistant turn's text is separated from
its first ```` by a blank line, emitted outside the range and pinned by
``BLOCK``/``BLOCK_REPLAY_ALL`` (0.9.7; before that the separator was the range's own
leading newline). The grouping matches the sibling's chat_template.jinja
and Qwen's own format; before 2026-09-18 this template emitted a fresh user
turn per tool result and a single newline before the tool call, so the same
conversation rendered differently on the two paths. Checked against Ollama's
own renderer (``/api/chat`` with ``"_debug_render_only": true``, 0.33.3).
5. tool signatures inside ```` go through ``json``. Since Ollama 0.14.0
``.Function`` is a struct with no String() method, so a bare
``{{ .Function }}`` prints Go struct syntax, not JSON (verified by rendering
with Ollama's own template package at v0.33.3; still so in 0.34.0 and main).
6. the reasoning-effort arms, verbatim: the ``high``/``max`` think level sets
the xhigh instruction and ``low`` sets the low one - the mapping README
"Reasoning effort" documents as a table;
7. nothing else assigns ``$effort`` - no ``else`` arm, no ``medium`` arm - so
the medium level, which is also what Ollama sends for an unset value, adds
no line. That is this repo's documented default tier;
8. the system block still emits ``$effort``, so the arms above render rather
than assigning a variable nothing prints.
Checks 6-8 are structure, not a render - same reason as the rest of this file,
there is no Go template engine in these checks - and their labels say so. What
the arms produce at run time is covered live instead, by scripts/live_check.sh,
which needs the bundled blob. Added 2026-09-18: until then the effort mapping
was the one documented template property with no offline guard.
Usage: python3 scripts/check_go_template.py [PATH] (default: ./template)
"""
import os
import re
import sys
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
path = sys.argv[1] if len(sys.argv) > 1 else os.path.join(ROOT, "template")
try:
t = open(path, encoding="utf-8").read()
except OSError as exc:
print(f"[!] cannot read {path}: {exc}")
sys.exit(1)
LOOP = ('{{- $lastUserIdx := -1 -}}\n'
'{{- range $idx, $msg := .Messages -}}\n'
'{{- if eq $msg.Role "user" }}{{ $lastUserIdx = $idx }}{{ end -}}\n'
'{{- end }}\n')
RANGE = '{{- range $i, $_ := .Messages }}'
BLOCK = ('{{ else if eq .Role "assistant" }}{{ $calls := .ToolCalls }}'
'{{ $blank := or .Content (not $calls) }}<|im_start|>assistant\n'
'{{ if (and $.IsThinkSet (and .Thinking (or $last (gt $i $lastUserIdx)))) -}}\n'
'\n{{ .Thinking }}\n\n'
'{{ end -}}\n'
'{{ if and $.IsThinkSet $blank }}\n'
'{{ end -}}\n'
'{{ if .Content }}{{ .Content }}{{ if $calls }}\n'
'{{ end }}{{ end }}')
BLOCK_REPLAY_ALL = BLOCK.replace('(and $.IsThinkSet (and .Thinking (or $last (gt $i $lastUserIdx))))',
'$.IsThinkSet')
THINK = '\n{{ .Thinking }}\n'
TOOL_GROUP = ('{{- else if eq .Role "tool" }}\n'
'{{- if $prevIsTool }}\n'
'{{ else }}<|im_start|>user\n'
'{{ end }}\n'
'{{ .Content }}\n'
'\n'
'{{- if not $nextIsTool }}<|im_end|>\n'
'{{ end }}\n')
# Ollama derives the tool-call PARSER from this template: it finds the range over
# .ToolCalls and takes the LITERAL TEXT at the start of the body as the prefix to
# scan for in the model's output. If the body starts with an action instead of
# text, the derived prefix is wrong and Ollama stops recognising tool calls -
# they arrive as plain text in `content` and `tool_calls` is empty. That shipped
# in Janus 0.9.5 / Thanatos 0.12.5: a `{{- if $j }}` separator at the head of the
# body silently broke first-turn tool calling on the real model (0/5 samples;
# v0.9.4 scored 2/2), while every render test still passed because the RENDERED
# bytes were identical. Keep the body literal-first.
TOOL_CALL_BODY = ('{{- range .ToolCalls }}\n'
'\n'
'{"name": "{{ .Function.Name }}", "arguments": {{ .Function.Arguments }}}\n'
'\n')
NEXT = '{{- $nextIsTool := and (gt (len $rest) 1) (eq (index $rest 1).Role "tool") -}}'
ACTION = re.compile(r'\{\{-?\s*(range|if|with|define|block|end)\b')
# Reasoning effort. Ollama folds /v1's eight reasoning_effort values into four
# think levels before any template runs - high/xhigh/max/ultra -> high or max,
# low/minimal -> low, medium and unset -> medium, none -> thinking off (Ollama's
# source, unchanged v0.33.3 through v0.34.2, read 2026-09-17) - so the template
# branches on levels, not on the request value, and only three levels can reach
# these arms.
EFFORT_INIT = '{{- $effort := "" }}\n'
EFFORT_GUARD = '{{- if $.ThinkLevel }}\n'
XHIGH_ARM = (
'{{- if or (eq $.ThinkLevel "high") (eq $.ThinkLevel "max") }}'
'{{ $effort = "Reasoning effort is set to xhigh. Please think carefully'
' through the task, validate key assumptions, consider plausible'
' alternatives, and prioritize correctness, consistency, and clarity in'
' the final answer." }}\n')
LOW_ARM = (
'{{- else if eq $.ThinkLevel "low" }}'
'{{ $effort = "Reasoning effort is set to low. Keep your thinking brief'
' and focused, moving directly to the conclusion without unnecessary'
' elaboration." }}\n')
# Pins the whole system block: the $effort emission, the .System emission that
# carries this repo's shipped persona (deleting it was previously undetectable -
# live_check compares token deltas that shift equally, smoke_test only greps for
# leaked control tokens, bridge sync compares the two copies to each other), and
# the two conditional separators, which exist so the block does not end on a
# blank line when nothing follows. Qwen 3.6 renders '<|im_start|>system\n' +
# content + '<|im_end|>\n' with nothing between.
EMIT = ('{{- if or .System .Tools $effort }}<|im_start|>system\n'
'{{ if $effort }}{{ $effort }}{{ if or .System .Tools }}\n'
'\n'
'{{ end }}{{ end }}{{ if .System }}{{ .System }}{{ if .Tools }}\n'
'\n'
'{{ end }}{{ end }}\n')
# This repo's documented default tier (README "Reasoning effort", table read
# 2026-09-18): medium, which adds no line. Janus ships no chat_template.jinja
# and Qwen 3.6's own embedded template has no reasoning_effort variable at all
# (read from the template, 2026-09-18), so medium is the default on every path
# this repo ships.
DEFAULT_TIER = "medium, on every path this repo ships"
def block_end(text, start):
"""Offset of the {{ end }} closing the block opened at `start`, or -1."""
depth = 0
for m in ACTION.finditer(text, start):
depth += -1 if m.group(1) == "end" else 1
if depth == 0:
return m.start()
return -1
# --- Ollama's own AST-derived behaviour, emulated -------------------------
# Two things Ollama reads out of this template's PARSE TREE, not its render, so
# no render test can see either one break:
# * the tool-call tag, from tools/template.go parseTag(): the first IfNode whose
# pipe holds a FieldNode named ToolCalls, then the first non-whitespace
# TextNode in its body, cut at '{' and trimmed. An ActionNode reached first
# yields no text and the tag falls back to '{', which stops tool calling
# (Janus 0.9.5 / Thanatos 0.12.5: 0/5 first-turn calls parsed). Note the
# search is for the FIRST such IfNode anywhere in the tree - so a condition
# that names .ToolCalls EARLIER than the tool-call block captures the search
# and breaks it just as thoroughly. That is why $calls aliases .ToolCalls
# before the branch: a VariableNode is not a FieldNode, so it is not matched.
# * the thinking tags, from thinking/template.go InferTags(): find a .Thinking
# FieldNode inside a range over .Messages, go up to the nearest ListNode and
# take its FIRST and LAST nodes - both must be TextNodes, which become the
# opening and closing tags. Put anything else last in that list (a trailing
# {{ if }}, say) and the closing tag comes back empty, the thinking
# capability disappears, and Ollama answers "does not support thinking".
# Both were read from ollama v0.33.3 and reproduced here; this port returns the
# known-correct verdicts for v0.9.4 (''), v0.9.5 ('{') and v0.9.6.
TEXT, ACTION_TOK = 'text', 'action'
def _lex(src):
"""Tokenize, applying Go's {{- / -}} whitespace trimming."""
toks, i, pending = [], 0, ''
while True:
j = src.find('{{', i)
if j < 0:
pending += src[i:]
break
pending += src[i:j]
k, in_str, esc = j + 2, None, False
while k < len(src):
c = src[k]
if in_str:
if esc:
esc = False
elif c == '\\':
esc = True
elif c == in_str:
in_str = None
elif c in '"`':
in_str = c
elif src.startswith('}}', k):
break
k += 1
inner = src[j + 2:k]
left, right = inner.startswith('-'), inner.endswith('-')
inner = inner[1:] if left else inner
inner = inner[:-1] if right else inner
if left:
pending = pending.rstrip()
if pending:
toks.append((TEXT, pending))
pending = ''
toks.append((ACTION_TOK, inner.strip()))
i = k + 2
if right:
m = re.match(r'\s*', src[i:])
i += m.end()
if pending:
toks.append((TEXT, pending))
return toks
class _Text:
def __init__(self, t):
self.text = t
class _Action:
def __init__(self, p):
self.pipe = p
class _Branch:
def __init__(self, kind, pipe):
self.kind, self.pipe, self.list, self.else_list = kind, pipe, [], None
def _parse(toks):
root, stack = [], []
cur = root
for kind, val in toks:
if kind == TEXT:
cur.append(_Text(val))
continue
parts = val.split(None, 1)
head = parts[0] if parts else ''
if head in ('if', 'range', 'with'):
b = _Branch(head, parts[1] if len(parts) > 1 else '')
cur.append(b)
stack.append((cur, b))
cur = b.list
elif head == 'else':
_, b = stack[-1]
rest = val[4:].strip()
if rest.startswith('if'):
nb = _Branch('if', rest[2:].strip())
b.else_list = [nb]
stack.append((cur, nb))
cur = nb.list
else:
b.else_list = []
cur = b.else_list
elif head == 'end':
prev, b = stack.pop()
while stack and stack[-1][1].else_list and stack[-1][1].else_list[0] is b:
prev, b = stack.pop()
cur = prev
else:
cur.append(_Action(val))
return root
_FIELD = r'(? block Qwen renders for a turn that did no reasoning, so a "
"tool-call chain renders differently here than through the sibling's chat_template.jinja "
"(this repo ships none); a missing block "
"makes Ollama switch to the GGUF's embedded template")
r, k = t.find(RANGE), t.find(THINK)
e = block_end(t, r) if r != -1 else -1
check("thinking block is inside range $i, $_ := .Messages", r != -1 and k != -1 and r < k < e,
"Ollama's thinking.InferTags only counts a .Thinking field inside a range over .Messages")
check(".Thinking wrapped in / text", THINK in t,
"InferTags needs text on both sides of the .Thinking field to infer the tags")
# Qwen renders a turn's reasoning as \n...\n followed by a blank
# line, groups consecutive tool results into ONE user turn, and separates an
# assistant turn's text from its first tool call by a blank line. Before 2026-09-18
# this template did none of the three: it emitted ... on one line,
# opened a fresh <|im_start|>user per tool result, and used a single newline before
# . Verified against the sibling's chat_template.jinja (this repo ships
# none) with Ollama's own renderer via
# /api/chat {"_debug_render_only": true} on 0.33.3.
check("consecutive tool results share one user turn",
TOOL_GROUP in t and NEXT in t and "{{- $prevIsTool = eq .Role \"tool\" }}" in t,
"a fresh <|im_start|>user per tool result is off-distribution for parallel tool calls; "
"Qwen groups them into one turn with one block each")
check("tool-call range body starts with literal text, not an action",
TOOL_CALL_BODY in t,
"Ollama derives its tool-call parser from the literal text at the head of the "
"`range .ToolCalls` body; an action there (a separator `{{- if $j }}`, say) makes it "
"stop parsing tool calls entirely - they come back as plain text in `content`. "
"Renders look identical, so only a live tool call catches it")
check("Ollama's parseTag derives the tool-call tag from this template",
derived_tool_tag(t) == "",
"emulating ollama v0.33.3 tools/template.go parseTag on this file: the tag came back "
f"{derived_tool_tag(t)!r}, not ''. '{{' means Ollama stopped recognising "
" blocks and every tool call arrives as plain text. Causes: an action at the "
"head of the `range .ToolCalls` body (Janus 0.9.5), or an EARLIER `{{ if }}` whose pipe "
"names .ToolCalls - parseTag takes the FIRST such if in the tree, so alias it "
"(`{{ $calls := .ToolCalls }}`) instead of naming the field in a condition")
check("Ollama's InferTags reads / as this template's thinking tags",
inferred_think_tags(t) == ("", ""),
"emulating ollama v0.33.3 thinking/template.go InferTags on this file: the tags came back "
f"{inferred_think_tags(t)!r}. InferTags takes the FIRST and LAST nodes of the list holding "
"{{ .Thinking }} and both must be text, so a trailing {{ if }} inside that block empties "
"the closing tag, drops the thinking capability, and Ollama then answers 'does not support "
"thinking' (a 400) on any request that sets think")
_ti = t.find("{{- if .Tools }}")
tools_block = t[_ti:block_end(t, _ti)] if _ti != -1 else ""
_tag = derived_tool_tag(t)
check("the tag advertised to the model is the tag Ollama parses",
_tag != "{" and f"<{_tag.strip('<>')}>" in tools_block and f"{_tag.strip('<>')}>" in tools_block,
f"the system block teaches the model one tool-call tag while Ollama's parser scans for "
f"another ({_tag!r}). Nothing else compares the two: the preamble is prose to every "
f"other check here, so the model can be told to emit while the parser waits "
f"for and every tool call comes back as plain text - 0.9.5's outage, reached by "
f"a different route")
for needle in (".Tools", ".ToolCalls", 'eq .Role "tool"', ""):
check(f"tool round trip keeps {needle}", needle in t,
"Ollama needs these to credit the Go template with tools and a tool round trip")
check("tool signatures rendered as JSON ({{ json .Function }})",
'{"type": "function", "function": {{ json .Function }}}' in t,
"since Ollama 0.14.0 a bare {{ .Function }} prints Go struct syntax inside , "
"not JSON, so the model sees garbled tool definitions")
# The effort mapping is user-facing behaviour, documented as a table in README
# "Reasoning effort". The four checks below assert its structure, not a render.
# What the arms produce at run time was established live instead
# (scripts/live_check.sh on Ollama 0.33.3, CPU only, 2026-09-18): high and max
# each added 38 prompt tokens, low added 26, medium and unset added none.
g = t.find(EFFORT_GUARD)
effort = t[g:block_end(t, g)] if g != -1 else ""
check("reasoning effort (structure): high/max think level sets the xhigh instruction",
XHIGH_ARM in effort,
"Ollama folds high/xhigh/max/ultra onto this level; drop the arm or edit its "
"string and the xhigh line the README's table promises silently stops reaching "
"the prompt")
check("reasoning effort (structure): low think level sets the low instruction",
LOW_ARM in effort,
"Ollama folds low/minimal onto this level; without the arm they render the same "
"prompt as medium")
check(f"reasoning effort (structure): medium and unset add no instruction "
f"(default tier: {DEFAULT_TIER})",
0 <= t.find(EFFORT_INIT) < g
and t.count("$effort = ") == 2
and "{{ else }}" not in effort and "{{- else }}" not in effort
and '"medium"' not in effort,
"$effort starts empty and only the two arms above assign it ANYWHERE in the file - "
"counting inside the $.ThinkLevel body only, as this check did before 0.9.9, missed an "
"assignment placed after that guard, e.g. {{- if not $.ThinkLevel }}{{ $effort = ... }}, "
"which moves the documented default tier while every check still passes - so the medium level "
"- which is also what Ollama sends for an unset value - adds no line; an else arm "
"or a medium arm would move this repo's documented default tier")
check("reasoning effort (structure): $effort reaches the system block",
EMIT in t,
"the system block has to be emitted for $effort as well as .System/.Tools, and to "
"print it first; lose either half and the arms above set a variable nothing renders")
print()
if failures:
print(f"[!] {len(failures)} Go template check(s) failed ({path})")
for name, why in failures:
print(f" - {name}: {why}")
sys.exit(1)
print("[+] Go template keeps Ollama's thinking detection, the replay condition "
"and the reasoning-effort mapping")