Created
July 20, 2026 13:23
-
-
Save joematthews/9be31aee86fb1e1c195ba14289b70c41 to your computer and use it in GitHub Desktop.
Gemma 4 12B pi chat template (KV-cache-stable retain build for the pi coding agent + llama.cpp)
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| {# | |
| Template: Gemma 4 12B -- KV-cache-stable "retain" build (WORKING / active) | |
| Base: unsloth/gemma-4-12b-it tokenizer_config.json chat_template, fetched 2026-07-20. | |
| The 12B sibling of gemma-4-e4b-pi.jinja -- same pi communication-flow deltas | |
| ported onto the 12B base (NOT copied from the E4B pi file). | |
| Changes vs the 12B base (four, matching the E4B pi build): | |
| 1. preserve_thinking now DEFAULTS TO true. llama.cpp does not pass this kwarg, and retaining | |
| reasoning across turns keeps the rendered prompt prefix byte-stable -- stops the KV-cache | |
| busting / full reprocessing described in ggml-org/llama.cpp#21912. | |
| Set preserve_thinking=false to restore upstream trim-on-new-turn behavior. | |
| 2. Thinking gate broadened from "(preserve_thinking and message.tool_calls)" to just | |
| "preserve_thinking", so reasoning on prior FINAL answers is retained too, not only | |
| tool-call turns -- maximal prefix stability. | |
| 3. Reasoning close-marker: '{reasoning}<channel|>' with NO newline before <channel|> (base | |
| emitted '\n<channel|>'). The model generates the close with no preceding newline, so the | |
| extra \n broke KV-cache prefix reuse at the reasoning->answer seam. | |
| 4. format_tool_response_block: detect bash's "Command exited with code" failure signature in | |
| a non-mapping tool response and hoist it to a front-loaded structural flag | |
| ('{error:true,value:...}') the model can't miss. Interim heuristic; swap for a real marker | |
| once isError survives the wire. | |
| Kept from 12B base (do NOT regress): the add_generation_prompt empty-thought channel emitted | |
| when enable_thinking is false -- absent from the E4B base, which is why the E4B pi template | |
| cannot be reused verbatim here. | |
| Also kept from base (do NOT regress): string/JSON tool-arg tolerance, null handling, filter_keys | |
| tool-schema serialization, multimodal placeholders, O(1) turn-continuation tracking, | |
| strip_thinking on visible content. | |
| #} | |
| {%- macro format_parameters(properties, required, filter_keys=false) -%} | |
| {%- set standard_keys = ['description', 'type', 'properties', 'required', 'nullable'] -%} | |
| {%- set ns = namespace(found_first=false) -%} | |
| {%- for key, value in properties | dictsort -%} | |
| {%- set add_comma = false -%} | |
| {%- if not filter_keys or key not in standard_keys -%} | |
| {%- if ns.found_first %},{% endif -%} | |
| {%- set ns.found_first = true -%} | |
| {{ key }}:{ | |
| {%- if value['description'] -%} | |
| description:<|"|>{{ value['description'] }}<|"|> | |
| {%- set add_comma = true -%} | |
| {%- endif -%} | |
| {%- if value['type'] | upper == 'STRING' -%} | |
| {%- if value['enum'] -%} | |
| {%- if add_comma %},{%- else -%} {%- set add_comma = true -%} {% endif -%} | |
| enum:{{ format_argument(value['enum']) }} | |
| {%- endif -%} | |
| {%- elif value['type'] | upper == 'ARRAY' -%} | |
| {%- if value['items'] is mapping and value['items'] -%} | |
| {%- if add_comma %},{%- else -%} {%- set add_comma = true -%} {% endif -%} | |
| items:{ | |
| {%- set ns_items = namespace(found_first=false) -%} | |
| {%- for item_key, item_value in value['items'] | dictsort -%} | |
| {%- if item_value is not none -%} | |
| {%- if ns_items.found_first %},{% endif -%} | |
| {%- set ns_items.found_first = true -%} | |
| {%- if item_key == 'properties' -%} | |
| properties:{ | |
| {%- if item_value is mapping -%} | |
| {{- format_parameters(item_value, value['items']['required'] | default([])) -}} | |
| {%- endif -%} | |
| } | |
| {%- elif item_key == 'required' -%} | |
| required:[ | |
| {%- for req_item in item_value -%} | |
| <|"|>{{- req_item -}}<|"|> | |
| {%- if not loop.last %},{% endif -%} | |
| {%- endfor -%} | |
| ] | |
| {%- elif item_key == 'type' -%} | |
| {%- if item_value is string -%} | |
| type:{{ format_argument(item_value | upper) }} | |
| {%- else -%} | |
| type:{{ format_argument(item_value | map('upper') | list) }} | |
| {%- endif -%} | |
| {%- else -%} | |
| {{ item_key }}:{{ format_argument(item_value) }} | |
| {%- endif -%} | |
| {%- endif -%} | |
| {%- endfor -%} | |
| } | |
| {%- endif -%} | |
| {%- endif -%} | |
| {%- if value['nullable'] %} | |
| {%- if add_comma %},{%- else -%} {%- set add_comma = true -%} {% endif -%} | |
| nullable:true | |
| {%- endif -%} | |
| {%- if value['type'] | upper == 'OBJECT' -%} | |
| {%- if value['properties'] is defined and value['properties'] is mapping -%} | |
| {%- if add_comma %},{%- else -%} {%- set add_comma = true -%} {% endif -%} | |
| properties:{ | |
| {{- format_parameters(value['properties'], value['required'] | default([])) -}} | |
| } | |
| {%- elif value is mapping -%} | |
| {%- if add_comma %},{%- else -%} {%- set add_comma = true -%} {% endif -%} | |
| properties:{ | |
| {{- format_parameters(value, value['required'] | default([]), filter_keys=true) -}} | |
| } | |
| {%- endif -%} | |
| {%- if value['required'] -%} | |
| {%- if add_comma %},{%- else -%} {%- set add_comma = true -%} {% endif -%} | |
| required:[ | |
| {%- for item in value['required'] | default([]) -%} | |
| <|"|>{{- item -}}<|"|> | |
| {%- if not loop.last %},{% endif -%} | |
| {%- endfor -%} | |
| ] | |
| {%- endif -%} | |
| {%- endif -%} | |
| {%- if add_comma %},{%- else -%} {%- set add_comma = true -%} {% endif -%} | |
| type:<|"|>{{ value['type'] | upper }}<|"|>} | |
| {%- endif -%} | |
| {%- endfor -%} | |
| {%- endmacro -%} | |
| {%- macro format_function_declaration(tool_data) -%} | |
| declaration:{{- tool_data['function']['name'] -}}{description:<|"|>{{- tool_data['function']['description'] -}}<|"|> | |
| {%- set params = tool_data['function']['parameters'] -%} | |
| {%- if params -%} | |
| ,parameters:{ | |
| {%- if params['properties'] -%} | |
| properties:{ {{- format_parameters(params['properties'], params['required']) -}} }, | |
| {%- endif -%} | |
| {%- if params['required'] -%} | |
| required:[ | |
| {%- for item in params['required'] -%} | |
| <|"|>{{- item -}}<|"|> | |
| {{- ',' if not loop.last -}} | |
| {%- endfor -%} | |
| ], | |
| {%- endif -%} | |
| {%- if params['type'] -%} | |
| type:<|"|>{{- params['type'] | upper -}}<|"|>} | |
| {%- endif -%} | |
| {%- endif -%} | |
| {%- if 'response' in tool_data['function'] -%} | |
| {%- set response_declaration = tool_data['function']['response'] -%} | |
| ,response:{ | |
| {%- if response_declaration['description'] -%} | |
| description:<|"|>{{- response_declaration['description'] -}}<|"|>, | |
| {%- endif -%} | |
| {%- if response_declaration['type'] | upper == 'OBJECT' -%} | |
| type:<|"|>{{- response_declaration['type'] | upper -}}<|"|>} | |
| {%- endif -%} | |
| {%- endif -%} | |
| } | |
| {%- endmacro -%} | |
| {%- macro format_argument(argument, escape_keys=True) -%} | |
| {%- if argument is none -%} | |
| {{- 'null' -}} | |
| {%- elif argument is string -%} | |
| {{- '<|"|>' + argument + '<|"|>' -}} | |
| {%- elif argument is boolean -%} | |
| {{- 'true' if argument else 'false' -}} | |
| {%- elif argument is mapping -%} | |
| {{- '{' -}} | |
| {%- set ns = namespace(found_first=false) -%} | |
| {%- for key, value in argument | dictsort -%} | |
| {%- if ns.found_first %},{% endif -%} | |
| {%- set ns.found_first = true -%} | |
| {%- if escape_keys -%} | |
| {{- '<|"|>' + key + '<|"|>' -}} | |
| {%- else -%} | |
| {{- key -}} | |
| {%- endif -%} | |
| :{{- format_argument(value, escape_keys=escape_keys) -}} | |
| {%- endfor -%} | |
| {{- '}' -}} | |
| {%- elif argument is sequence -%} | |
| {{- '[' -}} | |
| {%- for item in argument -%} | |
| {{- format_argument(item, escape_keys=escape_keys) -}} | |
| {%- if not loop.last %},{% endif -%} | |
| {%- endfor -%} | |
| {{- ']' -}} | |
| {%- else -%} | |
| {{- argument -}} | |
| {%- endif -%} | |
| {%- endmacro -%} | |
| {%- macro strip_thinking(text) -%} | |
| {%- set ns = namespace(result='') -%} | |
| {%- for part in text.split('<channel|>') -%} | |
| {%- if '<|channel>' in part -%} | |
| {%- set ns.result = ns.result + part.split('<|channel>')[0] -%} | |
| {%- else -%} | |
| {%- set ns.result = ns.result + part -%} | |
| {%- endif -%} | |
| {%- endfor -%} | |
| {{- ns.result | trim -}} | |
| {%- endmacro -%} | |
| {%- macro format_tool_response_block(tool_name, response) -%} | |
| {{- '<|tool_response>' -}} | |
| {%- if response is mapping -%} | |
| {{- 'response:' + tool_name + '{' -}} | |
| {%- for key, value in response | dictsort -%} | |
| {{- key -}}:{{- format_argument(value, escape_keys=False) -}} | |
| {%- if not loop.last %},{% endif -%} | |
| {%- endfor -%} | |
| {{- '}' -}} | |
| {%- else -%} | |
| {#- Interim: llama.cpp drops pi's isError at the wire, so detect bash's failure | |
| signature in-text and hoist it to a front-loaded structural flag the model | |
| can't miss. Swap this heuristic for a real marker once the tool-hook | |
| extension injects one (or once isError is forwarded on the wire). -#} | |
| {%- set rtext = response | string -%} | |
| {%- if 'Command exited with code' in rtext -%} | |
| {{- 'response:' + tool_name + '{error:true,value:' + format_argument(response, escape_keys=False) + '}' -}} | |
| {%- else -%} | |
| {{- 'response:' + tool_name + '{value:' + format_argument(response, escape_keys=False) + '}' -}} | |
| {%- endif -%} | |
| {%- endif -%} | |
| {{- '<tool_response|>' -}} | |
| {%- endmacro -%} | |
| {#- ===== SETUP ===== -#} | |
| {%- set ns = namespace(prev_message_type=None, prev_non_tool_role=None) -%} | |
| {%- set loop_messages = messages -%} | |
| {%- set enable_thinking = enable_thinking | default(false) -%} | |
| {%- set preserve_thinking = preserve_thinking | default(true) -%} {#- WORKING BUILD: default true (base was false) so retention holds even when llama.cpp omits the kwarg -#} | |
| {{- bos_token -}} | |
| {#- Handle System/Tool Definitions Block -#} | |
| {%- if enable_thinking or tools or (messages and messages[0]['role'] in ['system', 'developer']) -%} | |
| {{- '<|turn>system\n' -}} | |
| {#- Inject Thinking token at the very top of the FIRST system turn -#} | |
| {%- if enable_thinking -%} | |
| {{- '<|think|>\n' -}} | |
| {%- set ns.prev_message_type = 'think' -%} | |
| {%- endif -%} | |
| {%- if messages and messages[0]['role'] in ['system', 'developer'] -%} | |
| {%- if messages[0]['content'] is string -%} | |
| {{- messages[0]['content'] | trim -}} | |
| {%- elif messages[0]['content'] is sequence -%} | |
| {%- for item in messages[0]['content'] -%} | |
| {{- item['text'] | trim + ' '-}} | |
| {%- endfor -%} | |
| {%- endif -%} | |
| {%- set loop_messages = messages[1:] -%} | |
| {%- endif -%} | |
| {%- if tools -%} | |
| {%- for tool in tools %} | |
| {{- '<|tool>' -}} | |
| {{- format_function_declaration(tool) | trim -}} | |
| {{- '<tool|>' -}} | |
| {%- endfor %} | |
| {%- set ns.prev_message_type = 'tool' -%} | |
| {%- endif -%} | |
| {{- '<turn|>\n' -}} | |
| {%- endif %} | |
| {#- Pre-scan: find last user message index for reasoning guard -#} | |
| {%- set ns_turn = namespace(last_user_idx=-1) -%} | |
| {%- for i in range(loop_messages | length) -%} | |
| {%- if loop_messages[i]['role'] == 'user' -%} | |
| {%- set ns_turn.last_user_idx = i -%} | |
| {%- endif -%} | |
| {%- endfor -%} | |
| {#- Loop through messages -#} | |
| {%- for message in loop_messages -%} | |
| {%- if message['role'] != 'tool' -%} | |
| {%- set ns.prev_message_type = None -%} | |
| {%- set role = 'model' if message['role'] == 'assistant' else message['role'] -%} | |
| {#- Detect continuation using tracked state -- O(1) instead of O(n) backward scan -#} | |
| {%- set continue_same_model_turn = (role == 'model' and ns.prev_non_tool_role == 'assistant') -%} | |
| {%- if not continue_same_model_turn -%} | |
| {{- '<|turn>' + role + '\n' }} | |
| {%- endif -%} | |
| {#- Render reasoning/reasoning_content as thinking channel -#} | |
| {%- set thinking_text = message.get('reasoning') or message.get('reasoning_content') -%} | |
| {#- WORKING BUILD: retain ALL prior-turn reasoning (base gated this on tool_calls) so the KV-cache prefix stays byte-stable across turns -#} | |
| {%- set thinking_gate = (loop.index0 > ns_turn.last_user_idx) or preserve_thinking -%} | |
| {%- if thinking_text and thinking_gate -%} | |
| {{- '<|channel>thought\n' + thinking_text + '<channel|>' -}} | |
| {%- endif -%} | |
| {%- if message.get('tool_calls') -%} | |
| {%- for tool_call in message.get('tool_calls') -%} | |
| {%- set function = tool_call['function'] -%} | |
| {{- '<|tool_call>call:' + function['name'] + '{' -}} | |
| {%- if function['arguments'] is mapping -%} | |
| {%- set ns_args = namespace(found_first=false) -%} | |
| {%- for key, value in function['arguments'] | dictsort -%} | |
| {%- if ns_args.found_first %},{% endif -%} | |
| {%- set ns_args.found_first = true -%} | |
| {{- key -}}:{{- format_argument(value, escape_keys=False) -}} | |
| {%- endfor -%} | |
| {%- elif function['arguments'] is none -%} | |
| {%- elif function['arguments'] is string -%} | |
| {#- Pre-serialized args (e.g. an OpenAI JSON string). We cannot JSON-parse | |
| portably in-template, so render non-fatally instead of erroring. Strip an | |
| outer {...} so it composes with the DSL braces rather than double-wrapping. | |
| Prefer passing arguments as a mapping for exact Gemma DSL. -#} | |
| {%- set argstr = function['arguments'] | trim -%} | |
| {%- if argstr[:1] == '{' and argstr[-1:] == '}' -%} | |
| {{- argstr[1:-1] -}} | |
| {%- else -%} | |
| {{- function['arguments'] -}} | |
| {%- endif -%} | |
| {%- endif -%} | |
| {{- '}<tool_call|>' -}} | |
| {%- endfor -%} | |
| {%- set ns.prev_message_type = 'tool_call' -%} | |
| {%- endif -%} | |
| {%- set ns_tr_out = namespace(flag=false) -%} | |
| {%- if message.get('tool_responses') -%} | |
| {#- Legacy: tool_responses embedded on the assistant message (Google/Gemma native) -#} | |
| {%- for tool_response in message.get('tool_responses') -%} | |
| {{- format_tool_response_block(tool_response['name'] | default('unknown', true), tool_response['response']) -}} | |
| {%- set ns_tr_out.flag = true -%} | |
| {%- set ns.prev_message_type = 'tool_response' -%} | |
| {%- endfor -%} | |
| {%- elif message.get('tool_calls') -%} | |
| {#- OpenAI Chat Completions: forward-scan consecutive role:tool messages -#} | |
| {%- set ns_tool_scan = namespace(stopped=false) -%} | |
| {%- for k in range(loop.index0 + 1, loop_messages | length) -%} | |
| {%- if ns_tool_scan.stopped -%} | |
| {%- elif loop_messages[k]['role'] != 'tool' -%} | |
| {%- set ns_tool_scan.stopped = true -%} | |
| {%- else -%} | |
| {%- set follow = loop_messages[k] -%} | |
| {#- Resolve tool_call_id to function name -#} | |
| {%- set ns_tname = namespace(name=follow.get('name') or 'unknown') -%} | |
| {%- for tc in message.get('tool_calls') -%} | |
| {%- if tc.get('id') == follow.get('tool_call_id') -%} | |
| {%- set ns_tname.name = tc['function']['name'] -%} | |
| {%- endif -%} | |
| {%- endfor -%} | |
| {#- Handle content as string or content-parts array -#} | |
| {%- set tool_body = follow.get('content') -%} | |
| {%- if tool_body is string -%} | |
| {{- format_tool_response_block(ns_tname.name, tool_body) -}} | |
| {%- elif tool_body is sequence and tool_body is not string -%} | |
| {%- set ns_txt = namespace(s='') -%} | |
| {%- for part in tool_body -%} | |
| {%- if part.get('type') == 'text' -%} | |
| {%- set ns_txt.s = ns_txt.s + (part.get('text') | default('')) -%} | |
| {%- endif -%} | |
| {%- endfor -%} | |
| {{- format_tool_response_block(ns_tname.name, ns_txt.s) -}} | |
| {%- for part in tool_body -%} | |
| {%- if part.get('type') in ['image', 'image_url'] -%} | |
| {{- '<|image|>' -}} | |
| {%- elif part.get('type') in ['audio', 'input_audio'] -%} | |
| {{- '<|audio|>' -}} | |
| {%- elif part.get('type') == 'video' -%} | |
| {{- '<|video|>' -}} | |
| {%- endif -%} | |
| {%- endfor -%} | |
| {%- else -%} | |
| {{- format_tool_response_block(ns_tname.name, tool_body) -}} | |
| {%- endif -%} | |
| {%- set ns_tr_out.flag = true -%} | |
| {%- set ns.prev_message_type = 'tool_response' -%} | |
| {%- endif -%} | |
| {%- endfor -%} | |
| {%- endif -%} | |
| {%- set captured_content -%} | |
| {%- if message.get('content') is string -%} | |
| {%- if role == 'model' -%} | |
| {{- strip_thinking(message['content']) -}} | |
| {%- else -%} | |
| {{- message['content'] | trim -}} | |
| {%- endif -%} | |
| {%- elif message.get('content') is sequence -%} | |
| {%- for item in message['content'] -%} | |
| {%- if item.get('type') == 'text' -%} | |
| {%- if role == 'model' -%} | |
| {{- strip_thinking(item['text']) -}} | |
| {%- else -%} | |
| {{- item['text'] | trim -}} | |
| {%- endif -%} | |
| {%- elif item.get('type') in ['image', 'image_url'] -%} | |
| {{- '<|image|>' -}} | |
| {%- elif item.get('type') in ['audio', 'input_audio'] -%} | |
| {{- '<|audio|>' -}} | |
| {%- elif item.get('type') == 'video' -%} | |
| {{- '<|video|>' -}} | |
| {%- endif -%} | |
| {%- endfor -%} | |
| {%- endif -%} | |
| {%- endset -%} | |
| {{- captured_content -}} | |
| {%- set has_content = captured_content | trim | length > 0 -%} | |
| {#- Forward-scan: find next non-tool message role for continuation detection -#} | |
| {%- set next_nt = namespace(role=None, found=false) -%} | |
| {%- for j in range(loop.index0 + 1, loop_messages | length) -%} | |
| {%- if not next_nt.found -%} | |
| {%- if loop_messages[j]['role'] != 'tool' -%} | |
| {%- set next_nt.role = loop_messages[j]['role'] -%} | |
| {%- set next_nt.found = true -%} | |
| {%- endif -%} | |
| {%- endif -%} | |
| {%- endfor -%} | |
| {%- set continues_into_next = ( | |
| role == 'model' | |
| and next_nt.role == 'assistant' | |
| and (not message.get('tool_calls') or ns_tr_out.flag) | |
| ) -%} | |
| {%- if ns.prev_message_type == 'tool_call' and not ns_tr_out.flag -%} | |
| {{- '<|tool_response>' -}} | |
| {%- elif continues_into_next -%} | |
| {%- elif not (ns_tr_out.flag and not has_content and not next_nt.found) -%} | |
| {{- '<turn|>\n' -}} | |
| {%- endif -%} | |
| {#- Track previous non-tool role for next iteration (avoids O(n) backward scan) -#} | |
| {%- set ns.prev_non_tool_role = message['role'] -%} | |
| {%- endif -%} | |
| {%- endfor -%} | |
| {%- if add_generation_prompt -%} | |
| {%- if ns.prev_message_type != 'tool_response' and ns.prev_message_type != 'tool_call' -%} | |
| {{- '<|turn>model\n' -}} | |
| {%- if not enable_thinking -%} | |
| {{- '<|channel>thought\n<channel|>' -}} | |
| {%- endif -%} | |
| {%- elif ns.prev_message_type == 'tool_response' and enable_thinking -%} | |
| {{- '<|channel>thought\n' -}} | |
| {%- endif -%} | |
| {%- endif -%} |
Sign up for free
to join this conversation on GitHub.
Already have an account?
Sign in to comment