diff --git a/src/art/trajectories/_tokenize.py b/src/art/trajectories/_tokenize.py
index 9b692894c..83053e7a2 100644
--- a/src/art/trajectories/_tokenize.py
+++ b/src/art/trajectories/_tokenize.py
@@ -4464,6 +4464,58 @@ def _source_covers_complete_sampled_message(
) == normalize_chat_message(projected[0])
+def _preserve_literal_thinking_off_content(
+ history: ChatCompletionsHistory,
+ messages: list[dict[str, Any]],
+ template: object,
+ kwargs: Mapping[str, object],
+) -> None:
+ # This Qwen3.5 template treats any in unstructured content as a
+ # reasoning separator, even with thinking disabled. Restrict the render-copy
+ # adaptation to its exact preserved template; other templates may interpret
+ # an empty reasoning_content field differently.
+ if (
+ not isinstance(template, str)
+ or sha256(template.encode()).hexdigest()
+ != "098047d425a6673b1fe1a82a197a481616e53a283beaa8cb76cbb74d38ca6644"
+ or kwargs.get("enable_thinking") is not False
+ or kwargs.get("preserve_thinking") is not True
+ ):
+ return
+ for message, source in zip(messages, history.message_sources, strict=True):
+ if (
+ source is None
+ or not isinstance(source.exchange, ChatCompletionsExchange)
+ or source.choice_index is None
+ or message.get("role") != "assistant"
+ or not isinstance(content := message.get("content"), str)
+ or "" not in content
+ ):
+ continue
+ request_kwargs = source.exchange.request.get("chat_template_kwargs")
+ if (
+ not isinstance(request_kwargs, Mapping)
+ or request_kwargs.get("enable_thinking") is not False
+ ):
+ continue
+ choice = _chat_choice(source)
+ # Visible-only histories may omit structured reasoning present in the
+ # source response. Preserve both that source and normalized aliases.
+ if any(
+ value is not None and not (isinstance(value, str) and value == "")
+ for value in (
+ message.get("reasoning"),
+ message.get("reasoning_content"),
+ _field(choice.message, "reasoning"),
+ _field(choice.message, "reasoning_content"),
+ )
+ ):
+ continue
+ prompt, output, _ = _chat_choice_tokens(choice, source.exchange.response)
+ if prompt is not None and output is not None:
+ message["reasoning_content"] = ""
+
+
def _tokenize_chat_view(
history: ChatCompletionsHistory,
*,
@@ -4510,6 +4562,7 @@ def _tokenize_chat_view(
**default_chat_template_kwargs_for_template(template),
**explicit_kwargs,
}
+ _preserve_literal_thinking_off_content(history, messages, template, kwargs)
ends_with_assistant = bool(messages) and messages[-1].get("role") == "assistant"
segmented = False
diff --git a/tests/fixtures/qwen35_preserved_thinking.jinja b/tests/fixtures/qwen35_preserved_thinking.jinja
new file mode 100644
index 000000000..91dc43ca2
--- /dev/null
+++ b/tests/fixtures/qwen35_preserved_thinking.jinja
@@ -0,0 +1,154 @@
+{%- set preserve_thinking = preserve_thinking | default(true) -%}{%- set image_count = namespace(value=0) %}
+{%- set video_count = namespace(value=0) %}
+{%- macro render_content(content, do_vision_count, is_system_content=false) %}
+ {%- if content is string %}
+ {{- content }}
+ {%- elif content is iterable and content is not mapping %}
+ {%- for item in content %}
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
+ {%- if is_system_content %}
+ {{- raise_exception('System message cannot contain images.') }}
+ {%- endif %}
+ {%- if do_vision_count %}
+ {%- set image_count.value = image_count.value + 1 %}
+ {%- endif %}
+ {%- if add_vision_id %}
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
+ {%- endif %}
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
+ {%- elif 'video' in item or item.type == 'video' %}
+ {%- if is_system_content %}
+ {{- raise_exception('System message cannot contain videos.') }}
+ {%- endif %}
+ {%- if do_vision_count %}
+ {%- set video_count.value = video_count.value + 1 %}
+ {%- endif %}
+ {%- if add_vision_id %}
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
+ {%- endif %}
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
+ {%- elif 'text' in item %}
+ {{- item.text }}
+ {%- else %}
+ {{- raise_exception('Unexpected item type in content.') }}
+ {%- endif %}
+ {%- endfor %}
+ {%- elif content is none or content is undefined %}
+ {{- '' }}
+ {%- else %}
+ {{- raise_exception('Unexpected content type.') }}
+ {%- endif %}
+{%- endmacro %}
+{%- if not messages %}
+ {{- raise_exception('No messages provided.') }}
+{%- endif %}
+{%- if tools and tools is iterable and tools is not mapping %}
+ {{- '<|im_start|>system\n' }}
+ {{- "# Tools\n\nYou have access to the following functions:\n\n" }}
+ {%- for tool in tools %}
+ {{- "\n" }}
+ {{- tool | tojson }}
+ {%- endfor %}
+ {{- "\n" }}
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n\n\n\nvalue_1\n\n\nThis is the value for the second parameter\nthat can span\nmultiple lines\n\n\n\n\n\nReminder:\n- Function calls MUST follow the specified format: an inner block must be nested within XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n' }}
+ {%- if messages[0].role == 'system' %}
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
+ {%- if content %}
+ {{- '\n\n' + content }}
+ {%- endif %}
+ {%- endif %}
+ {{- '<|im_end|>\n' }}
+{%- else %}
+ {%- if messages[0].role == 'system' %}
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
+ {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
+ {%- endif %}
+{%- endif %}
+{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
+{%- for message in messages[::-1] %}
+ {%- set index = (messages|length - 1) - loop.index0 %}
+ {%- if ns.multi_step_tool and message.role == "user" %}
+ {%- set content = render_content(message.content, false)|trim %}
+ {%- if not(content.startswith('') and content.endswith('')) %}
+ {%- set ns.multi_step_tool = false %}
+ {%- set ns.last_query_index = index %}
+ {%- endif %}
+ {%- endif %}
+{%- endfor %}
+{%- if ns.multi_step_tool %}
+ {{- raise_exception('No user query found in messages.') }}
+{%- endif %}
+{%- for message in messages %}
+ {%- set content = (render_content(message.content, true) if preserve_thinking and message.role == 'assistant' else render_content(message.content, true)|trim) %}
+ {%- if message.role == "system" %}
+ {%- if not loop.first %}
+ {{- raise_exception('System message must be at the beginning.') }}
+ {%- endif %}
+ {%- elif message.role == "user" %}
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
+ {%- elif message.role == "assistant" %}
+ {%- set reasoning_content = '' %}
+ {%- if message.reasoning_content is string %}
+ {%- set reasoning_content = message.reasoning_content %}
+ {%- else %}
+ {%- if '' in content %}
+ {%- set reasoning_content = content.split('')[0].rstrip('\n').split('')[-1].lstrip('\n') %}
+ {%- set content = content.split('')[-1].lstrip('\n') %}
+ {%- endif %}
+ {%- endif %}
+ {%- if not preserve_thinking or message.reasoning_content is not string %}{%- set reasoning_content = reasoning_content|trim %}{%- endif %}
+ {%- if (preserve_thinking is defined and preserve_thinking is true) or (loop.index0 > ns.last_query_index) %}
+ {{- '<|im_start|>' + message.role + '\n\n' + reasoning_content + ('\n\n' if preserve_thinking and message.reasoning_content is string and reasoning_content else '\n\n\n') + content }}
+ {%- else %}
+ {{- '<|im_start|>' + message.role + '\n' + content }}
+ {%- endif %}
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
+ {%- for tool_call in message.tool_calls %}
+ {%- if tool_call.function is defined %}
+ {%- set tool_call = tool_call.function %}
+ {%- endif %}
+ {%- if loop.first %}
+ {%- if content|trim %}
+ {{- '\n\n\n\n' }}
+ {%- else %}
+ {{- '\n\n' }}
+ {%- endif %}
+ {%- else %}
+ {{- '\n\n\n' }}
+ {%- endif %}
+ {%- if tool_call.arguments is defined %}
+ {%- for args_name, args_value in tool_call.arguments|items %}
+ {{- '\n' }}
+ {%- set args_value = args_value | string if args_value is string else args_value | tojson | safe %}
+ {{- args_value }}
+ {{- '\n\n' }}
+ {%- endfor %}
+ {%- endif %}
+ {{- '\n' }}
+ {%- endfor %}
+ {%- endif %}
+ {{- '<|im_end|>\n' }}
+ {%- elif message.role == "tool" %}
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
+ {{- '<|im_start|>user' }}
+ {%- endif %}
+ {{- '\n\n' }}
+ {{- content }}
+ {{- '\n' }}
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
+ {{- '<|im_end|>\n' }}
+ {%- elif loop.last %}
+ {{- '<|im_end|>\n' }}
+ {%- endif %}
+ {%- else %}
+ {{- raise_exception('Unexpected message role.') }}
+ {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+ {{- '<|im_start|>assistant\n' }}
+ {%- if enable_thinking is defined and enable_thinking is false %}
+ {{- '\n\n\n\n' }}
+ {%- else %}
+ {{- '\n' }}
+ {%- endif %}
+{%- endif %}
\ No newline at end of file
diff --git a/tests/unit/trajectories/test_literal_thinking_off.py b/tests/unit/trajectories/test_literal_thinking_off.py
new file mode 100644
index 000000000..16188c733
--- /dev/null
+++ b/tests/unit/trajectories/test_literal_thinking_off.py
@@ -0,0 +1,443 @@
+from __future__ import annotations
+
+from copy import deepcopy
+from datetime import UTC, datetime
+from pathlib import Path
+import re
+from typing import Any, cast
+
+from jinja2.sandbox import ImmutableSandboxedEnvironment
+from openai.types.chat import ChatCompletion, ChatCompletionMessageParam
+import pytest
+
+import art.trajectories as tr
+from art.trajectories import _tokenize
+
+# Public Qwen3.5 template after ART's existing thinking-preservation rewrite.
+_TEMPLATE = (
+ Path(__file__).parents[2] / "fixtures/qwen35_preserved_thinking.jinja"
+).read_text()
+_LITERAL = "HEAD\nDISCARDED_PUBLIC_SEGMENT\n\nTAIL"
+
+
+class _TemplateTokenizer:
+ chat_template = _TEMPLATE
+
+ def __init__(self) -> None:
+ self.calls: list[list[dict[str, Any]]] = []
+ self.rendered: list[str] = []
+ self.env = ImmutableSandboxedEnvironment(
+ trim_blocks=True, lstrip_blocks=True, extensions=["jinja2.ext.loopcontrols"]
+ )
+
+ def __call__(self, text: str, **kwargs: Any) -> dict[str, Any]:
+ result: dict[str, Any] = {"input_ids": list(map(ord, text))}
+ if kwargs.get("return_offsets_mapping"):
+ result["offset_mapping"] = [(i, i + 1) for i in range(len(text))]
+ return result
+
+ def decode(self, token_ids: list[int], **kwargs: Any) -> str:
+ return "".join(map(chr, token_ids))
+
+ def apply_chat_template(
+ self,
+ messages: list[dict[str, Any]],
+ *,
+ tokenize: bool = True,
+ add_generation_prompt: bool = False,
+ chat_template: str | None = None,
+ **kwargs: Any,
+ ) -> str | list[int]:
+ self.calls.append(deepcopy(messages))
+ text = self.env.from_string(chat_template or self.chat_template).render(
+ messages=messages,
+ add_generation_prompt=add_generation_prompt,
+ **kwargs,
+ )
+ self.rendered.append(text)
+ return list(map(ord, text)) if tokenize else text
+
+
+def _history(
+ *,
+ thinking: bool | None = False,
+ content: str = _LITERAL,
+ reasoning: str | None = None,
+ reasoning_field: str = "reasoning_content",
+) -> tuple[tr.ChatCompletionsHistory, _TemplateTokenizer]:
+ tokenizer = _TemplateTokenizer()
+ request_kwargs: dict[str, Any] = {"preserve_thinking": True}
+ if thinking is not None:
+ request_kwargs["enable_thinking"] = thinking
+ prompt_messages = [{"role": "user", "content": "Public query."}]
+ prompt = tokenizer.apply_chat_template(
+ prompt_messages, add_generation_prompt=True, **request_kwargs
+ )
+ message: dict[str, Any] = {"role": "assistant", "content": content}
+ if reasoning is not None:
+ message[reasoning_field] = reasoning
+ output = list(map(ord, content))
+ response = ChatCompletion.model_validate(
+ {
+ "id": "public-repro",
+ "object": "chat.completion",
+ "created": 0,
+ "model": "public/qwen35",
+ "choices": [
+ {
+ "index": 0,
+ "finish_reason": "length",
+ "message": message,
+ "prompt_token_ids": prompt,
+ "token_ids": output,
+ "logprobs": {
+ "content": [
+ {
+ "token": f"token_id:{token}",
+ "logprob": -0.5,
+ "bytes": [],
+ "top_logprobs": [],
+ }
+ for token in output
+ ]
+ },
+ }
+ ],
+ }
+ )
+ exchange = tr.ChatCompletionsExchange(
+ request=tr.ChatCompletionsRequest(
+ model="public/qwen35",
+ messages=cast(list[ChatCompletionMessageParam], prompt_messages),
+ chat_template=_TEMPLATE,
+ chat_template_kwargs=request_kwargs,
+ ),
+ response=response,
+ start_time=datetime(2026, 1, 1, tzinfo=UTC),
+ end_time=datetime(2026, 1, 1, tzinfo=UTC),
+ )
+ history = tr.Trajectory(
+ exchanges=tr.TrajectoryExchanges(chat_completions=[exchange])
+ ).chat_completions_history()
+ tokenizer.calls.clear()
+ tokenizer.rendered.clear()
+ return history, tokenizer
+
+
+def _outcome(
+ history: tr.ChatCompletionsHistory, tokenizer: _TemplateTokenizer
+) -> object:
+ try:
+ value = history.tokenize(tokenizer=tokenizer)
+ except ValueError as error:
+ return type(error), str(error)
+ return value.tokens, value.flags, [None if x != x else x for x in value.logprobs]
+
+
+@pytest.mark.parametrize(
+ "content", [_LITERAL, "literal text", "πonetwoend"]
+)
+def test_native_thinking_off_retains_literal_content(
+ content: str, monkeypatch: pytest.MonkeyPatch
+) -> None:
+ history, tokenizer = _history(content=content)
+ original = history.model_dump(mode="python")
+ # The pre-fix history path misrenders literal content even when later native
+ # token splicing can recover the terminal output.
+ with monkeypatch.context() as patch:
+ patch.setattr(
+ _tokenize, "_preserve_literal_thinking_off_content", lambda *args: None
+ )
+ _outcome(history, tokenizer)
+ assert content not in tokenizer.rendered[0]
+ tokenizer.calls.clear()
+ tokenizer.rendered.clear()
+ tokenized = history.tokenize(tokenizer=tokenizer)
+ assert content in tokenizer.rendered[0]
+ sampled = [
+ i for i, flag in enumerate(tokenized.flags) if flag & tr.TokenFlag.SAMPLED
+ ]
+ assert "".join(chr(tokenized.tokens[i]) for i in sampled) == content
+ assert all(tokenized.logprobs[i] == -0.5 for i in sampled)
+ required = tr.TokenFlag.EXACT | tr.TokenFlag.ASSISTANT | tr.TokenFlag.OUTPUT
+ assert all(tokenized.flags[i] & required == required for i in sampled)
+ assert not any(flag & tr.TokenFlag.STOP for flag in tokenized.flags)
+ assert tokenizer.calls[0][-1]["reasoning_content"] == ""
+ assert tokenizer.calls[0][-1]["content"] == content
+ assert history.model_dump(mode="python") == original
+
+
+@pytest.mark.parametrize(
+ "case",
+ [
+ "source_on",
+ "source_unknown",
+ "effective_on",
+ "preserve_off",
+ "other_template",
+ "no_source",
+ "request_source",
+ "no_native_prompt",
+ "structured",
+ "alias",
+ "visible_only",
+ ],
+)
+def test_unrelated_histories_keep_original_rendering(
+ case: str, monkeypatch: pytest.MonkeyPatch
+) -> None:
+ history, tokenizer = _history(
+ thinking=True
+ if case == "source_on"
+ else None
+ if case == "source_unknown"
+ else False,
+ reasoning="explicit reasoning"
+ if case in {"structured", "alias", "visible_only"}
+ else None,
+ reasoning_field="reasoning" if case == "alias" else "reasoning_content",
+ )
+ assert history.chat_template_kwargs is not None
+ history.chat_template_kwargs["enable_thinking"] = case == "effective_on"
+ if case == "preserve_off":
+ history.chat_template_kwargs["preserve_thinking"] = False
+ if case == "other_template":
+ history.chat_template = _TEMPLATE + "{# different template #}"
+ if case == "no_source":
+ history.message_sources[-1] = None
+ source = history.message_sources[-1]
+ if case == "request_source":
+ assert source is not None
+ assert isinstance(source.exchange, tr.ChatCompletionsExchange)
+ source.exchange.request["messages"].append(deepcopy(history.messages[-1]))
+ history.message_sources[-1] = tr.ChatCompletionsMessageSource(
+ exchange=source.exchange, request_index=1
+ )
+ if case == "no_native_prompt":
+ assert source is not None
+ assert isinstance(source.exchange, tr.ChatCompletionsExchange)
+ extra = source.exchange.response.choices[0].model_extra
+ assert extra is not None
+ extra.pop("prompt_token_ids")
+ if case == "visible_only":
+ cast(dict[str, Any], history.messages[-1]).pop("reasoning")
+ original = history.model_dump(mode="python")
+ candidate = _outcome(history, tokenizer)
+ calls = deepcopy(tokenizer.calls)
+ tokenizer.calls.clear()
+ with monkeypatch.context() as patch:
+ patch.setattr(
+ _tokenize, "_preserve_literal_thinking_off_content", lambda *args: None
+ )
+ baseline = _outcome(history, tokenizer)
+ assert candidate == baseline
+ assert len(calls) == len(tokenizer.calls)
+ assert calls[0] == tokenizer.calls[0]
+ assert history.model_dump(mode="python") == original
+
+
+@pytest.mark.parametrize("field", ["reasoning_content", "reasoning"])
+def test_explicit_empty_reasoning_is_preserved(field: str) -> None:
+ history, tokenizer = _history(reasoning="", reasoning_field=field)
+ original = history.model_dump(mode="python")
+ tokenized = history.tokenize(tokenizer=tokenizer)
+ assert (
+ "".join(
+ chr(token)
+ for token, flag in zip(tokenized.tokens, tokenized.flags, strict=True)
+ if flag & tr.TokenFlag.SAMPLED
+ )
+ == _LITERAL
+ )
+ assert tokenizer.calls[0][-1]["reasoning_content"] == ""
+ assert history.model_dump(mode="python") == original
+
+
+def test_literal_adaptation_does_not_relax_source_validation() -> None:
+ history, tokenizer = _history()
+ cast(dict[str, Any], history.messages[-1])["content"] += "not in source"
+ with pytest.raises(ValueError, match="source"):
+ history.tokenize(tokenizer=tokenizer)
+ assert not tokenizer.calls
+
+
+@pytest.mark.parametrize("earlier_thinking", [True, None])
+def test_mixed_history_uses_each_generations_own_request(
+ earlier_thinking: bool | None,
+) -> None:
+ first, tokenizer = _history(thinking=earlier_thinking)
+ second, _ = _history()
+ history = tr.ChatCompletionsHistory(
+ model=second.model,
+ messages=[*first.messages, *second.messages],
+ message_sources=[*first.message_sources, *second.message_sources],
+ chat_template=second.chat_template,
+ chat_template_kwargs=second.chat_template_kwargs,
+ )
+ original = history.model_dump(mode="python")
+ _outcome(history, tokenizer)
+ rendered_messages = tokenizer.calls[0]
+ assert "reasoning_content" not in rendered_messages[1]
+ assert rendered_messages[3]["reasoning_content"] == ""
+ assert [message["content"] for message in rendered_messages] == [
+ message["content"] for message in history.messages
+ ]
+ assert history.model_dump(mode="python") == original
+
+
+class _NewlineRunTokenizer(_TemplateTokenizer):
+ """Reversible public codec that exposes the open/closed scaffold boundary."""
+
+ all_special_tokens = ["<|im_start|>", "<|im_end|>", "", ""]
+ all_special_ids = [200000, 200001, 200002, 200003]
+ eos_token_id = 200001
+
+ def __call__(self, text: str, **kwargs: Any) -> dict[str, Any]:
+ pieces = list(
+ re.finditer(r"<\|im_start\|>|<\|im_end\|>|||\n+|[^\n]", text)
+ )
+ result = {
+ "input_ids": [
+ self.all_special_ids[self.all_special_tokens.index(piece.group())]
+ if piece.group() in self.all_special_tokens
+ else 300000 + len(piece.group())
+ if piece.group().startswith("\n")
+ else ord(piece.group())
+ for piece in pieces
+ ]
+ }
+ if kwargs.get("return_offsets_mapping"):
+ result["offset_mapping"] = [piece.span() for piece in pieces]
+ return result
+
+ def decode(self, token_ids: list[int], **kwargs: Any) -> str:
+ return "".join(
+ self.all_special_tokens[self.all_special_ids.index(token)]
+ if token in self.all_special_ids
+ else "\n" * (token - 300000)
+ if token > 300000
+ else chr(token)
+ for token in token_ids
+ )
+
+ def convert_tokens_to_ids(self, token: str) -> int | None:
+ return (
+ self.all_special_ids[self.all_special_tokens.index(token)]
+ if token in self.all_special_tokens
+ else None
+ )
+
+ def apply_chat_template(
+ self, messages: list[dict[str, Any]], *, tokenize: bool = True, **kwargs: Any
+ ) -> str | list[int]:
+ text = super().apply_chat_template(messages, tokenize=False, **kwargs)
+ assert isinstance(text, str)
+ return self(text)["input_ids"] if tokenize else text
+
+
+def test_literal_next_turn_preserves_preceding_length_stop_boundary(
+ monkeypatch: pytest.MonkeyPatch,
+) -> None:
+ tokenizer = _NewlineRunTokenizer()
+ messages = [
+ {"role": "user", "content": "first"},
+ {"role": "assistant", "content": "unchanged preceding output"},
+ {"role": "user", "content": "next query"},
+ {"role": "assistant", "content": _LITERAL},
+ ]
+ exchanges = []
+ for index in (1, 3):
+ single, _ = _history(content=messages[index]["content"])
+ source = single.message_sources[-1]
+ assert source is not None and isinstance(
+ source.exchange, tr.ChatCompletionsExchange
+ )
+ exchange = source.exchange
+ exchange.request["messages"] = cast(
+ list[ChatCompletionMessageParam], deepcopy(messages[:index])
+ )
+ data = exchange.response.model_dump(mode="python")
+ data["id"] = f"public-length-{index}"
+ choice = data["choices"][0]
+ choice["prompt_token_ids"] = tokenizer.apply_chat_template(
+ messages[:index],
+ add_generation_prompt=True,
+ enable_thinking=False,
+ preserve_thinking=True,
+ )
+ choice["token_ids"] = tokenizer(messages[index]["content"])["input_ids"]
+ choice["logprobs"]["content"] = [
+ {
+ "token": f"token_id:{token}",
+ "logprob": -0.5,
+ "bytes": [],
+ "top_logprobs": [],
+ }
+ for token in choice["token_ids"]
+ ]
+ exchange.response = ChatCompletion.model_validate(data)
+ exchanges.append(exchange)
+ history = tr.Trajectory(
+ exchanges=tr.TrajectoryExchanges(chat_completions=exchanges)
+ ).chat_completions_history()
+ original = history.model_dump(mode="python")
+ first = exchanges[0].response.choices[0].model_extra
+ last = exchanges[1].response.choices[0].model_extra
+ assert first is not None and last is not None
+ end = len(first["prompt_token_ids"]) + len(first["token_ids"])
+ native_boundary = last["prompt_token_ids"][end:]
+ assert (
+ last["prompt_token_ids"][:end] == first["prompt_token_ids"] + first["token_ids"]
+ )
+ source = history.message_sources[1]
+ key = _tokenize._sampled_source_key(source)
+ builder = _tokenize._tokenize_exact_projected_chat_history
+ observed = []
+
+ def observe(*args: Any, **kwargs: Any) -> tr.TokenizedHistory | None:
+ result = builder(*args, **kwargs)
+ if boundary := kwargs.get("length_stop_boundaries", {}).get(key):
+ observed.append((boundary, result))
+ return result
+
+ monkeypatch.setattr(_tokenize, "_tokenize_exact_projected_chat_history", observe)
+ with monkeypatch.context() as patch:
+ patch.setattr(
+ _tokenize, "_preserve_literal_thinking_off_content", lambda *args: None
+ )
+ _outcome(history, tokenizer)
+ boundary, old_exact = observed[0]
+ stored = list(boundary.tail + boundary.following)
+ assert old_exact is None
+ assert len(native_boundary) - len(stored) == 2
+ assert stored[:-1] == native_boundary[:-3]
+ assert tokenizer.decode(stored[-1:]) == "\n"
+ assert tokenizer.decode(native_boundary[-3:]) == "\n\n\n\n"
+ observed.clear()
+
+ value = history.tokenize(tokenizer=tokenizer)
+ fixed_boundary, fixed_exact = observed[0]
+ assert fixed_exact is value
+ assert list(fixed_boundary.tail + fixed_boundary.following) == native_boundary
+ assert (
+ value.tokens[: len(last["prompt_token_ids"]) + len(last["token_ids"])]
+ == last["prompt_token_ids"] + last["token_ids"]
+ )
+ assert value.flags[end] == tr.TokenFlag.EXACT | tr.TokenFlag.STOP
+ assert not value.flags[end] & tr.TokenFlag.SAMPLED
+ for exchange in exchanges:
+ extra = exchange.response.choices[0].model_extra
+ assert extra is not None
+ start = len(extra["prompt_token_ids"])
+ stop = start + len(extra["token_ids"])
+ assert value.tokens[start:stop] == extra["token_ids"]
+ assert value.logprobs[start:stop] == [-0.5] * (stop - start)
+ assert all(
+ flag
+ == tr.TokenFlag.EXACT
+ | tr.TokenFlag.SAMPLED
+ | tr.TokenFlag.ASSISTANT
+ | tr.TokenFlag.OUTPUT
+ for flag in value.flags[start:stop]
+ )
+ assert history.model_dump(mode="python") == original