diff --git a/src/art/trajectories/_tokenize.py b/src/art/trajectories/_tokenize.py index 9b692894c..83053e7a2 100644 --- a/src/art/trajectories/_tokenize.py +++ b/src/art/trajectories/_tokenize.py @@ -4464,6 +4464,58 @@ def _source_covers_complete_sampled_message( ) == normalize_chat_message(projected[0]) +def _preserve_literal_thinking_off_content( + history: ChatCompletionsHistory, + messages: list[dict[str, Any]], + template: object, + kwargs: Mapping[str, object], +) -> None: + # This Qwen3.5 template treats any in unstructured content as a + # reasoning separator, even with thinking disabled. Restrict the render-copy + # adaptation to its exact preserved template; other templates may interpret + # an empty reasoning_content field differently. + if ( + not isinstance(template, str) + or sha256(template.encode()).hexdigest() + != "098047d425a6673b1fe1a82a197a481616e53a283beaa8cb76cbb74d38ca6644" + or kwargs.get("enable_thinking") is not False + or kwargs.get("preserve_thinking") is not True + ): + return + for message, source in zip(messages, history.message_sources, strict=True): + if ( + source is None + or not isinstance(source.exchange, ChatCompletionsExchange) + or source.choice_index is None + or message.get("role") != "assistant" + or not isinstance(content := message.get("content"), str) + or "" not in content + ): + continue + request_kwargs = source.exchange.request.get("chat_template_kwargs") + if ( + not isinstance(request_kwargs, Mapping) + or request_kwargs.get("enable_thinking") is not False + ): + continue + choice = _chat_choice(source) + # Visible-only histories may omit structured reasoning present in the + # source response. Preserve both that source and normalized aliases. + if any( + value is not None and not (isinstance(value, str) and value == "") + for value in ( + message.get("reasoning"), + message.get("reasoning_content"), + _field(choice.message, "reasoning"), + _field(choice.message, "reasoning_content"), + ) + ): + continue + prompt, output, _ = _chat_choice_tokens(choice, source.exchange.response) + if prompt is not None and output is not None: + message["reasoning_content"] = "" + + def _tokenize_chat_view( history: ChatCompletionsHistory, *, @@ -4510,6 +4562,7 @@ def _tokenize_chat_view( **default_chat_template_kwargs_for_template(template), **explicit_kwargs, } + _preserve_literal_thinking_off_content(history, messages, template, kwargs) ends_with_assistant = bool(messages) and messages[-1].get("role") == "assistant" segmented = False diff --git a/tests/fixtures/qwen35_preserved_thinking.jinja b/tests/fixtures/qwen35_preserved_thinking.jinja new file mode 100644 index 000000000..91dc43ca2 --- /dev/null +++ b/tests/fixtures/qwen35_preserved_thinking.jinja @@ -0,0 +1,154 @@ +{%- set preserve_thinking = preserve_thinking | default(true) -%}{%- set image_count = namespace(value=0) %} +{%- set video_count = namespace(value=0) %} +{%- macro render_content(content, do_vision_count, is_system_content=false) %} + {%- if content is string %} + {{- content }} + {%- elif content is iterable and content is not mapping %} + {%- for item in content %} + {%- if 'image' in item or 'image_url' in item or item.type == 'image' %} + {%- if is_system_content %} + {{- raise_exception('System message cannot contain images.') }} + {%- endif %} + {%- if do_vision_count %} + {%- set image_count.value = image_count.value + 1 %} + {%- endif %} + {%- if add_vision_id %} + {{- 'Picture ' ~ image_count.value ~ ': ' }} + {%- endif %} + {{- '<|vision_start|><|image_pad|><|vision_end|>' }} + {%- elif 'video' in item or item.type == 'video' %} + {%- if is_system_content %} + {{- raise_exception('System message cannot contain videos.') }} + {%- endif %} + {%- if do_vision_count %} + {%- set video_count.value = video_count.value + 1 %} + {%- endif %} + {%- if add_vision_id %} + {{- 'Video ' ~ video_count.value ~ ': ' }} + {%- endif %} + {{- '<|vision_start|><|video_pad|><|vision_end|>' }} + {%- elif 'text' in item %} + {{- item.text }} + {%- else %} + {{- raise_exception('Unexpected item type in content.') }} + {%- endif %} + {%- endfor %} + {%- elif content is none or content is undefined %} + {{- '' }} + {%- else %} + {{- raise_exception('Unexpected content type.') }} + {%- endif %} +{%- endmacro %} +{%- if not messages %} + {{- raise_exception('No messages provided.') }} +{%- endif %} +{%- if tools and tools is iterable and tools is not mapping %} + {{- '<|im_start|>system\n' }} + {{- "# Tools\n\nYou have access to the following functions:\n\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n" }} + {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n\n\n\nvalue_1\n\n\nThis is the value for the second parameter\nthat can span\nmultiple lines\n\n\n\n\n\nReminder:\n- Function calls MUST follow the specified format: an inner block must be nested within XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n' }} + {%- if messages[0].role == 'system' %} + {%- set content = render_content(messages[0].content, false, true)|trim %} + {%- if content %} + {{- '\n\n' + content }} + {%- endif %} + {%- endif %} + {{- '<|im_end|>\n' }} +{%- else %} + {%- if messages[0].role == 'system' %} + {%- set content = render_content(messages[0].content, false, true)|trim %} + {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %} +{%- for message in messages[::-1] %} + {%- set index = (messages|length - 1) - loop.index0 %} + {%- if ns.multi_step_tool and message.role == "user" %} + {%- set content = render_content(message.content, false)|trim %} + {%- if not(content.startswith('') and content.endswith('')) %} + {%- set ns.multi_step_tool = false %} + {%- set ns.last_query_index = index %} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if ns.multi_step_tool %} + {{- raise_exception('No user query found in messages.') }} +{%- endif %} +{%- for message in messages %} + {%- set content = (render_content(message.content, true) if preserve_thinking and message.role == 'assistant' else render_content(message.content, true)|trim) %} + {%- if message.role == "system" %} + {%- if not loop.first %} + {{- raise_exception('System message must be at the beginning.') }} + {%- endif %} + {%- elif message.role == "user" %} + {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {%- set reasoning_content = '' %} + {%- if message.reasoning_content is string %} + {%- set reasoning_content = message.reasoning_content %} + {%- else %} + {%- if '' in content %} + {%- set reasoning_content = content.split('')[0].rstrip('\n').split('')[-1].lstrip('\n') %} + {%- set content = content.split('')[-1].lstrip('\n') %} + {%- endif %} + {%- endif %} + {%- if not preserve_thinking or message.reasoning_content is not string %}{%- set reasoning_content = reasoning_content|trim %}{%- endif %} + {%- if (preserve_thinking is defined and preserve_thinking is true) or (loop.index0 > ns.last_query_index) %} + {{- '<|im_start|>' + message.role + '\n\n' + reasoning_content + ('\n\n' if preserve_thinking and message.reasoning_content is string and reasoning_content else '\n\n\n') + content }} + {%- else %} + {{- '<|im_start|>' + message.role + '\n' + content }} + {%- endif %} + {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {%- if loop.first %} + {%- if content|trim %} + {{- '\n\n\n\n' }} + {%- else %} + {{- '\n\n' }} + {%- endif %} + {%- else %} + {{- '\n\n\n' }} + {%- endif %} + {%- if tool_call.arguments is defined %} + {%- for args_name, args_value in tool_call.arguments|items %} + {{- '\n' }} + {%- set args_value = args_value | string if args_value is string else args_value | tojson | safe %} + {{- args_value }} + {{- '\n\n' }} + {%- endfor %} + {%- endif %} + {{- '\n' }} + {%- endfor %} + {%- endif %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if loop.previtem and loop.previtem.role != "tool" %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- content }} + {{- '\n' }} + {%- if not loop.last and loop.nextitem.role != "tool" %} + {{- '<|im_end|>\n' }} + {%- elif loop.last %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- else %} + {{- raise_exception('Unexpected message role.') }} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} + {%- if enable_thinking is defined and enable_thinking is false %} + {{- '\n\n\n\n' }} + {%- else %} + {{- '\n' }} + {%- endif %} +{%- endif %} \ No newline at end of file diff --git a/tests/unit/trajectories/test_literal_thinking_off.py b/tests/unit/trajectories/test_literal_thinking_off.py new file mode 100644 index 000000000..16188c733 --- /dev/null +++ b/tests/unit/trajectories/test_literal_thinking_off.py @@ -0,0 +1,443 @@ +from __future__ import annotations + +from copy import deepcopy +from datetime import UTC, datetime +from pathlib import Path +import re +from typing import Any, cast + +from jinja2.sandbox import ImmutableSandboxedEnvironment +from openai.types.chat import ChatCompletion, ChatCompletionMessageParam +import pytest + +import art.trajectories as tr +from art.trajectories import _tokenize + +# Public Qwen3.5 template after ART's existing thinking-preservation rewrite. +_TEMPLATE = ( + Path(__file__).parents[2] / "fixtures/qwen35_preserved_thinking.jinja" +).read_text() +_LITERAL = "HEAD\nDISCARDED_PUBLIC_SEGMENT\n\nTAIL" + + +class _TemplateTokenizer: + chat_template = _TEMPLATE + + def __init__(self) -> None: + self.calls: list[list[dict[str, Any]]] = [] + self.rendered: list[str] = [] + self.env = ImmutableSandboxedEnvironment( + trim_blocks=True, lstrip_blocks=True, extensions=["jinja2.ext.loopcontrols"] + ) + + def __call__(self, text: str, **kwargs: Any) -> dict[str, Any]: + result: dict[str, Any] = {"input_ids": list(map(ord, text))} + if kwargs.get("return_offsets_mapping"): + result["offset_mapping"] = [(i, i + 1) for i in range(len(text))] + return result + + def decode(self, token_ids: list[int], **kwargs: Any) -> str: + return "".join(map(chr, token_ids)) + + def apply_chat_template( + self, + messages: list[dict[str, Any]], + *, + tokenize: bool = True, + add_generation_prompt: bool = False, + chat_template: str | None = None, + **kwargs: Any, + ) -> str | list[int]: + self.calls.append(deepcopy(messages)) + text = self.env.from_string(chat_template or self.chat_template).render( + messages=messages, + add_generation_prompt=add_generation_prompt, + **kwargs, + ) + self.rendered.append(text) + return list(map(ord, text)) if tokenize else text + + +def _history( + *, + thinking: bool | None = False, + content: str = _LITERAL, + reasoning: str | None = None, + reasoning_field: str = "reasoning_content", +) -> tuple[tr.ChatCompletionsHistory, _TemplateTokenizer]: + tokenizer = _TemplateTokenizer() + request_kwargs: dict[str, Any] = {"preserve_thinking": True} + if thinking is not None: + request_kwargs["enable_thinking"] = thinking + prompt_messages = [{"role": "user", "content": "Public query."}] + prompt = tokenizer.apply_chat_template( + prompt_messages, add_generation_prompt=True, **request_kwargs + ) + message: dict[str, Any] = {"role": "assistant", "content": content} + if reasoning is not None: + message[reasoning_field] = reasoning + output = list(map(ord, content)) + response = ChatCompletion.model_validate( + { + "id": "public-repro", + "object": "chat.completion", + "created": 0, + "model": "public/qwen35", + "choices": [ + { + "index": 0, + "finish_reason": "length", + "message": message, + "prompt_token_ids": prompt, + "token_ids": output, + "logprobs": { + "content": [ + { + "token": f"token_id:{token}", + "logprob": -0.5, + "bytes": [], + "top_logprobs": [], + } + for token in output + ] + }, + } + ], + } + ) + exchange = tr.ChatCompletionsExchange( + request=tr.ChatCompletionsRequest( + model="public/qwen35", + messages=cast(list[ChatCompletionMessageParam], prompt_messages), + chat_template=_TEMPLATE, + chat_template_kwargs=request_kwargs, + ), + response=response, + start_time=datetime(2026, 1, 1, tzinfo=UTC), + end_time=datetime(2026, 1, 1, tzinfo=UTC), + ) + history = tr.Trajectory( + exchanges=tr.TrajectoryExchanges(chat_completions=[exchange]) + ).chat_completions_history() + tokenizer.calls.clear() + tokenizer.rendered.clear() + return history, tokenizer + + +def _outcome( + history: tr.ChatCompletionsHistory, tokenizer: _TemplateTokenizer +) -> object: + try: + value = history.tokenize(tokenizer=tokenizer) + except ValueError as error: + return type(error), str(error) + return value.tokens, value.flags, [None if x != x else x for x in value.logprobs] + + +@pytest.mark.parametrize( + "content", [_LITERAL, "literal text", "πonetwoend"] +) +def test_native_thinking_off_retains_literal_content( + content: str, monkeypatch: pytest.MonkeyPatch +) -> None: + history, tokenizer = _history(content=content) + original = history.model_dump(mode="python") + # The pre-fix history path misrenders literal content even when later native + # token splicing can recover the terminal output. + with monkeypatch.context() as patch: + patch.setattr( + _tokenize, "_preserve_literal_thinking_off_content", lambda *args: None + ) + _outcome(history, tokenizer) + assert content not in tokenizer.rendered[0] + tokenizer.calls.clear() + tokenizer.rendered.clear() + tokenized = history.tokenize(tokenizer=tokenizer) + assert content in tokenizer.rendered[0] + sampled = [ + i for i, flag in enumerate(tokenized.flags) if flag & tr.TokenFlag.SAMPLED + ] + assert "".join(chr(tokenized.tokens[i]) for i in sampled) == content + assert all(tokenized.logprobs[i] == -0.5 for i in sampled) + required = tr.TokenFlag.EXACT | tr.TokenFlag.ASSISTANT | tr.TokenFlag.OUTPUT + assert all(tokenized.flags[i] & required == required for i in sampled) + assert not any(flag & tr.TokenFlag.STOP for flag in tokenized.flags) + assert tokenizer.calls[0][-1]["reasoning_content"] == "" + assert tokenizer.calls[0][-1]["content"] == content + assert history.model_dump(mode="python") == original + + +@pytest.mark.parametrize( + "case", + [ + "source_on", + "source_unknown", + "effective_on", + "preserve_off", + "other_template", + "no_source", + "request_source", + "no_native_prompt", + "structured", + "alias", + "visible_only", + ], +) +def test_unrelated_histories_keep_original_rendering( + case: str, monkeypatch: pytest.MonkeyPatch +) -> None: + history, tokenizer = _history( + thinking=True + if case == "source_on" + else None + if case == "source_unknown" + else False, + reasoning="explicit reasoning" + if case in {"structured", "alias", "visible_only"} + else None, + reasoning_field="reasoning" if case == "alias" else "reasoning_content", + ) + assert history.chat_template_kwargs is not None + history.chat_template_kwargs["enable_thinking"] = case == "effective_on" + if case == "preserve_off": + history.chat_template_kwargs["preserve_thinking"] = False + if case == "other_template": + history.chat_template = _TEMPLATE + "{# different template #}" + if case == "no_source": + history.message_sources[-1] = None + source = history.message_sources[-1] + if case == "request_source": + assert source is not None + assert isinstance(source.exchange, tr.ChatCompletionsExchange) + source.exchange.request["messages"].append(deepcopy(history.messages[-1])) + history.message_sources[-1] = tr.ChatCompletionsMessageSource( + exchange=source.exchange, request_index=1 + ) + if case == "no_native_prompt": + assert source is not None + assert isinstance(source.exchange, tr.ChatCompletionsExchange) + extra = source.exchange.response.choices[0].model_extra + assert extra is not None + extra.pop("prompt_token_ids") + if case == "visible_only": + cast(dict[str, Any], history.messages[-1]).pop("reasoning") + original = history.model_dump(mode="python") + candidate = _outcome(history, tokenizer) + calls = deepcopy(tokenizer.calls) + tokenizer.calls.clear() + with monkeypatch.context() as patch: + patch.setattr( + _tokenize, "_preserve_literal_thinking_off_content", lambda *args: None + ) + baseline = _outcome(history, tokenizer) + assert candidate == baseline + assert len(calls) == len(tokenizer.calls) + assert calls[0] == tokenizer.calls[0] + assert history.model_dump(mode="python") == original + + +@pytest.mark.parametrize("field", ["reasoning_content", "reasoning"]) +def test_explicit_empty_reasoning_is_preserved(field: str) -> None: + history, tokenizer = _history(reasoning="", reasoning_field=field) + original = history.model_dump(mode="python") + tokenized = history.tokenize(tokenizer=tokenizer) + assert ( + "".join( + chr(token) + for token, flag in zip(tokenized.tokens, tokenized.flags, strict=True) + if flag & tr.TokenFlag.SAMPLED + ) + == _LITERAL + ) + assert tokenizer.calls[0][-1]["reasoning_content"] == "" + assert history.model_dump(mode="python") == original + + +def test_literal_adaptation_does_not_relax_source_validation() -> None: + history, tokenizer = _history() + cast(dict[str, Any], history.messages[-1])["content"] += "not in source" + with pytest.raises(ValueError, match="source"): + history.tokenize(tokenizer=tokenizer) + assert not tokenizer.calls + + +@pytest.mark.parametrize("earlier_thinking", [True, None]) +def test_mixed_history_uses_each_generations_own_request( + earlier_thinking: bool | None, +) -> None: + first, tokenizer = _history(thinking=earlier_thinking) + second, _ = _history() + history = tr.ChatCompletionsHistory( + model=second.model, + messages=[*first.messages, *second.messages], + message_sources=[*first.message_sources, *second.message_sources], + chat_template=second.chat_template, + chat_template_kwargs=second.chat_template_kwargs, + ) + original = history.model_dump(mode="python") + _outcome(history, tokenizer) + rendered_messages = tokenizer.calls[0] + assert "reasoning_content" not in rendered_messages[1] + assert rendered_messages[3]["reasoning_content"] == "" + assert [message["content"] for message in rendered_messages] == [ + message["content"] for message in history.messages + ] + assert history.model_dump(mode="python") == original + + +class _NewlineRunTokenizer(_TemplateTokenizer): + """Reversible public codec that exposes the open/closed scaffold boundary.""" + + all_special_tokens = ["<|im_start|>", "<|im_end|>", "", ""] + all_special_ids = [200000, 200001, 200002, 200003] + eos_token_id = 200001 + + def __call__(self, text: str, **kwargs: Any) -> dict[str, Any]: + pieces = list( + re.finditer(r"<\|im_start\|>|<\|im_end\|>|||\n+|[^\n]", text) + ) + result = { + "input_ids": [ + self.all_special_ids[self.all_special_tokens.index(piece.group())] + if piece.group() in self.all_special_tokens + else 300000 + len(piece.group()) + if piece.group().startswith("\n") + else ord(piece.group()) + for piece in pieces + ] + } + if kwargs.get("return_offsets_mapping"): + result["offset_mapping"] = [piece.span() for piece in pieces] + return result + + def decode(self, token_ids: list[int], **kwargs: Any) -> str: + return "".join( + self.all_special_tokens[self.all_special_ids.index(token)] + if token in self.all_special_ids + else "\n" * (token - 300000) + if token > 300000 + else chr(token) + for token in token_ids + ) + + def convert_tokens_to_ids(self, token: str) -> int | None: + return ( + self.all_special_ids[self.all_special_tokens.index(token)] + if token in self.all_special_tokens + else None + ) + + def apply_chat_template( + self, messages: list[dict[str, Any]], *, tokenize: bool = True, **kwargs: Any + ) -> str | list[int]: + text = super().apply_chat_template(messages, tokenize=False, **kwargs) + assert isinstance(text, str) + return self(text)["input_ids"] if tokenize else text + + +def test_literal_next_turn_preserves_preceding_length_stop_boundary( + monkeypatch: pytest.MonkeyPatch, +) -> None: + tokenizer = _NewlineRunTokenizer() + messages = [ + {"role": "user", "content": "first"}, + {"role": "assistant", "content": "unchanged preceding output"}, + {"role": "user", "content": "next query"}, + {"role": "assistant", "content": _LITERAL}, + ] + exchanges = [] + for index in (1, 3): + single, _ = _history(content=messages[index]["content"]) + source = single.message_sources[-1] + assert source is not None and isinstance( + source.exchange, tr.ChatCompletionsExchange + ) + exchange = source.exchange + exchange.request["messages"] = cast( + list[ChatCompletionMessageParam], deepcopy(messages[:index]) + ) + data = exchange.response.model_dump(mode="python") + data["id"] = f"public-length-{index}" + choice = data["choices"][0] + choice["prompt_token_ids"] = tokenizer.apply_chat_template( + messages[:index], + add_generation_prompt=True, + enable_thinking=False, + preserve_thinking=True, + ) + choice["token_ids"] = tokenizer(messages[index]["content"])["input_ids"] + choice["logprobs"]["content"] = [ + { + "token": f"token_id:{token}", + "logprob": -0.5, + "bytes": [], + "top_logprobs": [], + } + for token in choice["token_ids"] + ] + exchange.response = ChatCompletion.model_validate(data) + exchanges.append(exchange) + history = tr.Trajectory( + exchanges=tr.TrajectoryExchanges(chat_completions=exchanges) + ).chat_completions_history() + original = history.model_dump(mode="python") + first = exchanges[0].response.choices[0].model_extra + last = exchanges[1].response.choices[0].model_extra + assert first is not None and last is not None + end = len(first["prompt_token_ids"]) + len(first["token_ids"]) + native_boundary = last["prompt_token_ids"][end:] + assert ( + last["prompt_token_ids"][:end] == first["prompt_token_ids"] + first["token_ids"] + ) + source = history.message_sources[1] + key = _tokenize._sampled_source_key(source) + builder = _tokenize._tokenize_exact_projected_chat_history + observed = [] + + def observe(*args: Any, **kwargs: Any) -> tr.TokenizedHistory | None: + result = builder(*args, **kwargs) + if boundary := kwargs.get("length_stop_boundaries", {}).get(key): + observed.append((boundary, result)) + return result + + monkeypatch.setattr(_tokenize, "_tokenize_exact_projected_chat_history", observe) + with monkeypatch.context() as patch: + patch.setattr( + _tokenize, "_preserve_literal_thinking_off_content", lambda *args: None + ) + _outcome(history, tokenizer) + boundary, old_exact = observed[0] + stored = list(boundary.tail + boundary.following) + assert old_exact is None + assert len(native_boundary) - len(stored) == 2 + assert stored[:-1] == native_boundary[:-3] + assert tokenizer.decode(stored[-1:]) == "\n" + assert tokenizer.decode(native_boundary[-3:]) == "\n\n\n\n" + observed.clear() + + value = history.tokenize(tokenizer=tokenizer) + fixed_boundary, fixed_exact = observed[0] + assert fixed_exact is value + assert list(fixed_boundary.tail + fixed_boundary.following) == native_boundary + assert ( + value.tokens[: len(last["prompt_token_ids"]) + len(last["token_ids"])] + == last["prompt_token_ids"] + last["token_ids"] + ) + assert value.flags[end] == tr.TokenFlag.EXACT | tr.TokenFlag.STOP + assert not value.flags[end] & tr.TokenFlag.SAMPLED + for exchange in exchanges: + extra = exchange.response.choices[0].model_extra + assert extra is not None + start = len(extra["prompt_token_ids"]) + stop = start + len(extra["token_ids"]) + assert value.tokens[start:stop] == extra["token_ids"] + assert value.logprobs[start:stop] == [-0.5] * (stop - start) + assert all( + flag + == tr.TokenFlag.EXACT + | tr.TokenFlag.SAMPLED + | tr.TokenFlag.ASSISTANT + | tr.TokenFlag.OUTPUT + for flag in value.flags[start:stop] + ) + assert history.model_dump(mode="python") == original