Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 1 addition & 4 deletions .dialyzer_ignore.exs
Original file line number Diff line number Diff line change
Expand Up @@ -52,13 +52,10 @@
{"bench/imp/benchmark_truth/rlm_protocol.ex", :unused_fun, {133, 8}},
# defensive guard success typing proves redundant
{"bench/imp/benchmark_truth/runner.ex", :guard_fail, {651, 55}},
# defensive clause: ReqLLM.model/1 contracts to ok/error tuples only; the
# catch-all turns any unexpected registry result into a loud error (#75)
{"lib/imp/clients/req_llm.ex", :pattern_match_cov, {163, 7}},
# defensive guard: ReqLLM.Response types provider_meta as map() with a %{}
# default, but the struct does not enforce it (a caller can build one with
# nil), and ReqLLM's own OpenTelemetry attributes guard it with is_map/1.
{"lib/imp/clients/req_llm.ex", :guard_fail, 1868},
{"lib/imp/clients/req_llm.ex", :guard_fail, 1902},
# defensive error clause on an always-ok internal call
{"lib/imp/clients/training.ex", :pattern_match, {1215, 13}},
# defensive error clause on an always-ok internal call
Expand Down
11 changes: 11 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -23,6 +23,17 @@ User-visible changes to Imp are recorded here.
unknown charge, not a free one; a host that wants the old number for calls
with no reported charge reads `estimated_cost` for them, knowing it is an
estimate.
- `:reasoning_effort` accepts `max`, when an LM is built, on a call and in a
saved program. Imp's accepted efforts are read from ReqLLM's own
`reasoning_effort` option, so they are every effort ReqLLM accepts, on every
provider; ReqLLM's provider maps or clamps it to what that provider's API
takes. OpenRouter receives `"max"` in either wire field.
- A string effort such as `"high"` reaches ReqLLM as its atom. ReqLLM checks
the effort against its atom list before any provider sees it, so in 0.6.0 a
string effort, including every effort loaded from a saved program, failed
every call on most providers: OpenRouter without
`openrouter_reasoning_wire: :nested`, Anthropic, Google and Groq among
them. OpenAI and xAI were not affected.

## 0.6.0 — 2026-09-28

Expand Down
48 changes: 41 additions & 7 deletions lib/imp/clients/req_llm.ex
Original file line number Diff line number Diff line change
Expand Up @@ -26,11 +26,13 @@ defmodule Imp.Clients.ReqLLM do
completion.

`:reasoning_effort` is the one reasoning option, on the client or on a call.
It takes `none`, `minimal`, `low`, `medium`, `high`, `xhigh` or `default`, as
an atom or a string. A call naming `nil` spends no reasoning on that call
whatever the client is configured with. Native reasoning fields
(`Imp.Predict`) set the same option, so a client configured with an effort
and a program that asks for one never disagree.
It takes any value of ReqLLM's own `reasoning_effort` option, such as `high`,
`xhigh`, `max` or `default`, as an atom or a string, on every provider;
ReqLLM's provider maps or clamps it to what that provider's API takes. A
call naming `nil` spends no reasoning on that call whatever the client is
configured with. Native reasoning fields (`Imp.Predict`) set the same
option, so a client configured with an effort and a program that asks for
one never disagree.

OpenRouter accepts the effort in two wire fields, and its endpoint catalog
says which one an endpoint supports: ReqLLM's top-level `reasoning_effort`
Expand Down Expand Up @@ -156,11 +158,12 @@ defmodule Imp.Clients.ReqLLM do

defp resolve_model(%{capabilities: _} = model), do: {:ok, model}

# ReqLLM.model/1 returns ok/error tuples. Any other result is a
# CaseClauseError, which the rescue turns into an error like any other.
defp resolve_model(model_spec) do
case ReqLLM.model(model_spec) do
{:ok, model} -> {:ok, model}
{:error, reason} -> {:error, reason}
other -> {:error, {:unexpected_registry_result, other}}
end
rescue
error -> {:error, error}
Expand Down Expand Up @@ -351,6 +354,7 @@ defmodule Imp.Clients.ReqLLM do
opts =
opts
|> encode_openrouter_reasoning(lm.model)
|> atomize_reasoning_effort()
|> cap_transport_timeouts()
|> bind_to_caller()
|> keep_error_headers()
Expand Down Expand Up @@ -725,6 +729,7 @@ defmodule Imp.Clients.ReqLLM do
opts =
opts
|> encode_openrouter_reasoning(lm.model)
|> atomize_reasoning_effort()
|> cap_transport_timeouts()
|> enforce_explicit_no_retry()

Expand Down Expand Up @@ -823,7 +828,20 @@ defmodule Imp.Clients.ReqLLM do
end
end

@reasoning_efforts ~w(none minimal low medium high xhigh default)
# The accepted efforts are ReqLLM's own `reasoning_effort` option, read from
# its generation schema so Imp keeps no second list. The match fails the
# build if ReqLLM changes the option's shape.
{:in, req_llm_efforts} =
ReqLLM.Provider.Options.generation_schema().schema
|> Keyword.fetch!(:reasoning_effort)
|> Keyword.fetch!(:type)

@reasoning_efforts Enum.map(req_llm_efforts, &Atom.to_string/1)

@doc false
@spec reasoning_efforts() :: [String.t()]
def reasoning_efforts, do: @reasoning_efforts

@reasoning_wires ~w(top_level nested)

defp normalize_reasoning_effort_option!(opts, context) do
Expand Down Expand Up @@ -886,6 +904,22 @@ defmodule Imp.Clients.ReqLLM do
":reasoning_effort must be one of #{inspect(@reasoning_efforts)}, got: #{inspect(effort)}"
end

# ReqLLM checks `:reasoning_effort` against its atom list before most
# providers (OpenRouter, Anthropic, Google, Groq) see it; only OpenAI, xAI
# and Meta turn the string form into the atom first. A string effort would
# fail ReqLLM's option validation on the others. The effort is kept as
# given in the client's options, where it is part of the cache key, and
# becomes the atom only on its way into ReqLLM.
defp atomize_reasoning_effort(opts) do
case Keyword.fetch(opts, :reasoning_effort) do
{:ok, effort} when is_binary(effort) and effort in @reasoning_efforts ->
Keyword.put(opts, :reasoning_effort, String.to_existing_atom(effort))

_other ->
opts
end
end

# With `openrouter_reasoning_wire: :nested` the effort leaves the ReqLLM
# options here and a request step writes it into the body as the nested
# `reasoning` object. Otherwise ReqLLM's OpenRouter provider sends the
Expand Down
12 changes: 6 additions & 6 deletions lib/imp/saving.ex
Original file line number Diff line number Diff line change
Expand Up @@ -1586,13 +1586,13 @@ defmodule Imp.Saving do
"saved ReqLLM input_envelope must be a list, got: #{inspect(value)}"
end

defp require_reasoning_effort!(effort)
when effort in ~w(none minimal low medium high xhigh default),
do: effort

defp require_reasoning_effort!(effort) do
raise ArgumentError,
"saved ReqLLM reasoning_effort is unsupported: #{inspect(effort)}"
if is_binary(effort) and effort in Imp.Clients.ReqLLM.reasoning_efforts() do
effort
else
raise ArgumentError,
"saved ReqLLM reasoning_effort is unsupported: #{inspect(effort)}"
end
end

defp decode_req_http_options!(options) when is_list(options) do
Expand Down
146 changes: 145 additions & 1 deletion test/req_llm_client_test.exs
Original file line number Diff line number Diff line change
Expand Up @@ -1381,7 +1381,7 @@ defmodule ReqLLMClientTest do

assert_received {:req_llm_generate, "anthropic:claude-sonnet-4-6", messages, opts}
refute inspect(messages) =~ "reasoning"
assert Keyword.fetch!(opts, :reasoning_effort) == "low"
assert Keyword.fetch!(opts, :reasoning_effort) == :low
end

test "manual reasoning fields still work without provider-native thinking" do
Expand Down Expand Up @@ -2315,6 +2315,150 @@ defmodule ReqLLMClientTest do
assert top_level.opts[:reasoning_effort] == :low
end

test "reasoning_effort max reaches OpenRouter as \"max\" on either wire" do
owner = self()

adapter = fn request ->
send(owner, {:openrouter_max_transport, request.body})

body = %{
"id" => "openrouter-max-local",
"object" => "chat.completion",
"model" => "provider/snapshot",
"choices" => [
%{
"index" => 0,
"message" => %{"role" => "assistant", "content" => "ok"},
"finish_reason" => "stop"
}
],
"usage" => %{"prompt_tokens" => 1, "completion_tokens" => 1, "total_tokens" => 2}
}

{request, Req.Response.new(status: 200, body: body)}
end

model = %{
provider: :openrouter,
id: "provider/model",
model: "provider/model",
base_url: "https://provider-disabled.invalid/v1"
}

transport = [
api_key: "provider-disabled",
cache: false,
max_retries: 0,
req_http_options: [adapter: adapter, retry: false, max_retries: 0]
]

messages = [%{role: :user, content: "reason"}]

# Configured on the client, as a string: ReqLLM's top-level field.
configured = Imp.req_llm(model, [reasoning_effort: "max"] ++ transport)
assert {:ok, _response} = Imp.Clients.ReqLLM.generate(configured, messages, [])
assert_received {:openrouter_max_transport, request_body}
request = request_body |> IO.iodata_to_binary() |> Jason.decode!()
assert request["reasoning_effort"] == "max"

# Named on a call, as an atom, over a client with no effort.
plain = Imp.req_llm(model, transport)

assert {:ok, _response} =
Imp.Clients.ReqLLM.generate(plain, messages, reasoning_effort: :max)

assert_received {:openrouter_max_transport, request_body}
request = request_body |> IO.iodata_to_binary() |> Jason.decode!()
assert request["reasoning_effort"] == "max"

# The nested wire carries the same value.
nested =
Imp.req_llm(
model,
[reasoning_effort: :max, openrouter_reasoning_wire: :nested] ++ transport
)

assert {:ok, _response} = Imp.Clients.ReqLLM.generate(nested, messages, [])
assert_received {:openrouter_max_transport, request_body}
request = request_body |> IO.iodata_to_binary() |> Jason.decode!()
assert request["reasoning"] == %{"effort" => "max"}
refute Map.has_key?(request, "reasoning_effort")
end

test "a string reasoning_effort reaches OpenRouter and Anthropic requests" do
owner = self()

adapter = fn request ->
send(owner, {:string_effort_transport, request.body})

body =
if String.contains?(to_string(request.url), "anthropic") do
%{
"id" => "msg_local",
"type" => "message",
"role" => "assistant",
"model" => "claude-sonnet-4-6",
"content" => [%{"type" => "text", "text" => "ok"}],
"stop_reason" => "end_turn",
"usage" => %{"input_tokens" => 1, "output_tokens" => 1}
}
else
%{
"id" => "openrouter-string-local",
"object" => "chat.completion",
"model" => "provider/snapshot",
"choices" => [
%{
"index" => 0,
"message" => %{"role" => "assistant", "content" => "ok"},
"finish_reason" => "stop"
}
],
"usage" => %{"prompt_tokens" => 1, "completion_tokens" => 1, "total_tokens" => 2}
}
end

{request, Req.Response.new(status: 200, body: body)}
end

transport = [
api_key: "provider-disabled",
cache: false,
max_retries: 0,
reasoning_effort: "high",
req_http_options: [adapter: adapter, retry: false, max_retries: 0]
]

messages = [%{role: :user, content: "reason"}]

openrouter =
Imp.req_llm(
%{
provider: :openrouter,
id: "provider/model",
model: "provider/model",
base_url: "https://openrouter.invalid/v1"
},
transport
)

assert {:ok, _response} = Imp.Clients.ReqLLM.generate(openrouter, messages, [])
assert_received {:string_effort_transport, request_body}
request = request_body |> IO.iodata_to_binary() |> Jason.decode!()
assert request["reasoning_effort"] == "high"

anthropic =
Imp.req_llm(
"anthropic:claude-sonnet-4-6",
[base_url: "https://anthropic.invalid"] ++ transport
)

assert {:ok, _response} = Imp.Clients.ReqLLM.generate(anthropic, messages, [])
assert_received {:string_effort_transport, request_body}
request = request_body |> IO.iodata_to_binary() |> Jason.decode!()
assert request["thinking"] == %{"type" => "enabled", "budget_tokens" => 4096}
end

test "reasoning_effort rejects unknown values and duplicates early, on any provider" do
for effort <- [:invented, "invented", %{effort: :high}, 3] do
assert_raise ArgumentError, ~r/:reasoning_effort/, fn ->
Expand Down
8 changes: 8 additions & 0 deletions test/saving_req_llm_transport_test.exs
Original file line number Diff line number Diff line change
Expand Up @@ -107,6 +107,14 @@ defmodule Imp.SavingReqLLMTransportTest do
assert loaded.lm.opts[:reasoning_effort] == "high"
assert loaded.lm.opts[:openrouter_reasoning_wire] == :nested

max =
Imp.predict("question -> answer",
lm: Imp.req_llm("openrouter:provider/model", reasoning_effort: :max)
)

assert Imp.load!(max |> Imp.dump() |> json_round_trip()).lm.opts[:reasoning_effort] ==
"max"

root = tmp_dir("openrouter-reasoning-fresh-beam")
artifact = Path.join(root, "program.json")
receipt = Path.join(root, "receipt.bin")
Expand Down
Loading