diff --git a/.dialyzer_ignore.exs b/.dialyzer_ignore.exs index ede8f606..2c3d7955 100644 --- a/.dialyzer_ignore.exs +++ b/.dialyzer_ignore.exs @@ -57,9 +57,9 @@ # defensive clause/guard: ReqLLM.Response types `usage` as map() on the # struct, but its schema defaults the field to nil and Response.usage/1 is # `map() | nil`, so the nil clause is reachable at runtime. - {"lib/imp/clients/req_llm.ex", :guard_fail, 1414}, - {"lib/imp/clients/req_llm.ex", :pattern_match_cov, {1442, 8}}, - {"lib/imp/clients/req_llm.ex", :pattern_match_cov, {147, 7}}, + {"lib/imp/clients/req_llm.ex", :guard_fail, 1448}, + {"lib/imp/clients/req_llm.ex", :pattern_match_cov, {1476, 8}}, + {"lib/imp/clients/req_llm.ex", :pattern_match_cov, {150, 7}}, # defensive error clause on an always-ok internal call {"lib/imp/clients/training.ex", :pattern_match, {1215, 13}}, # defensive error clause on an always-ok internal call @@ -219,7 +219,7 @@ # Defensive fallbacks and MapSet opacity retained at the 0.3 cut. These are # individually pinned so a changed success type makes the gate ask again. {"bench/imp/benchmark_truth/multimodal_runner.ex", :pattern_match_cov, {340, 16}}, - {"lib/imp/adapter/chat.ex", :pattern_match_cov, {774, 8}}, + {"lib/imp/adapter/chat.ex", :pattern_match_cov, {787, 8}}, {"lib/imp/adapter/xml.ex", :pattern_match_cov, {675, 8}}, {"lib/imp/mcp.ex", :pattern_match_cov, {372, 8}}, {"lib/imp/optimizer/artifact.ex", :call_without_opaque, {745, 52}}, diff --git a/CHANGELOG.md b/CHANGELOG.md index 476a65b4..6f80cc04 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,33 +4,42 @@ User-visible changes to Imp are recorded here. ## Unreleased -- A `ReActV2` turn now ends when the model stops calling tools. A step that - comes back as prose with no tool call, for a task signature with exactly one - output of type `:string`, finishes the run with that prose as the output and - `termination_reason: :answered`, in that one request. It used to cost one more - request with `tool_choice` naming `submit`, which both spent a call and, when - the model had already acted with a tool, came back with a summary of what it - did rather than what it said. Every other mainstream loop — Anthropic's tool - runner, the OpenAI Agents SDK, LangGraph's ReAct, Pydantic AI — ends the turn - this way, so it is the default. A signature with several outputs, or one - non-text output, still takes the forced submit, because prose cannot fill - those fields, and so does a step that says nothing at all. The new option - `prose: :forced_submit` keeps the old behaviour for a single-output - signature. -- `ReActV2` gains `on_max_iters`, what the step limit does. The default, - `:forced_submit`, is what it did before: one more request with `tool_choice` - naming `submit`. `:last_prose` makes one more request with no tools in it at - all, so the only thing the model can do is speak, and that prose is the - single text output, with `termination_reason: :last_prose`. This is the - ending that fits a host whose model already finishes turns by writing prose: - it is never asked to call a tool it did not choose. A completion that says - nothing finishes with an empty answer rather than an error. `:last_prose` - needs a signature with exactly one output of type `:string`, and is refused - at construction otherwise. The companion option `last_prose_note`, a string, - puts one line of host text in front of that request as a user message and - keeps it in the returned history; Imp writes no sentence of its own. Both - options persist through `dump`/`load`, and a dump written before them loads - as `:forced_submit`. +- `Imp.Clients.ReqLLM` returns a response whose body carries a provider error + as `{:error, %ReqLLM.Error.API.Request{}}`. OpenRouter relays an upstream + provider's refusal as a successful HTTP response with an error object and no + choices, which ReqLLM decodes to an empty message; read as a completion, a + refused request was a model that said nothing. +- `ReActV2` offers `submit` only to a signature that needs it. A task + signature with exactly one output of type `:string` gets no `submit` tool: + a step that comes back as prose with no tool call is the answer, in that + one request, with `termination_reason: :answered`, which is how Anthropic's + tool runner, the OpenAI Agents SDK, LangGraph's ReAct and Pydantic AI end a + turn. Its history event carries the output, as a `submit`'s does, and the + answer is not also emitted as a `:reasoning` event. A signature with several + outputs, or one non-text output, keeps DSPy's `submit` unchanged. +- A step of a one-text-output signature that calls nothing and says nothing + is an empty answer: the turn ends there with `termination_reason: + :answered` and no further request, because saying nothing is how a model + declines to answer. +- An interrupted turn of a one-text-output signature (the step limit, a + failed request, prose the output does not accept) makes one more + request with the same tools as every step and `tool_choice: "none"`, so the + model can only write text, and that text is the answer, with `termination_reason: :last_prose` + and `termination_cause` naming the interruption (`:max_iters`, + `:prediction_error`, `:parse_error`, `:invalid_answer`). A completion that says nothing is an empty answer rather + than an error. A tool call the model makes on that request anyway is not + run; the text is the answer and the calls are listed in + `unexecuted_tool_calls`. `last_prose_note`, a string, puts one line of host text in + front of that request as a user message and keeps it in the returned + history; Imp writes no sentence of its own. If the process's `Imp.Deadline` + has already passed, no request is made and the run ends with + `termination_reason: :deadline_exceeded`. `forced_submit_notice` is for + signatures with `submit` and `last_prose_note` for those without; each is + refused at construction for the other. There is no `prose` or + `on_max_iters` option, and a dump no longer carries them. +- `Imp.Observability` reports a prediction that ended `:answered`, + `:last_prose` or `:finished_by_tool` as complete; it reported them as + incomplete. - `ReActV2` gains `finish_on`, a map from tool name to `fn arguments, result, inputs -> {:finish, outputs} | :continue end`. A tool named there ends the turn with the outputs the function returns, which are diff --git a/docs/IMP_FOR_DSPY_USERS.md b/docs/IMP_FOR_DSPY_USERS.md index bc7e7f8f..1aaf8a90 100644 --- a/docs/IMP_FOR_DSPY_USERS.md +++ b/docs/IMP_FOR_DSPY_USERS.md @@ -17,7 +17,7 @@ updates, telemetry, and fresh-runtime artifact application. | `dspy.Signature` / `"q -> a"` | `Imp.signature("q -> a")` — the same compact input/output idea with Imp type spellings such as `array[...]`, `enum[...]`, and `number` | | `dspy.Predict(sig)` | `Imp.predict(sig, lm: lm)` | | `dspy.ChainOfThought` | `Imp.chain_of_thought/2` | -| `dspy.ReAct(sig, tools=[...])` | `Imp.react_v2(sig, tools, tool_policy: [...])` — typed tools, structured observations, and validated `submit` | +| `dspy.ReAct(sig, tools=[...])` | `Imp.react_v2(sig, tools, tool_policy: [...])` — typed tools, structured observations, and validated `submit` for several or typed outputs; one text output is answered in prose | | `dspy.Example` / `.with_inputs` | `Imp.example/1` / `Imp.with_inputs/2` | | `dspy.Prediction` | `%Imp.Prediction{}` — read fields with `Imp.get/2` | | `dspy.Evaluate` | `Imp.evaluate/4` — returns score plus per-example rows | diff --git a/docs/differentials/REACT_V2_FIDELITY.md b/docs/differentials/REACT_V2_FIDELITY.md index 5730c4ef..5f2f403f 100644 --- a/docs/differentials/REACT_V2_FIDELITY.md +++ b/docs/differentials/REACT_V2_FIDELITY.md @@ -41,9 +41,8 @@ existing fail-fast `Imp.Predict.ReAct`. | History is structured rather than one growing trajectory string | `Imp.History` stores per-turn inputs, thought, typed calls, call results, and final fields | parallel, failure, serialization, and adapter replay tests | | Parallel tool calls preserve IDs and execute all calls | Every missing ID receives `call__`; results retain the corresponding ID | parallel call test | | Unknown tools and execution failures become observations | ReActV2 records error results and continues; existing ReAct remains fail-fast | recovery test | -| `submit` is reserved and validates final outputs | Constructor rejects user `submit`; the generated submit tool uses the task JSON schema | reserved-submit and missing-output tests | -| Empty calls, parse failure, context exhaustion, or budget exhaustion force one submit call | Parse failure, context exhaustion and budget exhaustion still do: the final predictor call pins provider `tool_choice` to `submit` and clears `reasoning_effort`, matching the pinned call configuration. A step of prose with no tool call does not, when the task declares exactly one text output: that prose is the output and the turn is over, as it is in Anthropic's tool runner, the OpenAI Agents SDK, LangGraph's ReAct and Pydantic AI. `prose: :forced_submit` restores the upstream shape | forced-submit test, prose-answer test | -| No upstream equivalent | `on_max_iters: :last_prose` ends a turn that reaches the step limit with one request that carries no tools, so the model can only speak; that prose is the single text output, with `termination_reason: :last_prose`, and an empty completion is an empty answer. `:last_prose_note` puts one line of host text in front of that request. The default stays the forced submit | `test/react_v2_last_prose_test.exs` | +| `submit` is reserved and validates final outputs | The name is reserved for every signature. A signature with several outputs, or one output that is not text, gets the upstream `submit` tool, built from the task JSON schema, with upstream's description and guidance text. A signature with exactly one output of type `:string` gets no `submit`: its answer is the prose the model writes when it stops calling tools (`termination_reason: :answered`), as in Anthropic's tool runner, the OpenAI Agents SDK, LangGraph's ReAct and Pydantic AI, and its guidance says to answer in plain text. Upstream needs `submit` because a signature can have several typed outputs; one text output does not | reserved-submit, missing-output and no-submit tests, `test/react_v2_request_shape_test.exs` | +| Empty calls, parse failure, context exhaustion, or budget exhaustion force one submit call | With `submit`, parse failure, empty calls and budget exhaustion still do: the final predictor call pins provider `tool_choice` to `submit` and clears `reasoning_effort`, matching the pinned call configuration. With one text output there is no `submit` to force, so every interruption makes one request with the same tools and `tool_choice: "none"` (encoded `"none"` for OpenAI and OpenRouter and `{"type": "none"}` for Anthropic; a call made anyway is not run and is named in `unexecuted_tool_calls`), and its text is the answer (`termination_reason: :last_prose`, `termination_cause` naming the interruption); `:last_prose_note` puts one line of host text in front of it, and a passed `Imp.Deadline` ends the run with `:deadline_exceeded` instead of making the request. Context exhaustion ends the run at once in both cases, since a further request would be refused the same way | forced-submit tests, `test/react_v2_last_prose_test.exs`, `test/react_v2_last_request_wire_test.exs` | | No upstream equivalent | `finish_on` names tools that end the turn with the outputs they carry, the shape Pydantic AI calls an output tool | `finish_on` tests | | Prior calls replay as native assistant/tool messages | Chat adapter emits assistant `tool_calls` and matching tool-result messages by call ID | native history adapter test and ReqLLM tests | diff --git a/examples/deployment/agent_optimization.exs b/examples/deployment/agent_optimization.exs index 93f9a3e6..f4078cd5 100644 --- a/examples/deployment/agent_optimization.exs +++ b/examples/deployment/agent_optimization.exs @@ -255,7 +255,7 @@ defmodule ImpDeployment.AgentOptimization.Runner do actions = events - |> Enum.filter(&(&1.kind == :tool_call and to_string(&1.tool_name) != "submit")) + |> Enum.filter(&(&1.kind == :tool_call)) |> Enum.map(&to_string(&1.tool_name)) errors = Enum.filter(events, &(&1.kind == :run_failed or not is_nil(&1.error))) @@ -285,7 +285,7 @@ defmodule ImpDeployment.AgentOptimization.Runner do grounded? = expected_result? and String.contains?(answer, row.account_id) clean? = errors == [] - valid_final? = completed and answer != "" and termination_reason == :submit + valid_final? = completed and answer != "" and termination_reason == :answered score = if expected? do @@ -307,7 +307,7 @@ defmodule ImpDeployment.AgentOptimization.Runner do if(expected_result?, do: nil, else: "expected action did not return its sandbox result"), if(valid_final? and clean? and grounded?, do: nil, - else: "run did not submit a clean answer grounded in account #{row.account_id}" + else: "run did not answer cleanly, grounded in account #{row.account_id}" ) ] |> Enum.reject(&is_nil/1) diff --git a/examples/workspace_agent/lib/workspace_agent.ex b/examples/workspace_agent/lib/workspace_agent.ex index c81283ca..b2be504d 100644 --- a/examples/workspace_agent/lib/workspace_agent.ex +++ b/examples/workspace_agent/lib/workspace_agent.ex @@ -202,8 +202,8 @@ defmodule WorkspaceAgent do "observed files and cite relative file paths. You may create files, replace exact " <> "text, and run argument-vector commands when the task requires it; those effects " <> "require explicit client approval. Start with the README, inspect only what is needed, " <> - "make the smallest coherent change, run the relevant check, then synthesize and call " <> - "submit. Use the provider's named function calls rather than serializing a tool call " <> + "make the smallest coherent change, run the relevant check, then write the answer " <> + "as plain text without calling a tool. Use the provider's named function calls rather than serializing a tool call " <> "as response text, and pass a JSON object matching the selected tool schema. For " <> "commands, pass the executable once, for example run_command with " <> "{\"command\":\"cat\",\"args\":[\"README.md\"]}; never repeat the executable inside " <> @@ -303,9 +303,7 @@ defmodule WorkspaceAgent do :none -> "No workspace content was observed." end - tool_turn("Return only the grounded smoke result.", "submit", "smoke-submit", %{ - answer: answer - }) + answer end end ) diff --git a/examples/workspace_agent/test/workspace_agent_test.exs b/examples/workspace_agent/test/workspace_agent_test.exs index 7c42d50b..c1a06021 100644 --- a/examples/workspace_agent/test/workspace_agent_test.exs +++ b/examples/workspace_agent/test/workspace_agent_test.exs @@ -14,7 +14,7 @@ defmodule WorkspaceAgentTest do %{root: root} end - test "provider-free ReAct factory reads and submits workspace evidence", %{root: root} do + test "provider-free ReAct factory reads and answers with workspace evidence", %{root: root} do assert {:ok, program, %{cleanup: cleanup}} = WorkspaceAgent.program(%{cwd: root}, provider: :static, program: :react) diff --git a/lib/imp/adapter/chat.ex b/lib/imp/adapter/chat.ex index 3bc4ab28..a1936547 100644 --- a/lib/imp/adapter/chat.ex +++ b/lib/imp/adapter/chat.ex @@ -102,7 +102,8 @@ defmodule Imp.Adapter.Chat do output: output_renderer, input_section: input_renderer, tool_result: tool_result_renderer, - history_note: Keyword.get(opts, :history_note_renderer) || (&no_history_note/2) + history_note: Keyword.get(opts, :history_note_renderer) || (&no_history_note/2), + submit_is_text?: submit_is_text?(Keyword.get(opts, :guidance)) } {history_messages, history_fields} = extract_history(signature, inputs, renderers) @@ -618,18 +619,30 @@ defmodule Imp.Adapter.Chat do end # Renders loop guidance passed as data, appended after the program's own - # instructions so the two have separate owners. + # instructions so the two have separate owners. With a finish tool the text + # is DSPy ReActV2's, byte for byte. A `finish_tool` of nil means the loop has + # no finish tool and the answer is the text the model writes when it stops + # calling tools, so that one line says so instead; this is Imp's divergence + # for a signature with one text output (`Imp.Predict.ReActV2`). defp with_guidance(instructions, nil), do: instructions defp with_guidance(instructions, %{} = guidance) do names = fn key -> guidance |> Map.get(key, []) |> Enum.map_join(", ", &"`#{&1}`") end - finish = Map.get(guidance, :finish_tool, :submit) + + finish = + case Map.get(guidance, :finish_tool, :submit) do + nil -> + "When the final answer is ready, write it as plain text without calling a tool." + + tool -> + "When the final answer is ready, call `#{tool}` with #{names.(:output_names)}." + end """ #{instructions} You are an Agent. Use the supplied tools to produce #{names.(:output_names)} from #{names.(:input_names)}. Call tools when more information is needed. - When the final answer is ready, call `#{finish}` with #{names.(:output_names)}. + #{finish} The available tools are: #{names.(:tool_names)}. """ |> String.trim() @@ -1053,7 +1066,7 @@ defmodule Imp.Adapter.Chat do messages = if native_tool_history_turn?(turn) do - render_native_tool_history_turn(signature, turn, renderers.tool_result) + render_native_tool_history_turn(signature, turn, renderers) else [ %{ @@ -1093,13 +1106,31 @@ defmodule Imp.Adapter.Chat do defp native_tool_history_turn?(turn), do: not is_nil(fetch_field(turn, :tool_calls)) - defp render_native_tool_history_turn(signature, turn, tool_result_renderer) do + # A loop whose guidance names no finish tool answers in plain text, and has + # no `submit` for a recorded call to name. + defp submit_is_text?(%{} = guidance), + do: Map.has_key?(guidance, :finish_tool) and is_nil(guidance.finish_tool) + + defp submit_is_text?(_guidance), do: false + + defp render_native_tool_history_turn(signature, turn, renderers) do calls = normalize_history_tool_calls(fetch_field(turn, :tool_calls)) results = List.wrap(fetch_field(turn, :tool_call_results)) + {calls, results, answer} = + if renderers.submit_is_text?, + do: submit_as_text(calls, results), + else: {calls, results, nil} + + tool_result_renderer = renderers.tool_result + user = %{ role: :user, - content: render_inputs(signature, turn, skip: history_input_fields(signature)) + content: + render_inputs(signature, turn, + skip: history_input_fields(signature), + section_renderer: renderers.input_section + ) } thought = turn |> fetch_field(:next_thought) |> blank_to_empty() @@ -1125,13 +1156,106 @@ defmodule Imp.Adapter.Chat do } end) - [user, assistant | tool_messages] + answered = + cond do + is_nil(answer) -> [] + calls == [] -> [] + true -> [%{role: :assistant, content: answer}] + end + + assistant = + if is_binary(answer) and calls == [], + do: %{assistant | content: join_text(thought, answer)}, + else: assistant + + ([user, assistant | tool_messages] ++ answered) |> Enum.reject(fn %{role: :assistant, tool_calls: calls} -> calls == [] message -> blank_message?(message) end) end + # A recorded `submit` call, replayed to a loop that has none. The call was + # the turn's answer, so it is shown as the answer: assistant text, with its + # result dropped. Shown as a call to a tool the request does not offer, some + # providers' models imitate it and write the raw tool-call markup as text. + # A submit the loop rejected was not the answer, so it is dropped with its + # error; the loop took the last accepted one, and so does this. A call + # recorded without an id is matched to its result by name. + defp submit_as_text(calls, results) do + {submits, calls} = Enum.split_with(calls, &submit_call?/1) + + case submits do + [] -> + {calls, results, nil} + + submits -> + {submit_results, results} = Enum.split_with(results, &submit_result?(&1, submits)) + + answer = + submits + |> Enum.zip(submit_results_in_order(submits, submit_results)) + |> Enum.reject(fn {_submit, result} -> rejected?(result) end) + |> List.last() + |> case do + nil -> nil + {submit, _result} -> submit |> get_in([:function, :arguments]) |> submitted_text() + end + + {calls, results, answer} + end + end + + defp submit_call?(call), do: get_in(call, [:function, :name]) == "submit" + + defp submit_result?(result, submits) do + case fetch_field(result, :id) do + nil -> to_string(fetch_field(result, :name)) == "submit" + id -> Enum.any?(submits, &(Map.get(&1, :id) == id)) + end + end + + # Each submit's result, or nil when none was recorded: by id when the call + # has one, otherwise the next id-less result, since the loop records results + # in call order. + defp submit_results_in_order(submits, submit_results) do + idless = Enum.filter(submit_results, &is_nil(fetch_field(&1, :id))) + + {paired, _idless} = + Enum.map_reduce(submits, idless, fn submit, idless -> + case Map.get(submit, :id) do + nil -> + case idless do + [result | rest] -> {result, rest} + [] -> {nil, []} + end + + id -> + {Enum.find(submit_results, &(fetch_field(&1, :id) == id)), idless} + end + end) + + paired + end + + defp rejected?(nil), do: false + + defp rejected?(result), + do: fetch_field(result, :error) == true or match?({:error, _}, fetch_field(result, :result)) + + defp submitted_text(%{} = arguments) do + case Map.values(arguments) do + [text] when is_binary(text) -> text + _other -> Jason.encode!(arguments) + end + end + + defp submitted_text(text) when is_binary(text), do: text + defp submitted_text(_arguments), do: "" + + defp join_text("", answer), do: answer + defp join_text(thought, answer), do: thought <> "\n\n" <> answer + defp normalize_history_tool_calls(%Imp.Adapter.Types.ToolCalls{tool_calls: calls}), do: Enum.map(calls, &Imp.Adapter.Types.ToolCall.format/1) diff --git a/lib/imp/clients/req_llm.ex b/lib/imp/clients/req_llm.ex index f84861ca..848008db 100644 --- a/lib/imp/clients/req_llm.ex +++ b/lib/imp/clients/req_llm.ex @@ -18,6 +18,9 @@ defmodule Imp.Clients.ReqLLM do `error.code = "context_length_exceeded"` become `Imp.ContextWindowExceededError`. Other provider errors retain their original shape; prose and generic HTTP 400 responses do not trigger context recovery. + A successful HTTP response whose body carries a provider error, which is how + OpenRouter relays an upstream refusal, is returned as that error + (`ReqLLM.Error.API.Request`), never as an empty completion. `:reasoning_effort` is the one reasoning option, on the client or on a call. It takes `none`, `minimal`, `low`, `medium`, `high`, `xhigh` or `default`, as @@ -317,7 +320,10 @@ defmodule Imp.Clients.ReqLLM do case lm.req_module.generate_text(lm.model, to_req_messages(messages), opts) do {:ok, response} -> - {:ok, from_response(response, lm.model)} + case relayed_error(response) do + nil -> {:ok, from_response(response, lm.model)} + error -> {:error, normalize_context_refusal(error)} + end {:error, reason} -> {:error, normalize_context_refusal(reason)} @@ -331,6 +337,34 @@ defmodule Imp.Clients.ReqLLM do kind, reason -> {:error, {:req_llm_generate_failed, error_message({kind, reason})}} end + # OpenRouter relays an upstream provider's refusal as a successful HTTP + # response whose body is an error object with no choices, and ReqLLM decodes + # that into a response with an empty message and the error in + # `provider_meta`. Read as a completion it says nothing, and a model that + # says nothing has declined to answer (`Imp.Predict.ReActV2`), so a refused + # request would be recorded as a choice. It is the failed request it reports, + # in the shape ReqLLM gives an HTTP error. + defp relayed_error(%ReqLLM.Response{provider_meta: %{} = meta}) do + case Map.get(meta, "error") || Map.get(meta, :error) do + %{} = error -> + code = Map.get(error, "code") || Map.get(error, :code) + + %ReqLLM.Error.API.Request{ + reason: Map.get(error, "message") || Map.get(error, :message) || inspect(error), + status: if(is_integer(code), do: code), + response_body: %{"error" => error} + } + + message when is_binary(message) and message != "" -> + %ReqLLM.Error.API.Request{reason: message, response_body: %{"error" => message}} + + _none -> + nil + end + end + + defp relayed_error(_response), do: nil + # OpenAI-compatible providers name this refusal in the structured error code. # General HTTP 400s and prose mentioning context are not safe retry signals. defp normalize_context_refusal( diff --git a/lib/imp/observability.ex b/lib/imp/observability.ex index 92dd0baf..e07d37ce 100644 --- a/lib/imp/observability.ex +++ b/lib/imp/observability.ex @@ -411,11 +411,25 @@ defmodule Imp.Observability do %{event_count: length(trace), actions: actions} end + # The termination reasons of a run that ended with its outputs. ReAct and + # ReActV2 end with `:submit`, `:forced_submit` or `:direct`; ReActV2 also + # ends with prose (`:answered`), the text of the last request of an + # interrupted turn (`:last_prose`) and a terminal tool (`:finished_by_tool`). + # Every other reason names why a run stopped without them. + @complete_terminations [ + :submit, + :forced_submit, + :direct, + :answered, + :last_prose, + :finished_by_tool, + nil + ] + defp prediction_status(%Imp.Prediction{} = prediction) do - case Imp.Prediction.get(prediction, :termination_reason) do - reason when reason in [:submit, :forced_submit, :direct, nil] -> :ok - _reason -> :incomplete - end + if Imp.Prediction.get(prediction, :termination_reason) in @complete_terminations, + do: :ok, + else: :incomplete end defp result_status({:error, _reason}), do: :error diff --git a/lib/imp/predict/react_v2.ex b/lib/imp/predict/react_v2.ex index d3a1b98c..77323d6a 100644 --- a/lib/imp/predict/react_v2.ex +++ b/lib/imp/predict/react_v2.ex @@ -5,19 +5,21 @@ defmodule Imp.Predict.ReActV2 do ReActV2 preserves parallel tool call IDs and results in `Imp.History` and keeps unknown and failed tool calls as observations. + ## Which signatures get `submit` + + A task signature with exactly one output of type `:string` has an answer + that the model can write as plain text, so its loop offers no `submit` tool. + Every other signature (several outputs, or one output that is not text) gets + the reserved `submit` tool, whose parameters are the signature's outputs, + exactly as DSPy's ReActV2 has it. The name `submit` is reserved for every + signature, so a user tool cannot take it. + ## How a turn ends - * `submit`. The model calls the reserved `submit` tool with the signature's - outputs. `termination_reason: :submit`. - * Prose. The step comes back as text with no tool call, and the task - signature has exactly one output of type `:string`. That prose is the - output, and the run finishes in that one request, with - `termination_reason: :answered`. This is what every other mainstream tool - loop does, so it is the default; `prose: :forced_submit` restores the - older behaviour for a single-output signature. A signature with several - outputs, or one non-text output, always takes the forced submit, because - prose cannot fill those fields. A step that says nothing at all also takes - the forced submit: there is no answer in an empty completion. + * Prose (one text output). A step that says something and calls no tool is + the answer, in that one request: `termination_reason: :answered`. + * `submit` (every other signature). The model calls `submit` with the + signature's outputs: `termination_reason: :submit`. * A terminal tool. `finish_on` maps a tool name to `fn arguments, result, inputs -> {:finish, outputs} | :continue end`. It runs after that tool's call executes; `{:finish, outputs}` validates @@ -28,19 +30,49 @@ defmodule Imp.Predict.ReActV2 do execute and are recorded, and a `submit` in the same step still wins. Outputs that fail validation are recorded as that call's result, the same error a bad `submit` records, and the loop continues. - * `max_iters`. What the step limit does is `on_max_iters`. The default, - `:forced_submit`, makes one more request with `tool_choice` naming - `submit` (`termination_reason: :forced_submit`). `:last_prose` makes one - more request with no tools in it at all, so the only thing the model can - do is speak; that prose is the single text output and - `termination_reason: :last_prose`, and a completion that says nothing is - an empty answer rather than an error. `:last_prose` needs a signature - with exactly one output of type `:string`, and is refused at - construction otherwise. `:last_prose_note` puts one line of host text in - front of that request as a user message; Imp writes no sentence of its - own. - * A prediction error. The loop forces one more request with `tool_choice` - naming `submit` (`termination_reason: :forced_submit`). + + ## When a turn is interrupted + + A turn is interrupted when it reaches `max_iters`, when a step's request + fails (`:prediction_error`, `:parse_error`), or when a step calls no tool and + gives no answer (`:invalid_answer` for prose the text output does not + accept, and `:empty_tool_calls` for a signature with `submit`). + + With one text output, a step that calls no tool and says nothing is not an + interruption: it is an empty answer, and the turn ends there + (`termination_reason: :answered`). Saying nothing is how a model declines + to answer, and asking it again would make declining cost a second request. + + With one text output, every interruption takes the same path: one more + request with `tool_choice: "none"`, so the model can only write text, and + that text is the answer, with `termination_reason: :last_prose`. The + request offers the same tools as every other step: a provider may refuse a + history of tool calls when no tools are declared (Anthropic does), and a + changed roster changes the prompt prefix a provider caches. If the model + calls a tool anyway, the call is not run; the completion's text, if any, is + the answer, and the calls are kept in `unexecuted_tool_calls`. A completion + that says nothing is an empty answer rather than an error. + `:last_prose_note` puts one line of host text in front of that request as a + user message; Imp writes no sentence of its own. If the process's + `Imp.Deadline` has already passed, no request is made and the prediction + ends with `termination_reason: :deadline_exceeded` and no answer. Either + way `termination_cause` names the interruption. + + With `submit`, an interruption forces one more request with `tool_choice` + naming `submit` (`termination_reason: :forced_submit`), as DSPy does. + `:forced_submit_notice`, a string or a 1-arity function of the termination + reason, adds one user-visible turn in front of that request saying why. If a + provider cannot honor that tool contract, a tools-disabled typed extractor + derives the task outputs from the original inputs and accumulated history. + + The last request says nothing about why it is being made unless a note is + given. Each note is kept in the returned history like any other turn. A note + given for the other kind of signature is refused at construction. + + A request refused because the context window is full is not an + interruption of this kind: a further request would be refused the same way, + so the prediction ends at once with `termination_reason: + :context_window_exceeded` (see below). A step's outputs are `next_thought` and `tool_calls`. The provider holds the tool roster natively, so a step normally comes back as native tool calls. A @@ -52,14 +84,7 @@ defmodule Imp.Predict.ReActV2 do history as its own turn. A tool call the model writes as JSON rather than calling natively is accepted with `tool` for `name` and `args` or `parameters` for `arguments` (`Imp.Adapter.Types.ToolCall`); a map - that names no tool at all is kept as a malformed-call observation. The - forced request says nothing - about why by default; `:forced_submit_notice`, a string or a 1-arity function - of the termination reason, adds one user-visible turn saying so, which is kept - in the returned history like any other turn, and `:last_prose_note` does the - same for the `:last_prose` request. If a provider cannot - honor that tool contract, a tools-disabled typed extractor derives the task - outputs from the original inputs and accumulated history. + that names no tool at all is kept as a malformed-call observation. On a recognized context-window refusal, up to eight smaller requests omit oldest prior episodes from the prompt, preserving their full durable history. @@ -85,8 +110,6 @@ defmodule Imp.Predict.ReActV2 do tools: %{}, max_iters: 20, tool_policy: :allow, - prose: :answer, - on_max_iters: :forced_submit, finish_on: %{} ] @@ -103,23 +126,13 @@ defmodule Imp.Predict.ReActV2 do metadata: [type: {:map, :any, :any}, default: %{}], max_iters: [type: :non_neg_integer, default: 20], tool_policy: [type: {:custom, Imp.ToolPolicy, :validate, []}, default: :allow], - # What to tell the model when the loop makes it submit. A 1-arity function - # of the termination reason, or a plain string; nil says nothing, which is - # what the loop did before this option existed. + # What to tell the model when the loop makes it submit, for a signature + # that has `submit`. A 1-arity function of the termination reason, or a + # plain string; nil says nothing. forced_submit_notice: [type: {:or, [{:fun, 1}, :string, nil]}, default: nil], - # What a step of plain prose with no tool call means. `:answer` ends the - # turn with that prose as the single text output, which is what every other - # mainstream tool loop does. `:forced_submit` keeps the older behaviour of - # one more request with `tool_choice` naming submit. - prose: [type: {:in, [:answer, :forced_submit]}, default: :answer], - # What the step limit does. `:forced_submit` makes one more request with - # `tool_choice` naming submit. `:last_prose` makes one more request with no - # tools in it, so the only thing the model can do is speak, and what it - # says is the answer; it needs a signature with one text output for that - # prose to be. - on_max_iters: [type: {:in, [:forced_submit, :last_prose]}, default: :forced_submit], - # One line of text put in front of the `:last_prose` request as a user - # message. nil says nothing, and Imp never writes a sentence of its own. + # One line of text put in front of the last request of an interrupted turn, + # for a signature with one text output. nil says nothing, and Imp never + # writes a sentence of its own. last_prose_note: [type: {:or, [:string, nil]}, default: nil], # Tools that end the turn with the outputs they carry, the shape Pydantic # AI calls an output tool. Name to @@ -136,15 +149,8 @@ defmodule Imp.Predict.ReActV2 do raise ArgumentError, "submit is reserved by Imp.Predict.ReActV2" end - if opts[:on_max_iters] == :last_prose and not single_text_output?(signature) do - raise ArgumentError, - "Imp.Predict.ReActV2.new/3: on_max_iters: :last_prose needs a signature with " <> - "exactly one output of type :string, got: " <> - inspect(Imp.Signature.output_names(signature)) - end - - submit = Imp.Tool.new(:submit, "Submit the final outputs for the task.", & &1) - tools = Map.put(tools, :submit, submit) + validate_notes!(signature, opts) + tools = put_submit(tools, signature) react_signature = %Imp.Signature{ @@ -162,7 +168,8 @@ defmodule Imp.Predict.ReActV2 do # A step answered in plain prose, with no native tool call, is a # thought that called nothing. `Imp.Adapter.Chat` reads a marker-free # completion as `next_thought`, and `tool_calls` takes its declared - # default of none, which ends the step at `forced_submit`. + # default of none, which ends the turn: as the answer when the + # signature has one text output, and at the forced submit otherwise. metadata: %{prose_step: :next_thought} } @@ -192,12 +199,51 @@ defmodule Imp.Predict.ReActV2 do tool_policy: opts[:tool_policy], forced_submit_notice: opts[:forced_submit_notice], last_prose_note: opts[:last_prose_note], - prose: opts[:prose], - on_max_iters: opts[:on_max_iters], finish_on: resolve_finish_on!(opts[:finish_on], tools) } end + # Each note belongs to one of the two ways an interrupted turn ends, and a + # note given for the other one would never be said. + defp validate_notes!(signature, opts) do + cond do + single_text_output?(signature) and opts[:forced_submit_notice] != nil -> + raise ArgumentError, + "Imp.Predict.ReActV2.new/3: :forced_submit_notice needs a signature with submit; " <> + "a signature with one text output has none, so use :last_prose_note" + + not single_text_output?(signature) and opts[:last_prose_note] != nil -> + raise ArgumentError, + "Imp.Predict.ReActV2.new/3: :last_prose_note needs a signature with " <> + "exactly one output of type :string, got: " <> + inspect(Imp.Signature.output_names(signature)) + + true -> + :ok + end + end + + @doc false + # Adds the reserved `submit` tool when the signature has one. Loading a saved + # program goes through here too, so the rule lives in one place. + # + # Divergence from DSPy's ReActV2, which offers `submit` for every signature: + # a signature with one text output gets none. Its answer is the text the + # model writes when it stops calling tools, which is how every other + # mainstream tool loop ends a turn, and a `submit` beside that would be a + # second way to say the same thing. DSPy needs `submit` because a signature + # can have several typed outputs, and those signatures still get it. + def put_submit(tools, signature) do + if single_text_output?(signature), + do: tools, + else: + Map.put( + tools, + :submit, + Imp.Tool.new(:submit, "Submit the final outputs for the task.", & &1) + ) + end + @doc false def validate_finish_on(finish_on) when is_map(finish_on) do invalid = @@ -300,47 +346,38 @@ defmodule Imp.Predict.ReActV2 do end end - defp run(react, history, inputs, pending, turn, max_iters, execution) when turn >= max_iters do - case react.on_max_iters do - :forced_submit -> - forced_submit(react, history, inputs, pending, :max_iters, turn, nil, execution) - - :last_prose -> - last_prose(react, history, pending, turn) - end - end + defp run(react, history, inputs, pending, turn, max_iters, execution) when turn >= max_iters, + do: interrupted(react, history, inputs, pending, :max_iters, turn, nil, execution) defp run(react, history, inputs, pending, turn, max_iters, execution) do case predict(react.react, react, history, pending) do {:ok, prediction, history} -> calls = prediction |> Imp.get(:tool_calls, []) |> normalize_calls(turn) - emit_reasoning(prediction, turn) + # Prose that is the answer is not a thought: the `:final` event carries + # it, and a `:reasoning` event would say it a second time. if calls.tool_calls == [] do - # The step said something and called nothing. What it said is part of - # the run, so it is appended as this turn's history event; the pending - # inputs it carries are then spent. - {history, pending} = append_thought_only_step(history, pending, prediction, calls) - - case prose_answer(react, prediction) do + # The step called nothing. What it said is part of the run, so it is + # appended as this turn's history event, with the outputs when it is + # the answer, as a `submit`'s event carries them; the pending inputs + # it carries are then spent. + case parse_prose(react.signature, prediction) do {:ok, outputs} -> # The model stopped calling tools and said its answer. That is the # end of the turn, and it costs no further request. + history = + append_history(history, history_event(pending, prediction, calls, [], outputs)) + final_prediction(outputs, history, :answered) - :none -> - forced_submit( - react, - history, - inputs, - pending, - :empty_tool_calls, - turn, - nil, - execution - ) + {:none, cause} -> + emit_reasoning(prediction, turn) + {history, pending} = append_thought_only_step(history, pending, prediction, calls) + interrupted(react, history, inputs, pending, cause, turn, nil, execution) end else + emit_reasoning(prediction, turn) + case execute_calls(react, calls, execution, inputs) do {:cancel, reason} -> {:error, {:execution_cancelled, reason}} @@ -372,7 +409,7 @@ defmodule Imp.Predict.ReActV2 do if context_window_exceeded?(reason) do incomplete_prediction(history, :context_window_exceeded, reason) else - forced_submit( + interrupted( react, history, inputs, @@ -386,6 +423,14 @@ defmodule Imp.Predict.ReActV2 do end end + # The one place an interrupted turn goes: the last request with + # `tool_choice: "none"` for a signature with one text output, the forced submit for every other. + defp interrupted(react, history, inputs, pending, cause, turn, error, execution) do + if single_text_output?(react.signature), + do: last_prose(react, history, pending, cause, turn, error), + else: forced_submit(react, history, inputs, pending, cause, turn, error, execution) + end + defp forced_submit( react, history, @@ -432,33 +477,80 @@ defmodule Imp.Predict.ReActV2 do end end - # The step limit under `on_max_iters: :last_prose`. The last request carries - # no tools, so the only thing the model can do is speak, and what it says is - # the single text output. A completion that says nothing is an empty answer: - # the run is over either way, and there is nothing to force. - defp last_prose(react, history, pending, turn) do - history = append_note(history, react.signature, react.last_prose_note) + # An interrupted turn of a signature with one text output. The last request + # says `tool_choice: "none"`, so the model can only write text, and that + # text is the single text output. A tool call it makes anyway is not run, + # and is kept out of the history, where it would replay as a call with no + # result; the prediction names it in `unexecuted_tool_calls` instead. A completion that says nothing is an + # empty answer: the run is over either way, and there is nothing to force. + # A deadline that has already passed leaves no time for that request, so + # none is made. + defp last_prose(react, history, pending, cause, turn, initial_error) do + if deadline_passed?() do + incomplete_prediction(history, :deadline_exceeded, initial_error, cause) + else + {history, pending} = note_after_inputs(history, pending, react) + + case predict(last_prose_program(react), react, history, pending) do + {:ok, prediction, history} -> + calls = prediction |> Imp.get(:tool_calls, []) |> normalize_calls(turn) + outputs = last_prose_outputs(react.signature, prediction) + no_calls = %ToolCalls{tool_calls: []} + history = append_last_step(history, pending, prediction, no_calls, outputs) + + outputs + |> Map.put(:termination_cause, cause) + |> put_unexecuted(calls) + |> final_prediction(history, :last_prose) + + {:error, reason, history} -> + termination = + cond do + deadline_passed?() -> :deadline_exceeded + context_window_exceeded?(reason) -> :context_window_exceeded + true -> cause + end + + incomplete_prediction( + history, + termination, + %{initial: initial_error, last_prose: reason}, + cause + ) + end + end + end - case predict(last_prose_program(react), react, history, pending) do - {:ok, prediction, history} -> - calls = prediction |> Imp.get(:tool_calls, []) |> normalize_calls(turn) - emit_reasoning(prediction, turn) - history = append_last_step(history, pending, prediction, calls) - final_prediction(last_prose_outputs(react.signature, prediction), history, :last_prose) + # The note is the last thing the model reads. Inputs no step has spent yet + # (the first step failed) would otherwise render after it, so they go into + # the history first, as the user turn they are. + defp note_after_inputs(history, pending, %{last_prose_note: note} = react) + when is_binary(note) and note != "" and map_size(pending) > 0 do + history = history |> append_history(pending) |> append_note(react.signature, note) + {history, %{}} + end - {:error, reason, history} -> - termination = - if context_window_exceeded?(reason), do: :context_window_exceeded, else: :max_iters + defp note_after_inputs(history, pending, react), + do: {append_note(history, react.signature, react.last_prose_note), pending} - incomplete_prediction(history, termination, reason) - end - end + defp deadline_passed?, do: Imp.Deadline.expired?(Imp.Deadline.current()) - # A request with no tools. Both provider keys go: the chat completions body - # carries `tools` and `tool_choice` together or not at all, and a request - # that names a tool choice without a roster is invalid. + # The same roster as every step, with `tool_choice: "none"`. ReqLLM encodes + # it as `"none"` for OpenAI and OpenRouter and `{"type": "none"}` for + # Anthropic (`test/react_v2_last_request_wire_test.exs`). defp last_prose_program(react) do - %{react.react | config: Keyword.drop(react.react.config, [:tools, :tool_choice])} + %{react.react | config: Keyword.put(react.react.config, :tool_choice, "none")} + end + + defp put_unexecuted(outputs, %ToolCalls{tool_calls: []}), do: outputs + + defp put_unexecuted(outputs, %ToolCalls{tool_calls: calls}) do + unexecuted = + Enum.map(calls, fn call -> + %{id: call.id, name: call.name, arguments: Imp.Tool.normalize_arguments(call.arguments)} + end) + + Map.put(outputs, :unexecuted_tool_calls, Imp.Redaction.redact(unexecuted)) end defp last_prose_outputs(signature, prediction) do @@ -466,7 +558,7 @@ defmodule Imp.Predict.ReActV2 do {:ok, outputs} -> outputs - :none -> + {:none, _cause} -> [%Imp.Signature.Field{name: name}] = signature.outputs %{name => nil} end @@ -538,7 +630,7 @@ defmodule Imp.Predict.ReActV2 do submit_calls = %ToolCalls{tool_calls: Enum.filter(calls.tool_calls, &submit?/1)} if submit_calls.tool_calls == [] do - history = append_last_step(history, pending, prediction, calls) + history = append_last_step(history, pending, prediction, calls, nil) extract_final(react, inputs, history, reason, initial_error) else case execute_calls(react, submit_calls, execution, inputs) do @@ -567,15 +659,15 @@ defmodule Imp.Predict.ReActV2 do end # The completion of the run's last request, thought and any calls, as this - # turn's history event. A completion that said nothing and called nothing - # adds no turn. - defp append_last_step(history, pending, prediction, calls) do + # turn's history event, with the outputs when they are the answer. A + # completion that said nothing and called nothing adds no turn. + defp append_last_step(history, pending, prediction, calls, outputs) do thought = Imp.get(prediction, :next_thought) if thought in [nil, ""] and calls.tool_calls == [] do history else - event = history_event(pending, prediction, calls, [], nil) + event = history_event(pending, prediction, calls, [], outputs) append_history(history, event) end end @@ -839,24 +931,27 @@ defmodule Imp.Predict.ReActV2 do # A step that stops calling tools and says something has answered, when the # task declares exactly one text output for that prose to be. Several outputs, - # or one that is not text, cannot be filled from prose, and an empty - # completion says nothing, so both still take the forced submit. The prose is - # validated through the same parse a `submit`'s arguments go through, so a - # constrained output is not quietly filled with something it excludes. - defp prose_answer(%__MODULE__{prose: :forced_submit}, _prediction), do: :none - defp prose_answer(react, prediction), do: parse_prose(react.signature, prediction) - + # or one that is not text, cannot be filled from prose and take the forced + # submit. The prose is validated through the same parse a `submit`'s + # arguments go through, so a constrained output is not quietly filled with + # something it excludes. What is not an answer carries the interruption it + # is. defp parse_prose(signature, prediction) do - with true <- single_text_output?(signature), - [%Imp.Signature.Field{name: name}] <- signature.outputs, - prose when is_binary(prose) and prose != "" <- Imp.get(prediction, :next_thought), + with {:text, [%Imp.Signature.Field{name: name}]} <- text_output(signature), + {:prose, prose} when is_binary(prose) and prose != "" <- + {:prose, Imp.get(prediction, :next_thought)}, {:ok, parsed} <- Imp.Adapter.Chat.parse(signature, %{name => prose}, []) do {:ok, Imp.Prediction.to_map(parsed)} else - _not_an_answer -> :none + :submit -> {:none, :empty_tool_calls} + {:prose, _nothing} -> {:ok, %{hd(signature.outputs).name => nil}} + {:error, _reason} -> {:none, :invalid_answer} end end + defp text_output(signature), + do: if(single_text_output?(signature), do: {:text, signature.outputs}, else: :submit) + defp single_text_output?(%Imp.Signature{outputs: [%Imp.Signature.Field{type: type}]}), do: type in [:string, "string"] @@ -976,8 +1071,9 @@ defmodule Imp.Predict.ReActV2 do {:ok, prediction} end - defp incomplete_prediction(history, reason, error) do + defp incomplete_prediction(history, reason, error, cause \\ nil) do fields = projection_metadata(%{history: history.full, termination_reason: reason}, history) + fields = if cause, do: Map.put(fields, :termination_cause, cause), else: fields fields = if error, @@ -1142,10 +1238,11 @@ defmodule Imp.Predict.ReActV2 do defp maybe_put(map, key, value), do: Map.put(map, key, value) # What the adapter needs to say about the loop, as data. `finish_tool` is the - # tool that ends the turn, so a renderer never has to know its name. + # tool that ends the turn, so a renderer never has to know its name, and nil + # when the signature has no `submit` and the answer is plain text. defp guidance(signature, tools) do %{ - finish_tool: :submit, + finish_tool: if(single_text_output?(signature), do: nil, else: :submit), input_names: Imp.Signature.input_names(signature), output_names: Imp.Signature.output_names(signature), tool_names: tools |> Map.keys() |> Enum.sort() diff --git a/lib/imp/saving.ex b/lib/imp/saving.ex index 19df99f0..0d438fd2 100644 --- a/lib/imp/saving.ex +++ b/lib/imp/saving.ex @@ -239,8 +239,6 @@ defmodule Imp.Saving do "react" => dump(react.react), "tools" => dump_tools(Map.delete(react.tools, :submit), "ReActV2"), "max_iters" => react.max_iters, - "prose" => Atom.to_string(react.prose), - "on_max_iters" => Atom.to_string(react.on_max_iters), "last_prose_note" => react.last_prose_note, "finish_on" => dump_finish_on(react.finish_on), "tool_policy" => dump_tool_policy(react.tool_policy, "ReActV2 tool policy") @@ -568,15 +566,13 @@ defmodule Imp.Saving do def load(%{"type" => "react_v2"} = state) do require_keys!(state, ["type", "signature", "react", "tools", "max_iters", "tool_policy"]) tools = load_tools!(state["tools"], "ReActV2") - submit = Imp.Tool.new(:submit, "Submit the final outputs for the task.", & &1) + signature = Imp.Signature.load(state["signature"]) %Imp.Predict.ReActV2{ - signature: Imp.Signature.load(state["signature"]), + signature: signature, react: require_predict!(load(state["react"]), "ReActV2"), - tools: Map.put(tools, :submit, submit), + tools: Imp.Predict.ReActV2.put_submit(tools, signature), max_iters: require_non_negative_integer!(state["max_iters"], "ReActV2 max_iters"), - prose: load_react_v2_prose!(state["prose"]), - on_max_iters: load_react_v2_on_max_iters!(state["on_max_iters"]), last_prose_note: load_react_v2_last_prose_note!(state["last_prose_note"]), finish_on: load_finish_on!(state["finish_on"]), tool_policy: load_tool_policy!(state["tool_policy"], "ReActV2 tool policy") @@ -1158,24 +1154,6 @@ defmodule Imp.Saving do defp load_finish_on!(other), do: raise(ArgumentError, "invalid saved ReActV2 finish_on: #{inspect(other)}") - # A dump written before ReActV2 had the option carries no "prose" key, and - # the older behaviour it was written under is the forced submit. - defp load_react_v2_prose!(nil), do: :forced_submit - defp load_react_v2_prose!("answer"), do: :answer - defp load_react_v2_prose!("forced_submit"), do: :forced_submit - - defp load_react_v2_prose!(other), - do: raise(ArgumentError, "invalid saved ReActV2 prose: #{inspect(other)}") - - # A dump written before ReActV2 had the option carries no "on_max_iters" key, - # and the step limit forced a submit then. - defp load_react_v2_on_max_iters!(nil), do: :forced_submit - defp load_react_v2_on_max_iters!("forced_submit"), do: :forced_submit - defp load_react_v2_on_max_iters!("last_prose"), do: :last_prose - - defp load_react_v2_on_max_iters!(other), - do: raise(ArgumentError, "invalid saved ReActV2 on_max_iters: #{inspect(other)}") - defp load_react_v2_last_prose_note!(nil), do: nil defp load_react_v2_last_prose_note!(note) when is_binary(note), do: note diff --git a/lib/mix/tasks/imp_acp.demo.ex b/lib/mix/tasks/imp_acp.demo.ex index f45e8bf3..afd7de70 100644 --- a/lib/mix/tasks/imp_acp.demo.ex +++ b/lib/mix/tasks/imp_acp.demo.ex @@ -24,28 +24,10 @@ defmodule Mix.Tasks.ImpAcp.Demo do handler: fn messages, _opts -> case Imp.ACP.DemoMessages.current_tool_result(messages) do {:error, _reason} -> - %{ - next_thought: "Respect the failed workspace inspection.", - tool_calls: [ - %{ - id: "submit-workspace-denied", - name: "submit", - arguments: %{answer: "Workspace inspection was not authorized."} - } - ] - } + "Workspace inspection was not authorized." {:ok, _content} -> - %{ - next_thought: "The workspace tool supplied the answer.", - tool_calls: [ - %{ - id: "submit-workspace", - name: "submit", - arguments: %{answer: "Imp ReActV2 is running in #{workspace}."} - } - ] - } + "Imp ReActV2 is running in #{workspace}." :none -> %{ diff --git a/lib/mix/tasks/imp_acp.host_demo.ex b/lib/mix/tasks/imp_acp.host_demo.ex index f98b9350..629404c8 100644 --- a/lib/mix/tasks/imp_acp.host_demo.ex +++ b/lib/mix/tasks/imp_acp.host_demo.ex @@ -22,14 +22,10 @@ defmodule Mix.Tasks.ImpAcp.HostDemo do {:ok, content} -> answer = content |> String.split("\n") |> List.first() - tool_turn("Return the host-supplied evidence.", "submit", "host-submit", %{ - answer: "ACP host supplied: #{answer}" - }) + "ACP host supplied: #{answer}" {:error, reason} -> - tool_turn("Report the failed host read.", "submit", "host-failed", %{ - answer: "ACP host read failed: #{inspect(reason)}" - }) + "ACP host read failed: #{inspect(reason)}" :none -> tool_turn("Ask the ACP host for the mounted README.", "read_file", "host-read", %{ diff --git a/lib/mix/tasks/imp_acp.host_effects_demo.ex b/lib/mix/tasks/imp_acp.host_effects_demo.ex index c6a97c47..f52d8f45 100644 --- a/lib/mix/tasks/imp_acp.host_effects_demo.ex +++ b/lib/mix/tasks/imp_acp.host_effects_demo.ex @@ -46,9 +46,7 @@ defmodule Mix.Tasks.ImpAcp.HostEffectsDemo do :none -> "host effect produced no result" end - tool_turn("Return the host-observed result.", "submit", "host-submit", %{ - answer: result - }) + result end end ) diff --git a/lib/mix/tasks/imp_acp.host_terminal_cancel_demo.ex b/lib/mix/tasks/imp_acp.host_terminal_cancel_demo.ex index b4923005..8e9c3245 100644 --- a/lib/mix/tasks/imp_acp.host_terminal_cancel_demo.ex +++ b/lib/mix/tasks/imp_acp.host_terminal_cancel_demo.ex @@ -26,12 +26,10 @@ defmodule Mix.Tasks.ImpAcp.HostTerminalCancelDemo do }) {:ok, result} -> - tool_turn("Return the terminal result.", "submit", "host-submit", %{answer: result}) + result {:error, reason} -> - tool_turn("Report the terminal failure.", "submit", "host-failed", %{ - answer: "host terminal failed: #{inspect(reason)}" - }) + "host terminal failed: #{inspect(reason)}" end end ) diff --git a/lib/mix/tasks/imp_acp.mcp_demo.ex b/lib/mix/tasks/imp_acp.mcp_demo.ex index 290881f3..e4b284a9 100644 --- a/lib/mix/tasks/imp_acp.mcp_demo.ex +++ b/lib/mix/tasks/imp_acp.mcp_demo.ex @@ -40,28 +40,10 @@ defmodule Mix.Tasks.ImpAcp.McpDemo do handler: fn messages, _opts -> case Imp.ACP.DemoMessages.current_tool_result(messages) do {:error, _reason} -> - %{ - next_thought: "Respect the failed MCP request.", - tool_calls: [ - %{ - id: "submit-mcp-denied", - name: "submit", - arguments: %{answer: "The MCP tool request was not authorized."} - } - ] - } + "The MCP tool request was not authorized." {:ok, content} -> - %{ - next_thought: "Return the MCP observation.", - tool_calls: [ - %{ - id: "submit-mcp-workspace", - name: "submit", - arguments: %{answer: "MCP returned workspace #{content}."} - } - ] - } + "MCP returned workspace #{content}." :none -> %{ diff --git a/lib/mix/tasks/imp_acp.rich_demo.ex b/lib/mix/tasks/imp_acp.rich_demo.ex index dba6e86b..9c1099d8 100644 --- a/lib/mix/tasks/imp_acp.rich_demo.ex +++ b/lib/mix/tasks/imp_acp.rich_demo.ex @@ -22,28 +22,10 @@ defmodule Mix.Tasks.ImpAcp.RichDemo do handler: fn messages, _opts -> case Imp.ACP.DemoMessages.current_tool_result(messages) do {:error, _reason} -> - %{ - next_thought: "Respect the failed workspace inspection.", - tool_calls: [ - %{ - id: "submit-workspace-denied-1", - name: "submit", - arguments: %{answer: "Workspace inspection was not authorized."} - } - ] - } + "Workspace inspection was not authorized." {:ok, content} -> - %{ - next_thought: "Return the grounded observation.", - tool_calls: [ - %{ - id: "submit-workspace-1", - name: "submit", - arguments: %{answer: "Imp observed workspace #{content}."} - } - ] - } + "Imp observed workspace #{content}." :none -> %{ diff --git a/priv/public_api.json b/priv/public_api.json index 47089d5c..eed3c9cf 100644 --- a/priv/public_api.json +++ b/priv/public_api.json @@ -7374,8 +7374,6 @@ "forced_submit_notice", "last_prose_note", "max_iters", - "on_max_iters", - "prose", "react", "signature", "tool_policy", diff --git a/test/acp_imp_acp_test.exs b/test/acp_imp_acp_test.exs index 77e49c55..2ea4a645 100644 --- a/test/acp_imp_acp_test.exs +++ b/test/acp_imp_acp_test.exs @@ -1552,13 +1552,7 @@ defmodule Imp.ACPTest do Imp.LM.Static.new( handler: fn messages, _opts -> send(test_pid, {:react_messages, messages}) - - %{ - next_thought: "answer", - tool_calls: [ - %{id: "submit", name: "submit", arguments: %{answer: "ok"}} - ] - } + "ok" end ) @@ -1570,12 +1564,12 @@ defmodule Imp.ACPTest do {:ok, %{"sessionId" => session_id}} = Client.new_session(client, "/tmp/project") assert {:ok, %{"stopReason" => "end_turn"}} = Client.prompt(client, session_id, "first") assert_receive {:react_messages, first_messages} - refute inspect(first_messages) =~ "answer: \"ok\"" + refute inspect(first_messages) =~ "content: \"ok\"" assert {:ok, %{"stopReason" => "end_turn"}} = Client.prompt(client, session_id, "second") assert_receive {:react_messages, second_messages} assert inspect(second_messages) =~ "first" - assert inspect(second_messages) =~ "answer: \"ok\"" + assert inspect(second_messages) =~ "content: \"ok\"" end test "durable ReAct sessions list, load, replay, continue, and delete across agent restart" do @@ -1593,11 +1587,7 @@ defmodule Imp.ACPTest do Imp.LM.Static.new( handler: fn messages, _opts -> send(test_pid, {:durable_react_messages, messages}) - - %{ - next_thought: "answer", - tool_calls: [%{id: "submit", name: "submit", arguments: %{answer: "ok"}}] - } + "ok" end ) @@ -1615,7 +1605,7 @@ defmodule Imp.ACPTest do Client.prompt(first_client, session_id, "first") assert_receive {:durable_react_messages, first_messages} - refute inspect(first_messages) =~ "answer: \"ok\"" + refute inspect(first_messages) =~ "content: \"ok\"" stop_if_alive(first_client) stop_if_alive(first_agent) @@ -1653,7 +1643,7 @@ defmodule Imp.ACPTest do assert_receive {:durable_react_messages, second_messages} assert inspect(second_messages) =~ "first" - assert inspect(second_messages) =~ "answer: \"ok\"" + assert inspect(second_messages) =~ "content: \"ok\"" assert {:ok, %{}} = Client.delete_session(second_client, session_id) assert {:ok, %{"sessions" => []}} = Client.list_sessions(second_client, cwd: workspace) @@ -1664,14 +1654,13 @@ defmodule Imp.ACPTest do lm = Imp.LM.Static.new( - handler: fn _messages, _opts -> - %{ - next_thought: "checking the workspace", - tool_calls: [ - %{id: "lookup-live-1", name: "lookup", arguments: %{query: "beam"}}, - %{id: "submit-live-1", name: "submit", arguments: %{answer: "BEAM"}} - ] - } + handler: fn messages, _opts -> + tool_then_answer( + messages, + "checking the workspace", + %{id: "lookup-live-1", name: "lookup", arguments: %{query: "beam"}}, + "BEAM" + ) end ) @@ -1714,7 +1703,6 @@ defmodule Imp.ACPTest do assert tool_call_id == authorized_tool_call_id assert tool_call_id == result_tool_call_id assert String.ends_with?(tool_call_id, ":lookup-live-1") - refute Enum.any?(updates, &String.ends_with?(&1["toolCallId"] || "", ":submit-live-1")) end test "ACP tool-call IDs remain unique when source IDs repeat across session turns" do @@ -1722,14 +1710,13 @@ defmodule Imp.ACPTest do lm = Imp.LM.Static.new( - handler: fn _messages, _opts -> - %{ - next_thought: "checking", - tool_calls: [ - %{id: "reused-source-id", name: "lookup", arguments: %{query: "beam"}}, - %{id: "reused-submit-id", name: "submit", arguments: %{answer: "BEAM"}} - ] - } + handler: fn messages, _opts -> + tool_then_answer( + messages, + "checking", + %{id: "reused-source-id", name: "lookup", arguments: %{query: "beam"}}, + "BEAM" + ) end ) @@ -2082,21 +2069,28 @@ submit(%{answer: observed <> ":" <> scratch})| %DelayedProgram{signature: Imp.signature("question -> answer"), test_pid: test_pid} end + # A scripted step that makes `call` until a tool result is the newest + # message, and then answers in prose. + defp tool_then_answer(messages, thought, call, answer) do + if List.last(messages)[:role] == :tool, + do: answer, + else: %{next_thought: thought, tool_calls: [call]} + end + defp one_tool_program(tool, answer) do lm = Imp.LM.Static.new( - handler: fn _messages, _opts -> - %{ - next_thought: "request the effect", - tool_calls: [ - %{id: "external-call", name: "external", arguments: %{value: "x"}}, - %{id: "submit-call", name: "submit", arguments: %{answer: answer}} - ] - } + handler: fn messages, _opts -> + tool_then_answer( + messages, + "request the effect", + %{id: "external-call", name: "external", arguments: %{value: "x"}}, + answer + ) end ) - Imp.react_v2("question -> answer", [tool], lm: lm, max_iters: 1) + Imp.react_v2("question -> answer", [tool], lm: lm, max_iters: 2) end defp mcp_program(tools) do @@ -2113,12 +2107,7 @@ submit(%{answer: observed <> ":" <> scratch})| } %{content: content} -> - %{ - next_thought: "submit MCP observation", - tool_calls: [ - %{id: "mcp-submit", name: "submit", arguments: %{answer: "mcp:#{content}"}} - ] - } + "mcp:#{content}" end end ) @@ -2153,9 +2142,7 @@ submit(%{answer: observed <> ":" <> scratch})| }) _ -> - tool_turn("finish", "submit", "host-submit", %{ - answer: "host capabilities complete" - }) + "host capabilities complete" end end ) diff --git a/test/adapter_chat_tool_result_renderer_test.exs b/test/adapter_chat_tool_result_renderer_test.exs index da3d6e56..da67b7aa 100644 --- a/test/adapter_chat_tool_result_renderer_test.exs +++ b/test/adapter_chat_tool_result_renderer_test.exs @@ -24,7 +24,7 @@ defmodule Imp.Adapter.ChatToolResultRendererTest do next_thought: "look", tool_calls: [%{id: "c1", name: "look", arguments: %{}}] }, - else: %{tool_calls: [%{id: "s", name: "submit", arguments: %{answer: "ok"}}]} + else: "ok" end ) end diff --git a/test/deployment_agent_optimization_example_test.exs b/test/deployment_agent_optimization_example_test.exs index 2af45ac4..24d6a341 100644 --- a/test/deployment_agent_optimization_example_test.exs +++ b/test/deployment_agent_optimization_example_test.exs @@ -21,18 +21,21 @@ defmodule Imp.DeploymentAgentOptimizationExampleTest do test "packaged agent story applies descriptions without replacing trusted tools" do lm = Imp.LM.Static.new( - handler: fn _messages, _opts -> - %{ - next_thought: "refund the duplicate charge", - tool_calls: [ - %{ - id: "billing-1", - name: "billing_remediation", - arguments: %{account_id: "A-104"} - }, - %{id: "submit-1", name: "submit", arguments: %{answer: "Refund queued for A-104"}} - ] - } + handler: fn messages, _opts -> + if List.last(messages)[:role] == :tool do + "Refund queued for A-104" + else + %{ + next_thought: "refund the duplicate charge", + tool_calls: [ + %{ + id: "billing-1", + name: "billing_remediation", + arguments: %{account_id: "A-104"} + } + ] + } + end end ) @@ -60,7 +63,8 @@ defmodule Imp.DeploymentAgentOptimizationExampleTest do assert {:ok, prediction} = Imp.call(updated, %{request: "Refund duplicate charge on A-104"}) assert Imp.get(prediction, :answer) == "Refund queued for A-104" - [event] = Imp.get(prediction, :history).messages + assert Imp.get(prediction, :termination_reason) == :answered + [event, _answer] = Imp.get(prediction, :history).messages assert Enum.any?(event.tool_call_results, fn result -> result.name == "billing_remediation" and result.result == "REFUND_QUEUED:A-104" diff --git a/test/history_test.exs b/test/history_test.exs index 4159620b..06cf9133 100644 --- a/test/history_test.exs +++ b/test/history_test.exs @@ -167,7 +167,7 @@ defmodule Imp.HistoryTest do if n == 0 do %{tool_calls: [%{id: "old-call", name: "retired_fetch", arguments: %{}}]} else - %{tool_calls: [%{id: "done", name: "submit", arguments: %{answer: "earlier answer"}}]} + "earlier answer" end end) tool = Imp.tool(:retired_fetch, "retired capability", fn _ -> @@ -204,7 +204,7 @@ defmodule Imp.HistoryTest do end), do: raise("old tool observation missing from next turn") unless Enum.any?(messages, &(Map.get(&1, :content, "") =~ "earlier question")), do: raise("old intent missing") - %{tool_calls: [%{id: "new-done", name: "submit", arguments: %{answer: "continued"}}]} + "continued" end) program = Imp.react_v2("intent -> answer", [], lm: lm) {:ok, result} = Imp.call(program, %{intent: "continue", history: history}) diff --git a/test/observability_inspection_test.exs b/test/observability_inspection_test.exs index b05f1b4d..42d50880 100644 --- a/test/observability_inspection_test.exs +++ b/test/observability_inspection_test.exs @@ -45,6 +45,25 @@ defmodule Imp.ObservabilityInspectionTest do refute Kernel.inspect(inspection) =~ @secret end + # Each of these is a ReActV2 turn that ended with its answer; only the way it + # ended differs. A turn that ran out of steps, time or context has none. + test "a prediction that ended with its answer is complete however the turn ended" do + for reason <- [:submit, :forced_submit, :answered, :last_prose, :finished_by_tool] do + prediction = Imp.Prediction.new(%{answer: "Paris", termination_reason: reason}) + + assert %Inspection{status: :ok} = Imp.Observability.inspect_artifact(prediction), + "#{reason} should be complete" + + assert %Status{state: :succeeded} = Imp.Observability.status(prediction) + end + + for reason <- [:max_iters, :deadline_exceeded, :context_window_exceeded] do + prediction = Imp.Prediction.new(%{termination_reason: reason}) + assert %Inspection{status: :incomplete} = Imp.Observability.inspect_artifact(prediction) + assert %Status{state: :failed} = Imp.Observability.status(prediction) + end + end + test "provider inspection mirrors recent prompt, messages, outputs, and timestamps" do history = [ %{timestamp: "old", prompt: "old prompt", outputs: ["old output"]}, diff --git a/test/optimizer_parameter_contract_test.exs b/test/optimizer_parameter_contract_test.exs index 23676590..c71ae0b2 100644 --- a/test/optimizer_parameter_contract_test.exs +++ b/test/optimizer_parameter_contract_test.exs @@ -182,9 +182,10 @@ defmodule Imp.Optimizer.ParameterContractTest do schema: %{"type" => "object", "properties" => %{"query" => %{"type" => "string"}}} ) + # ReActV2 has `submit` only for a signature that is not one text output. programs = [ ReAct.new("question -> answer", [tool]), - ReActV2.new("question -> answer", [tool]) + ReActV2.new("question -> answer, confidence: float", [tool]) ] for program <- programs do diff --git a/test/react_v2_context_test.exs b/test/react_v2_context_test.exs index a5998130..0a472074 100644 --- a/test/react_v2_context_test.exs +++ b/test/react_v2_context_test.exs @@ -29,21 +29,8 @@ defmodule Imp.ReActV2ContextTest do "choices" => [ %{ "index" => 0, - "message" => %{ - "role" => "assistant", - "content" => nil, - "tool_calls" => [ - %{ - "id" => "done", - "type" => "function", - "function" => %{ - "name" => "submit", - "arguments" => Jason.encode!(%{answer: "done"}) - } - } - ] - }, - "finish_reason" => "tool_calls" + "message" => %{"role" => "assistant", "content" => "done"}, + "finish_reason" => "stop" } ] }} @@ -163,8 +150,7 @@ defmodule Imp.ReActV2ContextTest do if n < 2, do: {:error, %Imp.ContextWindowExceededError{message: "limit"}}, - else: - {:ok, %{tool_calls: [%{id: "done", name: "submit", arguments: %{answer: "continued"}}]}} + else: {:ok, "continued"} end prior = @@ -217,7 +203,7 @@ defmodule Imp.ReActV2ContextTest do end) refute inspect(messages) =~ "prior-answer" - {:ok, %{tool_calls: [%{id: "done", name: "submit", arguments: %{answer: "done"}}]}} + {:ok, "done"} end end diff --git a/test/react_v2_forced_submit_notice_test.exs b/test/react_v2_forced_submit_notice_test.exs index bcd915ff..83575738 100644 --- a/test/react_v2_forced_submit_notice_test.exs +++ b/test/react_v2_forced_submit_notice_test.exs @@ -1,6 +1,10 @@ defmodule ReActV2ForcedSubmitNoticeTest do use ExUnit.Case, async: true + # The forced submit belongs to a signature with `submit`: more than one + # output, or one that is not text. + @signature "intent -> answer, confidence: float" + # The forced submit re-asks the model with `tool_choice: submit` and says # nothing about why. A host that wants the model told why it is being made to # finish sets `:forced_submit_notice`; the notice is a user message in the @@ -19,7 +23,7 @@ defmodule ReActV2ForcedSubmitNoticeTest do send(owner, {:request, n, messages, opts}) if forced?(opts) do - %{tool_calls: [%{id: "s", name: "submit", arguments: %{answer: "ok"}}]} + %{tool_calls: [%{id: "s", name: "submit", arguments: %{answer: "ok", confidence: 1.0}}]} else %{ next_thought: "look first", @@ -42,7 +46,7 @@ defmodule ReActV2ForcedSubmitNoticeTest do notice = "You have used every turn. Submit the answer you have now." program = - Imp.react_v2("intent -> answer", [look()], + Imp.react_v2(@signature, [look()], lm: recording_lm(owner), max_iters: 1, forced_submit_notice: fn reason -> @@ -68,7 +72,7 @@ defmodule ReActV2ForcedSubmitNoticeTest do owner = self() program = - Imp.react_v2("intent -> answer", [look()], + Imp.react_v2(@signature, [look()], lm: recording_lm(owner), max_iters: 1, forced_submit_notice: "Submit now." @@ -87,7 +91,7 @@ defmodule ReActV2ForcedSubmitNoticeTest do owner = self() program = - Imp.react_v2("intent -> answer", [look()], lm: recording_lm(owner), max_iters: 1) + Imp.react_v2(@signature, [look()], lm: recording_lm(owner), max_iters: 1) assert {:ok, prediction} = Imp.call(program, %{intent: "hello"}) [{first, _}, {forced, _}] = requests(2) @@ -102,7 +106,7 @@ defmodule ReActV2ForcedSubmitNoticeTest do owner = self() program = - Imp.react_v2("intent -> answer", [look()], + Imp.react_v2(@signature, [look()], lm: recording_lm(owner), max_iters: 1, forced_submit_notice: fn _reason -> nil end @@ -115,11 +119,11 @@ defmodule ReActV2ForcedSubmitNoticeTest do test "the option rejects anything that is not a string or a 1-arity function" do assert_raise ArgumentError, ~r/forced_submit_notice/, fn -> - Imp.react_v2("intent -> answer", [look()], forced_submit_notice: fn -> "no" end) + Imp.react_v2(@signature, [look()], forced_submit_notice: fn -> "no" end) end assert_raise ArgumentError, ~r/forced_submit_notice/, fn -> - Imp.react_v2("intent -> answer", [look()], forced_submit_notice: 7) + Imp.react_v2(@signature, [look()], forced_submit_notice: 7) end end end diff --git a/test/react_v2_last_prose_test.exs b/test/react_v2_last_prose_test.exs index 91a3b148..67a1f914 100644 --- a/test/react_v2_last_prose_test.exs +++ b/test/react_v2_last_prose_test.exs @@ -1,10 +1,12 @@ defmodule ReActV2LastProseTest do use ExUnit.Case, async: true - # `on_max_iters: :last_prose` ends a turn that reaches the step limit with one - # request that carries no tools, so the only thing the model can do is speak. - # What it says is the single text output; what it does not say is an empty - # answer, not an error. + # A signature with one text output ends every interrupted turn (the step + # limit, a failed request, a step that calls nothing and says nothing) with + # one request with `tool_choice: "none"` and the same tools as every step, so + # the model can only write text. What it writes is the single text output; what it does not say + # is an empty answer, not an error. A deadline that has already passed + # leaves no time for that request. defp look, do: Imp.tool(:look, "Look at a thing", fn _arguments -> %{"seen" => true} end) @@ -17,7 +19,7 @@ defmodule ReActV2LastProseTest do :counters.put(counter, 1, n) send(owner, {:request, n, messages, opts}) - if Keyword.has_key?(opts, :tools) do + if opts[:tool_choice] != "none" do %{ next_thought: "look first", tool_calls: [%{id: "c#{n}", name: "look", arguments: %{}}] @@ -34,27 +36,28 @@ defmodule ReActV2LastProseTest do defp user_contents(messages), do: messages |> Enum.filter(&(&1[:role] == :user)) |> Enum.map(& &1[:content]) - test "the step limit spends one request with no tools and takes its prose as the answer" do + test "the step limit spends one request that allows no tool call and takes its prose as the answer" do owner = self() prose = "Two looks were enough: the thing is there." program = Imp.react_v2("intent -> answer", [look()], lm: recording_lm(owner, prose), - max_iters: 2, - on_max_iters: :last_prose + max_iters: 2 ) assert {:ok, prediction} = Imp.call(program, %{intent: "hello"}) assert Imp.get(prediction, :answer) == prose assert Imp.get(prediction, :termination_reason) == :last_prose + assert Imp.get(prediction, :termination_cause) == :max_iters [{_first, first_opts}, {_second, _}, {last, last_opts}] = requests(3) refute_received {:request, 4, _messages, _opts} assert Keyword.fetch!(first_opts, :tool_choice) == "auto" - refute Keyword.has_key?(last_opts, :tools) - refute Keyword.has_key?(last_opts, :tool_choice) + assert last_opts[:tool_choice] == "none" + # The roster is the one every step sent, so the prompt prefix is unchanged. + assert last_opts[:tools] == first_opts[:tools] # Nothing was said on the model's behalf: the last request is the second # request plus that step's exchange. @@ -73,7 +76,6 @@ defmodule ReActV2LastProseTest do Imp.react_v2("intent -> answer", [look()], lm: recording_lm(owner, "The thing is there."), max_iters: 1, - on_max_iters: :last_prose, last_prose_note: note ) @@ -93,8 +95,7 @@ defmodule ReActV2LastProseTest do program = Imp.react_v2("intent -> answer", [look()], lm: recording_lm(owner, "The thing is there."), - max_iters: 1, - on_max_iters: :last_prose + max_iters: 1 ) assert {:ok, _prediction} = Imp.call(program, %{intent: "hello"}) @@ -111,8 +112,7 @@ defmodule ReActV2LastProseTest do program = Imp.react_v2("intent -> answer", [look()], lm: recording_lm(owner, ""), - max_iters: 1, - on_max_iters: :last_prose + max_iters: 1 ) assert {:ok, prediction} = Imp.call(program, %{intent: "hello"}) @@ -128,8 +128,7 @@ defmodule ReActV2LastProseTest do program = Imp.react_v2("intent -> answer", [look()], lm: recording_lm(owner, prose), - max_iters: 1, - on_max_iters: :last_prose + max_iters: 1 ) assert {:ok, run} = @@ -155,46 +154,211 @@ defmodule ReActV2LastProseTest do end end - test "a signature that is not one text output refuses the option" do - assert_raise ArgumentError, ~r/exactly one output of type :string/, fn -> - Imp.react_v2("intent -> answer, confidence: float", [look()], on_max_iters: :last_prose) + test "a failed step takes the same last request, and the cause is recorded" do + owner = self() + counter = :counters.new(1, []) + + lm = + Imp.LM.Static.new( + handler: fn messages, opts -> + n = :counters.get(counter, 1) + 1 + :counters.put(counter, 1, n) + send(owner, {:request, n, messages, opts}) + + if n == 1, + do: raise(RuntimeError, "provider unavailable"), + else: "I could not look, so from memory: it is there." + end + ) + + program = Imp.react_v2("intent -> answer", [look()], lm: lm, last_prose_note: "Last one.") + + assert {:ok, prediction} = Imp.call(program, %{intent: "hello"}) + assert Imp.get(prediction, :answer) == "I could not look, so from memory: it is there." + assert Imp.get(prediction, :termination_reason) == :last_prose + assert Imp.get(prediction, :termination_cause) == :prediction_error + + [_failed, {last, last_opts}] = requests(2) + assert last_opts[:tool_choice] == "none" + + # The inputs no step spent come first, and the note is the last thing said. + [inputs, note] = Enum.take(user_contents(last), -2) + assert inputs =~ "hello" + assert note =~ "Last one." + end + + test "a step that calls nothing and says nothing is an empty answer, with no further request" do + owner = self() + counter = :counters.new(1, []) + + lm = + Imp.LM.Static.new( + handler: fn messages, opts -> + n = :counters.get(counter, 1) + 1 + :counters.put(counter, 1, n) + send(owner, {:request, n, messages, opts}) + if n == 1, do: %{tool_calls: []}, else: "Said at last." + end + ) + + program = Imp.react_v2("intent -> answer", [look()], lm: lm) + + assert {:ok, prediction} = Imp.call(program, %{intent: "hello"}) + assert Imp.get(prediction, :answer) == nil + assert Imp.get(prediction, :termination_reason) == :answered + assert [{_only, _opts}] = requests(1) + refute_received {:request, 2, _, _} + end + + # OpenRouter relays an upstream provider's refusal as a successful response + # whose body is an error object: ReqLLM decodes it to an empty message with + # the error in `provider_meta`. The last request answers in prose. + defmodule RelayedErrorReqLLM do + def generate_text(model, messages, opts) do + send(Keyword.fetch!(opts, :owner), {:tool_choice, opts[:tool_choice]}) + + {message, meta} = + if opts[:tool_choice] == "none", + do: {"Answered after all.", %{}}, + else: {"", %{"error" => %{"code" => 400, "message" => "Upstream error"}}} + + {:ok, + %ReqLLM.Response{ + id: "unknown", + model: to_string(model), + context: ReqLLM.Context.new(messages), + message: ReqLLM.Context.assistant(message), + provider_meta: meta + }} end end - test "the default still forces a submit at the step limit" do + test "a request the provider refused in its response body is a failed step, not an empty answer" do + lm = + Imp.req_llm("openrouter:test/model", req_module: RelayedErrorReqLLM, owner: self()) + + program = Imp.react_v2("intent -> answer", [look()], lm: lm) + + assert {:ok, prediction} = Imp.call(program, %{intent: "hello"}) + assert Imp.get(prediction, :answer) == "Answered after all." + assert Imp.get(prediction, :termination_reason) == :last_prose + assert Imp.get(prediction, :termination_cause) == :prediction_error + assert_received {:tool_choice, "auto"} + assert_received {:tool_choice, "none"} + end + + test "a deadline that has already passed makes no last request" do owner = self() + # The deadline passes during the first step's tool call. + slow_look = + Imp.tool(:look, "Look at a thing", fn _arguments -> + Process.sleep(20) + %{"seen" => true} + end) + program = - Imp.react_v2("intent -> answer", [look()], lm: recording_lm(owner, ""), max_iters: 1) + Imp.react_v2("intent -> answer", [slow_look], + lm: recording_lm(owner, "never asked"), + max_iters: 1, + last_prose_note: "Last one." + ) - assert {:ok, _prediction} = Imp.call(program, %{intent: "hello"}) - [_first, {_forced, forced_opts}] = requests(2) - assert forced_opts[:tool_choice] == %{type: "tool", name: "submit"} + assert {:ok, prediction} = + Imp.Deadline.with_deadline(10, fn -> Imp.call(program, %{intent: "hello"}) end) + + assert Imp.get(prediction, :termination_reason) == :deadline_exceeded + assert Imp.get(prediction, :termination_cause) == :max_iters + assert Imp.get(prediction, :answer) == nil + + [_first] = requests(1) + refute_received {:request, 2, _messages, _opts} + + # The note is what the model would have been told; no request, no note. + history = Imp.get(prediction, :history) + refute Enum.any?(Imp.History.messages(history), &(Map.get(&1, :intent) == "Last one.")) end - test "dump and load round-trip the options, and an older dump forces a submit" do + # `tool_choice: "none"` is a request, not a guarantee. A call the model makes + # anyway is not run and is not replayed as a call with no result; its text + # is the answer and the call is named in `unexecuted_tool_calls`. + test "a tool call on the last request is not run, and its text is the answer" do + owner = self() + counter = :counters.new(1, []) + + look = + Imp.tool(:look, "Look at a thing", fn _arguments -> + send(owner, :looked) + %{"seen" => true} + end) + + lm = + Imp.LM.Static.new( + handler: fn _messages, opts -> + n = :counters.get(counter, 1) + 1 + :counters.put(counter, 1, n) + + if opts[:tool_choice] == "none", + do: %{ + next_thought: "One more look, then: it is there.", + tool_calls: [%{id: "late", name: "look", arguments: %{"where" => "shelf"}}] + }, + else: %{tool_calls: [%{id: "c#{n}", name: "look", arguments: %{}}]} + end + ) + + program = Imp.react_v2("intent -> answer", [look], lm: lm, max_iters: 1) + assert {:ok, prediction} = Imp.call(program, %{intent: "hello"}) + + assert_received :looked + refute_received :looked + + assert Imp.get(prediction, :answer) == "One more look, then: it is there." + assert Imp.get(prediction, :termination_reason) == :last_prose + + assert [%{id: "late", name: "look", arguments: %{where: "shelf"}}] = + Imp.get(prediction, :unexecuted_tool_calls) + + [_first, last] = Imp.History.messages(Imp.get(prediction, :history)) + assert last.answer == "One more look, then: it is there." + assert last.tool_calls.tool_calls == [] + end + + test "each note is refused for the signature it does not belong to" do + assert_raise ArgumentError, + ~r/:last_prose_note needs a signature with exactly one output/, + fn -> + Imp.react_v2("intent -> answer, confidence: float", [look()], + last_prose_note: "Now." + ) + end + + assert_raise ArgumentError, ~r/:forced_submit_notice needs a signature with submit/, fn -> + Imp.react_v2("intent -> answer", [look()], forced_submit_notice: "Now.") + end + end + + test "dump and load round-trip the note, and a loaded program has no submit" do runner = fn _arguments -> %{"seen" => true} end registry = Imp.Saving.Registry.new(look_runner: runner) tool = Imp.tool(:look, "Look at a thing", runner) dumped = - Imp.react_v2("intent -> answer", [tool], - on_max_iters: :last_prose, - last_prose_note: "Answer now." - ) + Imp.react_v2("intent -> answer", [tool], last_prose_note: "Answer now.") |> Imp.dump(registry: registry) - assert dumped["on_max_iters"] == "last_prose" assert dumped["last_prose_note"] == "Answer now." + refute Map.has_key?(dumped, "on_max_iters") loaded = Imp.load(dumped, registry: registry) - assert loaded.on_max_iters == :last_prose assert loaded.last_prose_note == "Answer now." + refute Map.has_key?(loaded.tools, :submit) - older = - Imp.load(Map.drop(dumped, ["on_max_iters", "last_prose_note"]), registry: registry) + with_submit = + Imp.react_v2("intent -> answer, confidence: float", [tool]) + |> Imp.dump(registry: registry) + |> Imp.load(registry: registry) - assert older.on_max_iters == :forced_submit - assert older.last_prose_note == nil + assert Map.has_key?(with_submit.tools, :submit) end end diff --git a/test/react_v2_last_request_wire_test.exs b/test/react_v2_last_request_wire_test.exs new file mode 100644 index 00000000..8f06eb19 --- /dev/null +++ b/test/react_v2_last_request_wire_test.exs @@ -0,0 +1,111 @@ +defmodule Imp.ReActV2LastRequestWireTest do + use ExUnit.Case + + # The last request of an interrupted turn says `tool_choice: "none"` and + # keeps the roster every step sent. What that means on the wire is the + # provider's encoding, so each provider path is run through the real ReqLLM + # request stack against a local server and the encoded body is read back. + + @providers [ + {:openai, "/v1", "none"}, + {:openrouter, "", "none"}, + {:anthropic, "", %{"type" => "none"}} + ] + + for {provider, prefix, expected_choice} <- @providers do + test "#{provider} encodes the last request as tool_choice none with the same tools" do + provider = unquote(provider) + {:ok, bodies} = Agent.start_link(fn -> [] end) + + url = + Imp.Test.LocalHTTP.start(fn request -> + body = Jason.decode!(request.body) + n = Agent.get_and_update(bodies, &{length(&1), &1 ++ [body]}) + {200, response(provider, body["model"], n)} + end) + + lm = + Imp.req_llm( + %{ + provider: provider, + id: "fixture", + model: "fixture", + base_url: url <> unquote(prefix) + }, + api_key: "fixture", + cache: false, + req_http_options: [retry: false, max_retries: 0] + ) + + look = Imp.tool(:look, "Look at a thing", fn _arguments -> "it is there" end) + program = Imp.react_v2("intent -> answer", [look], lm: lm, max_iters: 1) + + assert {:ok, prediction} = Imp.call(program, %{intent: "hello"}) + assert Imp.get(prediction, :answer) == "It is there." + assert Imp.get(prediction, :termination_reason) == :last_prose + + [first, last] = Agent.get(bodies, & &1) + assert last["tool_choice"] == unquote(Macro.escape(expected_choice)) + assert first["tool_choice"] != last["tool_choice"] + assert [%{} | _] = last["tools"] + assert last["tools"] == first["tools"] + end + end + + defp response(:anthropic, model, 0) do + %{ + "id" => "msg_1", + "type" => "message", + "role" => "assistant", + "model" => model, + "content" => [%{"type" => "tool_use", "id" => "toolu_1", "name" => "look", "input" => %{}}], + "stop_reason" => "tool_use", + "usage" => %{"input_tokens" => 1, "output_tokens" => 1} + } + end + + defp response(:anthropic, model, _n) do + %{ + "id" => "msg_2", + "type" => "message", + "role" => "assistant", + "model" => model, + "content" => [%{"type" => "text", "text" => "It is there."}], + "stop_reason" => "end_turn", + "usage" => %{"input_tokens" => 1, "output_tokens" => 1} + } + end + + defp response(_chat, model, 0) do + chat(model, %{ + "role" => "assistant", + "content" => nil, + "tool_calls" => [ + %{ + "id" => "call_1", + "type" => "function", + "function" => %{"name" => "look", "arguments" => "{}"} + } + ] + }) + end + + defp response(_chat, model, _n), + do: chat(model, %{"role" => "assistant", "content" => "It is there."}) + + defp chat(model, message) do + %{ + "id" => "chat", + "object" => "chat.completion", + "model" => model, + "choices" => [ + %{ + "index" => 0, + "message" => message, + "finish_reason" => if(message["tool_calls"], do: "tool_calls", else: "stop") + } + ], + "usage" => %{"prompt_tokens" => 1, "completion_tokens" => 1, "total_tokens" => 2} + } + end +end diff --git a/test/react_v2_request_shape_test.exs b/test/react_v2_request_shape_test.exs index 33d32a65..4d4fb1d6 100644 --- a/test/react_v2_request_shape_test.exs +++ b/test/react_v2_request_shape_test.exs @@ -21,7 +21,7 @@ defmodule ReActV2RequestShapeTest do next_thought: "look #{n}", tool_calls: [%{id: "c#{n}", name: "look", arguments: %{}}] }, - else: %{tool_calls: [%{id: "s", name: "submit", arguments: %{answer: "ok"}}]} + else: "ok" end ) end @@ -51,7 +51,7 @@ defmodule ReActV2RequestShapeTest do assert {:ok, _} = Imp.call(program, %{intent: "hello"}) for {messages, opts} <- requests(2) do - assert Enum.sort(Enum.map(opts[:tools], & &1.function.name)) == ["look", "submit"] + assert Enum.map(opts[:tools], & &1.function.name) == ["look"] refute Enum.any?(messages, &(&1[:content] =~ "[[ ## tools ## ]]")) refute Enum.any?( @@ -67,12 +67,40 @@ defmodule ReActV2RequestShapeTest do [{[system | _], _}] = requests(1) assert system.role == :system - assert system.content =~ "call `submit` with `answer`" - assert system.content =~ "The available tools are: `look`, `submit`." + + assert system.content =~ + "When the final answer is ready, write it as plain text without calling a tool." + + assert system.content =~ "The available tools are: `look`." + refute system.content =~ "submit" # And the program's own instructions carry none of it. refute program.signature.instructions =~ "You are an Agent" end + # With `submit`, the guidance is DSPy ReActV2's text. + test "a signature with submit is told to call it, in DSPy's words" do + submit_lm = + Imp.LM.Static.new( + handler: fn messages, _opts -> + send(self(), {:system, hd(messages)}) + %{tool_calls: [%{name: "submit", arguments: %{answer: "ok", confidence: 1.0}}]} + end + ) + + program = Imp.react_v2("intent -> answer, confidence: float", [look()], lm: submit_lm) + assert {:ok, _} = Imp.call(program, %{intent: "hello"}) + assert_received {:system, system} + + for line <- [ + "You are an Agent. Use the supplied tools to produce `answer`, `confidence` from `intent`.", + "Call tools when more information is needed.", + "When the final answer is ready, call `submit` with `answer`, `confidence`.", + "The available tools are: `look`, `submit`." + ] do + assert system.content =~ line + end + end + test "a host can replace the system message and keep parsing" do owner = self() @@ -94,9 +122,9 @@ defmodule ReActV2RequestShapeTest do assert_received {:rendered, %{ - finish_tool: :submit, + finish_tool: nil, output_names: [:answer], - tool_names: [:look, :submit] + tool_names: [:look] }} [{[system | _], _} | _] = requests(2) diff --git a/test/react_v2_test.exs b/test/react_v2_test.exs index 43c74900..9658fd80 100644 --- a/test/react_v2_test.exs +++ b/test/react_v2_test.exs @@ -1,6 +1,11 @@ defmodule ReActV2Test do use ExUnit.Case, async: true + # A signature with more than one output keeps DSPy's `submit`; the tests of + # the submit path use it. `question -> answer` has one text output, so its + # loop has no `submit` and the answer is the prose the model writes. + @submit_signature "question -> answer, confidence: float" + defmodule NativeToolStub do def generate_text(model, messages, opts) do state = Keyword.fetch!(opts, :state) @@ -30,7 +35,11 @@ defmodule ReActV2Test do message: ReqLLM.Context.assistant("", tool_calls: [ - ReqLLM.ToolCall.new("toolu_submit", "submit", ~s({"answer":"Paris"})) + ReqLLM.ToolCall.new( + "toolu_submit", + "submit", + ~s({"answer":"Paris","confidence":0.9}) + ) ] ), object: nil, @@ -84,7 +93,7 @@ defmodule ReActV2Test do "resp_submit", "toolu_submit", "submit", - ~s({"answer":"Paris"}) + ~s({"answer":"Paris","confidence":0.9}) ) {{:ok, response}, :done} @@ -138,7 +147,11 @@ defmodule ReActV2Test do context: ReqLLM.Context.new(messages), message: ReqLLM.Context.assistant( - Jason.encode!(%{reasoning: "The gathered evidence supports this.", answer: answer}) + Jason.encode!(%{ + reasoning: "The gathered evidence supports this.", + answer: answer, + confidence: 0.9 + }) ), object: nil, finish_reason: :stop @@ -172,7 +185,11 @@ defmodule ReActV2Test do tool_calls: [ %{id: "lookup-1", name: "lookup", arguments: %{query: "beam"}}, %{id: "missing-1", name: "missing", arguments: %{}}, - %{id: "submit-1", name: "submit", arguments: %{answer: "BEAM"}} + %{ + id: "submit-1", + name: "submit", + arguments: %{answer: "BEAM", confidence: 1.0} + } ] } ], @@ -180,7 +197,7 @@ defmodule ReActV2Test do ) assert {:ok, prediction} = - Imp.react_v2("question -> answer", [lookup], lm: lm) + Imp.react_v2(@submit_signature, [lookup], lm: lm) |> Imp.call(%{question: "What runtime?"}) assert Imp.get(prediction, :answer) == "BEAM" @@ -203,10 +220,7 @@ defmodule ReActV2Test do test "warns loudly on extra input keys but ignores them and still runs" do lm = action_lm([ - %{ - next_thought: "answer directly", - tool_calls: [%{id: "submit-1", name: "submit", arguments: %{answer: "BEAM"}}] - } + "BEAM" ]) program = Imp.react_v2("question -> answer", [], lm: lm) @@ -249,7 +263,7 @@ defmodule ReActV2Test do lm = action_lm([ %{tool_calls: [%{name: "broken", arguments: %{}}, %{name: "unknown", arguments: %{}}]}, - %{tool_calls: [%{name: "submit", arguments: %{answer: "recovered"}}]} + "recovered" ]) assert {:ok, prediction} = @@ -271,13 +285,13 @@ defmodule ReActV2Test do %{ tool_calls: [ %{name: "write", arguments: %{path: "notes.txt"}}, - %{name: "submit", arguments: %{answer: "recovered"}} + %{name: "submit", arguments: %{answer: "recovered", confidence: 1.0}} ] } ]) assert {:ok, prediction} = - Imp.react_v2("question -> answer", [write], lm: lm, max_iters: 2) + Imp.react_v2(@submit_signature, [write], lm: lm, max_iters: 2) |> Imp.call(%{question: "write and verify"}) assert Imp.get(prediction, :answer) == "recovered" @@ -299,13 +313,13 @@ defmodule ReActV2Test do action_lm( [ %{next_thought: "ready", tool_calls: []}, - %{tool_calls: [%{name: "submit", arguments: %{answer: "forced"}}]} + %{tool_calls: [%{name: "submit", arguments: %{answer: "forced", confidence: 1.0}}]} ], parent ) assert {:ok, prediction} = - Imp.react_v2("question -> answer", [], lm: lm, prose: :forced_submit) + Imp.react_v2(@submit_signature, [], lm: lm) |> Imp.call(%{question: "answer"}) assert Imp.get(prediction, :answer) == "forced" @@ -334,7 +348,7 @@ defmodule ReActV2Test do ) assert {:ok, prediction} = - Imp.react_v2("question -> answer", [], lm: lm, max_iters: 1) + Imp.react_v2(@submit_signature, [], lm: lm, max_iters: 1) |> Imp.call(%{question: "Capital of France?"}) assert Imp.get(prediction, :answer) == "Paris" @@ -364,7 +378,7 @@ defmodule ReActV2Test do ) assert {:ok, prediction} = - Imp.react_v2("question -> answer", [], lm: lm, max_iters: 1) + Imp.react_v2(@submit_signature, [], lm: lm, max_iters: 1) |> Imp.call(%{question: "Capital of France?"}) assert Imp.get(prediction, :answer) == "Paris" @@ -396,7 +410,7 @@ defmodule ReActV2Test do lookup = Imp.tool(:lookup, "Look up a fact", fn _args -> "unused" end) assert {:ok, prediction} = - Imp.react_v2("question -> answer", [lookup], + Imp.react_v2(@submit_signature, [lookup], lm: lm, max_iters: 1, config: [json_retries: 0] @@ -432,7 +446,7 @@ defmodule ReActV2Test do ) assert {:ok, prediction} = - Imp.react_v2("question -> answer", [], + Imp.react_v2(@submit_signature, [], lm: lm, max_iters: 1, config: [json_retries: 0] @@ -460,7 +474,7 @@ defmodule ReActV2Test do lookup = Imp.tool(:lookup, "Look up a fact", fn _args -> "unused" end) assert {:ok, prediction} = - Imp.react_v2("question -> answer", [lookup], + Imp.react_v2(@submit_signature, [lookup], lm: lm, max_iters: 1, config: [json_retries: 0] @@ -524,7 +538,7 @@ defmodule ReActV2Test do ) assert {:ok, prediction} = - Imp.react_v2("question -> answer", [], + Imp.react_v2(@submit_signature, [], lm: lm, max_iters: 1, config: [json_retries: 0] @@ -545,13 +559,20 @@ defmodule ReActV2Test do test "normalizes atom- and string-keyed tool-call collection wrappers" do for wrapped <- [ - %{tool_calls: [%{name: "submit", arguments: %{answer: "atom"}}]}, - %{"tool_calls" => [%{"name" => "submit", "arguments" => %{"answer" => "string"}}]}, + %{tool_calls: [%{name: "submit", arguments: %{answer: "atom", confidence: 1.0}}]}, + %{ + "tool_calls" => [ + %{ + "name" => "submit", + "arguments" => %{"answer" => "string", "confidence" => 1.0} + } + ] + }, %{ "tool_calls" => [ %{ "recipient_name" => "functions.submit", - "parameters" => %{"answer" => "recipient"} + "parameters" => %{"answer" => "recipient", "confidence" => 1.0} } ] } @@ -559,7 +580,7 @@ defmodule ReActV2Test do lm = action_lm([Imp.Prediction.new(%{tool_calls: wrapped})]) assert {:ok, prediction} = - Imp.react_v2("question -> answer", [], lm: lm) + Imp.react_v2(@submit_signature, [], lm: lm) |> Imp.call(%{question: "q"}) assert Imp.get(prediction, :answer) in ["atom", "string", "recipient"] @@ -577,7 +598,7 @@ defmodule ReActV2Test do action_lm( [ %{tool_calls: [%{name: "lookup", arguments: %{}}]}, - %{tool_calls: [%{name: "submit", arguments: %{answer: "forced"}}]} + "last words" ], parent ) @@ -588,8 +609,9 @@ defmodule ReActV2Test do assert {:ok, prediction} = Imp.call(program, Map.put(%{question: "q"}, max_iters_key, 1)) - assert Imp.get(prediction, :answer) == "forced" - assert Imp.get(prediction, :termination_reason) == :forced_submit + assert Imp.get(prediction, :answer) == "last words" + assert Imp.get(prediction, :termination_reason) == :last_prose + assert Imp.get(prediction, :termination_cause) == :max_iters assert_received {:lm_call, _normal_opts} assert_received {:lm_call, _forced_opts} refute_received {:lm_call, _extra_opts} @@ -617,14 +639,19 @@ defmodule ReActV2Test do ]) assert {:ok, prediction} = - Imp.react_v2("question -> answer", [], lm: lm, max_iters: 1) + Imp.react_v2(@submit_signature, [], lm: lm, max_iters: 1) |> Imp.call(%{question: "q"}) assert Imp.get(prediction, :answer) == nil assert Imp.get(prediction, :termination_reason) == :max_iters assert %Imp.History{messages: [event]} = Imp.get(prediction, :history) - assert [%{error: true, result: {:error, {:missing_output_fields, [:answer]}}}] = + assert [ + %{ + error: true, + result: {:error, {:missing_output_fields, [:answer, :confidence]}} + } + ] = event.tool_call_results end @@ -686,13 +713,14 @@ defmodule ReActV2Test do end history = %{"messages" => [%{"question" => "prior", "answer" => "prior answer"}]} - lm = action_lm([%{tool_calls: [%{name: "submit", arguments: %{answer: "continued"}}]}]) + lm = action_lm(["continued"]) assert {:ok, prediction} = Imp.react_v2("question -> answer", [], lm: lm, max_iters: 0) |> Imp.call(%{question: "next", history: history}) assert Imp.get(prediction, :answer) == "continued" + assert Imp.get(prediction, :termination_reason) == :last_prose assert %Imp.History{messages: [prior, current]} = Imp.get(prediction, :history) assert prior.question == "prior" assert current.answer == "continued" @@ -738,6 +766,187 @@ defmodule ReActV2Test do assert call.function.name == "lookup" end + # A turn recorded while the loop still offered `submit` is replayed to a loop + # that has none as what it was: the answer, in plain text. Shown as a call to + # a tool the request does not offer, a model can imitate it and write the + # raw tool-call markup as its answer. + test "a recorded submit is replayed as the answer's text to a loop without submit" do + submitted = fn thought, calls, results -> + Imp.History.new([ + %{ + question: "prior", + next_thought: thought, + tool_calls: Imp.Adapter.Types.ToolCalls.new(calls) |> Imp.Redaction.redact(), + tool_call_results: results + } + ]) + end + + no_submit = [guidance: %{finish_tool: nil, input_names: [], output_names: [], tool_names: []}] + signature = Imp.react_v2("question -> answer", []).react.signature + + alone = + submitted.( + "", + [%{id: "s-1", name: "submit", arguments: %{answer: "Seven, exactly."}}], + [%{id: "s-1", name: "submit", result: "Completed.", error: false}] + ) + + assert [%{role: :system}, %{role: :user}, answer, %{role: :user}] = + Imp.Adapter.Chat.format(signature, %{history: alone, tools: []}, no_submit) + + assert answer == %{role: :assistant, content: "Seven, exactly."} + + beside = + submitted.( + "checking", + [ + %{id: "l-1", name: "lookup", arguments: %{query: "beam"}}, + %{id: "s-1", name: "submit", arguments: %{answer: "BEAM."}} + ], + [ + %{id: "l-1", name: "lookup", result: "BEAM", error: false}, + %{id: "s-1", name: "submit", result: "Completed.", error: false} + ] + ) + + assert [ + %{role: :system}, + %{role: :user}, + %{role: :assistant, content: "checking", tool_calls: [call]}, + %{role: :tool, content: "BEAM"}, + %{role: :assistant, content: "BEAM."}, + %{role: :user} + ] = Imp.Adapter.Chat.format(signature, %{history: beside, tools: []}, no_submit) + + assert call.function.name == "lookup" + + # A loop that still has submit replays the call as recorded. + assert [ + %{role: :system}, + %{role: :user}, + %{tool_calls: [kept]}, + %{role: :tool}, + %{role: :user} + ] = + Imp.Adapter.Chat.format(signature, %{history: alone, tools: []}, []) + + assert kept.function.name == "submit" + end + + # A submit the loop rejected was not the answer: the loop told the model so + # and kept going. Replayed as text it would read as an answer the model gave + # and then gave again. A call recorded without an id is matched to its result + # by name, so the results of the step's other calls are kept. + test "a rejected or id-less recorded submit is replayed only as what the loop accepted" do + no_submit = [guidance: %{finish_tool: nil, input_names: [], output_names: [], tool_names: []}] + signature = Imp.react_v2("question -> answer", []).react.signature + + step = fn fields, calls, results -> + Map.merge(fields, %{ + tool_calls: Imp.Adapter.Types.ToolCalls.new(calls) |> Imp.Redaction.redact(), + tool_call_results: results + }) + end + + history = + Imp.History.new([ + step.( + %{question: "prior", next_thought: "trying"}, + [%{id: "s-1", name: "submit", arguments: %{reply: "wrong key"}}], + [ + %{ + id: "s-1", + name: "submit", + result: {:error, {:missing_output_fields, [:answer]}}, + error: true + } + ] + ), + step.( + %{next_thought: "again", answer: "Right key."}, + [%{id: "s-2", name: "submit", arguments: %{answer: "Right key."}}], + [%{id: "s-2", name: "submit", result: %{answer: "Right key."}, error: false}] + ) + ]) + |> Imp.History.dump() + |> Imp.History.load() + + messages = Imp.Adapter.Chat.format(signature, %{history: history, tools: []}, no_submit) + + assert [ + %{role: :system}, + %{role: :user}, + %{role: :assistant, content: "trying"}, + %{role: :assistant, content: "again\n\nRight key."}, + %{role: :user} + ] = messages + + refute Enum.any?(messages, &String.contains?(inspect(&1), "wrong key")) + + idless = + Imp.History.new([ + step.( + %{question: "prior", next_thought: ""}, + [ + %{name: "lookup", arguments: %{query: "beam"}}, + %{name: "submit", arguments: %{answer: "BEAM."}} + ], + [ + %{name: "lookup", result: "BEAM", error: false}, + %{name: "submit", result: %{answer: "BEAM."}, error: false} + ] + ) + ]) + + assert [ + %{role: :system}, + %{role: :user}, + %{role: :assistant, tool_calls: [%{function: %{name: "lookup"}}]}, + %{role: :tool, content: "BEAM"}, + %{role: :assistant, content: "BEAM."}, + %{role: :user} + ] = Imp.Adapter.Chat.format(signature, %{history: idless, tools: []}, no_submit) + end + + # A host that renders input sections its own way gets the same rendering for + # past turns as for the current one, whether or not the past turn called a + # tool; otherwise the model reads its history in one format and its present + # in another. + test "a history turn with tool calls uses the host's input section renderer" do + signature = Imp.react_v2("question -> answer", []).react.signature + plain = fn _field, value -> value end + + history = + Imp.History.new([ + %{ + question: "prior", + next_thought: "", + tool_calls: + Imp.Adapter.Types.ToolCalls.new([ + %{id: "l-1", name: "lookup", arguments: %{query: "beam"}} + ]) + |> Imp.Redaction.redact(), + tool_call_results: [%{id: "l-1", name: "lookup", result: "BEAM", error: false}] + } + ]) + + assert [ + %{role: :system}, + %{role: :user, content: past}, + %{role: :assistant}, + %{role: :tool}, + _ + ] = + Imp.Adapter.Chat.format( + signature, + %{history: history, tools: []}, + input_section_renderer: plain + ) + + assert past == "prior" + end + test "participates in LM demo and registry-backed persistence lifecycle" do runner = fn %{query: query} -> query end registry = Imp.Saving.Registry.new(lookup_runner: runner) @@ -749,7 +958,7 @@ defmodule ReActV2Test do |> Imp.with_demos([demo]) |> Imp.dump(registry: registry) |> Imp.load(registry: registry) - |> Imp.with_lm(action_lm([%{tool_calls: [%{name: "submit", arguments: %{answer: "ok"}}]}])) + |> Imp.with_lm(action_lm(["ok"])) assert program.react.demos == [demo] assert {:ok, prediction} = Imp.call(program, %{question: "q"}) @@ -811,39 +1020,44 @@ defmodule ReActV2Test do assert_received {:lm_call, _forced} end - # The opt-out for a single-output signature: `prose: :forced_submit` is the - # behaviour before a prose step ended the turn, and it costs two requests. - test "prose: :forced_submit keeps the second request for a single-output signature" do + # A signature with one text output has no `submit`: the roster the provider + # is sent names only the user's tools, and the guidance says to answer in + # plain text. A model that calls `submit` anyway is calling a tool that does + # not exist, which is an observation like any other unknown tool. + test "a signature with one text output is offered no submit tool" do parent = self() lm = action_lm( [ - "I already know this one, no lookup needed.", - %{tool_calls: [%{name: "submit", arguments: %{answer: "Paris"}}]} + %{tool_calls: [%{id: "s1", name: "submit", arguments: %{answer: "Paris"}}]}, + "Paris" ], parent ) lookup = Imp.tool(:lookup, "lookup", fn _arguments -> "unused" end) + program = Imp.react_v2("question -> answer", [lookup], lm: lm) - assert {:ok, prediction} = - Imp.react_v2("question -> answer", [lookup], lm: lm, prose: :forced_submit) - |> Imp.call(%{question: "Capital of France?"}) + refute Map.has_key?(program.tools, :submit) + assert program.react.adapter_opts[:guidance].finish_tool == nil + assert {:ok, prediction} = Imp.call(program, %{question: "Capital of France?"}) assert Imp.get(prediction, :answer) == "Paris" - assert Imp.get(prediction, :termination_reason) == :forced_submit + assert Imp.get(prediction, :termination_reason) == :answered - messages = prediction |> Imp.get(:history) |> Imp.History.messages() + assert %Imp.History{messages: [first, second]} = Imp.get(prediction, :history) + assert [%{error: true, result: {:error, {:unknown_tool, "submit"}}}] = first.tool_call_results + # The answered step's event carries the output, as a submit's event does. + assert second.answer == "Paris" - assert Enum.any?( - messages, - &(Map.get(&1, :next_thought) == "I already know this one, no lookup needed.") - ) + assert_received {:lm_call, opts} + assert Enum.map(opts[:tools], & &1.function.name) == ["lookup"] - assert_received {:lm_call, _normal} - assert_received {:lm_call, _forced} - refute_received {:lm_call, _third} + # With several outputs the same roster carries `submit`. + submit_program = Imp.react_v2(@submit_signature, [lookup]) + assert Map.has_key?(submit_program.tools, :submit) + assert submit_program.react.adapter_opts[:guidance].finish_tool == :submit end # A terminal tool ends the turn with the outputs it carries, the shape @@ -896,7 +1110,7 @@ defmodule ReActV2Test do action_lm( [ %{tool_calls: [%{id: "r1", name: "reply", arguments: %{"text" => "wait"}}]}, - %{tool_calls: [%{name: "submit", arguments: %{answer: "Paris"}}]} + "Paris" ], parent ) @@ -909,7 +1123,7 @@ defmodule ReActV2Test do |> Imp.call(%{question: "Capital of France?"}) assert Imp.get(prediction, :answer) == "Paris" - assert Imp.get(prediction, :termination_reason) == :submit + assert Imp.get(prediction, :termination_reason) == :answered assert %Imp.History{messages: [first, _second]} = Imp.get(prediction, :history) assert [%{name: "reply", error: false}] = first.tool_call_results end @@ -922,7 +1136,7 @@ defmodule ReActV2Test do lm = action_lm([ %{tool_calls: [%{id: "r1", name: "reply", arguments: %{}}]}, - %{tool_calls: [%{name: "submit", arguments: %{answer: "Paris"}}]} + "Paris" ]) assert {:ok, prediction} = @@ -933,7 +1147,7 @@ defmodule ReActV2Test do |> Imp.call(%{question: "Capital of France?"}) assert Imp.get(prediction, :answer) == "Paris" - assert Imp.get(prediction, :termination_reason) == :submit + assert Imp.get(prediction, :termination_reason) == :answered assert %Imp.History{messages: [first, _second]} = Imp.get(prediction, :history) @@ -969,7 +1183,7 @@ defmodule ReActV2Test do end assert_raise ArgumentError, ~r/submit already ends the turn/, fn -> - Imp.react_v2("question -> answer", [reply], + Imp.react_v2(@submit_signature, [reply], finish_on: %{submit: fn _a, _r, _i -> :continue end} ) end @@ -990,7 +1204,7 @@ defmodule ReActV2Test do lm = action_lm([ %{tool_calls: [call]}, - %{tool_calls: [%{name: "submit", arguments: %{answer: "done"}}]} + "done" ]) assert {:ok, prediction} = @@ -1039,8 +1253,12 @@ defmodule ReActV2Test do send(owner, {:request, messages}) Agent.get_and_update(state, fn - :first -> {prose, :second} - :second -> {%{tool_calls: [%{name: "submit", arguments: %{answer: "Paris"}}]}, :done} + :first -> + {prose, :second} + + :second -> + {%{tool_calls: [%{name: "submit", arguments: %{answer: "Paris", confidence: 1.0}}]}, + :done} end) end ) @@ -1048,9 +1266,8 @@ defmodule ReActV2Test do lookup = Imp.tool(:lookup, "lookup", fn _arguments -> "unused" end) assert {:ok, prediction} = - Imp.react_v2("question -> answer", [lookup], + Imp.react_v2(@submit_signature, [lookup], lm: lm, - prose: :forced_submit, forced_submit_notice: "Submit now." ) |> Imp.call(%{question: "Capital of France?"}) @@ -1095,7 +1312,7 @@ defmodule ReActV2Test do Imp.LM.Static.new( handler: fn messages, _opts -> send(owner, {:request, messages}) - %{tool_calls: [%{name: "submit", arguments: %{answer: "ok"}}]} + "ok" end ) diff --git a/test/req_llm_client_test.exs b/test/req_llm_client_test.exs index 35279347..d44ea74c 100644 --- a/test/req_llm_client_test.exs +++ b/test/req_llm_client_test.exs @@ -344,6 +344,26 @@ defmodule ReqLLMClientTest do end end + # What ReqLLM's OpenAI-format decoder returns for a successful HTTP response + # whose body is an error object and no choices, which is how OpenRouter + # relays an upstream provider's refusal: an empty message, no finish reason, + # and the error in `provider_meta`. + defmodule RelayedErrorStub do + def generate_text(model, messages, opts) do + error = Keyword.fetch!(opts, :relayed_error) + + {:ok, + %ReqLLM.Response{ + id: "unknown", + model: to_string(model), + context: ReqLLM.Context.new(messages), + message: ReqLLM.Context.assistant(""), + finish_reason: nil, + provider_meta: %{"error" => error} + }} + end + end + defmodule InvalidStub do def generate_text(_model, _messages, _opts), do: :not_a_req_llm_response end @@ -514,6 +534,16 @@ defmodule ReqLLMClientTest do end) end + test "a response that carries a provider error is a failed request, not an empty completion" do + lm = Imp.req_llm("openrouter:thinkingmachines/inkling", req_module: RelayedErrorStub) + message = "Upstream error from DeepInfra: Failed to compile structural_tag grammar" + + assert {:error, %ReqLLM.Error.API.Request{status: 400, reason: ^message}} = + Imp.Clients.ReqLLM.generate(lm, [%{role: :user, content: "hello"}], + relayed_error: %{"code" => 400, "message" => message} + ) + end + test "provider metadata preserves semantic schema descriptors while redacting credentials" do lm = Imp.req_llm("openai:gpt-test", req_module: ProviderMetadataStub) diff --git a/test/run_test.exs b/test/run_test.exs index 2e732f7c..a53622e5 100644 --- a/test/run_test.exs +++ b/test/run_test.exs @@ -35,13 +35,17 @@ defmodule Imp.RunTest do next_thought: "look it up", tool_calls: [ %{id: "provider-call-1", name: "lookup", arguments: %{query: "beam"}}, - %{id: "provider-submit-1", name: "submit", arguments: %{answer: "BEAM"}} + %{ + id: "provider-submit-1", + name: "submit", + arguments: %{answer: "BEAM", confidence: 1.0} + } ] } end ) - program = Imp.react_v2("question -> answer", [lookup], lm: lm) + program = Imp.react_v2("question -> answer, confidence: float", [lookup], lm: lm) assert {:ok, run} = Imp.start_run(program, %{question: "runtime?"}, @@ -233,7 +237,11 @@ defmodule Imp.RunTest do {%{ tool_calls: [ %{id: "lookup-2", name: "lookup", arguments: %{query: "beam"}}, - %{id: "submit-2", name: "submit", arguments: %{answer: "denied safely"}} + %{ + id: "submit-2", + name: "submit", + arguments: %{answer: "denied safely", confidence: 1.0} + } ] }, :done} end) @@ -255,7 +263,8 @@ defmodule Imp.RunTest do } ) - program = Imp.react_v2("question -> answer", [lookup], lm: lm, max_iters: 2) + program = + Imp.react_v2("question -> answer, confidence: float", [lookup], lm: lm, max_iters: 2) assert {:ok, run} = Imp.start_run(program, %{question: "lookup"}, diff --git a/test/tool_schema_runtime_test.exs b/test/tool_schema_runtime_test.exs index 559468ee..11455dd5 100644 --- a/test/tool_schema_runtime_test.exs +++ b/test/tool_schema_runtime_test.exs @@ -52,7 +52,7 @@ defmodule ToolSchemaRuntimeTest do refute_received {:schema_tool_called, _input} end - test "ReActV2 records validation errors and permits a later submit" do + test "ReActV2 records validation errors and permits a later answer" do parent = self() {:ok, turns} = Agent.start_link(fn -> 0 end) @@ -60,7 +60,7 @@ defmodule ToolSchemaRuntimeTest do static_lm(fn _messages -> Agent.get_and_update(turns, fn 0 -> {%{tool_calls: [%{name: "lookup", arguments: %{}}]}, 1} - _ -> {%{tool_calls: [%{name: "submit", arguments: %{answer: "recovered"}}]}, 2} + _ -> {"recovered", 2} end) end) diff --git a/test/trajectory_test.exs b/test/trajectory_test.exs index ee709751..d8ecd002 100644 --- a/test/trajectory_test.exs +++ b/test/trajectory_test.exs @@ -12,14 +12,14 @@ defmodule Imp.TrajectoryTest do next_thought: "look it up", tool_calls: [ %{id: "lookup-1", name: "lookup", arguments: %{query: "beam"}}, - %{id: "submit-1", name: "submit", arguments: %{answer: "BEAM"}} + %{id: "submit-1", name: "submit", arguments: %{answer: "BEAM", confidence: 1.0}} ] } end ) {:ok, run} = - Imp.Run.start(Imp.react_v2("question -> answer", [lookup], lm: lm), %{ + Imp.Run.start(Imp.react_v2("question -> answer, confidence: float", [lookup], lm: lm), %{ question: "runtime?", api_key: secret })