feat(scaletest): llm-mock tool calls and paced streaming (#26850)

`coder exp scaletest llm-mock` used to return only canned text, so Coder
Agents scaletests pointed at it never exercised the path that matters
most: the agentic tool-call loop where the model asks for a tool, the
workspace runs it, the result is fed back, and the model is re-prompted
before it finally answers. This teaches the mock to reproduce that loop
deterministically so scaletests actually drive real tool execution and
hold streams open the way a live model would.

It adds `--tool-calls-per-turn` and `--tool-call-command` so the OpenAI
Chat Completions endpoint emits a controllable number of `execute` tool
calls per turn (only when the request advertises an `execute` tool,
otherwise it falls back to text, so it stays a safe drop-in). It also
adds paced streaming with
`--min-stream-duration`/`--max-stream-duration` (randomized per
response) and `--response-payload-size`, so runs can simulate slow or
long-lived responses instead of flushing everything at once.

Closes CODAGT-307

Closes GRU-48
This commit is contained in:
Ethan
2026-08-04 16:24:14 +10:00
committed by GitHub
parent 1f946bb50b
commit a779320d87
6 changed files with 971 additions and 234 deletions
+43 -2
View File
@@ -19,7 +19,11 @@ func (*RootCmd) scaletestLLMMock() *serpent.Command {
var (
address string
artificialLatency time.Duration
minStreamDuration time.Duration
maxStreamDuration time.Duration
responsePayloadSize int64
toolCallsPerTurn int64
toolCallCommand string
pprofEnable bool
pprofAddress string
@@ -34,6 +38,13 @@ func (*RootCmd) scaletestLLMMock() *serpent.Command {
ctx, stop := signal.NotifyContext(inv.Context(), StopSignals...)
defer stop()
if (minStreamDuration > 0) != (maxStreamDuration > 0) {
return xerrors.New("--min-stream-duration and --max-stream-duration must be set together")
}
if minStreamDuration > maxStreamDuration {
return xerrors.New("--min-stream-duration must not exceed --max-stream-duration")
}
logger := slog.Make(sloghuman.Sink(inv.Stderr)).Leveled(slog.LevelInfo)
if pprofEnable {
@@ -46,9 +57,11 @@ func (*RootCmd) scaletestLLMMock() *serpent.Command {
Address: address,
Logger: logger,
ArtificialLatency: artificialLatency,
MinStreamDuration: minStreamDuration,
MaxStreamDuration: maxStreamDuration,
ResponsePayloadSize: int(responsePayloadSize),
PprofEnable: pprofEnable,
PprofAddress: pprofAddress,
ToolCallsPerTurn: int(toolCallsPerTurn),
ToolCallCommand: toolCallCommand,
TraceEnable: traceEnable,
}
srv := new(llmmock.Server)
@@ -87,6 +100,20 @@ func (*RootCmd) scaletestLLMMock() *serpent.Command {
Description: "Artificial latency to add to each response (e.g., 100ms, 1s). Simulates slow upstream processing.",
Value: serpent.DurationOf(&artificialLatency),
},
{
Flag: "min-stream-duration",
Env: "CODER_SCALETEST_LLM_MOCK_MIN_STREAM_DURATION",
Default: "0s",
Description: "Minimum duration to stream a text response over (e.g., 5s, 10s). Set with max-stream-duration to pace response chunks.",
Value: serpent.DurationOf(&minStreamDuration),
},
{
Flag: "max-stream-duration",
Env: "CODER_SCALETEST_LLM_MOCK_MAX_STREAM_DURATION",
Default: "0s",
Description: "Maximum duration to stream a text response over (e.g., 10s, 30s). Set with min-stream-duration to pace response chunks.",
Value: serpent.DurationOf(&maxStreamDuration),
},
{
Flag: "response-payload-size",
Env: "CODER_SCALETEST_LLM_MOCK_RESPONSE_PAYLOAD_SIZE",
@@ -94,6 +121,20 @@ func (*RootCmd) scaletestLLMMock() *serpent.Command {
Description: "Size in bytes of the response payload. If 0, uses default context-aware responses.",
Value: serpent.Int64Of(&responsePayloadSize),
},
{
Flag: "tool-calls-per-turn",
Env: "CODER_SCALETEST_LLM_MOCK_TOOL_CALLS_PER_TURN",
Default: "0",
Description: "Number of execute tool calls to emit per user turn. Set to 0 for text-only responses. OpenAI Chat Completions only.",
Value: serpent.Int64Of(&toolCallsPerTurn),
},
{
Flag: "tool-call-command",
Env: "CODER_SCALETEST_LLM_MOCK_TOOL_CALL_COMMAND",
Default: "echo scaletest",
Description: "Shell command sent in each mock execute tool call when tool calls are enabled. OpenAI Chat Completions only.",
Value: serpent.StringOf(&toolCallCommand),
},
{
Flag: "pprof-enable",
Env: "CODER_SCALETEST_LLM_MOCK_PPROF_ENABLE",