mirror of
https://github.com/coder/coder.git
synced 2026-09-24 15:04:27 +08:00
feat: add computer-use provider selection for AI agents (#24772)
Adds a deployment-wide setting to select the computer-use provider (Anthropic or OpenAI) for AI agents, plus the OpenAI computer-use runner needed to honor that selection. The setting is stored in `site_configs` under `agents_computer_use_provider`, defaults to Anthropic when unset, and is exposed via experimental GET/PUT endpoints under `/api/experimental/chats/config/computer-use-provider`. The chatd computer-use tool now dispatches to either `runAnthropicComputerUse` or `runOpenAIComputerUse` based on the resolved provider, with provider-specific result metadata for OpenAI screenshots. Frontend adds a provider dropdown to the Agents Experiments settings page nested under the virtual desktop toggle, with disabled state handling while virtual desktop is off and skeleton loaders while config queries are in flight. Hugo and Codex review follow-up: - Uses shared provider validation and clearer computer-use constant names. - Removes stale OpenAI pending-safety-checks commentary. - Documents why provider result metadata is needed for OpenAI screenshots. - Keeps the computer-use subagent visible when provider credentials are missing, then returns a clear spawn-time configuration error. - Uses OpenAI's recommended 1600x900 screenshot geometry to preserve the native 16:9 aspect ratio. - Moves OpenAI-specific computer-use helpers into `coderd/x/chatd/chatopenai/computeruse` after rebasing onto the provider package refactor in `main`. - Converts OpenAI pixel scroll deltas to Coder desktop wheel-click amounts. - Preserves OpenAI pointer modifiers with key down/up desktop actions and rejects unsupported non-left double-click buttons explicitly. - Maps OpenAI back/forward side-button clicks to browser navigation key actions. - Defaults omitted OpenAI click buttons to left-click. - Retries mouse release cleanup if the final OpenAI drag release fails. - Keeps computer-use subagent availability messages stable when provider config cannot be loaded, while logging the backend error. - Releases remaining OpenAI modifier keys if a synthetic key-up cleanup action fails. - Updates Storybook interaction stories so provider snapshots show the selected final provider. > Mux updated this PR description on behalf of Mike.
This commit is contained in:
@@ -195,6 +195,11 @@ type RunOptions struct {
|
||||
type ProviderTool struct {
|
||||
Definition fantasy.Tool
|
||||
Runner fantasy.AgentTool
|
||||
// ResultProviderMetadata extracts provider-specific metadata from successful
|
||||
// local runner responses. The chat loop attaches returned metadata to the tool
|
||||
// result sent back to the model. OpenAI computer-use uses this to request
|
||||
// original screenshot detail for image results.
|
||||
ResultProviderMetadata func(response fantasy.ToolResponse) fantasy.ProviderMetadata
|
||||
}
|
||||
|
||||
// stepResult holds the accumulated output of a single streaming
|
||||
@@ -1020,13 +1025,22 @@ func executeTools(
|
||||
toolMap[t.Info().Name] = t
|
||||
}
|
||||
providerRunnerNames := make(map[string]struct{}, len(providerTools))
|
||||
resultProviderMetadata := make(
|
||||
map[string]func(fantasy.ToolResponse) fantasy.ProviderMetadata,
|
||||
len(providerTools),
|
||||
)
|
||||
// Include runners from provider tools so locally-executed
|
||||
// provider tools (e.g. computer use) can be dispatched.
|
||||
for _, pt := range providerTools {
|
||||
if pt.Runner != nil {
|
||||
name := pt.Runner.Info().Name
|
||||
toolMap[name] = pt.Runner
|
||||
providerRunnerNames[name] = struct{}{}
|
||||
if pt.Runner == nil {
|
||||
continue
|
||||
}
|
||||
|
||||
name := pt.Runner.Info().Name
|
||||
toolMap[name] = pt.Runner
|
||||
providerRunnerNames[name] = struct{}{}
|
||||
if pt.ResultProviderMetadata != nil {
|
||||
resultProviderMetadata[name] = pt.ResultProviderMetadata
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1052,7 +1066,19 @@ func executeTools(
|
||||
// accurate individual completion times.
|
||||
completedAt[i] = dbtime.Now()
|
||||
}()
|
||||
results[i] = executeSingleTool(ctx, toolMap, tc, metrics, logger, provider, model, builtinToolNames, activeTools, providerRunnerNames)
|
||||
results[i] = executeSingleTool(
|
||||
ctx,
|
||||
toolMap,
|
||||
tc,
|
||||
metrics,
|
||||
logger,
|
||||
provider,
|
||||
model,
|
||||
builtinToolNames,
|
||||
activeTools,
|
||||
providerRunnerNames,
|
||||
resultProviderMetadata,
|
||||
)
|
||||
}()
|
||||
}
|
||||
wg.Wait()
|
||||
@@ -1349,6 +1375,7 @@ func executeSingleTool(
|
||||
builtinToolNames map[string]bool,
|
||||
activeTools []string,
|
||||
providerRunnerNames map[string]struct{},
|
||||
resultProviderMetadata map[string]func(fantasy.ToolResponse) fantasy.ProviderMetadata,
|
||||
) fantasy.ToolResultContent {
|
||||
result := fantasy.ToolResultContent{
|
||||
ToolCallID: tc.ToolCallID,
|
||||
@@ -1430,6 +1457,18 @@ func executeSingleTool(
|
||||
Text: strings.ToValidUTF8(resp.Content, "\uFFFD"),
|
||||
}
|
||||
}
|
||||
|
||||
if _, isError := result.Result.(fantasy.ToolResultOutputContentError); isError {
|
||||
return result
|
||||
}
|
||||
if len(result.ProviderMetadata) == 0 {
|
||||
if callback := resultProviderMetadata[tc.ToolName]; callback != nil {
|
||||
metadata := callback(resp)
|
||||
if len(metadata) > 0 {
|
||||
result.ProviderMetadata = metadata
|
||||
}
|
||||
}
|
||||
}
|
||||
return result
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user