mirror of
https://github.com/coder/coder.git
synced 2026-09-24 15:04:27 +08:00
feat: limit concurrent chat agents with pooled admission (#27902)
Limits concurrent chat generation on capped deployments to 5 root chats and 10 delegated subagent chats. The pools are deployment-wide and independent, so delegated work can continue while root capacity is full. The default caps live in AGPL code. Enterprise contributes only a licensing unlock, so unlicensed deployments stay capped and cannot fail open. Licensed deployments are uncapped while Agent Hours usage stays below an explicit hard limit. Deployments without a hard limit remain uncapped, and reaching the Agent Hours allocation only triggers warnings. Admission happens before a worker takes chat ownership. Capped deployments serialize admission across replicas with a transaction-scoped advisory lock and derive active and queued state from current ownership plus fresh runner heartbeats, rather than persisted queue markers or per-replica state. The acquisition query returns a bounded, pool-interleaved candidate set instead of ranking the whole backlog; a migration replaces the acquisition index with a pool-aware one. Refused chats stay running but unowned, and interrupt requests bypass admission so users can stop queued or over-cap chats. The single-chat API derives `queued_for_capacity` from live pool state; list endpoints do not report it. The UI polls that value every 5 seconds while a chat is running and shows a callout when the chat is waiting for capacity. Updates the administrator documentation and deployment-wide Prometheus gauges for active and queued agents. Replica-level values must be aggregated with `max`, not `sum`. > Mux updated this PR on Mike's behalf.
This commit is contained in:
Generated
+8
-11
@@ -93,6 +93,9 @@ type sqlcQuerier interface {
|
||||
CleanupDeletedMCPServerIDsFromChats(ctx context.Context) error
|
||||
CountAIBridgeSessions(ctx context.Context, arg CountAIBridgeSessionsParams) (int64, error)
|
||||
CountAuditLogs(ctx context.Context, arg CountAuditLogsParams) (int64, error)
|
||||
// Excluding the candidate keeps ownership takeover capacity-neutral.
|
||||
CountChatCapacityActiveByPool(ctx context.Context, arg CountChatCapacityActiveByPoolParams) (CountChatCapacityActiveByPoolRow, error)
|
||||
CountChatCapacityQueuedByPool(ctx context.Context, staleSeconds int32) (CountChatCapacityQueuedByPoolRow, error)
|
||||
// Cheap queue-length check used by ChatMachine.Update when deciding
|
||||
// whether the chat is in a "1" sub-state.
|
||||
CountChatQueuedMessages(ctx context.Context, chatID uuid.UUID) (int64, error)
|
||||
@@ -477,6 +480,8 @@ type sqlcQuerier interface {
|
||||
// personal chat model overrides. It defaults to false when unset.
|
||||
GetChatPersonalModelOverridesEnabled(ctx context.Context) (bool, error)
|
||||
GetChatPlanModeInstructions(ctx context.Context) (string, error)
|
||||
// Pool fullness distinguishes capacity waits from worker pickup delays.
|
||||
GetChatQueuedForCapacity(ctx context.Context, arg GetChatQueuedForCapacityParams) (bool, error)
|
||||
GetChatQueuedMessageByID(ctx context.Context, arg GetChatQueuedMessageByIDParams) (ChatQueuedMessage, error)
|
||||
// Returns the queue head (lowest position, then lowest id).
|
||||
GetChatQueuedMessageHead(ctx context.Context, chatID uuid.UUID) (ChatQueuedMessage, error)
|
||||
@@ -507,17 +512,9 @@ type sqlcQuerier interface {
|
||||
// jsonb_array_elements never raises "cannot extract elements from a
|
||||
// scalar". Backed by idx_chat_messages_user_prompts.
|
||||
GetChatUserPromptsByChatID(ctx context.Context, arg GetChatUserPromptsByChatIDParams) ([]GetChatUserPromptsByChatIDRow, error)
|
||||
// Returns chats that workers may try to acquire. Candidates must be:
|
||||
// - in a worker-runnable execution status;
|
||||
// - unarchived; and
|
||||
// - missing ownership, carrying inconsistent ownership, or lacking a
|
||||
// fresh heartbeat for the assigned runner.
|
||||
//
|
||||
// Missing ownership is worker_id IS NULL. Inconsistent ownership is
|
||||
// runner_id IS NULL while worker_id is set. Stale ownership is no
|
||||
// heartbeat row for (chat_id, runner_id), or one older than
|
||||
// @stale_seconds by database time. Candidates are ordered by oldest
|
||||
// updated_at first so workers drain stale runnable chats predictably.
|
||||
// Returns a bounded, pool-interleaved set of chats that workers may acquire.
|
||||
// Interrupting chats finish active work first. Requires-action chats follow so
|
||||
// their runner can enforce the action deadline before new generations start.
|
||||
GetChatWorkerAcquisitionCandidates(ctx context.Context, arg GetChatWorkerAcquisitionCandidatesParams) ([]GetChatWorkerAcquisitionCandidatesRow, error)
|
||||
// Returns the global TTL for chat workspaces as a Go duration string.
|
||||
// Returns "0s" (disabled) when no value has been configured.
|
||||
|
||||
Reference in New Issue
Block a user