mirror of
https://github.com/coder/coder.git
synced 2026-09-24 15:04:27 +08:00
perf(coderd): reduce duplicated reads in push and webpush paths (#23115)
## Background A 5000-chat scaletest (~50k turns, ~2m45s wall time) completed successfully, but the main bottleneck was **DB pool starvation from repeated reads**, not individually expensive SQL. The push/webpush path showed a few especially noisy reads: - `GetLastChatMessageByRole` for push body generation - `GetEnabledChatProviders` + `GetChatModelConfigByID` for push summary model resolution - `GetWebpushSubscriptionsByUserID` for every webpush dispatch This PR keeps the optimizations that remove those duplicate reads while leaving stream behavior unchanged. ## What changes in this PR ### 1. Reuse resolved chat state for push notifications `maybeSendPushNotification` used to re-read the last assistant message and re-resolve the chat model/provider after `runChat` had already done that work. Now `runChat` returns the final assistant text plus the already-resolved model and provider keys, and the push goroutine uses that state directly. That removes the extra push-path reads for: - `GetLastChatMessageByRole` - the second `resolveChatModel` path - the provider/model lookups that came with that second resolution ### 2. Cache webpush subscriptions during dispatch `Dispatch()` previously hit `GetWebpushSubscriptionsByUserID` on every push. A small per-user in-memory cache now avoids those repeated reads. The follow-up fix keeps that optimization correct: `InvalidateUser()` bumps a per-user generation so an older in-flight fetch cannot repopulate the cache with pre-mutation data after subscribe/unsubscribe. That preserves the cache win without letting local subscription changes be silently overwritten by stale fetch results. ## Why this is safe - The push change only reuses data already produced during the same chat run. It does not change notification semantics; if there is no assistant text to summarize, the existing fallback body still applies. - The webpush change keeps the existing TTL and `410 Gone` cleanup behavior. The generation guard only prevents stale in-flight fetches from poisoning the shared cache after invalidation. - The final PR does **not** change stream setup, pubsub/relay behavior, or chat status snapshot timing. ## Deliberately not included - No stream-path optimization in `Subscribe`. - No inline pubsub message payloads. - No distributed cross-replica webpush cache invalidation.
This commit is contained in:
+41
-28
@@ -2038,6 +2038,7 @@ func (p *Server) processChat(ctx context.Context, chat database.Chat) {
|
||||
status := database.ChatStatusWaiting
|
||||
wasInterrupted := false
|
||||
lastError := ""
|
||||
runResult := runChatResult{}
|
||||
remainingQueuedMessages := []database.ChatQueuedMessage{}
|
||||
shouldPublishQueueUpdate := false
|
||||
var promotedMessage *database.ChatMessage
|
||||
@@ -2144,11 +2145,12 @@ func (p *Server) processChat(ctx context.Context, chat database.Chat) {
|
||||
p.publishChatPubsubEvent(chat, coderdpubsub.ChatEventKindStatusChange, nil)
|
||||
|
||||
if !wasInterrupted {
|
||||
p.maybeSendPushNotification(cleanupCtx, chat, status, lastError, logger)
|
||||
p.maybeSendPushNotification(cleanupCtx, chat, status, lastError, runResult, logger)
|
||||
}
|
||||
}()
|
||||
|
||||
if err := p.runChat(chatCtx, chat, logger); err != nil {
|
||||
runResult, err := p.runChat(chatCtx, chat, logger)
|
||||
if err != nil {
|
||||
if errors.Is(err, chatloop.ErrInterrupted) || errors.Is(context.Cause(chatCtx), chatloop.ErrInterrupted) {
|
||||
logger.Info(ctx, "chat interrupted")
|
||||
status = database.ChatStatusWaiting
|
||||
@@ -2205,11 +2207,18 @@ func isShutdownCancellation(
|
||||
return errors.Is(context.Cause(chatCtx), context.Canceled)
|
||||
}
|
||||
|
||||
type runChatResult struct {
|
||||
FinalAssistantText string
|
||||
PushSummaryModel fantasy.LanguageModel
|
||||
ProviderKeys chatprovider.ProviderAPIKeys
|
||||
}
|
||||
|
||||
func (p *Server) runChat(
|
||||
ctx context.Context,
|
||||
chat database.Chat,
|
||||
logger slog.Logger,
|
||||
) error {
|
||||
) (runChatResult, error) {
|
||||
result := runChatResult{}
|
||||
var (
|
||||
model fantasy.LanguageModel
|
||||
modelConfig database.ChatModelConfig
|
||||
@@ -2241,14 +2250,16 @@ func (p *Server) runChat(
|
||||
return nil
|
||||
})
|
||||
if err := g.Wait(); err != nil {
|
||||
return err
|
||||
return result, err
|
||||
}
|
||||
result.PushSummaryModel = model
|
||||
result.ProviderKeys = providerKeys
|
||||
// Fire title generation asynchronously so it doesn't block the
|
||||
// chat response. It uses a detached context so it can finish
|
||||
// even after the chat processing context is canceled.
|
||||
// Snapshot model so the goroutine doesn't race with the
|
||||
// model = cuModel reassignment below.
|
||||
titleModel := model
|
||||
// Snapshot the original chat model so the goroutine doesn't
|
||||
// race with the model = cuModel reassignment below.
|
||||
titleModel := result.PushSummaryModel
|
||||
p.inflight.Add(1)
|
||||
go func() {
|
||||
defer p.inflight.Done()
|
||||
@@ -2257,7 +2268,7 @@ func (p *Server) runChat(
|
||||
|
||||
prompt, err := chatprompt.ConvertMessagesWithFiles(ctx, messages, p.chatFileResolver(), logger)
|
||||
if err != nil {
|
||||
return xerrors.Errorf("build chat prompt: %w", err)
|
||||
return result, xerrors.Errorf("build chat prompt: %w", err)
|
||||
}
|
||||
if chat.ParentChatID.Valid {
|
||||
prompt = chatprompt.InsertSystem(prompt, defaultSubagentInstruction)
|
||||
@@ -2389,9 +2400,11 @@ func (p *Server) runChat(
|
||||
prompt = chatprompt.InsertSystem(prompt, resolvedUserPrompt)
|
||||
}
|
||||
|
||||
// Use the model config's context_limit as a fallback when the LLM // provider doesn't include context_limit in its response metadata
|
||||
// Use the model config's context_limit as a fallback when the LLM
|
||||
// provider doesn't include context_limit in its response metadata
|
||||
// (which is the common case).
|
||||
modelConfigContextLimit := modelConfig.ContextLimit
|
||||
var finalAssistantText string
|
||||
|
||||
persistStep := func(persistCtx context.Context, step chatloop.PersistedStep) error {
|
||||
// If the chat context has been canceled, bail out before
|
||||
@@ -2455,6 +2468,7 @@ func (p *Server) runChat(
|
||||
for _, block := range assistantBlocks {
|
||||
sdkParts = append(sdkParts, chatprompt.PartFromContent(block))
|
||||
}
|
||||
finalAssistantText = strings.TrimSpace(contentBlocksToText(sdkParts))
|
||||
assistantContent, marshalErr := chatprompt.MarshalParts(sdkParts)
|
||||
if marshalErr != nil {
|
||||
return marshalErr
|
||||
@@ -2630,7 +2644,7 @@ func (p *Server) runChat(
|
||||
chatprovider.UserAgent(),
|
||||
)
|
||||
if cuErr != nil {
|
||||
return xerrors.Errorf("resolve computer use model: %w", cuErr)
|
||||
return result, xerrors.Errorf("resolve computer use model: %w", cuErr)
|
||||
}
|
||||
model = cuModel
|
||||
}
|
||||
@@ -2796,7 +2810,11 @@ func (p *Server) runChat(
|
||||
p.logger.Warn(ctx, "failed to persist interrupted chat step", slog.Error(err))
|
||||
},
|
||||
})
|
||||
return err
|
||||
if err != nil {
|
||||
return result, err
|
||||
}
|
||||
result.FinalAssistantText = finalAssistantText
|
||||
return result, nil
|
||||
}
|
||||
|
||||
// buildProviderTools creates provider-native tool definitions
|
||||
@@ -3301,6 +3319,7 @@ func (p *Server) maybeSendPushNotification(
|
||||
chat database.Chat,
|
||||
status database.ChatStatus,
|
||||
lastError string,
|
||||
runResult runChatResult,
|
||||
logger slog.Logger,
|
||||
) {
|
||||
if p.webpushDispatcher == nil || p.webpushDispatcher.PublicKey() == "" {
|
||||
@@ -3328,23 +3347,17 @@ func (p *Server) maybeSendPushNotification(
|
||||
defer p.inflight.Done()
|
||||
pushCtx := context.WithoutCancel(ctx)
|
||||
pushBody := "Agent has finished running."
|
||||
|
||||
msg, err := p.db.GetLastChatMessageByRole(pushCtx, database.GetLastChatMessageByRoleParams{
|
||||
ChatID: chat.ID,
|
||||
Role: database.ChatMessageRoleAssistant,
|
||||
})
|
||||
if err == nil {
|
||||
content, parseErr := chatprompt.ParseContent(msg)
|
||||
if parseErr == nil {
|
||||
assistantText := strings.TrimSpace(contentBlocksToText(content))
|
||||
if assistantText != "" {
|
||||
model, _, keys, resolveErr := p.resolveChatModel(pushCtx, chat)
|
||||
if resolveErr == nil {
|
||||
if summary := generatePushSummary(pushCtx, chat.Title, assistantText, model, keys, logger); summary != "" {
|
||||
pushBody = summary
|
||||
}
|
||||
}
|
||||
}
|
||||
assistantText := strings.TrimSpace(runResult.FinalAssistantText)
|
||||
if assistantText != "" && runResult.PushSummaryModel != nil {
|
||||
if summary := generatePushSummary(
|
||||
pushCtx,
|
||||
chat.Title,
|
||||
assistantText,
|
||||
runResult.PushSummaryModel,
|
||||
runResult.ProviderKeys,
|
||||
logger,
|
||||
); summary != "" {
|
||||
pushBody = summary
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -2234,12 +2234,10 @@ func TestSuccessfulChatSendsWebPushWithSummary(t *testing.T) {
|
||||
const assistantText = "I have completed the task successfully and all tests are passing now."
|
||||
const summaryText = "Completed task and verified all tests pass."
|
||||
|
||||
var nonStreamingRequests atomic.Int32
|
||||
openAIURL := chattest.NewOpenAI(t, func(req *chattest.OpenAIRequest) chattest.OpenAIResponse {
|
||||
if !req.Stream {
|
||||
// Non-streaming calls are used for title
|
||||
// generation and push summary generation.
|
||||
// Return the summary text for both — the title
|
||||
// result is irrelevant to this test.
|
||||
nonStreamingRequests.Add(1)
|
||||
return chattest.OpenAINonStreamingResponse(summaryText)
|
||||
}
|
||||
return chattest.OpenAIStreamingResponse(
|
||||
@@ -2286,6 +2284,63 @@ func TestSuccessfulChatSendsWebPushWithSummary(t *testing.T) {
|
||||
"push body should be the LLM-generated summary")
|
||||
require.NotEqual(t, "Agent has finished running.", msg.Body,
|
||||
"push body should not use the default fallback text")
|
||||
require.Equal(t, int32(1), nonStreamingRequests.Load(),
|
||||
"expected exactly one non-streaming request for push summary generation")
|
||||
}
|
||||
|
||||
func TestSuccessfulChatSendsWebPushFallbackWithoutSummaryForEmptyAssistantText(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
db, ps := dbtestutil.NewDB(t)
|
||||
ctx := testutil.Context(t, testutil.WaitLong)
|
||||
|
||||
var nonStreamingRequests atomic.Int32
|
||||
openAIURL := chattest.NewOpenAI(t, func(req *chattest.OpenAIRequest) chattest.OpenAIResponse {
|
||||
if !req.Stream {
|
||||
nonStreamingRequests.Add(1)
|
||||
return chattest.OpenAINonStreamingResponse("unexpected summary request")
|
||||
}
|
||||
return chattest.OpenAIStreamingResponse(
|
||||
chattest.OpenAITextChunks(" ")...,
|
||||
)
|
||||
})
|
||||
|
||||
mockPush := &mockWebpushDispatcher{}
|
||||
|
||||
logger := slogtest.Make(t, &slogtest.Options{IgnoreErrors: true})
|
||||
server := chatd.New(chatd.Config{
|
||||
Logger: logger,
|
||||
Database: db,
|
||||
ReplicaID: uuid.New(),
|
||||
Pubsub: ps,
|
||||
PendingChatAcquireInterval: 10 * time.Millisecond,
|
||||
InFlightChatStaleAfter: testutil.WaitSuperLong,
|
||||
WebpushDispatcher: mockPush,
|
||||
})
|
||||
t.Cleanup(func() {
|
||||
require.NoError(t, server.Close())
|
||||
})
|
||||
|
||||
user, model := seedChatDependencies(ctx, t, db)
|
||||
setOpenAIProviderBaseURL(ctx, t, db, openAIURL)
|
||||
|
||||
_, err := server.CreateChat(ctx, chatd.CreateOptions{
|
||||
OwnerID: user.ID,
|
||||
Title: "empty-summary-push-test",
|
||||
ModelConfigID: model.ID,
|
||||
InitialUserContent: []codersdk.ChatMessagePart{codersdk.ChatMessageText("do the thing")},
|
||||
})
|
||||
require.NoError(t, err)
|
||||
|
||||
testutil.Eventually(ctx, t, func(ctx context.Context) bool {
|
||||
return mockPush.dispatchCount.Load() >= 1
|
||||
}, testutil.IntervalFast)
|
||||
|
||||
msg := mockPush.getLastMessage()
|
||||
require.Equal(t, "Agent has finished running.", msg.Body,
|
||||
"push body should fall back when the final assistant text is empty")
|
||||
require.Equal(t, int32(0), nonStreamingRequests.Load(),
|
||||
"push summary should not be requested when final assistant text has no usable text")
|
||||
}
|
||||
|
||||
func TestComputerUseSubagentToolsAndModel(t *testing.T) {
|
||||
|
||||
Reference in New Issue
Block a user