fix(coderd/x/chatd): resolve inflight race (#26460)

Using `WaitGroup.Go` must be synchronized with `WaitGroup.Wait`
according to [go docs](https://pkg.go.dev/sync#WaitGroup.Go):

> If the WaitGroup is empty, Go must happen before a
[WaitGroup.Wait](https://pkg.go.dev/sync#WaitGroup.Wait).

There were a couple of places in chatd that violated this principle.
This was caught as a data race in
https://github.com/coder/internal/issues/1599. This PR ensures that all
functions that spawn inflight goroutines synchronize with each other.

I also noticed that inflight goroutines may be spawned after the server
is closed, which was surprising and looked like a bug. This PR therefore
also introduces a mechanism that disallows spawning inflight goroutines
after the server is closed, and ensures that any code that tries doing
it logs an error.

Closes https://github.com/coder/internal/issues/1599.
This commit is contained in:
Hugo Dutka
2026-06-17 18:29:43 +02:00
committed by GitHub
parent 87de6dc23e
commit 684d904c00
6 changed files with 94 additions and 49 deletions
+6 -10
View File
@@ -76,15 +76,7 @@ func (p *Server) scheduleDebugCleanup(
return
}
// Acquire inflightMu around the positive Add so Close() cannot
// call drainInflight concurrently when the counter is at zero.
// See drainInflight for the WaitGroup contract this preserves.
p.inflightMu.Lock()
p.inflight.Add(1)
p.inflightMu.Unlock()
go func() {
defer p.inflight.Done()
if err := p.goInflight(func() {
cleanupCtx := context.WithoutCancel(ctx)
for attempt := 0; attempt < debugCleanupAttempts; attempt++ {
if attempt > 0 {
@@ -106,7 +98,11 @@ func (p *Server) scheduleDebugCleanup(
logFields = append(logFields, slog.Error(err))
p.logger.Warn(cleanupCtx, logMessage, logFields...)
}
}()
}); err != nil {
logFields := append([]slog.Field{slog.F("cleanup", logMessage)}, fields...)
logFields = append(logFields, slog.Error(err))
p.logger.Error(context.WithoutCancel(ctx), "failed to schedule chat debug cleanup", logFields...)
}
}
func (p *Server) newDebugAwareModel(