Compare commits

...
Author SHA1 Message Date
candieduniverse 87b4f5c57c docs: add local latency observer plan 2026-03-17 11:04:07 -07:00
candieduniverse ec618b8456 docs: strengthen extraction guidance 2026-03-17 11:04:07 -07:00
candieduniverse a258b81248 docs: refine remote latency technique plans 2026-03-17 11:04:07 -07:00
candieduniverse 406f7a6bd3 docs: add remote latency technique plans 2026-03-17 11:04:07 -07:00
candieduniverse cd22717dc1 docs: analyze remote workspace latency branch 2026-03-17 11:04:07 -07:00
candieduniverse ef4e3ef31e test: fix latency validation completion scenario 2026-03-17 11:04:07 -07:00
candieduniverse 58fb643e01 test: add simulated remote latency validation harness 2026-03-17 11:04:07 -07:00
candieduniverse 3171f4594d docs: note task initialization telemetry summary support 2026-03-17 11:04:07 -07:00
candieduniverse 395179d8ae telemetry: track task initialization latency summaries 2026-03-17 11:04:07 -07:00
candieduniverse 5550f7edeb Remove mention of rollout from implementation plan doc 2026-03-17 11:03:56 -07:00
candieduniverse d0c4f4cc82 test: cover active task delta transport 2026-03-17 11:03:56 -07:00
candieduniverse 7dd39625ce docs: capture latency rollout tooling 2026-03-17 11:03:56 -07:00
candieduniverse 34693d1814 tooling: compare task latency summaries 2026-03-17 11:03:56 -07:00
candieduniverse 01c296d83c tooling: add task latency summary helper 2026-03-17 11:03:56 -07:00
candieduniverse b36efc325f test: cover persisted recovery after ephemeral flush 2026-03-17 11:03:56 -07:00
candieduniverse 7c71166af6 docs: mark rollout safeguards complete 2026-03-17 11:03:56 -07:00
candieduniverse 7af6eeca11 test: cover tab cache invalidation refresh 2026-03-17 11:03:56 -07:00
candieduniverse 25ea6d20b3 test: limit chat row rerenders 2026-03-17 11:03:56 -07:00
candieduniverse bb1de9a778 test: cover finalized request metadata state 2026-03-17 11:03:56 -07:00
candieduniverse c3dd7530d7 docs: mark tab prompt-quality coverage complete 2026-03-17 11:03:56 -07:00
candieduniverse 3350c3bbba docs: mark request metadata coverage complete 2026-03-17 11:03:56 -07:00
candieduniverse e9a3d4e25a test: cover api request metric finalization 2026-03-17 11:03:56 -07:00
candieduniverse d411ddf91a docs: record remote policy decision 2026-03-17 11:03:56 -07:00
candieduniverse 8e3be49f4e test: cover remote state update cadence path 2026-03-17 11:03:56 -07:00
candieduniverse 585104b216 test: cover task ui debug counters 2026-03-17 11:03:55 -07:00
candieduniverse eb31fc4132 docs: mark delta resync safeguards complete 2026-03-17 11:03:55 -07:00
candieduniverse 561edc901d dev: expose task ui debug counters 2026-03-17 11:03:55 -07:00
candieduniverse 037faf854a docs: mark debug logging support complete 2026-03-17 11:03:55 -07:00
candieduniverse f720177fc2 docs: mark hostbridge audit complete 2026-03-17 11:03:55 -07:00
candieduniverse 4af51bf928 test: cover controller state posting priorities 2026-03-17 11:03:55 -07:00
candieduniverse 69ad616362 docs: mark chat selector optimizations complete 2026-03-17 11:03:55 -07:00
candieduniverse be7019e380 refactor: memoize chat view render selectors 2026-03-17 11:03:55 -07:00
candieduniverse d788758bab test: cover coalesced low-stakes tool grouping 2026-03-17 11:03:55 -07:00
candieduniverse 60f620cd68 refactor: reduce request row render churn 2026-03-17 11:03:55 -07:00
candieduniverse 9a29714f20 test: cover periodic ephemeral message flushes 2026-03-17 11:03:55 -07:00
candieduniverse 591ff52910 docs: update latency plan test progress 2026-03-17 11:03:55 -07:00
candieduniverse 36a81ea62a test: cover usage update throttling 2026-03-17 11:03:55 -07:00
candieduniverse f2b30a718d docs: update latency plan progress 2026-03-17 11:03:55 -07:00
candieduniverse c7a9b3fd47 test: cover disabled telemetry latency instrumentation 2026-03-17 11:03:55 -07:00
candieduniverse 875e0b8f5f Add state update scheduler drain coverage 2026-03-17 11:03:55 -07:00
candieduniverse 6e1314f780 Add snapshot rehydration coverage 2026-03-17 11:03:55 -07:00
candieduniverse 205bd52f51 Cover partial completion stability 2026-03-17 11:03:55 -07:00
candieduniverse 5276df3431 Add scheduler drain coverage 2026-03-17 11:03:55 -07:00
candieduniverse ae813e6153 Add ordered task delta application coverage 2026-03-17 11:03:55 -07:00
candieduniverse dc03e1cb02 Record completed latency cache coverage 2026-03-17 11:03:55 -07:00
candieduniverse d0314185cb Cover delta resync fallback behavior 2026-03-17 11:03:55 -07:00
candieduniverse f74fd8d30f Strengthen partial message patch coverage 2026-03-17 11:03:55 -07:00
candieduniverse b7c1c233d1 Add delta snapshot convergence coverage 2026-03-17 11:03:55 -07:00
candieduniverse 670324cf17 Update latency plan progress tracking 2026-03-17 11:03:55 -07:00
candieduniverse 01a23c6a43 Cache static environment details for remote tasks 2026-03-17 11:03:55 -07:00
candieduniverse 75d19ef9a7 Reduce unnecessary terminal cooldown waits 2026-03-17 11:03:55 -07:00
candieduniverse ba3e9dcc18 Improve remote request-boundary tab caching 2026-03-17 11:03:54 -07:00
candieduniverse c1616e1470 telemetry: record latency hot-path histogram metrics 2026-03-17 11:03:54 -07:00
candieduniverse 82ef1877b6 perf: skip no-op partial message merges 2026-03-17 11:03:54 -07:00
candieduniverse ffb202fdfa perf: skip no-op full-state snapshot merges 2026-03-17 11:03:54 -07:00
candieduniverse 6a7d0d8379 perf: skip no-op task delta deletes 2026-03-17 11:03:54 -07:00
candieduniverse ca3883bc32 perf: avoid no-op webview task delta updates 2026-03-17 11:03:54 -07:00
candieduniverse 7ebb1f4a41 test: cover request-boundary cache load retries 2026-03-17 11:03:54 -07:00
candieduniverse dedf9dd427 test: cover controller scheduler disposal during flush 2026-03-17 11:03:54 -07:00
candieduniverse 5bdebf4efd test: cover scheduler disposal during active flush 2026-03-17 11:03:54 -07:00
candieduniverse 309a278131 test: cover scheduler follow-up flushes 2026-03-17 11:03:54 -07:00
candieduniverse 7863cd89ac test: cover partial completion persistence 2026-03-17 11:03:54 -07:00
candieduniverse b82d9b8686 telemetry: add callback delivery stats for task UI deltas 2026-03-17 11:03:54 -07:00
candieduniverse 1d26fc09b2 telemetry: instrument task UI delta delivery 2026-03-17 11:03:54 -07:00
candieduniverse 7b78f98571 test: cover task UI delta stream delivery 2026-03-17 11:03:54 -07:00
candieduniverse 4b203acb62 test: cover ephemeral message state flush behavior 2026-03-17 11:03:54 -07:00
candieduniverse c50220d8a5 Reduce chat rerenders during streaming updates 2026-03-17 11:03:54 -07:00
candieduniverse e8a0b99ada Reduce redundant state syncs during streaming 2026-03-17 11:03:54 -07:00
candieduniverse 76b77f7f2f Add task metadata delta sync for hot UI state 2026-03-17 11:03:54 -07:00
candieduniverse c8e4a355bc Reuse filtered tab snapshots in task prompts 2026-03-17 11:03:54 -07:00
candieduniverse 121a120ed6 Cache request-boundary tab queries 2026-03-17 11:03:54 -07:00
candieduniverse d498777e97 Add state and partial transport latency telemetry 2026-03-17 11:03:54 -07:00
candieduniverse 36b8f88253 Add latency rollout flags 2026-03-17 11:03:54 -07:00
candieduniverse cf27ba7307 Add latency cadence overrides 2026-03-17 11:03:54 -07:00
candieduniverse 1a081da9e7 Wire task UI delta stream through protobus 2026-03-17 11:03:54 -07:00
candieduniverse 845bcea85c Harden task UI delta resync handling 2026-03-17 11:03:53 -07:00
candieduniverse fcefb52b62 Resync task state after delta gaps 2026-03-17 11:03:53 -07:00
candieduniverse a52b0044ec Track remote latency chunk timing 2026-03-17 11:03:53 -07:00
candieduniverse b70549c39f Cache hostbridge tab queries 2026-03-17 11:03:53 -07:00
candieduniverse 9a2d4adfa6 Add task UI delta channel for active message updates 2026-03-17 11:03:53 -07:00
candieduniverse 988aa6209a Checking in progress on latency improvements for remote workspaces 2026-03-17 11:03:53 -07:00
candieduniverse 2e8a6ffe23 Add controller state update coalescing 2026-03-17 11:03:53 -07:00
candieduniverse c0430e47c2 Checking in progress on latency improvements for remote workspaces 2026-03-17 11:03:53 -07:00
candieduniverse be7ef15840 Add remote latency instrumentation foundations 2026-03-17 11:03:53 -07:00
candieduniverse a8d0e86dab Create implementation plan doc 2026-03-17 11:03:53 -07:00
74 changed files with 9924 additions and 377 deletions
+32
View File
@@ -125,6 +125,38 @@ ERROR_SERVICE_API_KEY=your-posthog-error-tracking-api-key
# E2E_TEST=true
# IS_TEST=true
# ============================================================================
# REMOTE WORKSPACE LATENCY TUNING / DEBUGGING
# ============================================================================
# These are internal rollout/debug flags for the remote-workspace latency work.
# Leave them unset unless you're explicitly validating latency behavior.
# Enable extra scheduler latency debug logging
# CLINE_DEBUG_LATENCY=1
# Disable individual latency features for A/B validation
# CLINE_DISABLE_PRESENTATION_SCHEDULER=true
# CLINE_DISABLE_EPHEMERAL_MESSAGE_PERSISTENCE=true
# CLINE_DISABLE_TASK_UI_DELTA_SYNC=true
# Override scheduler cadences (milliseconds)
# CLINE_PRESENTATION_CADENCE_MS=40
# CLINE_PRESENTATION_LOW_CADENCE_MS=125
# CLINE_REMOTE_PRESENTATION_CADENCE_MS=90
# CLINE_REMOTE_PRESENTATION_LOW_CADENCE_MS=125
# CLINE_STATE_UPDATE_CADENCE_MS=16
# CLINE_STATE_UPDATE_LOW_CADENCE_MS=150
# CLINE_REMOTE_STATE_UPDATE_CADENCE_MS=110
# CLINE_REMOTE_STATE_UPDATE_LOW_CADENCE_MS=150
# CLINE_USAGE_UPDATE_CADENCE_MS=250
# CLINE_REMOTE_USAGE_UPDATE_CADENCE_MS=400
# Override request-boundary cache TTLs (milliseconds)
# CLINE_REQUEST_BOUNDARY_CACHE_TTL_MS=500
# CLINE_REMOTE_REQUEST_BOUNDARY_CACHE_TTL_MS=1000
# CLINE_ENVIRONMENT_DETAILS_STATIC_CACHE_TTL_MS=30000
# CLINE_REMOTE_ENVIRONMENT_DETAILS_STATIC_CACHE_TTL_MS=60000
# ============================================================================
# USAGE INSTRUCTIONS
# ============================================================================
@@ -0,0 +1,775 @@
## Remote Workspace Latency Branch Analysis Report
This document analyzes the work implemented on branch `eve_troubleshooting-remote-workspaces` relative to `main`, with the goal of determining which changes most improve **user-perceived latency** for remote workspaces in Cline, and how to split that work into smaller, reviewable pull requests.
It is intentionally written as a decision document rather than an implementation plan. The branch already contains substantial implementation across backend scheduling, persistence, transport, frontend state application, telemetry, and validation harnessing. What we need now is a clear ranking of which pieces most improve day-to-day UX in remote workspaces, which are foundational but lower direct impact, and which should be separated into later or optional PRs.
---
## Executive Summary
The highest-impact improvements in this branch are the ones that **remove synchronous work from the streaming hot path** and **reduce the number and size of extension-host ↔ webview updates while a response is being generated**.
The most valuable changes, in order, are:
1. **Ephemeral partial message persistence split** — stops saving partial streaming mutations to disk on every update.
2. **Assistant presentation scheduling** — stops awaiting presentation work on every streamed chunk.
3. **Controller full-state coalescing** — reduces repeated full `ExtensionState` snapshot pushes.
4. **Task UI delta sync** — shifts active task execution away from snapshot-heavy transport toward targeted updates.
5. **Usage/metadata throttling** — removes low-value churn from token/cost updates.
Those five together represent the main UX win for remote workspaces. Everything else in the branch is either enabling infrastructure, measurement, correctness/safety support, or medium-value follow-on optimization.
---
## Scope Reviewed
Reviewed inputs:
- `docs/remote-workspace-latency-improvement-plan.md`
- branch diff stat versus `main`
- key implementation files in task execution, message state, controller state posting, latency utilities, delta transport, frontend delta application, request-boundary caching, and validation tooling
Notable implementation areas in the branch:
- `src/core/task/index.ts`
- `src/core/task/message-state.ts`
- `src/core/task/TaskPresentationScheduler.ts`
- `src/core/controller/index.ts`
- `src/core/controller/StateUpdateScheduler.ts`
- `src/shared/TaskUiDelta.ts`
- `webview-ui/src/context/ExtensionStateContext.tsx`
- `webview-ui/src/context/taskUiDeltaState.ts`
- `src/core/task/RequestBoundaryCache.ts`
- `src/core/task/latency.ts`
- validation/telemetry scripts and tests
---
## Evaluation Framework
Each technique is scored on a 15 scale in four dimensions:
- **User-perceived latency impact**: how much it improves responsiveness/smoothness in remote workspaces
- **Breadth of effect**: how often it helps across normal task execution
- **Extraction ease**: how cleanly it can be isolated into its own PR
- **Risk / coupling**: implementation and regression risk; 5 means low risk / low coupling
I also provide a **recommended priority tier**:
- **Tier 1**: highest ROI, should be split out first
- **Tier 2**: meaningful but more coupled or secondary
- **Tier 3**: foundational, supporting, or follow-on work
---
## Findings by Technique
### 1. Ephemeral partial message persistence split
**Primary files:**
- `src/core/task/message-state.ts`
- `src/core/task/index.ts`
- `src/core/task/EphemeralMessageFlushScheduler.ts`
**What changed**
The branch adds explicit ephemeral mutation APIs:
- `addToClineMessagesEphemeral(...)`
- `updateClineMessageEphemeral(...)`
- `flushClineMessagesAndUpdateHistory()`
Partial `say(...)` and `ask(...)` updates now use in-memory mutation plus change notification, rather than synchronously persisting every partial update. A periodic safety flush is added, and durable saves still happen at semantic boundaries.
**Why it matters for remote workspaces**
This is likely the single most important improvement in the branch.
Before this change, each partial token/reasoning/tool-progress update could trigger disk persistence plus task-history update work. In a remote workspace, that cost compounds:
- remote extension-host filesystem latency
- repeated JSON serialization / write amplification
- repeated history metadata recomputation
- possible indirect state-post churn caused by those updates
Persisting every animation frame is the wrong abstraction. Partial streaming text is ephemeral UI state; recovery correctness only requires persistence at durable checkpoints plus an occasional safety flush.
**User-visible effect**
- fewer stalls during streaming
- smoother “typing” feel
- less chattery behavior under remote host latency
- reduced pauses during reasoning and tool progress updates
**Why this ranks so high**
It attacks a source of latency that is both **expensive** and **unnecessary**. Unlike some optimizations that only reduce overhead indirectly, this one removes synchronous durability work directly from the hot path.
**Score**
- User-perceived latency impact: **5/5**
- Breadth of effect: **5/5**
- Extraction ease: **4/5**
- Risk / coupling: **3/5**
- **Tier: 1**
**Notes on PR extraction**
This can be its own focused PR if kept scoped to:
- message-state ephemeral APIs
- task streaming callsite conversion for partial updates
- safety flush scheduler
- targeted tests
It is somewhat coupled to presentation behavior, but not tightly coupled to task UI deltas.
---
### 2. Assistant presentation scheduling
**Primary files:**
- `src/core/task/TaskPresentationScheduler.ts`
- `src/core/task/index.ts`
- `src/core/task/latency.ts`
**What changed**
The branch introduces `TaskPresentationScheduler` and routes chunk-driven presentation through `scheduleAssistantPresentation(...)` instead of awaiting `presentAssistantMessage()` on every chunk. Immediate flushes are preserved for semantic boundaries.
Cadence is remote-aware:
- local: lower delay
- remote: more aggressive coalescing
**Why it matters for remote workspaces**
This directly decouples provider stream ingestion from UI presentation. That is the right systems design move. A model can emit tokens at machine cadence; the UI should update at a human-comfort cadence.
In the old model, each chunk could drag presentation work directly into the streaming loop. In remote mode, that means the chunk loop can become paced by downstream UI/persistence/transport side effects instead of by stream availability.
**User-visible effect**
- faster time to continued stream processing after first visible output
- reduced jitter when many small chunks arrive
- better smoothness under high RTT remote connections
**Limitations / caveats**
This improvement is substantial, but by itself it does not eliminate expensive work if the flush path still performs persistence or full-state posting too often. Its value is highest when combined with ephemeral persistence split and state-update coalescing.
**Score**
- User-perceived latency impact: **5/5**
- Breadth of effect: **5/5**
- Extraction ease: **4/5**
- Risk / coupling: **4/5**
- **Tier: 1**
**Notes on PR extraction**
This is one of the cleanest standalone PRs in the branch:
- scheduler class
- task integration
- remote-aware cadence helper
- scheduler tests
This should be one of the first PRs split out.
---
### 3. Controller full-state coalescing
**Primary files:**
- `src/core/controller/StateUpdateScheduler.ts`
- `src/core/controller/index.ts`
- `src/core/controller/state/subscribeToState.ts`
**What changed**
The branch changes `postStateToWebview()` from immediate-fire behavior into a scheduled/coalesced mechanism with priorities. Full-state flushes remain possible immediately when needed, but streaming-state churn is now collapsed.
**Why it matters for remote workspaces**
Full-state pushes are expensive in remote environments because they imply:
- building a large `ExtensionState`
- serializing it
- transferring it
- parsing it locally
- React state merge/reconciliation
Reducing the number of full snapshots has a direct impact on user-perceived smoothness, especially when a task is actively streaming.
**User-visible effect**
- fewer UI stalls caused by repeated snapshot delivery
- reduced burstiness in the chat UI while streaming
- lower CPU and transport overhead on both sides of the remote boundary
**Why it ranks slightly below the first two**
This is a major improvement, but it mainly attacks the *transport/snapshot* part of the problem. The first two changes remove synchronous work even earlier in the hot path.
**Score**
- User-perceived latency impact: **4.5/5**
- Breadth of effect: **5/5**
- Extraction ease: **4/5**
- Risk / coupling: **4/5**
- **Tier: 1**
**Notes on PR extraction**
Very suitable for its own PR:
- scheduler class
- controller integration
- state-post metrics
- tests validating coalescing and dirty-followup behavior
---
### 4. Task UI delta sync for active task execution
**Primary files:**
- `src/shared/TaskUiDelta.ts`
- `src/core/controller/ui/subscribeToTaskUiDeltas.ts`
- `src/core/task/message-state.ts`
- `webview-ui/src/context/ExtensionStateContext.tsx`
- `webview-ui/src/context/taskUiDeltaState.ts`
**What changed**
The branch adds a delta transport for active task execution:
- message added / updated / deleted
- metadata updated
- resync event
- sequence-based ordering and resync fallback
The frontend applies deltas incrementally and falls back to full-state resync on sequence mismatch.
**Why it matters for remote workspaces**
Architecturally, this is the biggest long-term win, because it changes the transport model from “keep re-sending snapshots” to “send only what changed.” In remote environments, that is exactly the right direction.
However, its *incremental user-visible gain* depends on whether the earlier coalescing changes already brought snapshot traffic down far enough. If snapshot coalescing plus partial message updates already produce acceptable smoothness, deltas become more of a scalability and polish improvement than the first-order fix.
**User-visible effect**
- fewer unnecessary full-list state replacements
- smaller payloads during active streaming
- better frontend patch locality
- improved headroom as tasks become longer/more active
**Why this is not #1 despite strong architecture**
Because it is more coupled and more invasive. The first wave of user QoL improvement likely comes from not doing expensive work too often. Delta sync then further reduces the remaining transport cost. It is highly valuable, but not the best *first* minimal PR unless the goal is architectural modernization rather than fastest latency win.
**Score**
- User-perceived latency impact: **4/5**
- Breadth of effect: **4/5**
- Extraction ease: **2.5/5**
- Risk / coupling: **2.5/5**
- **Tier: 2**
**Notes on PR extraction**
This should likely be split as a later PR after the foundational hot-path reductions are isolated.
Recommended sub-scope:
1. shared delta types + backend publisher infrastructure
2. frontend delta application + resync logic
3. enable specific message mutation classes to publish deltas
Trying to ship all delta-related work in the first PR set would likely obscure the simpler, higher-ROI changes.
---
### 5. Usage / metadata throttling
**Primary files:**
- `src/core/task/TaskUsageUpdateScheduler.ts`
- `src/core/task/index.ts`
- related request-row/frontend handling
**What changed**
Usage token/cost updates are no longer pushed at raw chunk cadence. Internal accounting remains accurate, but UI flushing is scheduled at a slower cadence with final flush on completion.
**Why it matters for remote workspaces**
This is a clean example of removing low-value churn. Users do not care if token counts animate 20 times per second. They care that the answer feels live.
By throttling metadata updates separately, the branch reduces incidental full-state or delta churn during generation.
**User-visible effect**
- less UI thrash in request headers and usage rows
- fewer distracting metadata changes during streaming
- slight improvement in perceived smoothness
**Score**
- User-perceived latency impact: **3.5/5**
- Breadth of effect: **4/5**
- Extraction ease: **4/5**
- Risk / coupling: **4/5**
- **Tier: 2**
**Notes on PR extraction**
Good candidate for a focused PR, especially after presentation + state coalescing are split out.
---
### 6. Webview render churn reduction
**Primary files:**
- `webview-ui/src/context/ExtensionStateContext.tsx`
- `webview-ui/src/components/chat/RequestStartRow.tsx`
- `webview-ui/src/components/chat/requestStartRowState.ts`
- `webview-ui/src/components/messages/MessageRenderer.tsx`
- message merge helpers
**What changed**
The branch improves frontend state merge patterns and factors request-row derivation logic to reduce unnecessary repeated scans and rerenders.
**Why it matters for remote workspaces**
Remote workspaces amplify backend transport cost first, but frontend churn still matters because every incoming update eventually becomes React work. If the frontend repaints too much, the user still experiences jitter.
That said, this is usually the second-order optimization after transport/event volume has already been reduced.
**User-visible effect**
- less flicker risk
- fewer unnecessary rerenders of unrelated rows
- more stable request status rendering
**Score**
- User-perceived latency impact: **3/5**
- Breadth of effect: **3.5/5**
- Extraction ease: **3/5**
- Risk / coupling: **4/5**
- **Tier: 2**
**Notes on PR extraction**
This should probably be a dedicated frontend optimization PR and not mixed into the first backend-focused latency PRs.
---
### 7. Request-boundary caching and environment-detail cost reduction
**Primary files:**
- `src/core/task/RequestBoundaryCache.ts`
- `src/core/task/index.ts`
- `src/hosts/vscode/hostbridge/window/getOpenTabs.ts`
- `src/hosts/vscode/hostbridge/window/getVisibleTabs.ts`
**What changed**
The branch caches open tabs, visible tabs, workspace config, CLI tool detection, and related request-boundary data with remote-aware TTLs.
**Why it matters for remote workspaces**
This reduces repeated request-start overhead rather than streaming-path overhead. It matters, especially for tasks that trigger repeated request loops, but it does not change the “live answer feels sluggish” symptom as much as the hot-path fixes.
**User-visible effect**
- somewhat faster transition into the next request
- less repeated overhead when building environment details
- improved prompt setup latency in remote mode
**Score**
- User-perceived latency impact: **2.5/5**
- Breadth of effect: **4/5**
- Extraction ease: **4/5**
- Risk / coupling: **4.5/5**
- **Tier: 3**
**Notes on PR extraction**
This is actually an excellent small PR because it is easy to isolate and low risk, but it should not be described as the primary remote latency fix. It is a supportive optimization.
---
### 8. Instrumentation, telemetry, and validation harnessing
**Primary files:**
- `src/core/task/latency.ts`
- `src/services/telemetry/TelemetryService.ts`
- `src/services/telemetry/taskLatencySummary.ts`
- `scripts/analyze-task-latency-metrics.mjs`
- `scripts/compare-task-latency-metrics.mjs`
- `scripts/validate-latency-scenarios.ts`
- `.env.example`
**What changed**
The branch adds broad instrumentation for presentation frequency, state payloads, partial message traffic, persistence latency, chunk-to-webview timing, and remote-awareness. It also adds JSONL analysis helpers and a validation harness that can simulate local vs remote behavior and feature-flag variants.
**Why it matters for remote workspaces**
This is essential for confidence and rollout, but it is not itself the main direct latency improvement. Its value is that it lets the team quantify which changes matter most and compare candidate splits.
The current validation output already suggests that simulated remote mode reduces full-state update count and bytes, but the scripted scenario still times out before a clean completion signal. So this infrastructure is promising, but not yet a complete “proof harness.”
**User-visible effect**
- indirect only
**Score**
- User-perceived latency impact: **1/5**
- Breadth of effect: **5/5** (for engineering decision-making)
- Extraction ease: **5/5**
- Risk / coupling: **5/5**
- **Tier: 3**
**Notes on PR extraction**
This should almost certainly be split into an early standalone PR or first PR in the stack. It makes later PRs easier to justify and safer to evaluate.
---
### 9. Remote-aware policy / cadence configuration
**Primary files:**
- `src/core/task/latency.ts`
- `.env.example`
- scheduler integration points
**What changed**
The branch centralizes remote detection and exposes remote/local cadence defaults plus env var overrides.
**Why it matters for remote workspaces**
This is important productization work: the same cadence should not be assumed optimal in local and remote environments. But by itself, configuration does not help unless the underlying schedulers exist.
**Score**
- User-perceived latency impact: **2/5**
- Breadth of effect: **4/5**
- Extraction ease: **4/5**
- Risk / coupling: **4.5/5**
- **Tier: 3**
---
## Ranked Impact Table
| Rank | Technique | Latency Impact | Breadth | Extraction Ease | Risk/Coupling | Tier |
|---|---|---:|---:|---:|---:|---|
| 1 | Ephemeral partial persistence split | 5.0 | 5.0 | 4.0 | 3.0 | Tier 1 |
| 2 | Assistant presentation scheduling | 5.0 | 5.0 | 4.0 | 4.0 | Tier 1 |
| 3 | Controller full-state coalescing | 4.5 | 5.0 | 4.0 | 4.0 | Tier 1 |
| 4 | Task UI delta sync | 4.0 | 4.0 | 2.5 | 2.5 | Tier 2 |
| 5 | Usage/metadata throttling | 3.5 | 4.0 | 4.0 | 4.0 | Tier 2 |
| 6 | Webview render churn reduction | 3.0 | 3.5 | 3.0 | 4.0 | Tier 2 |
| 7 | Request-boundary caching | 2.5 | 4.0 | 4.0 | 4.5 | Tier 3 |
| 8 | Remote-aware cadence/config | 2.0 | 4.0 | 4.0 | 4.5 | Tier 3 |
| 9 | Instrumentation / validation harness | 1.0 direct | 5.0 eng value | 5.0 | 5.0 | Tier 3 |
---
## If We Could Only Land One or Two Changes
This branch is strongest as a bundle, but if schedule or review bandwidth forces an even more minimal extraction, I would prioritize as follows:
### If only one change can land
Land **ephemeral partial persistence split** first.
Reason: it removes the most clearly unnecessary synchronous work from the streaming path. Even if presentation/state update behavior remains imperfect, stopping per-partial durability work should still materially reduce remote stalls and write amplification.
### If two changes can land
Land:
1. **Assistant presentation scheduler**
2. **Ephemeral partial persistence split**
Reason: together they decouple the stream from both presentation cadence and persistence cadence. That combination most directly changes how “live” the product feels.
### If three or four changes can land
Add, in order:
3. **Controller full-state coalescing**
4. **Usage/metadata throttling**
At that point, most of the hot-path churn should be removed even before delta sync lands.
---
## Most Important Conclusion
If the goal is to improve **user quality of life in remote workspaces**, the branchs highest-value idea is:
> **Stop treating every streamed chunk as a durable, full-state, immediately-presented event.**
The strongest improvements all follow from that principle:
- coalesce presentation
- defer durability for ephemeral updates
- coalesce full-state snapshots
- move active task execution toward deltas instead of snapshots
Everything else is either support for that model or further optimization around it.
---
## Recommended PR Decomposition
Below is the decomposition I would recommend for turning this branch into smaller minimal changesets.
### PR 1 — Instrumentation and latency analysis scaffolding
**Include:**
- telemetry additions for task latency metrics
- summary/compare scripts
- env flags/docs for validation
- minimal validation harness if it can be kept self-contained
**Why first:**
- lowest risk
- improves confidence in all later PRs
- provides before/after proof points
**Keep out:**
- scheduler behavior changes
- delta transport
- persistence changes
---
### PR 2 — Assistant presentation scheduler
**Include:**
- `TaskPresentationScheduler`
- task integration / scheduling of `presentAssistantMessage`
- remote-aware cadence helper for presentation
- tests
**Why second:**
- high impact
- conceptually narrow
- easiest major UX win to explain
**Keep out:**
- message-state ephemeral persistence
- controller snapshot coalescing
- task UI deltas
---
### PR 3 — Ephemeral partial message persistence split
**Include:**
- ephemeral message APIs
- dirty tracking
- periodic flush scheduler
- task streaming callsite conversion
- persistence-focused tests
**Why third:**
- likely the largest raw hot-path cost reduction
- still explainable as a coherent architecture change
**Keep out:**
- delta transport
- frontend state delta logic
---
### PR 4 — Controller full-state coalescing
**Include:**
- `StateUpdateScheduler`
- controller integration
- state posting priority behavior
- full-state payload instrumentation updates
**Why fourth:**
- complements PR2 and PR3
- directly targets remote snapshot pressure
---
### PR 5 — Usage/metadata throttling
**Include:**
- `TaskUsageUpdateScheduler`
- slower cadence for token/cost UI updates
- final flush behavior
**Why fifth:**
- relatively small and low risk
- easy to explain
- avoids muddying the bigger PRs
---
### PR 6 — Request-boundary caching / environment-detail optimization
**Include:**
- `RequestBoundaryCache`
- tab query caching
- workspace config / CLI tool caching
- environment detail TTL logic
**Why sixth:**
- good small PR
- nice setup-latency improvement
- not critical to the core “streaming feels slow” problem
---
### PR 7 — Task UI delta sync backend + frontend
**Include:**
- shared delta types
- backend delta subscription/publishing
- frontend delta application and resync
- debug counters
**Why later:**
- highest coupling across backend + transport + frontend
- best introduced after the simpler hot-path wins land
---
### PR 8 — Frontend render optimization polish
**Include:**
- request row derivation refactor
- targeted render-optimization helpers
- any row memoization / merge-polish changes
**Why last:**
- frontend-only polish is easiest to evaluate once transport patterns are stable
---
## What I Would Emphasize in the Eventual Write-Up / Review Narrative
When socializing these changes with reviewers or maintainers, I would frame them this way:
### Primary narrative
Remote workspaces make every synchronous persistence and snapshot push more expensive. This branch improves UX primarily by decoupling three clocks that were previously too tightly bound:
1. provider chunk ingestion
2. UI presentation
3. durable persistence / snapshot sync
### Strongest concrete claims to make
- Partial updates should not synchronously persist on every mutation.
- UI should update at a deliberate human cadence, not at token cadence.
- Full-state snapshots are too expensive to use as the main streaming transport in remote mode.
- Metadata counters should update slower than answer text.
### Claims to make more carefully
- Delta sync is the long-term architecture direction, but it is not necessarily the first or simplest PR to land.
- Request-boundary caching helps, but it is secondary to fixing the streaming hot path.
---
## Gaps / Cautions
### 1. Validation is directionally useful but not yet final proof
The branch includes solid instrumentation and a promising validation harness, but the current scripted scenario still times out before detecting `completion_result`. That means the evidence is good enough to support prioritization, but not yet strong enough to claim complete end-to-end UX proof.
### 2. Delta sync is valuable but raises extraction complexity
It touches:
- backend mutation points
- transport contracts
- frontend sequence tracking
- resync semantics
That is a lot of surface area for one PR. It should be intentionally staged.
### 3. Some improvements are multiplicative
The biggest user win is not any one change in isolation. It is the combination of:
- fewer presentation flushes
- fewer durable writes
- fewer full-state snapshots
- fewer low-value metadata updates
So while we should split the work into smaller PRs, we should expect the biggest UX gain after the first few land together.
---
## Recommended Next Step
The best immediate next step is to split out the following three PRs first:
1. **Instrumentation / telemetry scaffolding**
2. **Assistant presentation scheduler**
3. **Ephemeral partial persistence split**
Then follow quickly with:
4. **Controller full-state coalescing**
That four-PR sequence captures the bulk of the likely user-perceived latency win while keeping each PR reasonably understandable and reviewable.
---
## Bottom Line
The branchs most effective techniques for improving remote workspace UX are the ones that reduce hot-path work frequency and payload size during streaming. The highest-ROI changes are not the broadest architectural additions; they are the disciplined changes that stop doing unnecessary synchronous work on every partial update.
If we want the smallest set of PRs that likely produce the largest user QoL improvement, the best extraction order is:
1. instrumentation,
2. presentation scheduling,
3. ephemeral partial persistence,
4. full-state coalescing,
5. then delta sync and follow-on polish.
@@ -0,0 +1,874 @@
# Remote Workspace Latency Improvement Plan for Cline
This document is a detailed implementation plan for improving user-perceived latency when Cline is running in VS Code attached to a remote workspace. It is written as a development guide, not just a task list. Each step includes its objective, the mental model behind it, the concrete code changes that should be made, and the tests needed to validate the work.
The core theme of this plan is that Cline currently couples three different kinds of work too tightly during task execution:
1. **Model stream ingestion** — handling small incoming chunks from the provider as quickly as possible.
2. **UI synchronization** — updating the chat and task state shown in the webview.
3. **Durable persistence** — saving messages/history to disk and updating task history metadata.
That coupling is manageable in a purely local environment, but becomes much more expensive when VS Code is attached to a remote workspace because the extension host, filesystem, and some VS Code APIs are remote while the user-facing webview is local. The result is that tiny stream deltas can repeatedly trigger expensive persistence and transport work, increasing latency and making the UI feel “chattery” or sluggish.
The goal of this plan is to decouple these clocks so that:
- chunk ingestion stays fast,
- the UI updates at a human-friendly cadence,
- and persistence happens at durable boundaries rather than on every partial mutation.
---
## Guiding Principles
Before starting implementation, keep these principles in mind:
- **Optimize for user perception, not raw event frequency.** Users care about smooth responsiveness, not whether every token is rendered immediately.
- **Semantic boundaries matter more than token boundaries.** First token, tool approval, tool completion, error, cancellation, and completion events should feel immediate. Text streaming can be coalesced.
- **Remote-mode costs are multiplicative.** Every full-state serialization and webview push may involve remote transport, local parsing, and React reconciliation.
- **Persistence should model recovery, not animation.** Partial UI states usually do not need to be persisted synchronously.
- **Measure before and after.** This plan includes instrumentation work because latency improvements should be proven, not inferred.
---
## Phase 0 — Baseline and Instrumentation
### Goal
Before changing architecture, establish hard measurements so that development can compare before/after behavior in both local and remote contexts. The mental model here is simple: if we do not measure call frequency, serialization size, persistence time, and end-to-end chunk-to-paint delay, we will not know which optimizations actually helped.
- [x] Add instrumentation for streaming hot-path events
- [x] Add instrumentation for state payload sizes and frequencies
- [x] Add instrumentation for persistence latency
- [x] Add instrumentation for end-to-end chunk-to-webview timing
- [x] Add remote-vs-local environment tagging to these metrics
### Code changes
#### 0.1 Instrument `presentAssistantMessage()` frequency and duration
In `src/core/task/index.ts`:
- Add counters/timers around `presentAssistantMessage()`.
- Track:
- invocation count per request,
- total time spent in the function,
- average time per invocation,
- whether the call was triggered by text, reasoning, tool delta, or finalization.
Suggested implementation approach:
- Introduce a small per-request in-memory stats object on `TaskState` or a request-scoped local metrics structure inside `recursivelyMakeClineRequests()`.
- Record timestamps with `performance.now()`.
- Emit summary telemetry at the end of the request rather than per call.
#### 0.2 Instrument `postStateToWebview()` and full state serialization
In `src/core/controller/index.ts` and `src/core/controller/state/subscribeToState.ts`:
- Record:
- number of `postStateToWebview()` calls per request,
- time spent in `getStateToPostToWebview()`,
- size in bytes of serialized state,
- send time for `sendStateUpdate()`.
Telemetry already exists for state response size in `subscribeToState.ts`; extend that to include:
- frequency,
- whether the app was currently streaming,
- whether the task was remote.
#### 0.3 Instrument partial message event traffic
In `src/core/controller/ui/subscribeToPartialMessage.ts`:
- Record count of partial events sent,
- payload size,
- time spent broadcasting.
This helps answer whether full state or partial stream traffic is the bigger transport cost.
#### 0.4 Instrument persistence latency
In `src/core/task/message-state.ts`:
- Measure time spent in:
- `saveClineMessagesAndUpdateHistoryInternal()`
- `saveApiConversationHistory(...)`
- `updateTaskHistory(...)`
- Break down persistence cost by operation type where possible.
This is important because partial updates currently go through `updateClineMessage()` and trigger persistence synchronously.
#### 0.5 Add remote-awareness to instrumentation
In host/environment-derived state, tag metrics with whether the extension is running against a remote workspace. If there is already an authoritative host/environment signal, use that; otherwise add one.
Places to inspect and potentially extend:
- `HostProvider.env.getHostVersion({})`
- `getStateToPostToWebview()`
- any existing platform/host metadata included in telemetry
### Tests
- [x] Add unit tests for metric aggregation helpers
- [x] Add tests to verify instrumentation does not throw when telemetry is disabled
- [x] Add tests to verify state/partial payload-size accounting is invoked
These tests should focus on ensuring instrumentation remains non-blocking and failure-safe.
---
## Phase 1 — Introduce a Presentation Scheduler
### Goal
The purpose of this phase is to stop calling the expensive presentation path on every incoming chunk. The mental model is: **the stream can run at machine speed, but the UI should repaint at human speed**. The scheduler becomes the boundary between those two clocks.
- [x] Design and implement a `TaskPresentationScheduler`
- [x] Replace direct hot-path `presentAssistantMessage()` invocation with scheduled flushes
- [x] Preserve immediate flushes for semantic boundaries
- [x] Add local vs remote cadence selection
- [x] Add final-drain behavior at stream completion and abort
### Code changes
#### 1.1 Create a scheduler class
Add a new file, for example:
- `src/core/task/TaskPresentationScheduler.ts`
Responsibilities:
- accept presentation requests from the streaming loop,
- coalesce repeated requests,
- flush at a bounded cadence,
- support priority levels.
Suggested interface:
```ts
type PresentationPriority = "immediate" | "normal" | "low"
class TaskPresentationScheduler {
requestFlush(priority?: PresentationPriority): void
flushNow(): Promise<void>
dispose(): Promise<void>
}
```
The scheduler should hold:
- whether a flush is scheduled,
- whether a flush is currently running,
- whether more updates arrived while flushing,
- the highest pending priority.
#### 1.2 Integrate scheduler into `Task`
In `src/core/task/index.ts`:
- add a scheduler instance as a `Task` field,
- initialize it in the constructor,
- make the scheduler call into a refactored internal method, e.g. `flushAssistantPresentation()`.
Refactor current `presentAssistantMessage()` into two layers:
- public scheduler-facing method: `scheduleAssistantPresentation(...)`
- internal drain method: `flushAssistantPresentation()` or keep the name `presentAssistantMessage()` and make callers go through the scheduler.
#### 1.3 Replace direct hot-path calls
In the main streaming loop in `recursivelyMakeClineRequests()`:
- replace per-chunk `await this.presentAssistantMessage()` with a non-blocking scheduling call for normal streaming text/reasoning/tool deltas,
- keep direct/forced flushing for semantic boundaries such as:
- first visible token,
- tool completion or tool approval state transitions,
- ask creation,
- finalization after stream completion,
- stream abort/error.
The important design choice here is that chunk ingestion should no longer await UI presentation by default.
#### 1.4 Add adaptive cadence
Use different flush intervals depending on environment.
Initial recommendation:
- local: 3350ms
- remote: 75125ms
If there is a reliable remote signal from VS Code host metadata, use it. Otherwise keep the scheduler configurable and default to a conservative value such as 75ms.
#### 1.5 Ensure final drain semantics
At request completion, cancellation, or stream failure:
- force a final synchronous drain,
- ensure any remaining partial content is presented or completed,
- prevent pending scheduled flushes from running after task disposal.
### Tests
- [x] Unit test: multiple requests within the cadence window produce one flush
- [x] Unit test: an immediate-priority request preempts/coalesces normal requests correctly
- [x] Unit test: scheduler drains final updates on completion
- [x] Unit test: scheduler ignores/disposes pending work after task abort/dispose
- [ ] Integration test: streaming many text chunks produces fewer presentation invocations than chunk count
These tests should be written around deterministic fake timers so cadence behavior is reproducible.
---
## Phase 2 — Separate Ephemeral UI Updates from Durable Persistence
### Goal
This phase is likely one of the highest-impact improvements. The current system often persists partial updates immediately, which is expensive and unnecessary for animation-like streaming states. The mental model is: **partial updates are for the live experience; durable saves are for crash recovery and task history**. Those are related but not the same thing.
- [x] Add non-persisting message mutation APIs for partial updates
- [x] Update streaming paths to use ephemeral mutations
- [x] Add explicit durable flush points
- [x] Add periodic safety flush for long-running streams
- [x] Preserve correctness for crash recovery and resume behavior
### Code changes
#### 2.1 Extend `MessageStateHandler` with ephemeral mutation APIs
In `src/core/task/message-state.ts`:
Introduce methods that mutate in-memory message state and emit change notifications **without** saving to disk immediately.
Suggested methods:
```ts
async updateClineMessageEphemeral(index: number, updates: Partial<ClineMessage>): Promise<void>
async addToClineMessagesEphemeral(message: ClineMessage): Promise<void>
async flushClineMessagesAndUpdateHistory(): Promise<void>
```
Implementation details:
- preserve mutex safety,
- emit `clineMessagesChanged` just like durable mutations do,
- do not call `saveClineMessagesAndUpdateHistoryInternal()` inside ephemeral operations,
- maintain an internal dirty flag so the handler knows whether there are unsaved UI mutations.
#### 2.2 Switch partial text/reasoning/tool-progress updates to ephemeral APIs
In `src/core/task/index.ts`, update these call paths to use ephemeral updates where appropriate:
- `say(..., partial: true)` when updating existing partial say messages,
- `ask(..., partial: true)` when updating existing partial ask messages,
- reasoning partial row updates,
- native-tool-call text partial finalization where no durable save is yet needed.
The goal is to keep streaming UI smooth while deferring disk work.
#### 2.3 Define durable flush boundaries
Persist synchronously at semantic boundaries such as:
- new full message insertion,
- partial → complete transition,
- tool completion,
- `api_req_started` finalization,
- request completion,
- task abort/cancel,
- task resume state changes,
- checkpoint-relevant events.
Document these boundaries clearly in code comments because this becomes an architectural contract.
#### 2.4 Add a safety flush timer for long streams
To reduce the risk of losing too much partial content on a crash, add a lightweight periodic flush during active streaming.
Initial proposal:
- if there are unsaved ephemeral message changes,
- flush them every 12 seconds during an active stream.
This gives much better performance than per-delta saves while still offering reasonable recovery.
#### 2.5 Preserve task history correctness
Because `saveClineMessagesAndUpdateHistoryInternal()` also updates task history metadata, ensure that deferred persistence still yields correct task history snapshots at meaningful points.
In practice this means task history may be slightly behind while tokens are streaming, which is acceptable, but it must be correct when:
- a request ends,
- the task is cancelled,
- the task is resumed,
- the user views history after execution.
### Tests
- [x] Unit test: ephemeral update mutates in-memory message and emits change without saving
- [x] Unit test: durable flush persists previously ephemeral changes
- [x] Unit test: partial → complete transition triggers persistence
- [x] Unit test: periodic safety flush persists pending ephemeral changes
- [x] Integration test: abort during stream still persists a recoverable final state
- [x] Regression test: resume-from-history still works after deferred partial persistence
---
## Phase 3 — Coalesce Full State Updates
### Goal
Even after partial-message improvements, the codebase still calls `postStateToWebview()` from many locations. In remote environments, full-state pushes are especially expensive because they build and serialize a large `ExtensionState` payload. The mental model here is: **full state should be treated like a snapshot sync, not like a token stream transport**.
- [x] Add a controller-level full-state update coalescer
- [x] Prevent repeated `postStateToWebview()` calls from flooding the transport during active streaming
- [x] Add priority/urgency categories for full-state pushes
- [x] Keep initial and terminal state updates immediate and reliable
### Code changes
#### 3.1 Add a `StateUpdateScheduler` or coalescer to `Controller`
In `src/core/controller/index.ts`:
- replace the current fire-immediately behavior of `postStateToWebview()` with a coalescing scheduler,
- preserve a method for callers that need an immediate flush.
Suggested split:
```ts
async postStateToWebview(options?: { priority?: "immediate" | "normal" | "low" }): Promise<void>
private async flushStateToWebview(): Promise<void>
```
The scheduler should:
- collapse multiple requests into one pending flush,
- mark state dirty if more changes arrive while a flush is running,
- re-run once if needed after completion.
#### 3.2 Use shorter intervals outside streaming, longer during streaming
When a task is actively streaming, coalesce more aggressively.
Initial heuristic:
- idle/non-streaming: next-tick or very small debounce
- streaming local: 50ms
- streaming remote: 100150ms
This can be implemented in the controller or driven from task state.
#### 3.3 Audit existing `postStateToWebview()` callsites
Use the already-discovered callsite inventory in the codebase and categorize them:
- must remain immediate,
- can be coalesced,
- should be replaced by deltas later.
Examples likely needing immediacy:
- task initialization,
- task clear/cancel completion,
- auth/login state changes,
- mode switch,
- explicit task switching.
Examples likely safe to coalesce:
- usage/token updates during stream,
- retry-status metadata refreshes,
- background-command state churn,
- focus-chain intermediate updates.
### Tests
- [x] Unit test: repeated `postStateToWebview()` calls within a short interval produce one flush
- [x] Unit test: a dirty state during flush causes exactly one follow-up flush
- [x] Unit test: immediate-priority post bypasses normal delay
- [x] Integration test: active streaming generates significantly fewer full-state pushes
---
## Phase 4 — Expand from Snapshot Sync to Delta Sync for Active Task Execution
### Goal
This is a larger architectural improvement. Cline already has a partial-message subscription, which proves that the system can move small targeted UI updates instead of full snapshots. This phase extends that idea so that active task execution mostly uses **delta events**, while full-state snapshots are reserved for initialization, resync, and coarse-grained transitions.
- [x] Design a delta event model for hot task execution state
- [x] Add backend publishers for message and request metadata deltas
- [x] Update webview state handling to apply deltas safely
- [x] Keep full snapshot subscription as initialization and recovery path
- [x] Add ordering/versioning to prevent stale delta application
### Code changes
#### 4.1 Define delta event types
Create a new shared type module, for example:
- `src/shared/TaskUiDelta.ts`
Include delta types such as:
- `message_added`
- `message_updated`
- `message_deleted`
- `message_completed`
- `api_request_updated`
- `background_command_updated`
- `focus_chain_updated`
- `task_state_resynced`
Each event should include:
- task id,
- monotonic sequence number or revision,
- minimal payload required to update the client.
#### 4.2 Add subscription and publisher infrastructure
Pattern after:
- `src/core/controller/ui/subscribeToPartialMessage.ts`
- `src/core/controller/state/subscribeToState.ts`
Add something like:
- `src/core/controller/ui/subscribeToTaskUiDeltas.ts`
This should support both streaming subscribers and callback subscribers if needed.
#### 4.3 Publish deltas from message state mutations
In `MessageStateHandler` and/or `Task`:
- when a message is added/updated/deleted/completed, emit a delta event,
- for `api_req_started` usage/cost updates, emit a specific metadata delta rather than requiring a full-state push.
This may be naturally integrated with `clineMessagesChanged` events already emitted by `MessageStateHandler`.
#### 4.4 Update the webview to consume deltas
In `webview-ui/src/context/ExtensionStateContext.tsx`:
- subscribe to the new delta stream,
- patch state incrementally,
- preserve the full-state subscription for initial load and recovery/resync.
Recommended mental model for the webview:
- full state provides the initial canonical snapshot,
- deltas advance that state incrementally,
- if sequence numbers are skipped or an invariant fails, request/resubscribe to full state.
#### 4.5 Reduce full-state payload dependence during active execution
Once delta sync is working, stop relying on `state.clineMessages` inside frequent full-state pushes for active task updates.
The full state can still include `clineMessages`, but active execution should mostly ride on deltas.
### Tests
- [x] Unit test: message add/update/delete produces correct delta shape
- [x] Unit test: sequence ordering rejects or resyncs stale/missing deltas
- [x] Webview test: applying deltas yields the same final UI state as a full snapshot
- [x] Integration test: task execution with streaming text/tool updates works with delta transport enabled
- [x] Regression test: reopening/resubscribing still hydrates from full state correctly
---
## Phase 5 — Reduce Webview Render Churn
### Goal
Transport improvements alone are not enough if the webview still performs expensive reconciliation on every small update. The mental model here is: **we want the frontend to patch the smallest possible UI surface, especially for the active streaming row**.
- [x] Audit React render behavior for active streaming messages
- [x] Ensure partial text updates overwrite in place rather than causing broader list churn
- [x] Reduce expensive derived computations on every incremental update
- [x] Add memoization and stable references where beneficial
### Code changes
#### 5.1 Audit `ExtensionStateContext` update behavior
In `webview-ui/src/context/ExtensionStateContext.tsx`:
- examine where full state merges replace large arrays/objects,
- ensure delta/partial paths do the minimum necessary mutation via state replacement patterns.
Current behavior already patches the last matching message by `ts` for partial updates. Preserve this model and make it the standard for all active task deltas.
#### 5.2 Ensure active message row is the primary repaint target
In chat list rendering components (for example `ChatView` and related rows):
- make sure the active partial message updates do not cause avoidable re-renders of unrelated rows,
- add memoization around row components if not already present,
- keep keys stable and avoid replacing the entire messages array unnecessarily in hot paths.
#### 5.3 Reduce repeated expensive derivations
Components such as request rows, grouped tool rows, and task headers may derive substantial state from the whole `clineMessages` array.
Audit components such as:
- `webview-ui/src/components/chat/RequestStartRow.tsx`
- task header components
- any grouping/aggregation helpers used on every render
Refactor to:
- memoize by relevant slices,
- move repeated scans into selectors,
- avoid recomputing expensive groupings when only the current active row changed.
### Tests
- [x] Webview test: partial text updates only rerender the active message row (or as few components as practical)
- [x] Regression test: no flicker during partial→complete transition
- [x] Regression test: tool rows still group/render correctly under coalesced updates
---
## Phase 6 — Throttle Usage and Metadata Updates Separately from Text Streaming
### Goal
Token/cost counters and request metadata do not need to update at text-stream cadence. The mental model is: **the user watches the answer, not the token counter**. This phase reduces low-value churn without sacrificing correctness.
- [x] Batch usage/token/cost updates on a slower cadence
- [x] Flush final usage data immediately when a request ends
- [x] Keep telemetry capture decoupled from UI update cadence
### Code changes
#### 6.1 Decouple UI update cadence from telemetry capture cadence
In `src/core/task/index.ts` near `queueUsageChunkSideEffects(...)`:
- keep internal metrics accumulation immediate,
- continue recording telemetry as appropriate,
- but do not call `postStateToWebview()` for every usage chunk.
Instead:
- update the UI on a slower schedule (for example 250500ms),
- always flush final metrics on request completion/abort.
#### 6.2 Publish lightweight request-metadata deltas
If Phase 4 is implemented, use `api_request_updated` deltas rather than full-state pushes for these metrics.
If Phase 4 is not yet implemented, at minimum route metrics through the controller-level coalescer from Phase 3.
### Tests
- [x] Unit test: many usage chunks produce fewer UI updates than chunk count
- [x] Integration test: final displayed token/cost values remain accurate at request completion
- [x] Regression test: retry/cancel/error flows still show correct request metadata
---
## Phase 7 — Reduce Hostbridge and Environment-Detail Overhead at Request Boundaries
### Goal
Not all latency comes from streaming. Some comes from request setup, environment detail gathering, open-tab queries, and terminal state inspection. The mental model here is: **request boundaries should avoid recomputing or refetching unchanged data**.
- [x] Audit request-boundary hostbridge calls
- [x] Cache visible/open tab data with a short TTL
- [x] Cache or coalesce terminal-state inspections where safe
- [x] Avoid redundant expensive environment-detail reconstruction
### Code changes
#### 7.1 Cache tab queries briefly
In the hostbridge/window integration path:
- `src/hosts/vscode/hostbridge/window/getVisibleTabs.ts`
- `src/hosts/vscode/hostbridge/window/getOpenTabs.ts`
or in the task/controller layer that consumes them:
- add a short-lived cache (e.g. 2501000ms),
- invalidate on known tab-change events if such hooks are available,
- avoid repeated identical calls within one request-setup window.
#### 7.2 Audit `getEnvironmentDetails(...)`
In `src/core/task/index.ts`:
- identify repeated expensive sections,
- cache/reuse values that do not change materially within a short time,
- avoid unnecessary terminal cooling waits if there is no relevant terminal activity.
#### 7.3 Avoid full file-list regeneration when not needed
The code already avoids expensive file listing except when `includeFileDetails` is true. Preserve that behavior and further document it. If additional optimizations are needed, consider memoizing recent file-list snapshots per cwd for the duration of a task turn.
### Tests
- [x] Unit test: repeated tab/environment requests within TTL reuse cached values
- [x] Integration test: environment details still reflect fresh changes after invalidation/TTL expiry
- [x] Regression test: open/visible tabs remain accurate enough for prompt quality
---
## Phase 8 — Remote-Aware Policy and Configuration
### Goal
Different environments deserve different tuning. The mental model here is: **remote mode is a different performance envelope, so the product should adapt rather than relying on one-size-fits-all behavior**.
- [x] Detect remote execution context reliably
- [x] Add remote-aware defaults for presentation/state-update cadence
- [x] Expose internal config flags for development and staged rollout
- [x] Decide whether any user-facing settings are appropriate
### Code changes
#### 8.1 Establish remote-context detection
Decide on the most reliable source of truth for remote-vs-local execution. Candidate sources include host version/environment metadata already available through `HostProvider.env.getHostVersion({})` or VS Code host APIs.
Normalize this into a helper so the rest of the codebase can ask one question such as:
```ts
isRemoteWorkspaceEnvironment(): boolean
```
#### 8.2 Tune scheduler defaults based on remote context
Apply remote-aware cadence in:
- `TaskPresentationScheduler`
- controller state-update coalescer
- usage/metadata UI update cadence
- optional safety-flush intervals if needed
#### 8.3 Add development flags
Use internal settings, feature flags, or env vars to control rollout, for example:
- enable/disable presentation scheduler,
- enable/disable ephemeral partial persistence,
- enable/disable delta sync,
- override cadence intervals.
This is important for safe rollout and A/B comparison.
At this stage, no additional user-facing settings are recommended. The latency controls added so far are implementation details best kept behind internal env/config flags until telemetry and dogfooding show a clear need for end-user customization.
### Tests
- [x] Unit test: remote detection helper behaves correctly for representative host metadata
- [x] Unit test: cadence selection changes in remote mode
- [x] Integration test: remote-mode config path enables the intended scheduler defaults
---
## Phase 9 — Rollout and Safety Strategy
### Goal
These changes touch the hottest and most stateful part of task execution. The mental model here is: **performance work must be staged so that correctness is preserved while the architecture evolves**.
- [ ] Compare telemetry between old and new paths
- [x] Add fallback/resync paths for delta-sync failures
- [x] Prepare debugging aids for support and dogfooding
### Rollout order recommendation
1. **Instrumentation only**
2. **Presentation scheduler**
3. **Ephemeral partial persistence split**
4. **Controller full-state coalescing**
5. **Usage/metadata throttling**
6. **Frontend render optimizations**
7. **Delta sync for active task execution**
8. **Request-boundary hostbridge caching**
The reason for this order is that the early steps are high-value and lower risk, while delta sync is the largest architectural change and should be done only after the team has visibility and confidence.
### Telemetry comparisons to monitor during rollout
Use `node scripts/analyze-task-latency-metrics.mjs <path-to-jsonl>` to summarize exported `task.latency_metrics` events, and `node scripts/compare-task-latency-metrics.mjs <baseline-jsonl> <candidate-jsonl>` to diff before/after runs. The summary helpers also normalize `task.initialization` events so request-start latency can be reviewed alongside per-request streaming telemetry. For rollout toggles and cadence overrides, see `.env.example`.
- [ ] average `presentAssistantMessage()` invocations per request
- [ ] average partial message events per request
- [ ] average `postStateToWebview()` calls per request
- [ ] average serialized full-state payload size
- [ ] average persistence time per request
- [ ] median and p95 chunk-to-visible-update time
- [ ] median and p95 task initialization/request-start latency
- [ ] regression indicators: cancellation failures, resume failures, message ordering issues
### Debugging and support aids
- [x] Add optional debug logging for scheduler flush/coalescing behavior
- [x] Add optional debug counters in webview devtools for delta/full-state application counts
- [x] Add a forced full-resync mechanism if delta state diverges
---
## Detailed File-by-File Implementation Map
This section gives a practical map of the files most likely to change and what each change should do.
### Backend / extension host
- [x] `src/core/task/index.ts`
- Introduce presentation scheduling hooks
- Replace direct hot-path presentation awaits
- Separate semantic-boundary flushes from normal streaming cadence
- Reduce usage-metadata UI push frequency
- [x] `src/core/task/message-state.ts`
- Add ephemeral mutation APIs
- Add dirty tracking and explicit flush behavior
- Preserve mutex correctness
- [x] `src/core/task/TaskState.ts`
- Add any scheduler/config/metrics state needed
- [x] `src/core/task/TaskPresentationScheduler.ts` (new)
- Implement coalesced assistant presentation scheduling
- [x] `src/core/controller/index.ts`
- Add state-update coalescer
- Differentiate immediate vs coalesced state flushes
- [x] `src/core/controller/state/subscribeToState.ts`
- Extend payload/frequency instrumentation
- Potentially support lighter-weight state categories if needed later
- [x] `src/core/controller/ui/subscribeToPartialMessage.ts`
- Add payload/frequency instrumentation
- Potentially evolve into or complement broader delta transport
- [x] `src/core/controller/ui/subscribeToTaskUiDeltas.ts` (new, later phase)
- Streaming task UI delta subscription channel
- [x] `src/shared/TaskUiDelta.ts` (new, later phase)
- Delta type definitions and versioning model
- [x] `src/hosts/vscode/hostbridge/window/getVisibleTabs.ts`
- Add optional caching/invalidation if implemented in host layer
- [x] `src/hosts/vscode/hostbridge/window/getOpenTabs.ts`
- Add optional caching/invalidation if implemented in host layer
### Webview / frontend
- [x] `webview-ui/src/context/ExtensionStateContext.tsx`
- Continue handling full state snapshots
- Add delta subscription path
- Apply partial/delta updates with minimal state churn
- [x] `webview-ui/src/components/chat/RequestStartRow.tsx`
- Optimize expensive recomputation on active request updates
- [x] `webview-ui/src/App.tsx` / `ChatView` and related chat row components
- Audit rerender boundaries
- Memoize rows/selectors where useful
- [x] Any shared message-grouping or selector utilities used during chat rendering
- Extract and memoize repeated scans over the full message list
---
## Test Plan Summary
Below is a consolidated development checklist for tests. This can be used as a progress tracker during implementation.
### Backend tests
- [x] Scheduler unit tests with fake timers
- [x] MessageStateHandler ephemeral-vs-durable mutation tests
- [x] Periodic ephemeral safety-flush scheduler tests
- [x] Coalesced controller state-posting tests
- [x] Usage/metadata throttling tests
- [x] Remote-mode cadence selection tests
- [x] Hostbridge/environment TTL caching tests
### Integration tests
- [ ] Streaming text request produces fewer presentation flushes than chunks
- [ ] Native tool calling request preserves correct tool execution behavior under coalesced presentation
- [ ] Abort/cancel during stream preserves recoverable task state
- [ ] Resume from history still works after deferred partial persistence
- [x] Full state and delta state converge to the same result
### Frontend tests
- [x] Partial message update patches active row correctly
- [x] Partial→complete transition does not flicker
- [x] Delta events update chat state correctly in order
- [x] Full snapshot resync repairs intentionally diverged state
### Performance/regression validation
- [ ] Benchmark before/after counts for `presentAssistantMessage()`
- [ ] Benchmark before/after counts for `postStateToWebview()`
- [ ] Benchmark before/after persistence latency
- [ ] Benchmark before/after full-state payload sizes
- [ ] Benchmark before/after perceived streaming smoothness in remote mode
#### 2026-03-12 scripted validation status
- Added `scripts/validate-latency-scenarios.ts` to exercise the standalone core + mock hostbridge locally in both `local` and simulated-`remote` modes by driving gRPC task creation and observing:
- full-state snapshot delivery,
- task UI delta delivery,
- payload byte totals,
- first-observed update timing,
- and feature-toggle variants (`presentation_disabled`, `ephemeral_disabled`, `delta_disabled`).
- Added `scripts/test-hostbridge-server.ts` support for overriding `getHostVersion()` so the validation harness can force remote detection without a literal remote environment.
- Initial script run produced a useful partial signal but not a full end-to-end completion signal:
- local/default: 13 full-state updates, 27 task UI deltas, ~99 KB state payload bytes, ~38.6 KB delta payload bytes, first state in ~9 ms, first delta in ~24 ms.
- remote/default: 10 full-state updates, 27 task UI deltas, ~70.6 KB state payload bytes, ~38.8 KB delta payload bytes, first state in ~9 ms, first delta in ~25 ms.
- delta-disabled variants correctly dropped task delta count to 0 while preserving full-state delivery.
- The current mock-driven scenario still times out before a detectable `completion_result`, and `subscribeToPartialMessage()` did not emit during this run, so these results should be treated as transport/coalescing validation rather than final UX proof.
- Net takeaway so far: simulated-remote mode is already showing fewer/lighter full-state snapshots than simulated-local mode, and the feature flags are switching behavior in the expected directions, but a stronger completion-aware scripted scenario is still needed before checking off the benchmark items above.
---
## Recommended First Milestone
If the team wants the fastest path to meaningful improvement, the first milestone should include only the highest-ROI, lowest-risk work:
- [x] Add instrumentation
- [x] Implement `TaskPresentationScheduler`
- [x] Convert partial streaming updates to ephemeral message mutations
- [x] Add controller-level `postStateToWebview()` coalescing
- [x] Add tests and collect before/after telemetry
### Why this milestone first?
This milestone addresses the main latency amplifier without requiring a full transport redesign. It should materially improve remote responsiveness by reducing the number of times Cline:
- enters the presentation path,
- persists partial streaming state,
- serializes and transports full snapshots.
It also lays the foundation for later delta-sync work because once the hot path is decoupled, it becomes much easier to evolve transport behavior safely.
---
## Developer Mental Model Recap
When implementing this plan, keep the following model in your head:
- **The provider stream is not the UI clock.** Let the model emit chunks freely.
- **The UI should repaint at a deliberate cadence.** Fast enough to feel live, slow enough to avoid thrash.
- **Persistence is a safety boundary, not an animation system.** Save when the state becomes meaningfully durable.
- **Full state is for hydration and recovery.** Deltas are for active execution.
- **Remote mode magnifies every unnecessary snapshot and every synchronous save.** Favor coalescing, patching, and deferred durability.
If the implementation consistently follows those ideas, Cline should feel substantially more responsive in remote workspace scenarios while preserving correctness and recoverability.
@@ -0,0 +1,475 @@
# Technique Plan: Ephemeral Partial Persistence Split
This document is the implementation plan for the **ephemeral partial persistence split** technique identified in `docs/remote-workspace-latency-branch-analysis-report.md` as the highest-impact improvement for remote-workspace user-perceived latency.
The core idea is simple:
> **Stop treating every streamed chunk as durable state that must be written to disk immediately.**
That principle matters everywhere in this project, but it is especially important here. In remote workspace mode, partial updates are expensive not because the bytes are individually large, but because there are many of them and each one can pull in remote filesystem I/O, history recomputation, serialization work, and follow-on transport churn. For the user, this shows up as sluggish or jittery streaming, especially during long answers or large-file operations where the agent emits lots of progress text.
This plan focuses on splitting **ephemeral UI mutations** from **durable persistence boundaries** while preserving crash recovery, history correctness, and the rest of Clines product behavior.
## How To Use This Plan
This plan should be executed **on a new development branch**, not by continuing to stack work directly onto `eve_troubleshooting-remote-workspaces`.
That branch, `eve_troubleshooting-remote-workspaces`, should be treated as the **fully developed reference implementation** for this technique. In other words, the work described here is not speculative greenfield design work; it is a structured plan for re-deriving, validating, and extracting the technique from the already-working reference implementation into a smaller, more reviewable changeset.
The most effective way to use this document is:
1. read the goal and mental model for the step,
2. inspect how the reference implementation branch already solved it,
3. extract the minimum coherent version of that change into your own branch,
4. run the tests and validation described here,
5. compare behavior against the reference implementation when in doubt.
Be smart about this. Do not re-invent behavior that already exists in the reference implementation unless there is a very clear reason to improve or simplify it. The branch already contains the end-state shape we want to learn from. Your job is to extract, verify, and explain the minimal version of that behavior safely.
## Document Type, Audience, and Quality Bar
This is an **extraction implementation plan**, not a greenfield design doc and not a PR description. It is intended for a **Staff+ level distributed systems / infrastructure engineer** who is extracting a production-worthy technique from a known-good integrated implementation.
That means the quality bar is:
- the plan should be executable with minimal ambiguity,
- the reasoning behind each step should be legible to a strong reviewer,
- and the extraction should preserve product behavior across the rest of Cline while improving remote-workspace latency.
If a step in this plan would not help a strong engineer quickly answer “what exactly am I changing, why now, what must remain true, and how do I verify it?”, then the step is not detailed enough.
## Artifact Stack and Dependency Position
This doc sits in the broader artifact stack as follows:
1. `docs/remote-workspace-latency-branch-analysis-report.md` explains **which techniques matter most and why**.
2. This document explains **how to extract and implement one technique in a smaller branch/PR**.
3. A later PR-slicing / execution phase should use this document to drive real implementation work.
When using this plan, always begin by re-reading the branch analysis report so the extraction stays aligned with the larger prioritization and product intent.
## Developer Operating Posture
This plan is written for a Staff+-level engineer who is expected to use judgment, not just mechanically check boxes.
While implementing each step:
- actively compare your changes to `eve_troubleshooting-remote-workspaces`,
- preserve the original user experience across non-remote product surfaces,
- prefer coherent extraction over literal copy-paste,
- and keep asking: **what hot-path work are we removing, and what correctness boundary are we preserving?**
The most important meta-principle is still:
> **Stop treating every streamed chunk as a durable, full-state, immediately-presented event.**
For this technique specifically, the emphasis is on the **durable** part of that sentence.
## Minimal Coherent Extraction Boundary
The smallest coherent PR for this technique should usually include:
- message-state ephemeral mutation APIs,
- dirty tracking plus explicit flush support,
- task callsite conversion for partial updates,
- periodic safety flush,
- and the minimum set of tests needed to prove correctness.
What should **not** be split away from this technique if avoidable:
- the dirty-bit / flush contract,
- the durable-boundary logic for partial → complete transitions,
- safety-flush lifecycle management,
- and the tests that prove resume/abort correctness.
If those pieces are separated too aggressively, reviewers will have a much harder time understanding whether the extraction is actually safe.
## Common Failure Modes While Extracting
Watch for these failure modes explicitly:
- extracting the ephemeral APIs without converting the real hot-path callsites,
- clearing the dirty bit too early,
- leaving abort/resume paths on stale assumptions about immediate persistence,
- making durable and ephemeral mutation paths diverge semantically,
- and validating only happy-path streaming while missing crash-recovery or resume regressions.
If any of those happen, the extraction may appear simpler while actually weakening the product.
---
## Why This Technique Matters
When Cline is writing a large file, updating a long reasoning trace, or streaming many incremental tool/progress updates, the old behavior effectively says:
1. mutate message state,
2. save messages to disk,
3. update task history,
4. possibly trigger more UI/state work,
5. repeat on the next partial chunk.
That is the wrong clock boundary.
The better mental model is:
- **Streaming partials are animation state.** They exist to help the user perceive progress.
- **Durable persistence is recovery state.** It exists so the task can survive restart, cancellation, or history resume.
- **Those are related, but they are not the same thing.**
So the goal here is to keep the UI feeling live while only persisting at meaningful boundaries plus an occasional safety flush.
---
## Success Criteria
- Partial `say(...)` and `ask(...)` updates no longer synchronously persist on every mutation.
- Durable persistence still occurs at semantic boundaries such as completion, cancel, tool completion, and request completion.
- Long-running streams periodically safety-flush unsaved partial changes.
- Resume-from-history and crash-recovery behavior remain correct and predictable.
- Message-state behavior remains mutex-safe and consistent under concurrent tool / stream / abort activity.
---
## Files Most Likely to Change
- `src/core/task/message-state.ts`
- `src/core/task/index.ts`
- `src/core/task/EphemeralMessageFlushScheduler.ts`
- `src/test/message-state-handler.test.ts`
- `src/core/task/__tests__/EphemeralMessageFlushScheduler.test.ts`
- `src/core/task/__tests__/latency.test.ts`
- possibly targeted resume / abort integration tests
---
## Step-by-Step Implementation Plan
## Step 1 — Define the durability contract for message mutations
### Goal
Write down the architectural contract before changing behavior, so future developers know which updates are ephemeral and which are durable.
### Mental model
If developers cannot quickly answer “does this mutation need to survive a crash immediately?”, they will accidentally route new hot-path updates back through synchronous persistence. This step prevents future regression.
### Work
- [ ] Add code comments near `MessageStateHandler` describing the distinction between ephemeral and durable mutations.
- [ ] Add code comments in `Task.say(...)` and `Task.ask(...)` documenting which partial flows are intentionally ephemeral.
- [ ] Define a durable-boundary checklist in comments or docstrings, including at minimum:
- [ ] partial → complete transition
- [ ] tool completion / tool result boundary
- [ ] request completion
- [ ] cancellation / abort
- [ ] resume-related state changes
- [ ] checkpoint-relevant events
### Detailed code changes
- In `src/core/task/message-state.ts`, add a short header comment near the class definition explaining:
- durable methods write immediately,
- ephemeral methods mutate in memory and emit change notifications,
- `flushClineMessagesAndUpdateHistory()` is the bridge between the two.
- In `src/core/task/index.ts`, add comments above the partial-update branches in `say(...)` and `ask(...)` explaining why partial updates are not always durable.
When doing this work, read the corresponding code in `eve_troubleshooting-remote-workspaces` first and copy the intent, not just the surface syntax. The point of this step is to make the extraction self-explanatory for future reviewers and maintainers.
### Tests
- [ ] No behavioral tests required for comments alone.
- [ ] Ensure any snapshot/doc-based linting or type-check flow still passes.
---
## Step 2 — Add explicit ephemeral mutation APIs to `MessageStateHandler`
### Goal
Create first-class APIs for in-memory message changes that emit message-change notifications without synchronously persisting.
### Mental model
The presence of dedicated APIs changes developer behavior. If the only available mutation helper is a durable save path, then everything becomes durable by default. We want the inverse for streaming partials: durable only when intentionally requested.
### Work
- [ ] Add `addToClineMessagesEphemeral(message)`.
- [ ] Add `updateClineMessageEphemeral(index, updates)`.
- [ ] Add internal dirty tracking for unsaved ephemeral changes.
- [ ] Ensure ephemeral methods emit the same `clineMessagesChanged` events and task UI deltas as durable methods.
- [ ] Ensure all of the above remain protected by the existing mutex.
### Detailed code changes
- In `src/core/task/message-state.ts`:
- [ ] Add a `hasDirtyEphemeralChanges` flag if it does not already exist.
- [ ] In `addToClineMessagesEphemeral(...)`:
- [ ] set `conversationHistoryIndex` and `conversationHistoryDeletedRange` exactly as durable add does,
- [ ] mutate `clineMessages`,
- [ ] mark dirty,
- [ ] emit `clineMessagesChanged`.
- [ ] In `updateClineMessageEphemeral(...)`:
- [ ] validate index,
- [ ] capture previous message,
- [ ] mutate in place,
- [ ] mark dirty,
- [ ] emit `clineMessagesChanged`.
- [ ] Keep delta emission behavior identical between ephemeral and durable changes so frontend live behavior stays consistent.
The smart way to execute this step is to compare the durable and ephemeral codepaths side by side in the reference implementation branch and preserve the shared invariants exactly. The extraction should reduce write amplification, not create a shadow message-state model with slightly different semantics.
### Tests
- [ ] Unit test: ephemeral add mutates in-memory state without calling persistence.
- [ ] Unit test: ephemeral update mutates in-memory state without calling persistence.
- [ ] Unit test: ephemeral mutation emits `clineMessagesChanged` with correct shape.
- [ ] Unit test: ephemeral mutation still emits task UI deltas when delta sync is enabled.
- [ ] Unit test: invalid index still throws in ephemeral update path.
---
## Step 3 — Add explicit flush behavior for previously-ephemeral changes
### Goal
Provide a single durable flush method that persists all dirty ephemeral changes and updates task history once.
### Mental model
We are not removing durability; we are **batching** durability at the right semantic times. The flush method is the “commit” for a burst of ephemeral UI activity.
### Work
- [ ] Add `flushClineMessagesAndUpdateHistory()` if not already present.
- [ ] Make it a no-op when no ephemeral changes are dirty.
- [ ] Ensure it reuses the same internal persistence logic as durable mutations.
### Detailed code changes
- In `src/core/task/message-state.ts`:
- [ ] Add `flushClineMessagesAndUpdateHistory()` guarded by `withStateLock(...)`.
- [ ] If `hasDirtyEphemeralChanges` is false, return early.
- [ ] Otherwise call `saveClineMessagesAndUpdateHistoryInternal()`.
- [ ] Ensure `saveClineMessagesAndUpdateHistoryInternal()` clears the dirty flag only after successful save/update-history flow.
This step is where the implementation starts to feel like a deliberate state machine rather than a collection of helper methods. Be smart about failure ordering here: if the code clears the dirty bit too early or conflates flush success with mutation success, recovery correctness will quietly degrade.
### Tests
- [ ] Unit test: flush persists previously-ephemeral changes.
- [ ] Unit test: flush is a cheap no-op when there are no dirty changes.
- [ ] Unit test: task history reflects flushed content after prior ephemeral mutation.
---
## Step 4 — Switch streaming partial update callsites to ephemeral APIs
### Goal
Move the actual hot-path streaming mutations onto the new ephemeral methods.
### Mental model
The APIs only matter if the streaming loop uses them. This is the step that converts theory into latency improvement.
### Work
- [ ] Audit all partial `say(...)` paths.
- [ ] Audit all partial `ask(...)` paths.
- [ ] Switch normal streaming partial updates from durable to ephemeral mutation methods.
- [ ] Keep complete/finalized messages on durable paths unless explicitly flushed immediately after ephemeral completion.
### Detailed code changes
- In `src/core/task/index.ts`, update `say(...)`:
- [ ] partial update of existing partial `say` message should use `updateClineMessageEphemeral(...)`.
- [ ] new partial `say` message insertion should use `addToClineMessagesEphemeral(...)` where appropriate.
- In `src/core/task/index.ts`, update `ask(...)`:
- [ ] partial update of existing partial `ask` message should use `updateClineMessageEphemeral(...)`.
- [ ] new partial `ask` insertion should use `addToClineMessagesEphemeral(...)` where appropriate.
- For reasoning/tool-progress-related live updates:
- [ ] ensure they remain visible to the UI through message-change events / partial-message events,
- [ ] but no longer synchronously persist each partial mutation.
This is the step where it becomes easy to accidentally under-extract or over-extract. Use the reference implementation branch aggressively here. If a callsite was made ephemeral in `eve_troubleshooting-remote-workspaces`, understand why. If a callsite remained durable, understand why. Preserve that distinction intentionally.
### Tests
- [ ] Integration-style test: many partial text updates do not trigger per-update durable saves.
- [ ] Unit test: partial `say(...)` path still emits live UI updates while skipping persistence.
- [ ] Unit test: partial `ask(...)` path still emits live UI updates while skipping persistence.
---
## Step 5 — Define and enforce durable flush boundaries
### Goal
Ensure the system persists at the correct semantic points so correctness is preserved.
### Mental model
This is the balancing step. We are intentionally reducing durability frequency, so we must be precise about where durability is still required.
### Work
- [ ] Identify all partial → complete transitions.
- [ ] Flush at request completion.
- [ ] Flush on abort / cancellation.
- [ ] Flush when tool execution reaches a stable durable boundary.
- [ ] Flush when history / resume semantics require consistency.
### Detailed code changes
- In `src/core/task/index.ts`:
- [ ] when `partial: false` finalizes a previously partial message, use durable `updateClineMessage(...)` or explicit flush right after ephemeral completion.
- [ ] in request-finalization logic, call `flushClineMessagesAndUpdateHistory()` before or alongside final durable save path where needed.
- [ ] in abort/cancel paths, ensure pending ephemeral changes are flushed before task shutdown is considered complete.
- [ ] in resume-related flows, ensure history-visible state is not left behind dirty.
This step is the heart of the technique. The way to think about it is: we are removing durability from the hot path only because we are reintroducing durability at the right semantic boundaries. If those boundaries are fuzzy, the technique is incomplete.
### Tests
- [ ] Unit test: partial → complete transition results in durable persistence.
- [ ] Integration test: abort during stream persists a recoverable state.
- [ ] Regression test: resume-from-history still works after deferred partial persistence.
- [ ] Regression test: tool result flows remain properly visible in history after finalization.
---
## Step 6 — Add the periodic safety-flush scheduler
### Goal
Bound the amount of live UI state that could be lost if the extension host crashes during a long-running stream.
### Mental model
The correct behavior is not “persist every partial” and not “never persist until the very end.” The practical middle ground is:
- partials stay ephemeral during normal live streaming,
- but dirty state gets checkpointed periodically at low frequency.
### Work
- [ ] Add or finish `EphemeralMessageFlushScheduler`.
- [ ] Start it when a request begins streaming.
- [ ] Stop it when the request completes or aborts.
- [ ] Make it call `flushClineMessagesAndUpdateHistory()` on cadence only when dirty.
### Detailed code changes
- In `src/core/task/EphemeralMessageFlushScheduler.ts`:
- [ ] ensure a single timer is active,
- [ ] prevent overlap between flushes,
- [ ] support clean start/stop/dispose semantics.
- In `src/core/task/index.ts`:
- [ ] start scheduler near request start,
- [ ] stop scheduler in both success and error/finally paths,
- [ ] make cadence conservative (for example ~1.5s) to preserve UX win while bounding recovery loss.
Use the reference implementations chosen cadence and lifecycle wiring as the default starting point. Only diverge if you can clearly articulate why a different extraction improves isolation or reviewability without changing the core behavior.
### Tests
- [ ] Unit test: scheduler flushes pending ephemeral changes on cadence.
- [ ] Unit test: scheduler does nothing when there is no dirty ephemeral state.
- [ ] Unit test: scheduler stops cleanly on request completion/abort.
---
## Step 7 — Validate large-file and long-stream scenarios explicitly
### Goal
Confirm that the technique helps exactly the user scenario we care about: long-running, high-churn operations such as writing or rewriting large files.
### Mental model
Large-file operations are effectively “progress-heavy” workloads. Even if the tool invocation itself is durable, the surrounding reasoning, progress, and partial text can create a huge amount of avoidable mutation churn.
This is the direct answer to the large-file-write scenario: this technique helps because large-file operations tend to produce many small intermediate UI updates, and those should not all be treated like crash-critical persistence boundaries.
### Work
- [ ] Add a targeted validation scenario for large streamed output / large-file write behavior.
- [ ] Confirm persistence flush count drops sharply versus baseline.
- [ ] Confirm the user still sees smooth progress and correct final durable state.
### Detailed code changes
- Extend existing latency validation or tests to simulate:
- [ ] long streaming response,
- [ ] tool progress and completion,
- [ ] large-file write workflow or equivalent sustained partial-update workload.
- Use telemetry fields already added in latency instrumentation to compare:
- [ ] persistence flush count,
- [ ] save durations,
- [ ] chunk-to-webview timing.
### Tests
- [ ] Performance/regression test: long stream causes far fewer persistence flushes than partial-update count.
- [ ] Validation harness comparison: baseline vs ephemeral-persistence-enabled variant.
- [ ] Regression test: final message history and resume state remain correct.
---
## Step 8 — Rollout safeguards and developer controls
### Goal
Make the feature easy to disable, validate, and debug during rollout.
### Mental model
Hot-path changes need escape hatches. If behavior regresses in an edge case, the team should be able to isolate the feature quickly.
### Work
- [ ] Keep or add env flag gating for ephemeral persistence behavior.
- [ ] Ensure telemetry can compare enabled vs disabled behavior.
- [ ] Add debug logging only if low-noise and useful.
### Detailed code changes
- In `src/core/task/latency.ts` / `.env.example`:
- [ ] preserve `CLINE_DISABLE_EPHEMERAL_MESSAGE_PERSISTENCE` or equivalent.
- [ ] document intended use for A/B validation.
Treat the feature flag and validation path as first-class extraction requirements, not afterthoughts. Since the reference implementation already exists, the smoother development process is to keep comparison easy between “extracted technique enabled” and “feature disabled” modes.
### Tests
- [ ] Unit test: disable flag routes behavior back to durable-per-update path where applicable.
- [ ] Validation harness variant: feature-disabled mode still behaves correctly.
---
## Developer Checklist Summary
- [ ] Document the durability contract
- [ ] Add explicit ephemeral mutation APIs
- [ ] Add dirty tracking and explicit flush support
- [ ] Convert streaming partial callsites to ephemeral mutations
- [ ] Enforce durable flushes at semantic boundaries
- [ ] Add periodic safety flush scheduler
- [ ] Validate large-file / long-stream scenarios
- [ ] Preserve rollout flags and debugging support
- [ ] Run unit, integration, and validation-harness checks
---
## Final Mental Model Recap
When implementing this technique, keep this in mind:
- **Partial updates are for perception.**
- **Durable saves are for recovery.**
- **Doing recovery work on every animation step is what makes remote mode feel bad.**
- **The fix is not to persist less carelessly; it is to persist more intentionally.**
If this plan is implemented cleanly, large-file writes, long reasoning streams, and other high-churn tasks should feel markedly smoother in remote workspaces without sacrificing correctness.
@@ -0,0 +1,438 @@
# Technique Plan: Assistant Presentation Scheduler
This document is the implementation plan for the **assistant presentation scheduler** technique identified in `docs/remote-workspace-latency-branch-analysis-report.md` as one of the highest-ROI improvements for remote-workspace responsiveness.
The central principle is:
> **The provider stream is not the UI clock.**
In other words, the model may emit chunks at machine cadence, but the user only needs the UI to update at a human-friendly cadence. Trying to present every chunk immediately is what turns remote-mode transport, persistence, and rendering overhead into visible jitter.
This plan explains how to introduce a scheduler that coalesces presentation work without compromising semantic immediacy at important boundaries like first token, tool transitions, errors, and final completion.
## How To Use This Plan
This plan should be implemented on its **own extraction branch**. Do not treat this document as a net-new design exercise.
The branch `eve_troubleshooting-remote-workspaces` already contains the **fully developed reference implementation** for this technique. That branch should be used constantly while executing this plan: inspect how it solves each subproblem, then extract the minimal coherent subset of that behavior into your branch with tests and clear review boundaries.
Be smart about this. The reference implementation has already paid the discovery cost. The goal now is to convert that integrated work into a smaller, comprehensible, reviewable, and verifiable technique PR. If something in the reference implementation looks surprising, understand it before simplifying it.
## Developer Operating Posture
This technique is not just “add a debounce.” It is a hot-path scheduling change in the core task-execution loop. Treat it with the care you would give any latency-sensitive distributed-systems control surface.
While implementing:
- keep one eye on the extracted branch and one on `eve_troubleshooting-remote-workspaces`,
- preserve semantic immediacy even while coalescing ordinary chunk churn,
- and continually ask whether the stream is being allowed to run at machine speed while the UI updates at human speed.
The cross-cutting wisdom from the analysis report applies directly here:
> **Stop treating every streamed chunk as a durable, full-state, immediately-presented event.**
For this technique, the emphasis is on the **immediately-presented** part.
## Document Type, Audience, and Quality Bar
This is an **extraction implementation plan** for a **Staff+ level distributed systems / infrastructure engineer**. It is not a request to invent a scheduler concept from scratch; it is a guide for extracting a production-worthy scheduler from the already-working reference implementation.
The quality bar is high:
- each step should be operationally clear,
- each behavioral tradeoff should be explainable to reviewers,
- and the extracted result should preserve semantic immediacy where it matters while reducing hot-path churn where it does not.
## Artifact Stack and Dependency Position
This document depends on the branch analysis report and should be used after reviewing:
1. `docs/remote-workspace-latency-branch-analysis-report.md` for the “why this is high ROI” framing.
2. `eve_troubleshooting-remote-workspaces` for the actual known-good implementation details.
3. This plan for the extraction sequence, tests, and safety boundaries.
That sequence matters because this technique is easiest to reason about when the business case, integrated implementation, and extraction steps are all visible at once.
## Minimal Coherent Extraction Boundary
The smallest coherent PR for this technique should usually include:
- the scheduler primitive,
- task integration,
- remote-aware cadence selection,
- final-drain / disposal correctness,
- and tests for cadence, preemption, overlap, and teardown.
What should **not** be split apart if avoidable:
- scheduler primitive from task integration,
- immediate-priority semantics from cadence logic,
- final-drain behavior from the scheduler extraction,
- and the tests that prove overlap/teardown correctness.
## Common Failure Modes While Extracting
Watch for these failure modes explicitly:
- introducing a timer but leaving direct hot-path awaits in place,
- over-coalescing semantic-boundary events that users expect to feel immediate,
- forgetting final-drain behavior at stream completion,
- teardown bugs that allow delayed flushes after task disposal,
- and tuning cadence values without validating against the reference implementation.
---
## Why This Technique Matters
The streaming loop is one of the hottest paths in the whole system. If every chunk does all of the following synchronously:
- parse/update assistant content,
- present to the UI,
- possibly trigger follow-on state posting,
- possibly interact with persistence or tool execution,
then the stream becomes paced by downstream work rather than by provider availability.
That problem gets worse in remote workspaces because UI presentation is no longer “just local work.” It often implies transport across host boundaries, local parsing, and frontend reconciliation.
The goal here is not to make the UI less live. The goal is to make it live at the **right cadence**.
---
## Success Criteria
- Normal text/reasoning/tool-progress chunk presentation is coalesced behind a scheduler.
- The streaming loop no longer awaits presentation on every chunk.
- Important semantic boundaries still flush immediately.
- Remote workspaces use more conservative cadences than local workspaces.
- Final drain behavior guarantees that no residual content is left unpresented.
---
## Files Most Likely to Change
- `src/core/task/TaskPresentationScheduler.ts`
- `src/core/task/index.ts`
- `src/core/task/latency.ts`
- `src/core/task/__tests__/TaskPresentationScheduler.test.ts`
- `src/core/task/__tests__/latency.test.ts`
---
## Step-by-Step Implementation Plan
## Step 1 — Define the presentation contract and priorities
### Goal
Establish which kinds of updates can be coalesced and which must feel immediate.
### Mental model
Not all updates are equal.
- A tenth text chunk arriving 20ms after the ninth is not urgent.
- The first visible token is urgent.
- A tool completion or approval transition is urgent.
- Finalization is urgent.
The scheduler works only if priority rules are intentional and documented.
### Work
- [ ] Define presentation priorities such as `immediate`, `normal`, and `low`.
- [ ] Document semantic boundaries that must flush immediately.
- [ ] Document which chunk types default to normal coalescing.
### Detailed code changes
- In `src/core/task/TaskPresentationScheduler.ts`:
- [ ] expose or preserve a `PresentationPriority` type.
- In `src/core/task/index.ts`:
- [ ] document priority mapping logic near `getPresentationPriorityForChunk(...)`.
- In comments/docstrings, explicitly call out immediate boundaries:
- [ ] first visible token,
- [ ] tool transitions,
- [ ] finalization,
- [ ] abort/error cleanup.
Use the reference implementation branch to understand where those boundaries were discovered empirically. Some of them exist because they matter for user perception; others exist because they matter for correctness or because delayed presentation would feel broken. Preserve that reasoning in the extraction.
### Tests
- [ ] Unit test: priority merge rules behave as expected.
---
## Step 2 — Implement the scheduler primitive
### Goal
Build a reusable scheduler that coalesces repeated requests, avoids overlapping flushes, and supports immediate preemption.
### Mental model
Think of the scheduler as a small state machine:
- a flush may be pending,
- a flush may be running,
- more work may arrive while the flush is running,
- the highest pending priority wins.
The implementation must be robust under bursty chunk arrival, not just simple timer-based debounce logic.
### Work
- [ ] Implement `requestFlush(priority)`.
- [ ] Implement `flushNow()`.
- [ ] Track pending priority, active flush, and pending-while-flushing state.
- [ ] Add disposal semantics so no timers survive task teardown.
### Detailed code changes
- In `src/core/task/TaskPresentationScheduler.ts`:
- [ ] keep a `scheduledTimer`.
- [ ] keep `pendingPriority`.
- [ ] keep `flushInProgress`.
- [ ] keep `pendingWhileFlushing`.
- [ ] when `requestFlush(immediate)` arrives, cancel scheduled timer and run now.
- [ ] when work arrives during flush, mark pending and re-run once afterward.
- [ ] support `dispose()` to clear timer and suppress future work.
Be smart about state-machine edge cases. This scheduler sits on a bursty asynchronous path; the real implementation value is in correct behavior under overlap, priority escalation, and teardown, not just in the existence of a timer.
### Tests
- [ ] Unit test: multiple requests inside the cadence window produce one flush.
- [ ] Unit test: immediate priority preempts pending normal work.
- [ ] Unit test: updates arriving during a flush produce exactly one follow-up flush.
- [ ] Unit test: dispose clears timers and suppresses future flushes.
---
## Step 3 — Integrate scheduler into `Task`
### Goal
Make the `Task` use the scheduler as the default path for presentation without breaking existing semantics.
### Mental model
`presentAssistantMessage()` should become the **drain implementation**, not the hot-path public API that every chunk directly awaits.
### Work
- [ ] Add a scheduler field to `Task`.
- [ ] Add a scheduling wrapper such as `scheduleAssistantPresentation(...)`.
- [ ] Refactor direct callers to go through the wrapper except where explicit immediate drain is needed.
### Detailed code changes
- In `src/core/task/index.ts`:
- [ ] instantiate `TaskPresentationScheduler` in the constructor.
- [ ] wire `flush: async () => this.flushAssistantPresentation()`.
- [ ] add `scheduleAssistantPresentation(trigger, priority)`.
- [ ] keep `flushAssistantPresentation()` as the method that actually calls `presentAssistantMessage()`.
When extracting this step, mirror the reference implementations structure closely enough that future diffs remain comparable. The cleanest extraction is one where a reviewer can trivially line up the extracted version with the reference implementation and see the same conceptual architecture.
### Tests
- [ ] Unit test: `scheduleAssistantPresentation(...)` increments request metrics correctly.
- [ ] Unit test: scheduling-disabled mode still drains immediately.
---
## Step 4 — Replace direct per-chunk presentation awaits in the streaming loop
### Goal
Remove the default `await presentAssistantMessage()` behavior from the chunk-ingestion hot path.
### Mental model
This is where the real latency win happens. If the chunk loop no longer blocks on presentation for normal chunk traffic, provider ingestion stays fast and the UI drains on its own cadence.
### Work
- [ ] Update text chunk path to schedule presentation instead of awaiting it.
- [ ] Update reasoning chunk path to schedule presentation instead of awaiting it.
- [ ] Update tool-progress/native-tool-call related chunk path similarly.
- [ ] Preserve immediate scheduling for first-token and tool-related semantic transitions.
### Detailed code changes
- In `src/core/task/index.ts`, inside streaming chunk handling:
- [ ] text chunks should update assistant content and then call `scheduleAssistantPresentation("text", priority)`.
- [ ] reasoning chunks should call `scheduleAssistantPresentation("reasoning", priority)`.
- [ ] tool-call chunks should call `scheduleAssistantPresentation("tool", priority)`.
- Ensure priority logic uses whether visible assistant content already exists.
This step is the actual latency win. Use the reference implementation to identify every place where the old flow awaited presentation inside streaming logic, then confirm whether that await was deliberately removed or preserved for a semantic boundary. Be explicit; do not guess.
### Tests
- [ ] Integration-style test: many streaming chunks produce fewer presentation invocations than chunk count.
- [ ] Regression test: first visible token still appears promptly.
- [ ] Regression test: tool execution order is preserved under scheduled presentation.
---
## Step 5 — Add remote-aware cadence selection
### Goal
Use different default cadences for local and remote environments.
### Mental model
Remote workspaces need more coalescing because each UI flush is more expensive. The right question is not “what is the minimum possible delay?” but “what cadence is imperceptibly live while materially reducing churn?”
### Work
- [ ] Centralize cadence lookup in `latency.ts`.
- [ ] Keep `immediate` priority at zero-delay.
- [ ] Use more conservative normal/low cadences in remote mode.
- [ ] Allow env-var overrides for tuning.
### Detailed code changes
- In `src/core/task/latency.ts`:
- [ ] add or preserve `getPresentationCadenceMs(isRemoteWorkspace, priority)`.
- [ ] keep override env vars for local and remote cadence values.
- In `Task` constructor:
- [ ] pass cadence callback into scheduler so it adapts automatically once remote detection is known.
Do not tune cadence values from first principles unless necessary. Start from the values already proven in `eve_troubleshooting-remote-workspaces`, then adjust only if the extraction boundary demands it or validation shows a problem.
### Tests
- [ ] Unit test: remote mode returns higher normal cadence than local mode.
- [ ] Unit test: env var override wins over default values.
- [ ] Unit test: immediate priority always returns zero.
---
## Step 6 — Preserve final-drain semantics
### Goal
Guarantee that all pending content is fully presented before request completion, abort, or disposal.
### Mental model
Schedulers are easy to add and easy to get subtly wrong at teardown. The user must never lose the last bit of visible content because it was still sitting in a pending timer when the request ended.
### Work
- [ ] Force a final synchronous drain when the stream completes.
- [ ] Force final drain on abort/error cleanup where appropriate.
- [ ] Dispose scheduler cleanly during task teardown.
### Detailed code changes
- In `src/core/task/index.ts`:
- [ ] after the streaming loop has completed and partial blocks are finalized, call `await this.presentationScheduler.flushNow()`.
- [ ] in abort/finally paths, ensure no pending scheduled flush survives past task shutdown.
- In `TaskPresentationScheduler`:
- [ ] make `dispose()` clear timers and suppress post-disposal flushes.
This is one of the places where smart engineering judgment matters most: the last 1% of scheduler teardown correctness often determines whether the feature is “production-grade” or “subtly flaky.” Compare end-of-stream and abort behavior carefully against the reference implementation.
### Tests
- [ ] Unit test: final `flushNow()` drains pending coalesced work.
- [ ] Unit test: task disposal suppresses delayed pending flushes.
- [ ] Regression test: final text is visible before next request starts.
---
## Step 7 — Add instrumentation and verify chunk-to-visible behavior
### Goal
Measure the schedulers actual effect so cadence tuning is based on data.
### Mental model
Scheduling is always a tradeoff between update frequency and perceived responsiveness. The only good tuning process is to measure:
- how many flushes occur,
- how long flushes take,
- what the chunk-to-visible delay looks like.
### Work
- [ ] Track presentation invocation count.
- [ ] Track total/average presentation duration.
- [ ] Track final chunk-to-webview delay distribution.
- [ ] Emit request-level telemetry summary.
### Detailed code changes
- In `src/core/task/index.ts`:
- [ ] accumulate presentation metrics in request-scoped latency metrics.
- [ ] record chunk-to-webview delay when state or partial-message updates occur.
- In telemetry summary helpers:
- [ ] ensure presentation-related fields are included and comparable.
### Tests
- [ ] Unit test: metrics aggregate correctly under multiple scheduler flushes.
- [ ] Unit test: instrumentation is failure-safe when telemetry is disabled/unavailable.
---
## Step 8 — Validate special high-churn scenarios such as large-file writes
### Goal
Ensure the scheduler meaningfully helps the scenarios users actually notice.
### Mental model
Large-file write scenarios often generate:
- lots of reasoning text,
- tool descriptions/progress,
- potential partial previews,
- repeated task-state churn.
The scheduler should reduce the “chatty” feel without making the operation feel frozen.
That is why this technique materially helps large-file writes: the user does not need every incremental progress mutation painted at model-chunk cadence. They need the operation to feel continuously alive, not hyperactive.
### Work
- [ ] Add validation scenario for long streamed response and/or large-file write workflow.
- [ ] Compare presentation flush count with scheduler enabled vs disabled.
- [ ] Confirm first-token latency remains acceptable.
### Tests
- [ ] Validation harness scenario: scheduler-enabled mode produces fewer presentation flushes than chunk count.
- [ ] Comparison run: scheduler-disabled variant shows meaningfully higher presentation activity.
---
## Developer Checklist Summary
- [ ] Define presentation priorities and semantic boundaries
- [ ] Implement the scheduler primitive
- [ ] Integrate scheduler into `Task`
- [ ] Replace direct per-chunk presentation awaits
- [ ] Add remote-aware cadence selection
- [ ] Preserve final-drain semantics
- [ ] Instrument and verify behavior
- [ ] Validate large-file / long-stream scenarios
---
## Final Mental Model Recap
- **Streams run at machine speed.**
- **People read at human speed.**
- **UI presentation should honor the latter without blocking the former.**
If developers hold that model throughout implementation, this technique will reliably reduce jitter and improve perceived responsiveness in remote workspaces.
@@ -0,0 +1,436 @@
# Technique Plan: Controller Full-State Coalescing
This document is the implementation plan for the **controller full-state coalescing** technique identified in `docs/remote-workspace-latency-branch-analysis-report.md` as one of the top four highest-impact improvements for remote-workspace UX.
The key idea is:
> **Full state is a snapshot transport, not a token transport.**
When Cline is actively streaming, repeatedly rebuilding and sending large `ExtensionState` snapshots is one of the biggest avoidable costs in remote mode. Even if each snapshot is “not that large,” the repeated serialization, transport, parsing, and frontend reconciliation create visible UI churn.
This technique reduces that cost by coalescing repeated `postStateToWebview()` requests into a scheduler-driven snapshot flow with priorities and remote-aware cadence.
## How To Use This Plan
This plan is for extracting a coherent technique into its **own branch**, while treating `eve_troubleshooting-remote-workspaces` as the **fully developed reference implementation**.
That means the branch work has already been done once. The goal now is not to rediscover the architecture from scratch; it is to produce a smaller, easier-to-review implementation plan that tells a developer exactly how to extract and verify the technique with high confidence.
Be smart about this. Continually compare your work to the reference implementation and actively take implementation details from it when executing each step. The reference branch should be your source of truth for subtle behaviors, edge-case handling, and interactions with the rest of the product surface.
## Developer Operating Posture
This technique sits at the boundary between extension-host state management and frontend hydration. That makes it highly leveraged and easy to get wrong in ways that only show up under load or during cross-surface interactions.
While implementing:
- keep snapshot semantics explicit,
- preserve immediate behavior where product correctness or UX requires it,
- and use the reference implementation to understand which callsites were intentionally allowed to coalesce.
The governing principle remains:
> **Stop treating every streamed chunk as a durable, full-state, immediately-presented event.**
For this technique, the emphasis is on the **full-state** part.
## Document Type, Audience, and Quality Bar
This is an **extraction implementation plan** written for a **Staff+ level distributed systems / infrastructure engineer**. Its purpose is to turn an already-integrated optimization into a smaller, understandable, safe-to-review change set.
The quality bar is:
- state-posting behavior must remain easy to reason about,
- the extracted scheduler must preserve correctness across non-streaming product surfaces,
- and the doc must make explicit where coalescing is appropriate versus dangerous.
## Artifact Stack and Dependency Position
This doc should be used in the following sequence:
1. read the branch analysis report for prioritization context,
2. inspect `eve_troubleshooting-remote-workspaces` for the integrated implementation,
3. use this plan to extract a smaller, coherent PR with clear verification boundaries.
This sequence reduces rework and helps ensure the extraction remains anchored to the actual behavior we already know works.
## Minimal Coherent Extraction Boundary
The smallest coherent PR for this technique should usually include:
- controller-level scheduler introduction,
- controller `postStateToWebview()` routing changes,
- build/send/payload instrumentation,
- priority selection defaults,
- and tests covering coalescing plus immediate bypass.
What should **not** be split away if avoidable:
- scheduler extraction from controller integration,
- payload/build/send instrumentation from the coalescing work,
- default-priority logic from the scheduler rollout,
- and product-surface regression coverage for init/cancel/auth/task switching.
## Common Failure Modes While Extracting
Watch for these failure modes explicitly:
- coalescing the scheduler mechanically without auditing callsite urgency,
- reducing full-state post count while accidentally delaying critical UI transitions,
- proving frequency reduction without measuring payload size or send/build time,
- treating streaming and non-streaming surfaces the same,
- and forgetting that snapshot bugs often manifest as broad UI inconsistency rather than narrow chat glitches.
---
## Why This Technique Matters
In remote workspaces, a full-state update can imply all of the following:
- gather current extension state,
- build `ExtensionState`,
- JSON serialize it,
- transport it from extension host to webview boundary,
- parse it on the receiving side,
- merge it into frontend state,
- trigger React render work.
That is acceptable for:
- initialization,
- task switches,
- mode changes,
- recovery/resync.
It is **not** acceptable as the default transport for high-frequency streaming churn.
---
## Success Criteria
- Repeated `postStateToWebview()` calls during active streaming are coalesced.
- Immediate flushes are still available where correctness or UX requires them.
- State payload size and posting frequency are instrumented.
- Streaming tasks generate significantly fewer full-state pushes.
- No regressions appear in task switching, auth changes, cancellation, or other non-streaming product surfaces.
---
## Files Most Likely to Change
- `src/core/controller/StateUpdateScheduler.ts`
- `src/core/controller/index.ts`
- `src/core/controller/state/subscribeToState.ts`
- `src/core/controller/postStateToWebview.test.ts`
- `src/core/controller/StateUpdateScheduler.test.ts`
- `src/core/task/index.ts` for request metrics hookup
---
## Step-by-Step Implementation Plan
## Step 1 — Define snapshot posting priorities and rules
### Goal
Make it explicit which state posts must remain immediate and which can be coalesced.
### Mental model
If every caller thinks its update is urgent, the scheduler will collapse back into immediate mode and lose its value. We need a principled split between:
- **must be immediate**,
- **safe to coalesce**,
- **background/low priority**.
### Work
- [ ] Define `immediate`, `normal`, and `low` state update priorities.
- [ ] Audit core state-posting callsites.
- [ ] Document which flows must bypass coalescing.
### Detailed code changes
- In `src/core/controller/index.ts`, document priority expectations near `postStateToWebview(...)`.
- Categorize at least these as typically immediate:
- [ ] task initialization,
- [ ] explicit task clear/switch,
- [ ] auth/login/logout state changes,
- [ ] mode switch.
- Categorize these as usually coalescible during streaming:
- [ ] usage/cost updates,
- [ ] background command state churn,
- [ ] focus-chain intermediate changes,
- [ ] repeated chat-state updates caused by streaming partials.
Use the reference implementation branch to validate these classifications before finalizing them in your extraction. The point is not to make an abstract list; it is to preserve the already-learned boundary between “must feel instant” and “safe to batch.”
### Tests
- [ ] No behavior test required yet beyond upcoming scheduler tests.
---
## Step 2 — Build the controller-level state update scheduler
### Goal
Create a scheduler that coalesces repeated full-state posting requests while preserving immediate flush semantics.
### Mental model
This scheduler is conceptually parallel to the presentation scheduler, but it controls **snapshot delivery**, not message presentation. The implementation must handle:
- pending work,
- in-progress flushes,
- higher-priority upgrades,
- rerun-on-dirty behavior.
### Work
- [ ] Implement `StateUpdateScheduler` with request/flush/dispose behavior.
- [ ] Support priority merging.
- [ ] Avoid overlapping snapshot flushes.
- [ ] Re-run once if additional work arrives during flush.
### Detailed code changes
- In `src/core/controller/StateUpdateScheduler.ts`:
- [ ] track `scheduledTimer`, `pendingPriority`, `flushInProgress`, `pendingWhileFlushing`, and `disposed`.
- [ ] support `requestFlush(priority)`.
- [ ] support `flushNow()`.
- [ ] support `dispose()`.
- [ ] ensure `immediate` preempts a pending delayed timer.
Be smart about the scheduler design here. A controller-level scheduler can look mechanically similar to the presentation scheduler, but the failure mode is different: a presentation bug is usually visible in one message stream, while a snapshot bug can destabilize the whole UI state model.
### Tests
- [ ] Unit test: repeated normal-priority calls inside cadence window produce one flush.
- [ ] Unit test: immediate-priority request bypasses delay.
- [ ] Unit test: updates arriving while flush is running trigger exactly one follow-up flush.
- [ ] Unit test: dispose clears scheduled work.
---
## Step 3 — Route `postStateToWebview()` through the scheduler
### Goal
Make the scheduler the default behavior for full-state posting without breaking existing callers.
### Mental model
The API should remain convenient for the rest of the codebase. Most callers should still say “post state,” but the controller decides whether that means immediate flush or scheduled coalescing.
### Work
- [ ] Add optional priority parameter to `postStateToWebview(...)`.
- [ ] Determine the default priority based on whether the task is actively streaming.
- [ ] Preserve an explicit immediate path.
### Detailed code changes
- In `src/core/controller/index.ts`:
- [ ] instantiate `StateUpdateScheduler` in the constructor.
- [ ] change `postStateToWebview(options?)` so it:
- [ ] flushes immediately for `priority: "immediate"`,
- [ ] otherwise requests scheduled flush.
- [ ] add `getDefaultStateUpdatePriority()` that returns `normal` while streaming and `immediate` when idle/non-task.
When extracting this step, prefer preserving the reference implementations method boundaries and control flow. That will make later comparison and debugging much smoother.
### Tests
- [ ] Unit test: no-task / idle state posts remain immediate by default.
- [ ] Unit test: active-streaming posts default to coalesced priority.
---
## Step 4 — Instrument full-state payload size, build time, and send time
### Goal
Quantify the actual cost of snapshot posting and verify coalescing reduces it.
### Mental model
Snapshot frequency alone is not enough. One giant expensive snapshot can be worse than several small ones. We want to measure:
- how often full-state posts happen,
- how large they are,
- how long they take to build,
- how long they take to send.
### Work
- [ ] Measure `getStateToPostToWebview()` build time.
- [ ] Measure serialized payload size.
- [ ] Measure send time.
- [ ] Feed these metrics into per-request latency telemetry.
### Detailed code changes
- In `src/core/controller/index.ts`:
- [ ] add `flushStateToWebview()` that records build duration and delivery stats.
- [ ] call `task?.noteStateUpdateMetrics(...)` with build duration, payload bytes, and send duration.
- In `src/core/controller/state/subscribeToState.ts`:
- [ ] ensure payload byte counting remains available and accurate.
This step is not just observability polish. It is what lets the team prove that the extracted technique is actually reducing snapshot churn rather than merely moving it around.
### Tests
- [ ] Unit test: state update metrics are recorded when a flush occurs.
- [ ] Unit test: payload byte accounting is invoked.
---
## Step 5 — Audit and tune state-posting callsites
### Goal
Ensure callsites use the right priority and are not silently undermining the scheduler.
### Mental model
The scheduler provides the mechanism; callsite audit provides the correctness. Without the audit, some codepaths will still over-post or misuse immediacy.
### Work
- [ ] Review all major `postStateToWebview()` callsites.
- [ ] Keep task initialization and state transitions immediate where appropriate.
- [ ] Allow hot streaming churn to use normal or low priority.
### Detailed code changes
- In controller and task flows, inspect callsites involving:
- [ ] task init / resume,
- [ ] cancel / clear,
- [ ] auth state changes,
- [ ] usage updates,
- [ ] focus-chain metadata,
- [ ] background command metadata,
- [ ] periodic stream-related updates.
- Where needed, pass explicit priority rather than relying only on defaults.
The smart move here is to audit callsites with the reference implementation open, because the coalescing behavior only makes sense in context. The subtle value of the reference branch is that it already captures where the team discovered hidden urgency requirements.
### Tests
- [ ] Regression test: task init still hydrates the UI immediately.
- [ ] Regression test: cancel/clear still updates UI promptly.
- [ ] Regression test: auth/settings flows are not delayed in a user-visible bad way.
---
## Step 6 — Validate behavior during high-churn scenarios, especially large-file writes
### Goal
Verify that the feature materially helps the scenarios that produce the most snapshot churn.
### Mental model
Large-file writes often produce:
- repeated tool/progress updates,
- repeated request metadata changes,
- repeated message-state changes,
- possible repeated snapshot posts.
This technique should reduce the transport overhead from that churn even if the write tool itself remains functionally the same.
That is the key link to the large-file-write scenario: even when the tool work is mostly backend-side, the surrounding UI state churn can still create a slow, noisy experience if snapshots are over-posted.
### Work
- [ ] Add or extend validation scenarios for high-churn task execution.
- [ ] Compare full-state update count enabled vs disabled.
- [ ] Compare payload-byte totals enabled vs disabled.
### Tests
- [ ] Validation harness scenario: coalescing reduces full-state count in long-running tasks.
- [ ] Validation harness scenario: remote mode benefits more than local mode.
- [ ] Regression test: final state still fully converges after coalesced posting.
---
## Step 7 — Add remote-aware cadence and tuning controls
### Goal
Choose defaults that are appropriate for remote environments and safe to tune during rollout.
### Mental model
Remote mode should intentionally trade a little more coalescing for much less transport thrash. That should be tunable, not hard-coded forever.
### Work
- [ ] Centralize cadence defaults in `latency.ts`.
- [ ] Expose env var overrides for local and remote state cadence.
- [ ] Document these in `.env.example`.
### Detailed code changes
- In `src/core/task/latency.ts`:
- [ ] add/preserve `getStateUpdateCadenceMs(isRemoteWorkspace, priority)`.
- In `.env.example`:
- [ ] document state update cadence overrides.
Keep the tuning hooks aligned with the reference implementation so extracted behavior can be compared apples-to-apples during rollout and validation.
### Tests
- [ ] Unit test: remote defaults are more conservative than local defaults.
- [ ] Unit test: env override behavior works as expected.
---
## Step 8 — Verify product-surface safety outside the streaming path
### Goal
Make sure coalescing full-state snapshots does not degrade other product surfaces.
### Mental model
The branch goal is not just “remote chat feels better.” It is “remote workspaces improve without harming the rest of Cline.” State posting is used by many surfaces, so safety checks matter.
### Work
- [ ] Test history/task switching behavior.
- [ ] Test settings/auth-related UI updates.
- [ ] Test onboarding / welcome state hydration.
- [ ] Test focus-chain/background-command metadata behavior.
### Tests
- [ ] Regression test: state hydration on startup remains correct.
- [ ] Regression test: switching tasks shows the correct snapshot.
- [ ] Regression test: metadata deltas + snapshot flow do not leave stale UI after task switch.
---
## Developer Checklist Summary
- [ ] Define snapshot posting priorities
- [ ] Build controller-level scheduler
- [ ] Route `postStateToWebview()` through scheduler
- [ ] Instrument build/send/payload metrics
- [ ] Audit state-posting callsites
- [ ] Validate high-churn and large-file-write scenarios
- [ ] Add remote-aware cadence tuning
- [ ] Verify non-streaming product-surface safety
---
## Final Mental Model Recap
- **Full state is for hydration and synchronization.**
- **It is too expensive to be the main streaming transport in remote mode.**
- **Coalescing snapshots preserves correctness while reducing transport thrash.**
That is the idea developers should keep front-of-mind while implementing this technique.
@@ -0,0 +1,477 @@
# Technique Plan: Task UI Delta Sync for Active Task Execution
This document is the implementation plan for the **task UI delta sync** technique identified in `docs/remote-workspace-latency-branch-analysis-report.md` as the fourth most impactful technique in the branch and the strongest long-term transport architecture improvement.
The key idea is:
> **During active task execution, send targeted deltas rather than repeated full snapshots.**
This technique is more invasive than the first three top-ranked improvements, but it is strategically important because it moves the system toward a better model for remote workspaces: full-state snapshots for hydration and recovery, targeted deltas for live execution.
## How To Use This Plan
This plan should be executed on a dedicated extraction branch, while `eve_troubleshooting-remote-workspaces` is treated as the **fully developed reference implementation**.
That distinction matters. This technique is not a hypothetical architecture proposal; it is a plan for extracting and verifying a technique that already exists in integrated form in the reference branch. Developers working this plan should actively inspect the reference implementation for each step and pull implementation details from it deliberately.
Be smart about this. Because delta sync spans backend mutation publishing, transport contracts, and frontend application logic, the fastest way to make the development process stronger and smoother is to let the reference branch answer the “how did we already solve this edge case?” question early, rather than rediscovering it late.
## Developer Operating Posture
This is the most architecturally ambitious of the top four techniques. It changes how the system thinks about live task transport. That means the correct mindset is not “build a clever delta layer,” but:
- preserve snapshots as canonical hydration and recovery,
- shrink active-execution transport to the minimum necessary changes,
- and fall back to resync aggressively when invariants are violated.
The cross-cutting project wisdom still applies here:
> **Stop treating every streamed chunk as a durable, full-state, immediately-presented event.**
For this technique, the emphasis is on moving active execution away from **full-state** and toward **targeted transport**.
## Document Type, Audience, and Quality Bar
This is an **extraction implementation plan** for a **Staff+ level distributed systems / infrastructure engineer**. It assumes the reader is capable of reasoning about transport contracts, ordering invariants, state hydration, and recovery semantics.
The quality bar is especially high here because this technique crosses backend, transport, and frontend boundaries. The plan must therefore make it easy to answer:
- what the transport contract is,
- what invariants must hold,
- what recovery behavior is expected,
- and how the extracted version will be validated against the reference implementation.
## Artifact Stack and Dependency Position
This doc should be read as part of the following artifact sequence:
1. `docs/remote-workspace-latency-branch-analysis-report.md` explains why delta sync is strategically valuable but later in the extraction order.
2. `eve_troubleshooting-remote-workspaces` shows the integrated end state and should be consulted constantly.
3. This document defines the extraction steps, invariants, and test strategy for a smaller implementation branch.
Because this technique is more coupled than the other top-four techniques, keeping that sequence explicit will make development much smoother.
## Minimal Coherent Extraction Boundary
The smallest coherent PR for this technique should usually include:
- shared delta type definitions,
- backend publish/subscribe infrastructure,
- message-state delta emission,
- frontend delta application with sequencing and resync,
- and tests covering ordering, divergence, and recovery.
What should **not** be split away if avoidable:
- sequence validation from delta application,
- resync path from initial delta rollout,
- backend emission from frontend application if the goal is an end-to-end usable slice,
- and the fallback snapshot path that preserves product correctness.
## Common Failure Modes While Extracting
Watch for these failure modes explicitly:
- treating deltas as a replacement for snapshots rather than a companion to them,
- making the reducer permissive instead of sequence-strict,
- emitting deltas from the wrong abstraction boundary,
- forgetting task-identity filtering and task-switch behavior,
- and validating only happy-path ordered deltas without aggressive resync/fallback testing.
---
## Why This Technique Matters
Even after presentation scheduling, deferred persistence, and state coalescing, active execution can still generate meaningful transport churn. Full snapshots are fundamentally a coarse-grained mechanism. They resend lots of state that did not change.
In remote mode, that means unnecessary work across the whole pipeline:
- backend snapshot construction,
- serialization,
- remote transport,
- frontend parsing,
- broad state replacement / render churn.
Delta sync fixes the shape of the transport itself by sending only the state mutations that matter:
- message added,
- message updated,
- message deleted,
- task metadata updated,
- explicit resync signal.
---
## Success Criteria
- Active task execution can advance the webview primarily through task UI deltas.
- Full-state snapshots remain the canonical initialization and recovery path.
- Delta application is sequence-safe and can resync on gap or divergence.
- Backend message mutations publish minimal targeted deltas.
- Frontend applies deltas with minimal state churn.
- Task switches and stale deltas do not corrupt the UI.
---
## Files Most Likely to Change
- `src/shared/TaskUiDelta.ts`
- `src/core/controller/ui/subscribeToTaskUiDeltas.ts`
- `src/core/task/message-state.ts`
- `src/core/controller/index.ts`
- `webview-ui/src/context/ExtensionStateContext.tsx`
- `webview-ui/src/context/taskUiDeltaState.ts`
- `webview-ui/src/context/taskUiDebugCounters.ts`
- related tests in backend and webview
---
## Step-by-Step Implementation Plan
## Step 1 — Define the delta model and sequencing contract
### Goal
Create a small, explicit, versionable transport contract for active task execution changes.
### Mental model
Delta systems fail when they are “implicit.” They need an explicit contract for:
- what changed,
- which task it belongs to,
- in what order it must be applied,
- what to do if order is broken.
### Work
- [ ] Define delta event types.
- [ ] Ensure every delta contains `taskId` and `sequence`.
- [ ] Define resync behavior on missing/stale sequence.
### Detailed code changes
- In `src/shared/TaskUiDelta.ts`:
- [ ] define or refine delta union including:
- [ ] `message_added`
- [ ] `message_updated`
- [ ] `message_deleted`
- [ ] `task_metadata_updated`
- [ ] `task_state_resynced`
- [ ] document the sequencing contract in comments.
- Decide that:
- [ ] deltas are only valid for the current task,
- [ ] sequence must increment monotonically by 1,
- [ ] a gap triggers full resync.
Do not improvise this contract from memory. Read the reference implementation branch carefully and preserve the exact mental model it uses for sequence monotonicity and recovery semantics.
### Tests
- [ ] Unit test: type helpers or guards behave correctly.
- [ ] Unit test: sequence mismatch triggers resync result.
---
## Step 2 — Build backend delta publishing infrastructure
### Goal
Provide a transport channel for task UI deltas parallel to existing state and partial-message subscriptions.
### Mental model
Full-state snapshots and deltas should coexist, not replace each other outright. The backend must be able to publish deltas cheaply while retaining the existing snapshot transport as a recovery path.
### Work
- [ ] Add subscription/publisher mechanism for task UI deltas.
- [ ] Ensure it is failure-safe and non-blocking.
- [ ] Keep transport format minimal, ideally serialized delta JSON.
### Detailed code changes
- In `src/core/controller/ui/subscribeToTaskUiDeltas.ts`:
- [ ] implement backend subscription registry / broadcaster.
- [ ] add `sendTaskUiDelta(...)` helper.
- [ ] record payload-size metrics if useful.
- If protobuf transport needs changes:
- [ ] ensure message contract is appropriately wired through `proto/cline/ui.proto` or equivalent.
This is a good example of where “be smart about this” matters. The developer should not just make the channel exist; they should make it easy to reason about, easy to debug, and obviously subordinate to the canonical snapshot path.
### Tests
- [ ] Unit test: subscribers receive published deltas.
- [ ] Unit test: publisher handles no-subscriber case safely.
- [ ] Unit test: serialized payload shape is stable.
---
## Step 3 — Publish deltas from message-state mutations
### Goal
Make the message-state layer emit task UI deltas whenever live task messages mutate.
### Mental model
The message-state layer is the natural source of truth for chat mutation events. If deltas are emitted here, the system stays aligned with actual message semantics rather than ad hoc UI-side guesses.
### Work
- [ ] Emit `message_added` on add.
- [ ] Emit `message_updated` on update.
- [ ] Emit `message_deleted` on delete.
- [ ] Emit `task_state_resynced` on full replacement/set flows.
- [ ] Increment a per-task delta sequence on each publish.
### Detailed code changes
- In `src/core/task/message-state.ts`:
- [ ] wire `emitClineMessagesChanged(...)` to publish deltas when delta sync is enabled.
- [ ] use `taskState.taskUiDeltaSequence` as the monotonic sequence source.
- [ ] send minimal payloads for each mutation type.
- Ensure ephemeral and durable mutations both publish the same deltas so live UI behavior does not depend on durability choice.
This step should be executed with the reference implementation branch open beside the extraction branch. The key engineering task is not simply “emit deltas,” but “emit deltas from the true state mutation boundary without creating semantic skew between durable and ephemeral paths.”
### Tests
- [ ] Unit test: add publishes `message_added` with correct sequence.
- [ ] Unit test: update publishes `message_updated` with correct sequence.
- [ ] Unit test: delete publishes `message_deleted` with correct sequence.
- [ ] Unit test: set/overwrite publishes `task_state_resynced`.
---
## Step 4 — Publish task metadata deltas outside message-state mutations
### Goal
Handle non-message hot-path changes, such as focus-chain and background-command metadata, without relying on full snapshots.
### Mental model
Some of the most annoying snapshot churn comes from small metadata updates that are orthogonal to the message list. These deserve their own lightweight path.
### Work
- [ ] Add controller helper for metadata delta publication.
- [ ] Route focus-chain and background-command metadata through it.
- [ ] Fall back to snapshot posting when no current task or invalid task context exists.
### Detailed code changes
- In `src/core/controller/index.ts`:
- [ ] add or refine `postTaskMetadataDelta(...)`.
- [ ] only publish deltas when target task matches current active task.
- [ ] otherwise request a normal full-state post as fallback.
Keep the fallback path boring and reliable. Smart engineering here means preferring explicit fallback to snapshot sync over any attempt to get fancy when task identity or activity context is ambiguous.
### Tests
- [ ] Unit test: metadata delta publishes for current active task.
- [ ] Unit test: mismatched/non-active task falls back to snapshot path.
---
## Step 5 — Implement frontend delta application and ordering safety
### Goal
Make the webview able to apply deltas incrementally while detecting sequence gaps and requesting resync.
### Mental model
Frontend delta handling must be strict, not permissive. If it misses a sequence or applies a delta for the wrong task, stale UI bugs will appear and be hard to debug.
### Work
- [ ] Track latest applied sequence in the frontend.
- [ ] Ignore deltas for non-current tasks.
- [ ] Trigger resync on sequence mismatch.
- [ ] Apply message add/update/delete with minimal array churn.
### Detailed code changes
- In `webview-ui/src/context/taskUiDeltaState.ts`:
- [ ] validate `delta.sequence === latestSequence + 1`.
- [ ] return `resync` on mismatch.
- [ ] ignore deltas for non-current tasks while still advancing sequence semantics intentionally if that is the chosen policy.
- [ ] apply message mutations minimally.
- In `webview-ui/src/context/ExtensionStateContext.tsx`:
- [ ] subscribe to delta stream,
- [ ] feed deltas into reducer/helper,
- [ ] trigger full-state resync when helper returns `resync`.
The frontend side should be implemented with a bias toward correctness and repairability. If you find yourself making the delta reducer permissive to “keep things working,” stop and compare with the reference implementation. The right answer is usually stricter sequencing plus easier resync.
### Tests
- [ ] Webview test: ordered deltas produce correct final state.
- [ ] Webview test: sequence gap triggers resync path.
- [ ] Webview test: stale/non-current-task delta is ignored safely.
---
## Step 6 — Keep full-state snapshots as canonical hydration and recovery path
### Goal
Ensure deltas complement snapshots rather than replacing them unsafely.
### Mental model
Snapshots are still the canonical state source for:
- initial load,
- task switch,
- reconnect/reopen,
- recovery after divergence.
Deltas should advance current state, not become the sole source of truth.
### Work
- [ ] Preserve `subscribeToState` as initialization path.
- [ ] Reset delta sequence on full snapshot hydration.
- [ ] Trigger snapshot fetch on divergence.
### Detailed code changes
- In `ExtensionStateContext.tsx`:
- [ ] after receiving a fresh full snapshot, reset latest delta sequence tracking.
- [ ] on resync request, fetch latest state and replace current state.
- Ensure startup / reload still works even if no deltas arrive.
This step is essential to keeping the rest of Clines product surfaces healthy. Delta sync should improve active execution, not quietly turn startup, reopen, or task switching into undefined behavior.
### Tests
- [ ] Regression test: initial load hydrates correctly without prior deltas.
- [ ] Regression test: reopening or task switching still works.
- [ ] Regression test: full snapshot repairs intentionally diverged delta state.
---
## Step 7 — Minimize frontend churn when applying deltas
### Goal
Capture the benefit of deltas by applying them with minimal structural churn in React state.
### Mental model
A delta transport is less valuable if the frontend responds by rebuilding large portions of state anyway. The frontend should patch the smallest possible region.
### Work
- [ ] Update only the changed message when possible.
- [ ] Avoid replacing `clineMessages` unless necessary.
- [ ] Keep metadata updates narrow.
### Detailed code changes
- In `webview-ui/src/context/taskUiDeltaState.ts`:
- [ ] add/update should preserve array identity only where safe and replace minimal slices.
- [ ] delete should only filter when message exists.
- [ ] metadata updates should shallow-merge only changed fields.
Be smart about this at the React-state level too: if the frontend re-renders large portions of the tree on every delta, then the transport win will be partially squandered.
### Tests
- [ ] Webview test: unchanged update payload does not cause unnecessary state replacement.
- [ ] Webview test: active message row updates correctly under repeated deltas.
---
## Step 8 — Instrument, debug, and validate remote-mode benefit
### Goal
Make the delta system observable and prove it reduces snapshot dependence during active execution.
### Mental model
Delta systems are harder to reason about than snapshots, so they need better visibility. Developers should be able to see:
- how many deltas were applied,
- how many full states were still applied,
- how often resync happened.
### Work
- [ ] Add debug counters for full-state applications, partial-message applications, delta applications, and resync requests.
- [ ] Compare default mode vs delta-disabled mode in validation harness.
- [ ] Ensure feature flag exists for safe staged rollout.
### Detailed code changes
- In `webview-ui/src/context/taskUiDebugCounters.ts`:
- [ ] add counters for delta application and resync requests.
- In `.env.example` / `latency.ts`:
- [ ] preserve `CLINE_DISABLE_TASK_UI_DELTA_SYNC` or equivalent.
- In validation tooling:
- [ ] compare `stateUpdateCount`, `taskDeltaCount`, and payload bytes across variants.
Since the reference implementation already exists, one of the strongest ways to smooth development is to validate the extracted technique against both disabled-mode behavior and the known-good reference branch behavior.
### Tests
- [ ] Validation harness scenario: delta-enabled mode reduces full-state payload bytes during active execution.
- [ ] Validation harness scenario: delta-disabled variant falls back cleanly to snapshot behavior.
- [ ] Unit test: debug counters increment correctly where applicable.
---
## Step 9 — Validate the technique in large-file-write and long-running task scenarios
### Goal
Confirm that delta sync specifically helps long, noisy task executions, including large-file operations.
### Mental model
Large-file writes are not only about the write tool itself. They often generate a lot of nearby live task activity that becomes expensive when transported as snapshots. Delta sync should reduce that excess movement.
That is why this technique still matters for large-file-write scenarios, even though it is not the first thing to land: it attacks the remaining active-execution transport cost after the first three higher-ROI techniques have already reduced hot-path churn.
### Work
- [ ] Add scenario coverage for long active execution with many message mutations.
- [ ] Compare delta-enabled vs delta-disabled behavior.
- [ ] Verify convergence at the end of execution.
### Tests
- [ ] Integration/validation scenario: long-running execution with many message updates works correctly under delta sync.
- [ ] Regression test: final UI state matches snapshot-based state.
- [ ] Regression test: no stale message duplication or ordering bug appears after many updates.
---
## Developer Checklist Summary
- [ ] Define delta model and sequencing contract
- [ ] Build backend delta subscription/publishing infrastructure
- [ ] Publish deltas from message-state mutations
- [ ] Publish metadata deltas for non-message hot paths
- [ ] Implement frontend delta application with strict ordering safety
- [ ] Preserve full snapshots for hydration and recovery
- [ ] Minimize frontend churn during delta application
- [ ] Add observability, flags, and validation coverage
- [ ] Validate large-file / long-running execution scenarios
---
## Final Mental Model Recap
- **Full snapshots establish truth.**
- **Deltas advance truth during active execution.**
- **If ordering breaks, resync instead of guessing.**
- **The point is not cleverness; the point is to avoid shipping unchanged state over and over in remote mode.**
That is the mindset developers should keep while implementing this technique.
@@ -0,0 +1,633 @@
# Local Latency Observer Plan for Remote-Workspace Comparison
This document describes a plan for building a **locally visible latency observer** for Cline that can be added to a branch derived from `main` **and** to a branch derived from `eve_troubleshooting-remote-workspaces`, so that behavior can be observed and compared under the same measurement mechanism.
The goal is not to recreate the old `arafatkatze/ping-pong-test` branch literally. That older work is useful as inspiration because it validated an important product-development idea:
> engineers need a fast, visible, low-friction way to observe latency behavior locally while iterating.
But the repository has changed substantially since then. The right move now is to design a measurement mechanism that fits the current architecture, is useful on both baseline and candidate branches, and helps engineers compare **user-perceived latency behavior**, not just transport RTT in isolation.
---
## What This Artifact Is
This is an **implementation plan / extraction plan** for a Staff+ level distributed systems and infrastructure engineer. It is not a greenfield “brainstorming” doc and it is not a narrow PR description.
Its purpose is to define a branch-portable observer mechanism that can:
- be added to `main`,
- be added to `eve_troubleshooting-remote-workspaces`,
- surface useful measurements locally,
- and support apples-to-apples comparison of baseline vs improved behavior.
The quality bar for this plan is:
- the mechanism must be easy to reason about,
- useful during iterative development,
- minimally invasive to product behavior,
- and explicit about what is being measured and what is not.
---
## Reference Inputs and Inspiration
This plan should be read alongside:
1. `docs/remote-workspace-latency-branch-analysis-report.md`
2. `docs/remote-workspace-latency-improvement-plan.md`
3. the current telemetry/validation tooling already present in this repo
4. the older inspiration branch `origin/arafatkatze/ping-pong-test`
The old branch appears to have included a visible `LatencyTester` UI in settings that measured gRPC ping/pong latency with variable payload sizes. That is a useful inspiration because it made latency locally visible and easy to probe. However, it is not enough by itself for the current task, because the current problem is broader than pure transport RTT.
Today we care about multiple layers of user-perceived latency, including:
- UI transport latency,
- snapshot / delta delivery behavior,
- chunk-to-visible-update timing,
- request-start latency,
- and hot-path churn during active execution.
So the modern observer should retain the spirit of “easy visible local latency testing” while measuring the parts of the system that matter most for the remote-workspace problem.
---
## Core Goal
Build a **single observer mechanism** that can be added to both baseline and candidate branches and used to compare their behavior in a way that is:
- visible to the developer locally,
- useful during manual iteration,
- scriptable enough to support repeatable experiments,
- and robust to architectural differences between `main` and `eve_troubleshooting-remote-workspaces`.
---
## Non-Goals
To keep the effort focused, this observer should **not** try to be all of the following at once:
- a benchmark harness for every product surface,
- a production telemetry dashboard,
- a load-testing framework,
- or a branch-specific debugging toy that only works on one side of the comparison.
Instead, it should be a **branch-portable latency observation layer** with a small visible UI and enough instrumentation hooks to answer the comparison questions we care about.
---
## Guiding Principles
Before implementation, keep these principles in mind:
- **Optimize for comparability over cleverness.** A simpler observer that works identically on both branches is better than a richer observer that only works on one.
- **Measure user-perceived boundaries, not just transport boundaries.** Ping/pong RTT is useful, but it is only one slice of the problem.
- **Separate observer plumbing from branch-specific latency improvements.** The observer should not depend on the candidate branchs optimizations in order to function.
- **Make local visibility first-class.** Developers should be able to see measurements in the UI during manual testing, not only in exported logs.
- **Be smart about backward compatibility.** If the observer can gracefully detect missing richer metrics on `main` and still provide useful output, that is preferable to forcing two divergent observer implementations.
---
## Recommended High-Level Shape
The best shape for this feature is a **three-layer observer**:
1. **Transport probe layer** — lightweight ping/pong style measurements with variable payload sizes.
2. **Task-execution latency layer** — timing measurements around request start, first visible update, state update frequency/size, and optional chunk-to-visible metrics where supported.
3. **Local observer UI layer** — a developer-facing panel or settings section that displays current values, rolling history, and recent logs in a human-comprehensible way.
This approach preserves the useful simplicity of the old `LatencyTester` while expanding it into something aligned with the current problem space.
---
## Why a Single Branch-Portable Observer Matters
If the observer mechanism differs substantially between baseline and candidate branches, then comparison quality degrades immediately. Engineers start asking questions like:
- is the difference real or an artifact of the measurement tool?
- does one branch surface richer metrics only because it has extra plumbing?
- are we comparing the same event boundaries?
The smart move is to design the observer so that:
- a **minimum shared metric set** works on both branches,
- richer metrics can appear opportunistically where supported,
- and the UI makes clear which metrics are available vs unavailable on the current branch.
That way a single observer branch can be rebased/cherry-picked onto both `main` and `eve_troubleshooting-remote-workspaces` with minimal divergence.
---
## Recommended Measurement Categories
## 1. Transport probe metrics
These are inspired most directly by the old ping-pong tester.
Measure:
- [ ] round-trip UI service ping latency
- [ ] effect of varying payload sizes
- [ ] repeated sample min/max/avg/current
- [ ] continuous test mode for drift/jitter observation
Why it matters:
- helps reveal raw extension-host ↔ webview transport overhead,
- useful in remote environments like Codespaces / SSH / remote containers,
- easy for a developer to understand immediately.
Limitations:
- does **not** directly measure task-execution UX,
- should be treated as a lower-level signal, not the only KPI.
## 2. Task lifecycle metrics
Measure:
- [ ] task creation / initialization latency
- [ ] request-start latency
- [ ] time to first visible assistant update
- [ ] time to first full-state update
- [ ] time to first partial or delta update where available
Why it matters:
- these metrics align better with the users lived experience than transport RTT alone,
- they can show whether the candidate branch improves perceived responsiveness at important boundaries.
## 3. Hot-path churn metrics
Measure:
- [ ] number of full-state pushes per request
- [ ] total full-state payload bytes per request
- [ ] number of partial-message events per request
- [ ] number of task UI deltas per request where supported
- [ ] persistence flush counts where supported
Why it matters:
- these are the metrics most closely tied to the remote-workspace improvements described in the analysis report,
- they help explain *why* one branch feels faster, not just whether it does.
## 4. Comparison session metadata
Record:
- [ ] branch / commit identity
- [ ] local vs remote environment marker
- [ ] selected scenario / payload size / cadence mode
- [ ] timestamped session logs
Why it matters:
- makes manual comparisons auditable,
- prevents confusion when developers are switching between baseline and candidate runs.
---
## Branch-Portability Strategy
This is the most important design constraint.
The observer should be implemented so that it degrades gracefully:
### Minimum cross-branch baseline
These should work on both `main` and `eve_troubleshooting-remote-workspaces` with little or no branch-specific logic:
- [ ] ping/pong RTT with configurable payload size
- [ ] task initialization timing if a simple observer hook can be added
- [ ] first state update timing
- [ ] visible branch/commit/session labeling
- [ ] local UI for logs and rolling stats
### Opportunistic richer metrics
These may only exist natively on `eve_troubleshooting-remote-workspaces` or may require additional light plumbing on `main`:
- [ ] chunk-to-webview timing
- [ ] full-state post counts / bytes
- [ ] partial-message event counts
- [ ] task UI delta counts
- [ ] persistence flush metrics
### Recommended implementation rule
The observer UI should not fail if a metric is unavailable.
Instead it should display something like:
- “supported and active”
- “unsupported on this branch”
- or “observer hook not installed”
That makes the same branch usable on both baselines.
---
## Recommended Developer UX
The observer should be easy enough to use that engineers actually use it during iteration.
Recommended UI features:
- [ ] a dedicated dev-facing section in Settings or a dev/debug panel
- [ ] buttons for:
- [ ] single ping
- [ ] continuous ping
- [ ] test all payload sizes
- [ ] start observed task scenario
- [ ] reset stats
- [ ] visible stats cards / table for current/min/max/avg
- [ ] recent logs panel
- [ ] session metadata display (branch, commit, environment)
- [ ] explicit note describing which metrics are branch-portable vs richer-on-candidate
The point is to make the observer *pleasant enough* that it becomes part of the engineering workflow rather than a script that only gets used once.
---
## Recommended Implementation Plan
## Step 1 — Define the minimum shared metric contract
### Goal
Create the smallest metric surface that can work on both `main` and `eve_troubleshooting-remote-workspaces`.
### Mental model
The observer succeeds if its core contract is branch-portable. Everything richer is an enhancement.
### Work
- [ ] Define a shared observer metric model for:
- [ ] ping RTT samples
- [ ] task initialization / request-start samples
- [ ] first-visible-update samples
- [ ] optional richer counters
- [ ] Mark each metric as either:
- [ ] required/shared
- [ ] optional/richer
### Detailed code changes
- Add a shared type module, e.g. `src/shared/LatencyObserver.ts` or similar, that defines:
- [ ] sample types,
- [ ] rolling stats shape,
- [ ] branch capability flags,
- [ ] session metadata shape.
### Tests
- [ ] Unit test: metric aggregation shape is stable.
- [ ] Unit test: missing optional metrics do not break the model.
---
## Step 2 — Reintroduce a modern ping/pong transport probe
### Goal
Port the useful idea from the old `LatencyTester` branch into the current architecture in a modernized form.
### Mental model
This is the simplest locally visible measurement and should serve as the “does the pipe feel slow?” test.
### Work
- [ ] Add or confirm a simple UI-service ping endpoint that accepts payload size.
- [ ] Measure round-trip latency in the webview using `performance.now()`.
- [ ] Support multiple payload sizes and continuous testing.
### Detailed code changes
- Inspect whether the current codebase already has a UI ping path analogous to the old branchs `UiServiceClient.ping(...)` flow.
- If missing, add:
- [ ] a lightweight request/response endpoint in the UI service layer,
- [ ] optional payload-size expansion to simulate message size effects.
- In the webview, add a dev-facing component similar in spirit to the old `LatencyTester.tsx`, but keep it isolated behind a dev/debug visibility gate.
### Tests
- [ ] Unit/integration test: ping returns successfully.
- [ ] Test: payload size selection is reflected in the request.
- [ ] Regression test: continuous mode handles pending request overlap safely.
---
## Step 3 — Add a task-observer hook layer that is intentionally branch-portable
### Goal
Create a small observer hook interface that can be invoked from task lifecycle boundaries on both branches.
### Mental model
Do **not** couple the observer directly to candidate-branch-only telemetry structures. Instead create a tiny observer API that can be wired into either branch with minimal intrusion.
### Work
- [ ] Define a branch-portable observer service or callback layer.
- [ ] Add hooks for:
- [ ] task initialization start/end
- [ ] request start
- [ ] first visible update
- [ ] request completion
- [ ] Keep richer optional hooks for state-post counts, chunk-to-webview, etc.
### Detailed code changes
- Add a small service such as `LatencyObserverService` or similarly named utility in `src/services/` or `src/core/controller/`.
- It should:
- [ ] record timestamps,
- [ ] aggregate per-session/per-request values,
- [ ] expose results to the UI layer,
- [ ] not require the full candidate telemetry pipeline to exist.
### Tests
- [ ] Unit test: initialization and request lifecycle timings aggregate correctly.
- [ ] Unit test: optional hooks can be absent without crashing.
---
## Step 4 — Expose richer metrics when available, without making them required
### Goal
Allow the same observer branch to become more informative on `eve_troubleshooting-remote-workspaces` without breaking on `main`.
### Mental model
Think of this as capability detection, not branch forking.
### Work
- [ ] Integrate existing richer latency metrics when present:
- [ ] `task.latency_metrics`
- [ ] initialization telemetry
- [ ] payload-size accounting
- [ ] chunk-to-webview summaries
- [ ] Surface unsupported metrics as unavailable instead of failing.
### Detailed code changes
- Where current richer telemetry exists (candidate branch or current branch after instrumentation work), add adapters that feed it into the observer UI.
- On `main`, either:
- [ ] wire minimal equivalents if easy,
- [ ] or leave those fields explicitly unsupported.
### Tests
- [ ] Unit test: richer metrics adapter populates observer state when data exists.
- [ ] Unit test: missing richer metrics are displayed as unavailable.
---
## Step 5 — Build a visible local observer UI
### Goal
Create a developer-facing UI for running probes and reading results locally.
### Mental model
If the observer is only visible in logs, engineers will underuse it. The UI should make latency behavior tangible.
### Work
- [ ] Add a dev/debug observer panel or settings section.
- [ ] Show transport stats, lifecycle stats, optional richer metrics, and logs.
- [ ] Display branch/session/environment metadata.
### Detailed code changes
- Add a component in `webview-ui/src/components/settings/` or a more appropriate dev/debug location.
- Prefer a design inspired by the old `LatencyTester`:
- [ ] simple controls,
- [ ] rolling stats,
- [ ] recent logs,
- [ ] easy reset.
- Add session identity fields:
- [ ] git branch or commit label if available from backend,
- [ ] environment marker (local vs remote),
- [ ] capability flags for richer metrics.
### Tests
- [ ] Component test: controls trigger expected actions.
- [ ] Component test: unavailable metrics render intelligibly.
- [ ] Regression test: UI remains hidden or low-noise outside intended dev/debug usage.
---
## Step 6 — Add scenario-driven observation, not just passive measurement
### Goal
Give developers a way to drive comparable scenarios, not just watch idle telemetry.
### Mental model
A latency observer becomes far more valuable when it can help the developer answer:
- how does branch A behave during a simple request?
- how does branch B behave during the same request?
- what happens under larger payloads or high-churn streaming?
### Work
- [ ] Define a small set of recommended scenarios:
- [ ] pure ping test
- [ ] short assistant response
- [ ] long streaming response
- [ ] tool-heavy / high-churn scenario
- [ ] large-file-write-adjacent scenario if feasible
- [ ] Add UI affordances or documented steps for running those scenarios repeatedly.
### Detailed code changes
- This does not necessarily require the UI to generate tasks itself.
- At minimum, add a documented scenario matrix and a way for the observer to reset/session-label around a manual run.
- If practical, add a small “start known validation task” action in dev mode.
### Tests
- [ ] Validation test: scenario runs produce observer output.
- [ ] Regression test: observer reset cleanly separates sessions.
---
## Step 7 — Support export and comparison workflow
### Goal
Make it easy to compare baseline vs candidate observations after local runs.
### Mental model
Local visibility is great, but engineers also need artifacts they can compare side by side.
### Work
- [ ] Add export of observer session data to JSON.
- [ ] Include branch/commit/environment metadata in export.
- [ ] Make exported structure easy to diff or post-process.
### Detailed code changes
- Add an export button or command that writes a session artifact.
- Keep schema intentionally simple:
- [ ] session metadata,
- [ ] ping samples and aggregates,
- [ ] lifecycle measurements,
- [ ] richer metrics if available,
- [ ] recent logs or event markers.
### Tests
- [ ] Unit test: exported schema is stable.
- [ ] Regression test: export works even when optional metrics are unavailable.
---
## Step 8 — Make the observer safe for use on both `main` and candidate branches
### Goal
Minimize branch-specific surgery so the same observer work can be reused for baseline and future comparison runs.
### Mental model
The observer should behave like a thin compatibility layer, not a second product architecture.
### Work
- [ ] Avoid relying on candidate-only classes for core functionality.
- [ ] Gate richer behavior via capability detection.
- [ ] Keep backend hooks shallow and intentionally placed.
### Detailed code changes
- Prefer hooks at stable abstraction boundaries:
- [ ] UI service layer for ping/pong,
- [ ] controller/task lifecycle boundaries for timing,
- [ ] optional adapters for richer telemetry.
- Avoid deep invasive branch-specific assumptions where possible.
### Tests
- [ ] Manual validation on `main`.
- [ ] Manual validation on `eve_troubleshooting-remote-workspaces`.
- [ ] Confirm that the same observer branch can be adapted onto both with minimal or no code drift.
---
## Step 9 — Document how to interpret the measurements
### Goal
Prevent engineers from misusing the observer or over-interpreting low-level numbers.
### Mental model
Numbers without interpretation guidance lead to bad product decisions.
### Work
- [ ] Document what ping RTT does and does not mean.
- [ ] Document how to compare first-visible-update and state-churn metrics.
- [ ] Document that the most important measurement is user-perceived latency, not just a single low-level counter.
### Detailed code changes
- Add an adjacent doc section or inline UI help text explaining:
- [ ] transport RTT is a lower-level signal,
- [ ] task lifecycle timings are closer to user perception,
- [ ] churn metrics help explain cause,
- [ ] branch comparison should use the same scenario and environment.
### Tests
- [ ] No code-heavy tests needed; verify docs/UI copy is clear.
---
## Minimal Coherent Extraction Boundary
The smallest useful PR for this observer should probably include:
- a modernized ping/pong probe,
- the shared observer metric contract,
- a small local UI panel,
- a minimal task lifecycle timing hook layer,
- and an export path.
What should not be split apart if avoidable:
- observer metric model from the UI,
- ping probe from the visible stats surface,
- branch/session metadata from exported artifacts,
- and capability labeling from optional richer metrics.
---
## Common Failure Modes
Watch for these explicitly:
- building a tester that only measures transport RTT and not user-perceived task latency,
- building a tester that only works on one branch,
- tightly coupling the observer to candidate-only telemetry plumbing,
- surfacing metrics without clarifying whether they are unavailable vs zero,
- and creating a debug UI that is so hidden or awkward that engineers stop using it.
---
## Recommended Validation Workflow
Once implemented, a good manual workflow would be:
1. apply the observer branch to `main`
2. run a fixed set of scenarios and export results
3. apply the same observer branch to `eve_troubleshooting-remote-workspaces`
4. run the same scenarios in the same environment
5. compare:
- [ ] ping RTT behavior
- [ ] initialization and first-visible-update timing
- [ ] state push counts/bytes where available
- [ ] partial/delta event behavior where available
- [ ] subjective smoothness during the same manual workflow
That gives both a human and a machine-readable comparison story.
---
## Developer Checklist Summary
- [ ] Define a branch-portable shared metric contract
- [ ] Implement a modernized ping/pong transport probe
- [ ] Add a shallow branch-portable task observer hook layer
- [ ] Adapt richer metrics where available without making them required
- [ ] Build a visible local observer UI
- [ ] Support scenario-driven observation
- [ ] Add export/comparison workflow support
- [ ] Validate portability on both `main` and `eve_troubleshooting-remote-workspaces`
- [ ] Document how to interpret the results
---
## Final Mental Model Recap
- **The old ping-pong tester is inspiration, not a blueprint.**
- **The modern observer should measure both pipe latency and task-experience latency.**
- **A single branch-portable observer is much more valuable than two branch-specific testers.**
- **The best observer is one engineers will actually use while iterating.**
If implemented well, this mechanism will give the team a practical way to compare current behavior and future behavior using the same local visibility tool, which is exactly what the next stage of this work needs.
+7
View File
@@ -232,6 +232,10 @@ message ShowWebviewEvent {
bool preserve_editor_focus = 1; // When true, webview should not steal focus from editor
}
message TaskUiDeltaEvent {
string delta_json = 1;
}
// UiService provides methods for managing UI interactions
service UiService {
// Scrolls to a specific settings section in the settings view
@@ -267,6 +271,9 @@ service UiService {
// Subscribe to partial message updates (streaming Cline messages as they're built)
rpc subscribeToPartialMessage(EmptyRequest) returns (stream ClineMessage);
// Subscribe to task UI delta updates for active task execution state
rpc subscribeToTaskUiDeltas(EmptyRequest) returns (stream TaskUiDeltaEvent);
// Initialize webview when it launches
rpc initializeWebview(EmptyRequest) returns (Empty);
+2
View File
@@ -48,6 +48,8 @@ message GetHostVersionResponse {
optional string platform = 1;
// The version of the host platform, e.g. 1.103.0 for VSCode, or 2025.1.1.1 for JetBrains IDEs.
optional string version = 2;
// Remote identifier for host environments that support remote workspaces, e.g. ssh-remote.
optional string remote_name = 5;
// The type of the cline host environment, e.g. 'VSCode Extension', 'Cline for JetBrains', 'CLI'
// This is different from the platform because there are many JetBrains IDEs, but they all use the same
// plugin.
+32
View File
@@ -0,0 +1,32 @@
import fs from "node:fs/promises"
import path from "node:path"
import { summarizeTaskLatencyEvents } from "../src/services/telemetry/taskLatencySummary"
function parseEventLines(raw) {
return raw
.split(/\r?\n/)
.map((line) => line.trim())
.filter(Boolean)
.map((line) => JSON.parse(line))
.map((entry) => entry.properties ?? entry)
.filter((entry) => entry)
}
async function main() {
const inputPath = process.argv[2]
if (!inputPath) {
console.error("Usage: node scripts/analyze-task-latency-metrics.mjs <path-to-jsonl>")
process.exit(1)
}
const absolutePath = path.resolve(process.cwd(), inputPath)
const raw = await fs.readFile(absolutePath, "utf8")
const events = parseEventLines(raw)
const summary = summarizeTaskLatencyEvents(events)
console.log(JSON.stringify(summary, null, 2))
}
main().catch((error) => {
console.error(error)
process.exit(1)
})
+37
View File
@@ -0,0 +1,37 @@
import fs from "node:fs/promises"
import path from "node:path"
import { compareTaskLatencySummaries, summarizeTaskLatencyEvents } from "../src/services/telemetry/taskLatencySummary"
function parseEventLines(raw) {
return raw
.split(/\r?\n/)
.map((line) => line.trim())
.filter(Boolean)
.map((line) => JSON.parse(line))
.map((entry) => entry.properties ?? entry)
.filter(Boolean)
}
async function loadSummary(inputPath) {
const absolutePath = path.resolve(process.cwd(), inputPath)
const raw = await fs.readFile(absolutePath, "utf8")
return summarizeTaskLatencyEvents(parseEventLines(raw))
}
async function main() {
const baselinePath = process.argv[2]
const candidatePath = process.argv[3]
if (!baselinePath || !candidatePath) {
console.error("Usage: node scripts/compare-task-latency-metrics.mjs <baseline-jsonl> <candidate-jsonl>")
process.exit(1)
}
const baseline = await loadSummary(baselinePath)
const candidate = await loadSummary(candidatePath)
console.log(JSON.stringify(compareTaskLatencySummaries(baseline, candidate), null, 2))
}
main().catch((error) => {
console.error(error)
process.exit(1)
})
+9
View File
@@ -61,6 +61,15 @@ function createMockService<T extends grpc.UntypedServiceImplementation>(serviceN
// Special cases that need specific return values
switch (prop) {
case "getHostVersion":
callback(null, {
clineVersion: process.env.TEST_HOSTBRIDGE_CLINE_VERSION || "test-cline-version",
version: process.env.TEST_HOSTBRIDGE_IDE_VERSION || "1.0.0",
platform: process.env.TEST_HOSTBRIDGE_PLATFORM || "VS Code",
clineType: process.env.TEST_HOSTBRIDGE_CLINE_TYPE || "vscode",
remoteName: process.env.TEST_HOSTBRIDGE_REMOTE_NAME || "",
})
return
case "getWorkspacePaths":
const workspaceDir = process.env.TEST_HOSTBRIDGE_WORKSPACE_DIR || "/test-workspace"
callback(null, {
+323
View File
@@ -0,0 +1,323 @@
#!/usr/bin/env npx tsx
import { type ChildProcess, spawn } from "node:child_process"
import { once } from "node:events"
import net from "node:net"
import path from "node:path"
import { fileURLToPath } from "node:url"
import { credentials } from "@grpc/grpc-js"
import { AccountServiceClient } from "../src/generated/grpc-js/cline/account"
import { StateServiceClient } from "../src/generated/grpc-js/cline/state"
import { TaskServiceClient } from "../src/generated/grpc-js/cline/task"
import { UiServiceClient } from "../src/generated/grpc-js/cline/ui"
type ValidationMode = "local" | "remote"
type ValidationVariant = {
name: string
env: Record<string, string>
}
type ScenarioResult = {
variant: string
mode: ValidationMode
newTaskRpcMs: number
firstStateMs: number | null
firstPartialMessageMs: number | null
firstTaskDeltaMs: number | null
completionMs: number | null
stateUpdateCount: number
partialMessageCount: number
taskDeltaCount: number
statePayloadBytes: number
taskDeltaPayloadBytes: number
messageCountAtCompletion: number | null
completed: boolean
error?: string
}
const SCRIPT_DIR = path.dirname(fileURLToPath(import.meta.url))
const PROJECT_ROOT = path.resolve(SCRIPT_DIR, "..")
const variants: ValidationVariant[] = [
{ name: "default", env: {} },
{
name: "presentation_disabled",
env: {
CLINE_DISABLE_PRESENTATION_SCHEDULER: "true",
},
},
{
name: "ephemeral_disabled",
env: {
CLINE_DISABLE_EPHEMERAL_MESSAGE_PERSISTENCE: "true",
},
},
{
name: "delta_disabled",
env: {
CLINE_DISABLE_TASK_UI_DELTA_SYNC: "true",
},
},
]
async function waitForPort(port: number, timeoutMs = 20_000): Promise<void> {
const startedAt = Date.now()
while (Date.now() - startedAt < timeoutMs) {
try {
await new Promise<void>((resolve, reject) => {
const socket = net.connect(port, "127.0.0.1", () => {
socket.destroy()
resolve()
})
socket.on("error", reject)
})
return
} catch {
await new Promise((resolve) => setTimeout(resolve, 100))
}
}
throw new Error(`Timed out waiting for port ${port}`)
}
async function getFreePort(): Promise<number> {
return await new Promise((resolve, reject) => {
const server = net.createServer()
server.listen(0, "127.0.0.1", () => {
const address = server.address()
if (!address || typeof address === "string") {
server.close(() => reject(new Error("Unable to allocate port")))
return
}
const { port } = address
server.close((error) => {
if (error) {
reject(error)
return
}
resolve(port)
})
})
server.on("error", reject)
})
}
function unaryCall<T>(fn: (callback: (error: Error | null, response: T) => void) => void): Promise<T> {
return new Promise((resolve, reject) => {
fn((error, response) => {
if (error) {
reject(error)
return
}
resolve(response)
})
})
}
async function startServer(mode: ValidationMode, envOverrides: Record<string, string>) {
const grpcPort = await getFreePort()
const hostbridgePort = await getFreePort()
const env: Record<string, string> = {
...process.env,
PROTOBUS_PORT: String(grpcPort),
HOSTBRIDGE_PORT: String(hostbridgePort),
E2E_TEST: "true",
CLINE_ENVIRONMENT: "local",
TEST_HOSTBRIDGE_REMOTE_NAME: mode === "remote" ? "ssh-remote" : "",
TEST_HOSTBRIDGE_PLATFORM: mode === "remote" ? "VS Code Remote" : "VS Code",
GRPC_RECORDER_ENABLED: "false",
...envOverrides,
}
const child = spawn("npx", ["tsx", path.join(PROJECT_ROOT, "scripts", "test-standalone-core-api-server.ts")], {
cwd: PROJECT_ROOT,
env,
stdio: ["ignore", "pipe", "pipe"],
})
child.stdout?.on("data", () => {
// Drain stdout so the spawned server cannot block on a full pipe buffer.
})
let stderr = ""
child.stderr?.on("data", (chunk) => {
stderr += chunk.toString()
})
await waitForPort(grpcPort)
return { child, grpcPort, hostbridgePort, getStderr: () => stderr }
}
async function stopServer(child: ChildProcess) {
if (child.killed || child.exitCode !== null) {
return
}
child.kill("SIGINT")
try {
await Promise.race([once(child, "exit"), new Promise((resolve) => setTimeout(resolve, 5_000))])
} catch {
child.kill("SIGKILL")
}
}
async function runScenario(mode: ValidationMode, variant: ValidationVariant): Promise<ScenarioResult> {
const server = await startServer(mode, variant.env)
const address = `127.0.0.1:${server.grpcPort}`
const accountClient = new AccountServiceClient(address, credentials.createInsecure())
const stateClient = new StateServiceClient(address, credentials.createInsecure())
const taskClient = new TaskServiceClient(address, credentials.createInsecure())
const uiClient = new UiServiceClient(address, credentials.createInsecure())
let currentTaskId: string | undefined
const startedAt = Date.now()
let firstStateMs: number | null = null
let firstPartialMessageMs: number | null = null
let firstTaskDeltaMs: number | null = null
let completionMs: number | null = null
let stateUpdateCount = 0
let partialMessageCount = 0
let taskDeltaCount = 0
let statePayloadBytes = 0
let taskDeltaPayloadBytes = 0
let messageCountAtCompletion: number | null = null
let completed = false
const stateStream = stateClient.subscribeToState({})
const partialStream = uiClient.subscribeToPartialMessage({})
const deltaStream = uiClient.subscribeToTaskUiDeltas({})
for (const stream of [stateStream, partialStream, deltaStream]) {
stream.on("error", (streamError: any) => {
if (streamError?.code === 1 || streamError?.details === "Cancelled on client") {
return
}
console.error("validation stream error", streamError)
})
}
stateStream.on("data", (response: { stateJson?: string }) => {
stateUpdateCount += 1
const stateJson = response.stateJson || "{}"
statePayloadBytes += Buffer.byteLength(stateJson, "utf8")
if (firstStateMs === null) {
firstStateMs = Date.now() - startedAt
}
try {
const state = JSON.parse(stateJson)
const activeTaskId = state.currentTaskItem?.id
if (activeTaskId) {
currentTaskId = activeTaskId
}
const clineMessages = Array.isArray(state.clineMessages) ? state.clineMessages : []
const hasCompletion = clineMessages.some(
(message: any) => message.ask === "completion_result" || message.ask === "resume_completed_task",
)
if (hasCompletion && completionMs === null) {
completionMs = Date.now() - startedAt
messageCountAtCompletion = clineMessages.length
completed = true
}
} catch {
// ignore parse errors in validation harness
}
})
partialStream.on("data", (message: { say?: string; text?: string }) => {
partialMessageCount += 1
if (firstPartialMessageMs === null && (message.say === "text" || message.say === "reasoning")) {
firstPartialMessageMs = Date.now() - startedAt
}
})
deltaStream.on("data", (event: { deltaJson?: string }) => {
taskDeltaCount += 1
const deltaJson = event.deltaJson || ""
taskDeltaPayloadBytes += Buffer.byteLength(deltaJson, "utf8")
if (firstTaskDeltaMs === null) {
try {
const delta = JSON.parse(deltaJson)
if (delta.type?.startsWith("message_") || delta.type === "task_metadata_updated") {
firstTaskDeltaMs = Date.now() - startedAt
}
} catch {
// ignore parse errors
}
}
})
let newTaskRpcMs = 0
let error: string | undefined
try {
await unaryCall<{ value?: string }>((callback) => accountClient.accountLoginClicked({}, callback as any))
await unaryCall((callback) => accountClient.getUserOrganizations({}, callback as any))
const rpcStartedAt = Date.now()
const newTaskResponse = await unaryCall<{ value?: string }>((callback) =>
taskClient.newTask(
{
metadata: undefined,
text: "latency_validation",
images: [],
files: [],
taskSettings: undefined,
},
callback as any,
),
)
newTaskRpcMs = Date.now() - rpcStartedAt
currentTaskId = newTaskResponse.value || currentTaskId
const timeoutAt = Date.now() + 20_000
while (!completed && Date.now() < timeoutAt) {
await new Promise((resolve) => setTimeout(resolve, 100))
}
if (!completed) {
error = "Scenario timed out before completion"
}
} catch (scenarioError) {
error = scenarioError instanceof Error ? scenarioError.message : String(scenarioError)
}
stateStream.cancel()
partialStream.cancel()
deltaStream.cancel()
accountClient.close()
stateClient.close()
taskClient.close()
uiClient.close()
await stopServer(server.child)
return {
variant: variant.name,
mode,
newTaskRpcMs,
firstStateMs,
firstPartialMessageMs,
firstTaskDeltaMs,
completionMs,
stateUpdateCount,
partialMessageCount,
taskDeltaCount,
statePayloadBytes,
taskDeltaPayloadBytes,
messageCountAtCompletion,
completed,
error,
}
}
async function main() {
const results: ScenarioResult[] = []
for (const mode of ["local", "remote"] as const) {
for (const variant of variants) {
results.push(await runScenario(mode, variant))
}
}
console.log(JSON.stringify({ results }, null, 2))
}
main().catch((error) => {
console.error(error)
process.exit(1)
})
+147
View File
@@ -0,0 +1,147 @@
type StateUpdatePriority = "immediate" | "normal" | "low"
type StateUpdateSchedulerOptions = {
flush: () => Promise<void>
getDelayMs: (priority: StateUpdatePriority) => number
setTimeoutFn?: typeof setTimeout
clearTimeoutFn?: typeof clearTimeout
onFlushError?: (error: unknown) => void
getNow?: () => number
metrics?: {
onFlushStarted?: (priority: StateUpdatePriority) => void
onFlushCompleted?: (durationMs: number, priority: StateUpdatePriority) => void
}
}
export class StateUpdateScheduler {
private scheduledTimer: ReturnType<typeof setTimeout> | undefined
private pendingPriority: StateUpdatePriority | undefined
private flushInProgress = false
private disposed = false
private pendingWhileFlushing = false
private readonly flush: () => Promise<void>
private readonly getDelayMs: (priority: StateUpdatePriority) => number
private readonly setTimeoutFn: typeof setTimeout
private readonly clearTimeoutFn: typeof clearTimeout
private readonly onFlushError?: (error: unknown) => void
private readonly getNow: () => number
private readonly metrics?: StateUpdateSchedulerOptions["metrics"]
constructor(options: StateUpdateSchedulerOptions) {
this.flush = options.flush
this.getDelayMs = options.getDelayMs
this.setTimeoutFn = options.setTimeoutFn ?? setTimeout
this.clearTimeoutFn = options.clearTimeoutFn ?? clearTimeout
this.onFlushError = options.onFlushError
this.getNow = options.getNow ?? (() => performance.now())
this.metrics = options.metrics
}
requestFlush(priority: StateUpdatePriority = "normal"): void {
if (this.disposed) {
return
}
this.pendingPriority = this.mergePriority(this.pendingPriority, priority)
if (this.flushInProgress) {
this.pendingWhileFlushing = true
return
}
if (this.pendingPriority === "immediate") {
if (this.scheduledTimer) {
this.clearTimeoutFn(this.scheduledTimer)
this.scheduledTimer = undefined
}
void this.runFlushCycle()
return
}
if (this.scheduledTimer) {
return
}
const nextPriority = this.pendingPriority ?? "normal"
const delayMs = this.getDelayMs(nextPriority)
this.scheduledTimer = this.setTimeoutFn(() => {
this.scheduledTimer = undefined
void this.runFlushCycle()
}, delayMs)
}
async flushNow(): Promise<void> {
if (this.disposed) {
return
}
this.pendingPriority = this.mergePriority(this.pendingPriority, "immediate")
if (this.scheduledTimer) {
this.clearTimeoutFn(this.scheduledTimer)
this.scheduledTimer = undefined
}
await this.runFlushCycle()
}
async dispose(): Promise<void> {
this.disposed = true
if (this.scheduledTimer) {
this.clearTimeoutFn(this.scheduledTimer)
this.scheduledTimer = undefined
}
this.pendingPriority = undefined
this.pendingWhileFlushing = false
}
private async runFlushCycle(): Promise<void> {
if (this.disposed || this.flushInProgress || !this.pendingPriority) {
return
}
this.flushInProgress = true
const priority = this.pendingPriority
this.pendingPriority = undefined
this.pendingWhileFlushing = false
const startedAt = this.getNow()
this.metrics?.onFlushStarted?.(priority)
try {
await this.flush()
this.metrics?.onFlushCompleted?.(Math.max(0, this.getNow() - startedAt), priority)
} catch (error) {
this.onFlushError?.(error)
} finally {
this.flushInProgress = false
}
if (this.disposed) {
return
}
if (this.pendingPriority || this.pendingWhileFlushing) {
const priorityToRun = this.pendingPriority
if (priorityToRun === "immediate") {
await this.runFlushCycle()
} else {
this.requestFlush(priorityToRun ?? "normal")
}
}
}
private mergePriority(current: StateUpdatePriority | undefined, next: StateUpdatePriority): StateUpdatePriority {
if (!current) {
return next
}
const rank: Record<StateUpdatePriority, number> = {
low: 0,
normal: 1,
immediate: 2,
}
return rank[next] > rank[current] ? next : current
}
}
export type { StateUpdatePriority }
+88 -2
View File
@@ -1,6 +1,7 @@
import type { Anthropic } from "@anthropic-ai/sdk"
import { buildApiHandler } from "@core/api"
import { getHooksEnabledSafe } from "@core/hooks/hooks-utils"
import { getStateUpdateCadenceMs, isRemoteWorkspaceEnvironment } from "@core/task/latency"
import { tryAcquireTaskLockWithRetry } from "@core/task/TaskLockUtils"
import { detectWorkspaceRoots } from "@core/workspace/detection"
import { setupWorkspaceManager } from "@core/workspace/setup"
@@ -56,9 +57,11 @@ import { Task } from "../task"
import { sendMcpMarketplaceCatalogEvent } from "./mcp/subscribeToMcpMarketplaceCatalog"
import { getClineOnboardingModels } from "./models/getClineOnboardingModels"
import { appendClineStealthModels } from "./models/refreshOpenRouterModels"
import { type StateUpdatePriority, StateUpdateScheduler } from "./StateUpdateScheduler"
import { checkCliInstallation } from "./state/checkCliInstallation"
import { sendStateUpdate } from "./state/subscribeToState"
import { sendChatButtonClickedEvent } from "./ui/subscribeToChatButtonClicked"
import { sendTaskUiDelta } from "./ui/subscribeToTaskUiDeltas"
/*
https://github.com/microsoft/vscode-webview-ui-toolkit-samples/blob/main/default/weather-webview/src/providers/WeatherViewProvider.ts
@@ -85,6 +88,9 @@ export class Controller {
// Timer for periodic remote config fetching
private remoteConfigTimer?: NodeJS.Timeout
private isRemoteWorkspaceEnvironment = false
private readonly stateUpdateScheduler: StateUpdateScheduler
private readonly schedulerDebugLoggingEnabled = process.env.CLINE_DEBUG_LATENCY === "1"
// Public getter for workspace manager with lazy initialization - To get workspaces when task isn't initialized (Used by file mentions)
async ensureWorkspaceManager(): Promise<WorkspaceRootManager | undefined> {
@@ -118,6 +124,14 @@ export class Controller {
}
constructor(readonly context: ClineExtensionContext) {
void HostProvider.env
.getHostVersion({})
.then((hostVersion) => {
this.isRemoteWorkspaceEnvironment = isRemoteWorkspaceEnvironment(hostVersion)
})
.catch((error) => {
Logger.debug(`[Controller] Failed to detect remote workspace state: ${error}`)
})
Session.reset() // Reset session on controller initialization
PromptRegistry.getInstance() // Ensure prompts and tools are registered
this.stateManager = StateManager.get()
@@ -156,6 +170,24 @@ export class Controller {
// Check CLI installation status once on startup
checkCliInstallation(this)
this.stateUpdateScheduler = new StateUpdateScheduler({
flush: async () => this.flushStateToWebview(),
getDelayMs: (priority) => getStateUpdateCadenceMs(this.isRemoteWorkspaceEnvironment, priority),
onFlushError: (error) => Logger.debug(`[Controller] Failed scheduled state flush: ${error}`),
metrics: {
onFlushStarted: (priority) => {
if (this.schedulerDebugLoggingEnabled) {
Logger.debug(`[Controller] state flush started (${priority})`)
}
},
onFlushCompleted: (durationMs, priority) => {
if (this.schedulerDebugLoggingEnabled) {
Logger.debug(`[Controller] state flush completed (${priority}) in ${durationMs}ms`)
}
},
},
})
Logger.log("[Controller] ClineProvider instantiated")
}
@@ -172,6 +204,7 @@ export class Controller {
}
await this.clearTask()
await this.stateUpdateScheduler.dispose()
this.mcpHub.dispose()
Logger.error("Controller disposed")
@@ -499,9 +532,37 @@ export class Controller {
}
this.backgroundCommandRunning = running
this.backgroundCommandTaskId = nextTaskId
if (this.task && nextTaskId && this.task.taskId === nextTaskId) {
void this.postTaskMetadataDelta({
backgroundCommandRunning: running,
backgroundCommandTaskId: nextTaskId,
})
return
}
void this.postStateToWebview()
}
async postTaskMetadataDelta(
metadata: Partial<
Pick<ExtensionState, "currentFocusChainChecklist" | "backgroundCommandRunning" | "backgroundCommandTaskId">
>,
taskId?: string,
) {
const targetTask = this.task
const resolvedTaskId = taskId ?? targetTask?.taskId
if (!targetTask || !resolvedTaskId || targetTask.taskId !== resolvedTaskId) {
await this.postStateToWebview({ priority: "normal" })
return
}
await sendTaskUiDelta({
type: "task_metadata_updated",
taskId: resolvedTaskId,
sequence: ++targetTask.taskState.taskUiDeltaSequence,
metadata,
})
}
async cancelBackgroundCommand(): Promise<void> {
const didCancel = await this.task?.cancelBackgroundCommand()
if (!didCancel) {
@@ -837,9 +898,34 @@ export class Controller {
return updatedTaskHistory
}
async postStateToWebview() {
async postStateToWebview(options?: { priority?: StateUpdatePriority }) {
const priority = options?.priority ?? this.getDefaultStateUpdatePriority()
if (priority === "immediate") {
await this.stateUpdateScheduler.flushNow()
return
}
this.stateUpdateScheduler.requestFlush(priority)
}
private getDefaultStateUpdatePriority(): StateUpdatePriority {
if (!this.task) {
return "immediate"
}
return this.task.taskState.isStreaming ? "normal" : "immediate"
}
private async flushStateToWebview() {
const buildStartedAt = performance.now()
const state = await this.getStateToPostToWebview()
await sendStateUpdate(state)
const buildDurationMs = Math.max(0, performance.now() - buildStartedAt)
const deliveryStats = await sendStateUpdate(state)
this.task?.noteStateUpdateMetrics({
buildDurationMs,
serializedBytes: deliveryStats.payloadBytes,
sendDurationMs: deliveryStats.sendDurationMs,
})
}
async getStateToPostToWebview(): Promise<ExtensionState> {
@@ -0,0 +1,24 @@
import { strict as assert } from "assert"
import { sendStateUpdate, subscribeToState } from "./subscribeToState"
describe("subscribeToState", () => {
it("returns delivery stats and broadcasts updates to subscribers", async () => {
const received: string[] = []
const controller = {
getStateToPostToWebview: async () => ({ mode: "act", clineMessages: [] }),
} as any
await subscribeToState(controller, {} as any, async (message) => {
received.push(message.stateJson ?? "")
})
assert.equal(received.length, 1)
const stats = await sendStateUpdate({ mode: "plan", clineMessages: [] } as any)
assert.equal(received.length, 2)
assert.ok(stats.payloadBytes > 0)
assert.ok(stats.sendDurationMs >= 0)
assert.ok(stats.subscriberCount >= 1)
assert.ok(received[1]?.includes('"mode":"plan"'))
})
})
+21 -3
View File
@@ -9,6 +9,12 @@ import { Controller } from "../index"
// Keep track of active state subscriptions
const activeStateSubscriptions = new Set<StreamingResponseHandler<State>>()
export type StateUpdateDeliveryStats = {
payloadBytes: number
sendDurationMs: number
subscriberCount: number
}
/**
* Subscribe to state updates
* @param controller The controller instance
@@ -58,16 +64,22 @@ export async function subscribeToState(
* Send a state update to all active subscribers
* @param state The state to send
*/
export async function sendStateUpdate(state: ExtensionState): Promise<void> {
export async function sendStateUpdate(state: ExtensionState): Promise<StateUpdateDeliveryStats> {
let stateJson: string
try {
stateJson = JSON.stringify(state)
} catch (error) {
Logger.error("Error serializing state update:", error)
return
return {
payloadBytes: 0,
sendDurationMs: 0,
subscriberCount: activeStateSubscriptions.size,
}
}
recordStateSizeTelemetry(Buffer.byteLength(stateJson, "utf8"))
const payloadBytes = Buffer.byteLength(stateJson, "utf8")
recordStateSizeTelemetry(payloadBytes)
const startedAt = performance.now()
const promises = Array.from(activeStateSubscriptions).map(async (responseStream) => {
try {
@@ -84,6 +96,12 @@ export async function sendStateUpdate(state: ExtensionState): Promise<void> {
})
await Promise.all(promises)
return {
payloadBytes,
sendDurationMs: Math.max(0, performance.now() - startedAt),
subscriberCount: activeStateSubscriptions.size,
}
}
function recordStateSizeTelemetry(sizeBytes: number): void {
@@ -0,0 +1,24 @@
import { strict as assert } from "assert"
import { registerPartialMessageCallback, sendPartialMessageEvent, subscribeToPartialMessage } from "./subscribeToPartialMessage"
describe("subscribeToPartialMessage", () => {
it("returns delivery stats and broadcasts to stream and callback subscribers", async () => {
const callbackMessages: any[] = []
const streamMessages: any[] = []
const unsubscribe = registerPartialMessageCallback((message) => callbackMessages.push(message))
await subscribeToPartialMessage({} as any, {} as any, async (message) => {
streamMessages.push(message)
})
const stats = await sendPartialMessageEvent({ ts: 123, type: "say", say: "text", text: "hello" } as any)
unsubscribe()
assert.equal(callbackMessages.length, 1)
assert.equal(streamMessages.length, 1)
assert.ok(stats.payloadBytes > 0)
assert.ok(stats.broadcastDurationMs >= 0)
assert.ok(stats.streamSubscriberCount >= 1)
assert.ok(stats.callbackSubscriberCount >= 1)
})
})
@@ -1,5 +1,6 @@
import { EmptyRequest } from "@shared/proto/cline/common"
import { ClineMessage } from "@shared/proto/cline/ui"
import { telemetryService } from "@/services/telemetry"
import { Logger } from "@/shared/services/Logger"
import { getRequestRegistry, StreamingResponseHandler } from "../grpc-handler"
import { Controller } from "../index"
@@ -11,6 +12,13 @@ const activePartialMessageSubscriptions = new Set<StreamingResponseHandler<Cline
export type PartialMessageCallback = (message: ClineMessage) => void
const callbackSubscriptions = new Set<PartialMessageCallback>()
export type PartialMessageDeliveryStats = {
payloadBytes: number
broadcastDurationMs: number
streamSubscriberCount: number
callbackSubscriberCount: number
}
/**
* Subscribe to partial message events
* @param controller The controller instance
@@ -54,7 +62,11 @@ export function registerPartialMessageCallback(callback: PartialMessageCallback)
* Send a partial message event to all active subscribers
* @param partialMessage The ClineMessage to send
*/
export async function sendPartialMessageEvent(partialMessage: ClineMessage): Promise<void> {
export async function sendPartialMessageEvent(partialMessage: ClineMessage): Promise<PartialMessageDeliveryStats> {
const payloadBytes = Buffer.byteLength(JSON.stringify(partialMessage), "utf8")
telemetryService.captureGrpcResponseSize(payloadBytes, "cline.UiService", "subscribeToPartialMessage")
const startedAt = performance.now()
// Send to gRPC stream subscribers
const streamPromises = Array.from(activePartialMessageSubscriptions).map(async (responseStream) => {
try {
@@ -79,4 +91,11 @@ export async function sendPartialMessageEvent(partialMessage: ClineMessage): Pro
}
await Promise.all(streamPromises)
return {
payloadBytes,
broadcastDurationMs: Math.max(0, performance.now() - startedAt),
streamSubscriberCount: activePartialMessageSubscriptions.size,
callbackSubscriberCount: callbackSubscriptions.size,
}
}
@@ -0,0 +1,85 @@
import { EmptyRequest } from "@shared/proto/cline/common"
import { TaskUiDeltaEvent } from "@shared/proto/cline/ui"
import { strict as assert } from "assert"
import { registerTaskUiDeltaCallback, sendTaskUiDelta, subscribeToTaskUiDeltas } from "./subscribeToTaskUiDeltas"
describe("subscribeToTaskUiDeltas", () => {
it("broadcasts serialized task deltas to active stream subscribers", async () => {
const received: TaskUiDeltaEvent[] = []
const callbackReceived: Array<{ type: string; sequence: number }> = []
const unsubscribe = registerTaskUiDeltaCallback((delta) => {
callbackReceived.push({ type: delta.type, sequence: delta.sequence })
})
await subscribeToTaskUiDeltas({} as any, EmptyRequest.create({}), async (message) => {
received.push(message)
})
await sendTaskUiDelta({
type: "message_updated",
taskId: "task-1",
sequence: 1,
message: { ts: 123, type: "say", say: "text", text: "delta-text" },
})
const stats = await sendTaskUiDelta({
type: "message_updated",
taskId: "task-1",
sequence: 2,
message: { ts: 124, type: "say", say: "text", text: "delta-text-2" },
})
assert.equal(received.length, 2)
assert.ok(received[0]?.deltaJson)
assert.ok(stats)
assert.ok((stats?.payloadBytes ?? 0) > 0)
assert.ok((stats?.broadcastDurationMs ?? -1) >= 0)
assert.ok((stats?.streamSubscriberCount ?? 0) >= 1)
assert.ok((stats?.callbackSubscriberCount ?? 0) >= 1)
const parsed = JSON.parse(received[0]!.deltaJson)
assert.equal(parsed.type, "message_updated")
assert.equal(parsed.taskId, "task-1")
assert.equal(parsed.sequence, 1)
assert.equal(parsed.message.text, "delta-text")
const parsedSecond = JSON.parse(received[1]!.deltaJson)
assert.equal(parsedSecond.sequence, 2)
assert.equal(parsedSecond.message.text, "delta-text-2")
assert.deepStrictEqual(callbackReceived, [
{ type: "message_updated", sequence: 1 },
{ type: "message_updated", sequence: 2 },
])
unsubscribe()
})
it("removes stream subscribers that throw during delivery", async () => {
const received: TaskUiDeltaEvent[] = []
await subscribeToTaskUiDeltas({} as any, EmptyRequest.create({}), async () => {
throw new Error("stream disconnected")
})
await subscribeToTaskUiDeltas({} as any, EmptyRequest.create({}), async (message) => {
received.push(message)
})
await sendTaskUiDelta({
type: "task_state_resynced",
taskId: "task-1",
sequence: 1,
})
await sendTaskUiDelta({
type: "task_metadata_updated",
taskId: "task-1",
sequence: 2,
metadata: { backgroundCommandRunning: true, backgroundCommandTaskId: "task-1" },
})
assert.equal(received.length, 2)
const first = JSON.parse(received[0]!.deltaJson)
const second = JSON.parse(received[1]!.deltaJson)
assert.equal(first.type, "task_state_resynced")
assert.equal(second.type, "task_metadata_updated")
assert.equal(second.metadata.backgroundCommandRunning, true)
})
})
@@ -0,0 +1,82 @@
import { EmptyRequest } from "@shared/proto/cline/common"
import { TaskUiDeltaEvent } from "@shared/proto/cline/ui"
import { telemetryService } from "@/services/telemetry"
import { Logger } from "@/shared/services/Logger"
import type { TaskUiDelta } from "@/shared/TaskUiDelta"
import { getRequestRegistry, StreamingResponseHandler } from "../grpc-handler"
import type { Controller } from "../index"
const activeTaskUiDeltaSubscriptions = new Set<StreamingResponseHandler<TaskUiDeltaEvent>>()
export type TaskUiDeltaCallback = (delta: TaskUiDelta) => void
const callbackSubscriptions = new Set<TaskUiDeltaCallback>()
export type TaskUiDeltaDeliveryStats = {
payloadBytes: number
broadcastDurationMs: number
streamSubscriberCount: number
callbackSubscriberCount: number
}
export function registerTaskUiDeltaCallback(callback: TaskUiDeltaCallback): () => void {
callbackSubscriptions.add(callback)
return () => {
callbackSubscriptions.delete(callback)
}
}
export async function subscribeToTaskUiDeltas(
_controller: Controller,
_request: EmptyRequest,
responseStream: StreamingResponseHandler<TaskUiDeltaEvent>,
requestId?: string,
): Promise<void> {
activeTaskUiDeltaSubscriptions.add(responseStream)
const cleanup = () => {
activeTaskUiDeltaSubscriptions.delete(responseStream)
}
if (requestId) {
getRequestRegistry().registerRequest(requestId, cleanup, { type: "task_ui_delta_subscription" }, responseStream)
}
}
export async function sendTaskUiDelta(delta: TaskUiDelta): Promise<TaskUiDeltaDeliveryStats | undefined> {
let deltaJson: string
try {
deltaJson = JSON.stringify(delta)
} catch (error) {
Logger.error("Error serializing task UI delta:", error)
return undefined
}
const payloadBytes = Buffer.byteLength(deltaJson, "utf8")
telemetryService.captureGrpcResponseSize(payloadBytes, "cline.UiService", "subscribeToTaskUiDeltas")
const startedAt = performance.now()
const promises = Array.from(activeTaskUiDeltaSubscriptions).map(async (responseStream) => {
try {
await responseStream(TaskUiDeltaEvent.create({ deltaJson }), false)
} catch (error) {
Logger.error("Error sending task UI delta:", error)
activeTaskUiDeltaSubscriptions.delete(responseStream)
}
})
for (const callback of callbackSubscriptions) {
try {
callback(delta)
} catch (error) {
Logger.error("Error sending task UI delta to callback subscriber:", error)
}
}
await Promise.all(promises)
return {
payloadBytes,
broadcastDurationMs: Math.max(0, performance.now() - startedAt),
streamSubscriberCount: activeTaskUiDeltaSubscriptions.size,
callbackSubscriberCount: callbackSubscriptions.size,
}
}
@@ -0,0 +1,48 @@
type EphemeralMessageFlushSchedulerOptions = {
flush: () => Promise<void>
getDelayMs: () => number
setIntervalFn?: typeof setInterval
clearIntervalFn?: typeof clearInterval
onFlushError?: (error: unknown) => void
}
export class EphemeralMessageFlushScheduler {
private intervalHandle: ReturnType<typeof setInterval> | undefined
private readonly flush: () => Promise<void>
private readonly getDelayMs: () => number
private readonly setIntervalFn: typeof setInterval
private readonly clearIntervalFn: typeof clearInterval
private readonly onFlushError?: (error: unknown) => void
constructor(options: EphemeralMessageFlushSchedulerOptions) {
this.flush = options.flush
this.getDelayMs = options.getDelayMs
this.setIntervalFn = options.setIntervalFn ?? setInterval
this.clearIntervalFn = options.clearIntervalFn ?? clearInterval
this.onFlushError = options.onFlushError
}
start(): void {
if (this.intervalHandle) {
return
}
this.intervalHandle = this.setIntervalFn(() => {
void this.flush().catch((error) => this.onFlushError?.(error))
}, this.getDelayMs())
}
stop(): void {
if (!this.intervalHandle) {
return
}
this.clearIntervalFn(this.intervalHandle)
this.intervalHandle = undefined
}
dispose(): void {
this.stop()
}
}
+47
View File
@@ -0,0 +1,47 @@
type RequestBoundaryCacheOptions<T> = {
load: () => Promise<T>
ttlMs?: number
getTtlMs?: () => number
getNow?: () => number
}
export class RequestBoundaryCache<T> {
private value: T | undefined
private expiresAt = 0
private inFlight: Promise<T> | undefined
private readonly load: () => Promise<T>
private readonly getNow: () => number
private readonly getTtlMs: () => number
constructor(options: RequestBoundaryCacheOptions<T>) {
this.load = options.load
this.getNow = options.getNow ?? (() => Date.now())
this.getTtlMs = options.getTtlMs ?? (() => options.ttlMs ?? 0)
}
async get(): Promise<T> {
const now = this.getNow()
if (this.value !== undefined && now < this.expiresAt) {
return this.value
}
if (!this.inFlight) {
this.inFlight = this.load()
.then((value) => {
this.value = value
this.expiresAt = this.getNow() + this.getTtlMs()
return value
})
.finally(() => {
this.inFlight = undefined
})
}
return this.inFlight
}
clear(): void {
this.value = undefined
this.expiresAt = 0
}
}
+147
View File
@@ -0,0 +1,147 @@
type PresentationPriority = "immediate" | "normal" | "low"
type TaskPresentationSchedulerOptions = {
flush: () => Promise<void>
getDelayMs: (priority: PresentationPriority) => number
setTimeoutFn?: typeof setTimeout
clearTimeoutFn?: typeof clearTimeout
onFlushError?: (error: unknown) => void
getNow?: () => number
metrics?: {
onFlushStarted?: (priority: PresentationPriority) => void
onFlushCompleted?: (durationMs: number, priority: PresentationPriority) => void
}
}
export class TaskPresentationScheduler {
private scheduledTimer: ReturnType<typeof setTimeout> | undefined
private pendingPriority: PresentationPriority | undefined
private flushInProgress = false
private disposed = false
private pendingWhileFlushing = false
private readonly flush: () => Promise<void>
private readonly getDelayMs: (priority: PresentationPriority) => number
private readonly setTimeoutFn: typeof setTimeout
private readonly clearTimeoutFn: typeof clearTimeout
private readonly onFlushError?: (error: unknown) => void
private readonly getNow: () => number
private readonly metrics?: TaskPresentationSchedulerOptions["metrics"]
constructor(options: TaskPresentationSchedulerOptions) {
this.flush = options.flush
this.getDelayMs = options.getDelayMs
this.setTimeoutFn = options.setTimeoutFn ?? setTimeout
this.clearTimeoutFn = options.clearTimeoutFn ?? clearTimeout
this.onFlushError = options.onFlushError
this.getNow = options.getNow ?? (() => performance.now())
this.metrics = options.metrics
}
requestFlush(priority: PresentationPriority = "normal"): void {
if (this.disposed) {
return
}
this.pendingPriority = this.mergePriority(this.pendingPriority, priority)
if (this.flushInProgress) {
this.pendingWhileFlushing = true
return
}
if (this.pendingPriority === "immediate") {
if (this.scheduledTimer) {
this.clearTimeoutFn(this.scheduledTimer)
this.scheduledTimer = undefined
}
void this.runFlushCycle()
return
}
if (this.scheduledTimer) {
return
}
const nextPriority = this.pendingPriority ?? "normal"
const delayMs = this.getDelayMs(nextPriority)
this.scheduledTimer = this.setTimeoutFn(() => {
this.scheduledTimer = undefined
void this.runFlushCycle()
}, delayMs)
}
async flushNow(): Promise<void> {
if (this.disposed) {
return
}
this.pendingPriority = this.mergePriority(this.pendingPriority, "immediate")
if (this.scheduledTimer) {
this.clearTimeoutFn(this.scheduledTimer)
this.scheduledTimer = undefined
}
await this.runFlushCycle()
}
async dispose(): Promise<void> {
this.disposed = true
if (this.scheduledTimer) {
this.clearTimeoutFn(this.scheduledTimer)
this.scheduledTimer = undefined
}
this.pendingPriority = undefined
this.pendingWhileFlushing = false
}
private async runFlushCycle(): Promise<void> {
if (this.disposed || this.flushInProgress || !this.pendingPriority) {
return
}
this.flushInProgress = true
const priority = this.pendingPriority
this.pendingPriority = undefined
this.pendingWhileFlushing = false
const startedAt = this.getNow()
this.metrics?.onFlushStarted?.(priority)
try {
await this.flush()
this.metrics?.onFlushCompleted?.(Math.max(0, this.getNow() - startedAt), priority)
} catch (error) {
this.onFlushError?.(error)
} finally {
this.flushInProgress = false
}
if (this.disposed) {
return
}
if (this.pendingPriority || this.pendingWhileFlushing) {
const priorityToRun = this.pendingPriority
if (priorityToRun === "immediate") {
await this.runFlushCycle()
} else {
this.requestFlush(priorityToRun ?? "normal")
}
}
}
private mergePriority(current: PresentationPriority | undefined, next: PresentationPriority): PresentationPriority {
if (!current) {
return next
}
const rank: Record<PresentationPriority, number> = {
low: 0,
normal: 1,
immediate: 2,
}
return rank[next] > rank[current] ? next : current
}
}
export type { PresentationPriority }
+45
View File
@@ -7,6 +7,50 @@ export class TaskState {
// Task-level timing
taskStartTimeMs = Date.now()
taskFirstTokenTimeMs?: number
isRemoteWorkspace = false
// Request-scoped performance metrics
presentationMetrics = {
requestStartedAtMs: 0,
invocationCount: 0,
totalDurationMs: 0,
triggerCounts: {
text: 0,
reasoning: 0,
tool: 0,
finalization: 0,
other: 0,
},
}
statePostMetrics = {
requestStartedAtMs: 0,
callCount: 0,
coalescedCallCount: 0,
stateBuildDurationMs: 0,
serializedBytes: 0,
sendDurationMs: 0,
}
partialMessageMetrics = {
requestStartedAtMs: 0,
eventCount: 0,
payloadBytes: 0,
broadcastDurationMs: 0,
}
persistenceMetrics = {
requestStartedAtMs: 0,
saveMessagesDurationMs: 0,
saveConversationDurationMs: 0,
updateHistoryDurationMs: 0,
flushCount: 0,
}
chunkToWebviewMetrics = {
requestStartedAtMs: 0,
chunkCount: 0,
lastChunkReceivedAtMs: 0,
lastWebviewFlushCompletedAtMs: 0,
observedDelaysMs: [] as number[],
}
currentChunkReceivedAtMs?: number
// Streaming flags
isStreaming = false
@@ -62,6 +106,7 @@ export class TaskState {
apiRequestsSinceLastTodoUpdate = 0
currentFocusChainChecklist: string | null = null
todoListWasUpdatedByUser = false
taskUiDeltaSequence = 0
// Task Abort / Cancellation
abort = false
+86
View File
@@ -0,0 +1,86 @@
type TaskUsageUpdateSchedulerOptions = {
getDelayMs: () => number
setTimeoutFn?: typeof setTimeout
clearTimeoutFn?: typeof clearTimeout
onSideEffectError?: (error: unknown) => void
onUiFlushError?: (error: unknown) => void
}
type UsageUpdateWork = {
sideEffect?: () => Promise<void>
flushUi?: () => Promise<void>
}
export class TaskUsageUpdateScheduler {
private sideEffectsQueue = Promise.resolve()
private flushScheduled = false
private timer: ReturnType<typeof setTimeout> | undefined
private finalized = false
private readonly getDelayMs: () => number
private readonly setTimeoutFn: typeof setTimeout
private readonly clearTimeoutFn: typeof clearTimeout
private readonly onSideEffectError?: (error: unknown) => void
private readonly onUiFlushError?: (error: unknown) => void
constructor(options: TaskUsageUpdateSchedulerOptions) {
this.getDelayMs = options.getDelayMs
this.setTimeoutFn = options.setTimeoutFn ?? setTimeout
this.clearTimeoutFn = options.clearTimeoutFn ?? clearTimeout
this.onSideEffectError = options.onSideEffectError
this.onUiFlushError = options.onUiFlushError
}
enqueue(work: UsageUpdateWork): void {
if (this.finalized) {
return
}
if (work.sideEffect) {
this.sideEffectsQueue = this.sideEffectsQueue.then(work.sideEffect).catch((error) => this.onSideEffectError?.(error))
}
if (!work.flushUi || this.flushScheduled || this.finalized) {
return
}
this.flushScheduled = true
this.clearTimer()
this.timer = this.setTimeoutFn(() => {
this.timer = undefined
this.sideEffectsQueue = this.sideEffectsQueue
.then(async () => {
if (this.finalized) {
return
}
this.flushScheduled = false
await work.flushUi?.()
})
.catch((error) => {
this.flushScheduled = false
this.onUiFlushError?.(error)
})
}, this.getDelayMs())
}
async flushFinal(flushUi?: () => Promise<void>): Promise<void> {
this.finalized = true
this.flushScheduled = false
this.clearTimer()
await this.sideEffectsQueue
await flushUi?.()
}
dispose(): void {
this.finalized = true
this.flushScheduled = false
this.clearTimer()
}
private clearTimer(): void {
if (this.timer) {
this.clearTimeoutFn(this.timer)
this.timer = undefined
}
}
}
@@ -0,0 +1,102 @@
import { strict as assert } from "assert"
import { MessageStateHandler } from "../../task/message-state"
import { TaskState } from "../../task/TaskState"
import { EphemeralMessageFlushScheduler } from "../EphemeralMessageFlushScheduler"
class FakeIntervalController {
private now = 0
private nextId = 1
private intervals = new Map<number, { delay: number; nextRunAt: number; callback: () => void }>()
setInterval = (callback: () => void, delay: number) => {
const id = this.nextId++
this.intervals.set(id, { delay, nextRunAt: this.now + delay, callback })
return id as unknown as ReturnType<typeof setInterval>
}
clearInterval = (handle: ReturnType<typeof setInterval>) => {
this.intervals.delete(handle as unknown as number)
}
advance(ms: number) {
this.now += ms
let ran = true
while (ran) {
ran = false
for (const [id, interval] of [...this.intervals.entries()].sort((a, b) => a[1].nextRunAt - b[1].nextRunAt)) {
if (interval.nextRunAt <= this.now) {
interval.callback()
interval.nextRunAt += interval.delay
this.intervals.set(id, interval)
ran = true
}
}
}
}
}
describe("EphemeralMessageFlushScheduler", () => {
function createHandler(): MessageStateHandler {
return new MessageStateHandler({
taskId: "test-task-id",
ulid: "test-ulid",
taskState: new TaskState(),
updateTaskHistory: async () => [],
})
}
it("periodically flushes pending ephemeral message changes", async () => {
const timer = new FakeIntervalController()
const handler = createHandler()
const scheduler = new EphemeralMessageFlushScheduler({
flush: async () => handler.flushClineMessagesAndUpdateHistory(),
getDelayMs: () => 1500,
setIntervalFn: timer.setInterval as typeof setInterval,
clearIntervalFn: timer.clearInterval as typeof clearInterval,
})
await handler.addToClineMessagesEphemeral({
ts: Date.now(),
type: "say",
say: "text",
text: "streaming",
partial: true,
})
assert.equal(handler.consumeLatencyMetrics().persistenceFlushCount, 0)
scheduler.start()
timer.advance(1499)
await Promise.resolve()
assert.equal(handler.consumeLatencyMetrics().persistenceFlushCount, 0)
timer.advance(1)
await Promise.resolve()
await Promise.resolve()
assert.equal(handler.consumeLatencyMetrics().persistenceFlushCount, 1)
})
it("stops scheduling future flushes after stop is called", async () => {
const timer = new FakeIntervalController()
let flushCount = 0
const scheduler = new EphemeralMessageFlushScheduler({
flush: async () => {
flushCount += 1
},
getDelayMs: () => 100,
setIntervalFn: timer.setInterval as typeof setInterval,
clearIntervalFn: timer.clearInterval as typeof clearInterval,
})
scheduler.start()
timer.advance(100)
await Promise.resolve()
assert.equal(flushCount, 1)
scheduler.stop()
timer.advance(500)
await Promise.resolve()
assert.equal(flushCount, 1)
})
})
@@ -0,0 +1,122 @@
import { strict as assert } from "assert"
import { RequestBoundaryCache } from "../RequestBoundaryCache"
describe("RequestBoundaryCache", () => {
it("reuses cached values within the TTL and refreshes after expiry", async () => {
let now = 0
let loads = 0
const cache = new RequestBoundaryCache({
load: async () => {
loads += 1
return `value-${loads}`
},
ttlMs: 50,
getNow: () => now,
})
assert.equal(await cache.get(), "value-1")
assert.equal(await cache.get(), "value-1")
assert.equal(loads, 1)
now = 49
assert.equal(await cache.get(), "value-1")
assert.equal(loads, 1)
now = 50
assert.equal(await cache.get(), "value-2")
assert.equal(loads, 2)
})
it("shares in-flight loads across callers", async () => {
let loads = 0
let resolveLoad: ((value: string) => void) | undefined
const cache = new RequestBoundaryCache({
load: () => {
loads += 1
return new Promise<string>((resolve) => {
resolveLoad = resolve
})
},
ttlMs: 50,
})
const first = cache.get()
const second = cache.get()
assert.equal(loads, 1)
resolveLoad?.("shared-value")
assert.equal(await first, "shared-value")
assert.equal(await second, "shared-value")
})
it("can be cleared manually", async () => {
let loads = 0
const cache = new RequestBoundaryCache({
load: async () => {
loads += 1
return `value-${loads}`
},
ttlMs: 500,
})
assert.equal(await cache.get(), "value-1")
cache.clear()
assert.equal(await cache.get(), "value-2")
})
it("retries after a failed load instead of caching the failure", async () => {
let loads = 0
const cache = new RequestBoundaryCache({
load: async () => {
loads += 1
if (loads === 1) {
throw new Error("temporary failure")
}
return `value-${loads}`
},
ttlMs: 100,
})
await assert.rejects(cache.get(), /temporary failure/)
assert.equal(loads, 1)
assert.equal(await cache.get(), "value-2")
assert.equal(loads, 2)
assert.equal(await cache.get(), "value-2")
assert.equal(loads, 2)
})
it("uses the current TTL provider value for future cache refreshes", async () => {
let now = 0
let ttlMs = 50
let loads = 0
const cache = new RequestBoundaryCache({
load: async () => {
loads += 1
return `value-${loads}`
},
getTtlMs: () => ttlMs,
getNow: () => now,
})
assert.equal(await cache.get(), "value-1")
assert.equal(loads, 1)
now = 49
assert.equal(await cache.get(), "value-1")
assert.equal(loads, 1)
ttlMs = 200
now = 50
assert.equal(await cache.get(), "value-2")
assert.equal(loads, 2)
now = 249
assert.equal(await cache.get(), "value-2")
assert.equal(loads, 2)
now = 250
assert.equal(await cache.get(), "value-3")
assert.equal(loads, 3)
})
})
@@ -0,0 +1,200 @@
import { strict as assert } from "assert"
import { TaskPresentationScheduler } from "../TaskPresentationScheduler"
class FakeTimerController {
private now = 0
private nextId = 1
private timers = new Map<number, { time: number; callback: () => void }>()
setTimeout = (callback: () => void, delay: number) => {
const id = this.nextId++
this.timers.set(id, { time: this.now + delay, callback })
return id as unknown as ReturnType<typeof setTimeout>
}
clearTimeout = (handle: ReturnType<typeof setTimeout>) => {
this.timers.delete(handle as unknown as number)
}
advance(ms: number) {
this.now += ms
let ran = true
while (ran) {
ran = false
for (const [id, timer] of [...this.timers.entries()].sort((a, b) => a[1].time - b[1].time)) {
if (timer.time <= this.now) {
this.timers.delete(id)
timer.callback()
ran = true
}
}
}
}
getNow = () => this.now
}
describe("TaskPresentationScheduler", () => {
it("coalesces multiple requests within the cadence window into one flush", async () => {
const timer = new FakeTimerController()
let flushCount = 0
const scheduler = new TaskPresentationScheduler({
flush: async () => {
flushCount++
},
getDelayMs: () => 50,
setTimeoutFn: timer.setTimeout as typeof setTimeout,
clearTimeoutFn: timer.clearTimeout as typeof clearTimeout,
getNow: timer.getNow,
})
scheduler.requestFlush("normal")
scheduler.requestFlush("normal")
scheduler.requestFlush("low")
timer.advance(49)
assert.equal(flushCount, 0)
timer.advance(1)
await Promise.resolve()
assert.equal(flushCount, 1)
})
it("immediate flush preempts scheduled normal work", async () => {
const timer = new FakeTimerController()
let flushCount = 0
const scheduler = new TaskPresentationScheduler({
flush: async () => {
flushCount++
},
getDelayMs: () => 50,
setTimeoutFn: timer.setTimeout as typeof setTimeout,
clearTimeoutFn: timer.clearTimeout as typeof clearTimeout,
getNow: timer.getNow,
})
scheduler.requestFlush("normal")
await scheduler.flushNow()
assert.equal(flushCount, 1)
timer.advance(100)
await Promise.resolve()
assert.equal(flushCount, 1)
})
it("runs one follow-up flush when new work arrives during an active flush", async () => {
const timer = new FakeTimerController()
let flushCount = 0
let resolveFlush: (() => void) | undefined
const scheduler = new TaskPresentationScheduler({
flush: async () => {
flushCount++
await new Promise<void>((resolve) => {
resolveFlush = resolve
})
},
getDelayMs: () => 10,
setTimeoutFn: timer.setTimeout as typeof setTimeout,
clearTimeoutFn: timer.clearTimeout as typeof clearTimeout,
getNow: timer.getNow,
})
scheduler.requestFlush("normal")
timer.advance(10)
await Promise.resolve()
assert.equal(flushCount, 1)
scheduler.requestFlush("normal")
resolveFlush?.()
await Promise.resolve()
await Promise.resolve()
timer.advance(10)
await Promise.resolve()
assert.equal(flushCount, 2)
})
it("disposes pending work", async () => {
const timer = new FakeTimerController()
let flushCount = 0
const scheduler = new TaskPresentationScheduler({
flush: async () => {
flushCount++
},
getDelayMs: () => 50,
setTimeoutFn: timer.setTimeout as typeof setTimeout,
clearTimeoutFn: timer.clearTimeout as typeof clearTimeout,
getNow: timer.getNow,
})
scheduler.requestFlush("normal")
await scheduler.dispose()
timer.advance(100)
await Promise.resolve()
assert.equal(flushCount, 0)
})
it("does not schedule a follow-up flush after disposal during an active flush", async () => {
const timer = new FakeTimerController()
let flushCount = 0
let resolveFlush: (() => void) | undefined
const scheduler = new TaskPresentationScheduler({
flush: async () => {
flushCount++
await new Promise<void>((resolve) => {
resolveFlush = resolve
})
},
getDelayMs: () => 10,
setTimeoutFn: timer.setTimeout as typeof setTimeout,
clearTimeoutFn: timer.clearTimeout as typeof clearTimeout,
getNow: timer.getNow,
})
scheduler.requestFlush("normal")
timer.advance(10)
await Promise.resolve()
assert.equal(flushCount, 1)
scheduler.requestFlush("normal")
await scheduler.dispose()
resolveFlush?.()
await Promise.resolve()
await Promise.resolve()
timer.advance(20)
await Promise.resolve()
assert.equal(flushCount, 1)
})
it("flushNow drains pending updates immediately after the current flush completes", async () => {
const timer = new FakeTimerController()
let flushCount = 0
let resolveFlush: (() => void) | undefined
const scheduler = new TaskPresentationScheduler({
flush: async () => {
flushCount++
await new Promise<void>((resolve) => {
resolveFlush = resolve
})
},
getDelayMs: () => 25,
setTimeoutFn: timer.setTimeout as typeof setTimeout,
clearTimeoutFn: timer.clearTimeout as typeof clearTimeout,
getNow: timer.getNow,
})
scheduler.requestFlush("normal")
timer.advance(25)
await Promise.resolve()
assert.equal(flushCount, 1)
scheduler.requestFlush("normal")
const drainPromise = scheduler.flushNow()
resolveFlush?.()
await Promise.resolve()
await Promise.resolve()
assert.equal(flushCount, 2)
resolveFlush?.()
await drainPromise
timer.advance(50)
await Promise.resolve()
assert.equal(flushCount, 2)
})
})
@@ -0,0 +1,103 @@
import { strict as assert } from "assert"
import { TaskUsageUpdateScheduler } from "../TaskUsageUpdateScheduler"
class FakeTimerController {
private now = 0
private nextId = 1
private timers = new Map<number, { time: number; callback: () => void }>()
setTimeout = (callback: () => void, delay: number) => {
const id = this.nextId++
this.timers.set(id, { time: this.now + delay, callback })
return id as unknown as ReturnType<typeof setTimeout>
}
clearTimeout = (handle: ReturnType<typeof setTimeout>) => {
this.timers.delete(handle as unknown as number)
}
advance(ms: number) {
this.now += ms
let ran = true
while (ran) {
ran = false
for (const [id, timer] of [...this.timers.entries()].sort((a, b) => a[1].time - b[1].time)) {
if (timer.time <= this.now) {
this.timers.delete(id)
timer.callback()
ran = true
}
}
}
}
}
describe("TaskUsageUpdateScheduler", () => {
it("coalesces many usage chunks into fewer UI flushes than chunk count", async () => {
const flushMicrotasks = async (iterations = 20) => {
for (let i = 0; i < iterations; i++) {
await Promise.resolve()
}
}
const timer = new FakeTimerController()
let sideEffectCount = 0
let uiFlushCount = 0
const scheduler = new TaskUsageUpdateScheduler({
getDelayMs: () => 50,
setTimeoutFn: timer.setTimeout as typeof setTimeout,
clearTimeoutFn: timer.clearTimeout as typeof clearTimeout,
})
for (let i = 0; i < 5; i++) {
scheduler.enqueue({
sideEffect: async () => {
sideEffectCount += 1
},
flushUi: async () => {
uiFlushCount += 1
},
})
}
await flushMicrotasks()
assert.equal(uiFlushCount, 0)
assert.equal(sideEffectCount, 5)
timer.advance(50)
await flushMicrotasks()
assert.equal(sideEffectCount, 5)
assert.equal(uiFlushCount, 1)
assert.ok(uiFlushCount < sideEffectCount)
})
it("flushFinal immediately emits the final UI update after queued side effects complete", async () => {
const timer = new FakeTimerController()
const events: string[] = []
const scheduler = new TaskUsageUpdateScheduler({
getDelayMs: () => 100,
setTimeoutFn: timer.setTimeout as typeof setTimeout,
clearTimeoutFn: timer.clearTimeout as typeof clearTimeout,
})
scheduler.enqueue({
sideEffect: async () => {
events.push("side-effect")
},
flushUi: async () => {
events.push("debounced-ui")
},
})
await scheduler.flushFinal(async () => {
events.push("final-ui")
})
assert.deepStrictEqual(events, ["side-effect", "final-ui"])
timer.advance(200)
await Promise.resolve()
assert.deepStrictEqual(events, ["side-effect", "final-ui"])
})
})
+135
View File
@@ -0,0 +1,135 @@
import { strict as assert } from "assert"
import {
getEnvironmentDetailsStaticCacheTtlMs,
getPresentationCadenceMs,
getRequestBoundaryCacheTtlMs,
getStateUpdateCadenceMs,
getUsageUpdateCadenceMs,
isEphemeralMessagePersistenceDisabled,
isPresentationSchedulingDisabled,
isRemoteWorkspaceEnvironment,
isTaskUiDeltaSyncDisabled,
shouldWaitForTerminalCooldown,
summarizeChunkToWebviewDelays,
} from "../latency"
describe("task latency helpers", () => {
afterEach(() => {
delete process.env.CLINE_PRESENTATION_CADENCE_MS
delete process.env.CLINE_REMOTE_PRESENTATION_CADENCE_MS
delete process.env.CLINE_STATE_UPDATE_CADENCE_MS
delete process.env.CLINE_REMOTE_STATE_UPDATE_CADENCE_MS
delete process.env.CLINE_USAGE_UPDATE_CADENCE_MS
delete process.env.CLINE_REMOTE_USAGE_UPDATE_CADENCE_MS
delete process.env.CLINE_REQUEST_BOUNDARY_CACHE_TTL_MS
delete process.env.CLINE_REMOTE_REQUEST_BOUNDARY_CACHE_TTL_MS
delete process.env.CLINE_ENVIRONMENT_DETAILS_STATIC_CACHE_TTL_MS
delete process.env.CLINE_REMOTE_ENVIRONMENT_DETAILS_STATIC_CACHE_TTL_MS
delete process.env.CLINE_DISABLE_PRESENTATION_SCHEDULER
delete process.env.CLINE_DISABLE_EPHEMERAL_MESSAGE_PERSISTENCE
delete process.env.CLINE_DISABLE_TASK_UI_DELTA_SYNC
})
it("detects remote workspaces from remoteName, platform, and version metadata", () => {
assert.equal(isRemoteWorkspaceEnvironment({ remoteName: "ssh-remote" }), true)
assert.equal(isRemoteWorkspaceEnvironment({ platform: "VS Code Remote" }), true)
assert.equal(isRemoteWorkspaceEnvironment({ version: "Remote Server 1.0" }), true)
assert.equal(isRemoteWorkspaceEnvironment({ platform: "darwin", version: "1.0.0", remoteName: null }), false)
})
it("uses remote-aware presentation and state update cadences", () => {
assert.equal(getPresentationCadenceMs(false, "immediate"), 0)
assert.equal(getPresentationCadenceMs(false, "normal"), 40)
assert.equal(getPresentationCadenceMs(true, "normal"), 90)
assert.equal(getPresentationCadenceMs(true, "low"), 125)
assert.equal(getStateUpdateCadenceMs(false, "immediate"), 0)
assert.equal(getStateUpdateCadenceMs(false, "normal"), 16)
assert.equal(getStateUpdateCadenceMs(true, "normal"), 110)
assert.equal(getStateUpdateCadenceMs(true, "low"), 150)
assert.equal(getUsageUpdateCadenceMs(false), 250)
assert.equal(getUsageUpdateCadenceMs(true), 400)
assert.equal(getRequestBoundaryCacheTtlMs(false), 500)
assert.equal(getRequestBoundaryCacheTtlMs(true), 1000)
assert.equal(getEnvironmentDetailsStaticCacheTtlMs(false), 30_000)
assert.equal(getEnvironmentDetailsStaticCacheTtlMs(true), 60_000)
})
it("respects cadence overrides from environment variables", () => {
process.env.CLINE_PRESENTATION_CADENCE_MS = "22"
process.env.CLINE_REMOTE_PRESENTATION_CADENCE_MS = "77"
process.env.CLINE_STATE_UPDATE_CADENCE_MS = "18"
process.env.CLINE_REMOTE_STATE_UPDATE_CADENCE_MS = "99"
process.env.CLINE_USAGE_UPDATE_CADENCE_MS = "333"
process.env.CLINE_REMOTE_USAGE_UPDATE_CADENCE_MS = "555"
process.env.CLINE_REQUEST_BOUNDARY_CACHE_TTL_MS = "444"
process.env.CLINE_REMOTE_REQUEST_BOUNDARY_CACHE_TTL_MS = "888"
process.env.CLINE_ENVIRONMENT_DETAILS_STATIC_CACHE_TTL_MS = "1234"
process.env.CLINE_REMOTE_ENVIRONMENT_DETAILS_STATIC_CACHE_TTL_MS = "5678"
assert.equal(getPresentationCadenceMs(false, "normal"), 22)
assert.equal(getPresentationCadenceMs(true, "normal"), 77)
assert.equal(getStateUpdateCadenceMs(false, "normal"), 18)
assert.equal(getStateUpdateCadenceMs(true, "normal"), 99)
assert.equal(getUsageUpdateCadenceMs(false), 333)
assert.equal(getUsageUpdateCadenceMs(true), 555)
assert.equal(getRequestBoundaryCacheTtlMs(false), 444)
assert.equal(getRequestBoundaryCacheTtlMs(true), 888)
assert.equal(getEnvironmentDetailsStaticCacheTtlMs(false), 1234)
assert.equal(getEnvironmentDetailsStaticCacheTtlMs(true), 5678)
})
it("supports development flags for disabling schedulers and delta sync", () => {
process.env.CLINE_DISABLE_PRESENTATION_SCHEDULER = "true"
process.env.CLINE_DISABLE_EPHEMERAL_MESSAGE_PERSISTENCE = "1"
process.env.CLINE_DISABLE_TASK_UI_DELTA_SYNC = "yes"
assert.equal(isPresentationSchedulingDisabled(), true)
assert.equal(isEphemeralMessagePersistenceDisabled(), true)
assert.equal(isTaskUiDeltaSyncDisabled(), true)
})
it("waits for terminal cooldown only when there is active heat or a recent edit", () => {
assert.equal(
shouldWaitForTerminalCooldown({
busyTerminalIds: [],
isProcessHot: () => true,
didEditFile: false,
}),
false,
)
assert.equal(
shouldWaitForTerminalCooldown({
busyTerminalIds: [1, 2],
isProcessHot: () => false,
didEditFile: false,
}),
false,
)
assert.equal(
shouldWaitForTerminalCooldown({
busyTerminalIds: [1, 2],
isProcessHot: (terminalId) => terminalId === 2,
didEditFile: false,
}),
true,
)
assert.equal(
shouldWaitForTerminalCooldown({
busyTerminalIds: [1],
isProcessHot: () => false,
didEditFile: true,
}),
true,
)
})
it("summarizes chunk-to-webview delays with median and p95 percentiles", () => {
assert.deepStrictEqual(summarizeChunkToWebviewDelays([]), { medianMs: 0, p95Ms: 0 })
assert.deepStrictEqual(summarizeChunkToWebviewDelays([10, 20, 30, 40, 50]), { medianMs: 30, p95Ms: 50 })
assert.deepStrictEqual(summarizeChunkToWebviewDelays([5, 15, 25, 35]), { medianMs: 15, p95Ms: 35 })
})
})
+89
View File
@@ -0,0 +1,89 @@
import type { ApiHandler } from "@core/api"
import { strict as assert } from "assert"
import type { ClineApiReqInfo, ClineMessage } from "@/shared/ExtensionMessage"
import { MessageStateHandler } from "../message-state"
import { TaskState } from "../TaskState"
import { updateApiReqMsg } from "../utils"
describe("task utils", () => {
function createMessageStateHandler() {
return new MessageStateHandler({
taskId: "task-utils-test",
ulid: "task-utils-ulid",
taskState: new TaskState(),
updateTaskHistory: async () => [],
})
}
it("finalizes api request metrics with explicit total cost", async () => {
const handler = createMessageStateHandler()
const initialInfo: ClineApiReqInfo = {
request: "analyze files",
retryStatus: {
attempt: 2,
maxAttempts: 3,
delaySec: 4,
},
}
await handler.addToClineMessages({
ts: 1,
type: "say",
say: "api_req_started",
text: JSON.stringify(initialInfo),
} as ClineMessage)
await updateApiReqMsg({
messageStateHandler: handler,
lastApiReqIndex: 0,
inputTokens: 120,
outputTokens: 80,
cacheWriteTokens: 10,
cacheReadTokens: 5,
totalCost: 0.42,
api: {
getModel: () => ({ info: {} }),
} as ApiHandler,
})
const updatedInfo = JSON.parse(handler.getClineMessages()[0].text || "{}") as ClineApiReqInfo
assert.equal(updatedInfo.tokensIn, 120)
assert.equal(updatedInfo.tokensOut, 80)
assert.equal(updatedInfo.cacheWrites, 10)
assert.equal(updatedInfo.cacheReads, 5)
assert.equal(updatedInfo.cost, 0.42)
assert.equal(updatedInfo.retryStatus, undefined)
})
it("preserves request metadata while recording cancel and streaming failure details", async () => {
const handler = createMessageStateHandler()
await handler.addToClineMessages({
ts: 2,
type: "say",
say: "api_req_started",
text: JSON.stringify({ request: "retry request", model: "claude" }),
} as ClineMessage)
await updateApiReqMsg({
messageStateHandler: handler,
lastApiReqIndex: 0,
inputTokens: 50,
outputTokens: 10,
cacheWriteTokens: 0,
cacheReadTokens: 0,
api: {
getModel: () => ({ info: {} }),
} as ApiHandler,
cancelReason: "streaming_failed",
streamingFailedMessage: "network timeout",
totalCost: 0.1,
})
const updatedInfo = JSON.parse(handler.getClineMessages()[0].text || "{}") as ClineApiReqInfo
assert.equal(updatedInfo.request, "retry request")
assert.equal((updatedInfo as ClineApiReqInfo & { model?: string }).model, "claude")
assert.equal(updatedInfo.cancelReason, "streaming_failed")
assert.equal(updatedInfo.streamingFailedMessage, "network timeout")
assert.equal(updatedInfo.cost, 0.1)
})
})
+37 -34
View File
@@ -24,6 +24,7 @@ export interface FocusChainDependencies {
mode: Mode
stateManager: StateManager
postStateToWebview: () => Promise<void>
postTaskMetadataDelta: (metadata: { currentFocusChainChecklist?: string | null }) => Promise<void>
say: (type: ClineSay, text?: string, images?: string[], files?: string[], partial?: boolean) => Promise<number | undefined>
focusChainSettings: FocusChainSettings
}
@@ -33,6 +34,7 @@ export class FocusChainManager {
private taskState: TaskState
private stateManager: StateManager
private postStateToWebview: () => Promise<void>
private postTaskMetadataDelta: (metadata: { currentFocusChainChecklist?: string | null }) => Promise<void>
private say: (
type: ClineSay,
text?: string,
@@ -50,6 +52,7 @@ export class FocusChainManager {
this.taskState = dependencies.taskState
this.stateManager = dependencies.stateManager
this.postStateToWebview = dependencies.postStateToWebview
this.postTaskMetadataDelta = dependencies.postTaskMetadataDelta
this.say = dependencies.say
this.focusChainSettings = dependencies.focusChainSettings
}
@@ -85,7 +88,7 @@ export class FocusChainManager {
})
.on("unlink", async () => {
this.taskState.currentFocusChainChecklist = null
await this.postStateToWebview()
await this.postTaskMetadataDelta({ currentFocusChainChecklist: null })
})
.on("error", (error) => {
Logger.error(`[Task ${this.taskId}] Failed to watch focus chain file:`, error)
@@ -120,7 +123,7 @@ export class FocusChainManager {
this.taskState.currentFocusChainChecklist = markdownTodoList
this.taskState.todoListWasUpdatedByUser = true
await this.postStateToWebview()
await this.postTaskMetadataDelta({ currentFocusChainChecklist: markdownTodoList })
telemetryService.captureFocusChainListWritten(this.taskId)
} else {
Logger.log(
@@ -165,28 +168,28 @@ export class FocusChainManager {
`
// If there are no user changes, proceed with reminders based on list progress
} else {
let progressBasedMessageStub = ""
// If there are items on the list, but none have been completed yet, remind the model to update the list when appropriate
if (completedItems === 0 && totalItems > 0) {
progressBasedMessageStub =
"\n\n**Note:** No items are marked complete yet. As you work through the task, remember to mark items as complete when finished."
} else if (percentComplete >= 25 && percentComplete < 50) {
progressBasedMessageStub = `\n\n**Note:** ${percentComplete}% of items are complete.`
} else if (percentComplete >= 50 && percentComplete < 75) {
progressBasedMessageStub = `\n\n**Note:** ${percentComplete}% of items are complete. Proceed with the task.`
} else if (percentComplete >= 75) {
progressBasedMessageStub = `\n\n**Note:** ${percentComplete}% of items are complete! Focus on finishing the remaining items.`
}
// Every item on the list has been completed. Hooray!
else if (completedItems === totalItems && totalItems > 0) {
progressBasedMessageStub = FocusChainPrompts.completed
.replace("{{totalItems}}", totalItems.toString())
.replace("{{currentFocusChainChecklist}}", this.taskState.currentFocusChainChecklist)
}
}
let progressBasedMessageStub = ""
// If there are items on the list, but none have been completed yet, remind the model to update the list when appropriate
if (completedItems === 0 && totalItems > 0) {
progressBasedMessageStub =
"\n\n**Note:** No items are marked complete yet. As you work through the task, remember to mark items as complete when finished."
} else if (percentComplete >= 25 && percentComplete < 50) {
progressBasedMessageStub = `\n\n**Note:** ${percentComplete}% of items are complete.`
} else if (percentComplete >= 50 && percentComplete < 75) {
progressBasedMessageStub = `\n\n**Note:** ${percentComplete}% of items are complete. Proceed with the task.`
} else if (percentComplete >= 75) {
progressBasedMessageStub = `\n\n**Note:** ${percentComplete}% of items are complete! Focus on finishing the remaining items.`
}
// Every item on the list has been completed. Hooray!
else if (completedItems === totalItems && totalItems > 0) {
progressBasedMessageStub = FocusChainPrompts.completed
.replace("{{totalItems}}", totalItems.toString())
.replace("{{currentFocusChainChecklist}}", this.taskState.currentFocusChainChecklist)
}
// Return with progress-based stub
return `\n
// Return with progress-based stub
return `\n
${introUpdateRequired}\n
${listCurrentProgress}\n
${this.taskState.currentFocusChainChecklist}\n
@@ -194,25 +197,22 @@ export class FocusChainManager {
${FocusChainPrompts.reminder}\n
${progressBasedMessageStub}\n
`
}
}
// When switching from Plan to Act, request that a new list be generated
else if (this.taskState.didRespondToPlanAskBySwitchingMode) {
if (this.taskState.didRespondToPlanAskBySwitchingMode) {
return `${FocusChainPrompts.initial}`
}
// When in plan mode, lists are optional. TODO - May want to improve this soft prompt approach in a future version
else if (this.stateManager.getGlobalSettingsKey("mode") === "plan") {
if (this.stateManager.getGlobalSettingsKey("mode") === "plan") {
return FocusChainPrompts.planModeReminder
} else {
// Check if we're early in the task
const isEarlyInTask = this.taskState.apiRequestCount < 10
if (isEarlyInTask) {
return FocusChainPrompts.recommended
} else {
return FocusChainPrompts.apiRequestCount.replace("{{apiRequestCount}}", this.taskState.apiRequestCount.toString())
}
}
// Check if we're early in the task
const isEarlyInTask = this.taskState.apiRequestCount < 10
if (isEarlyInTask) {
return FocusChainPrompts.recommended
}
return FocusChainPrompts.apiRequestCount.replace("{{apiRequestCount}}", this.taskState.apiRequestCount.toString())
}
/**
@@ -301,12 +301,14 @@ export class FocusChainManager {
// Write the model's update to the markdown file
try {
await this.writeFocusChainToDisk(taskProgress.trim())
await this.postTaskMetadataDelta({ currentFocusChainChecklist: taskProgress.trim() })
// Send the task_progress message to the UI immediately
await this.say("task_progress", taskProgress.trim())
} catch (error) {
Logger.error(`[Task ${this.taskId}] focus chain list: Failed to write to markdown file:`, error)
// Fall back to creating a task_progress message directly if file write fails
await this.postTaskMetadataDelta({ currentFocusChainChecklist: taskProgress.trim() })
await this.say("task_progress", taskProgress.trim())
Logger.log(`[Task ${this.taskId}] focus chain list: Sent fallback task_progress message to UI`)
}
@@ -316,6 +318,7 @@ export class FocusChainManager {
if (markdownTodoList) {
const _previousList = this.taskState.currentFocusChainChecklist
this.taskState.currentFocusChainChecklist = markdownTodoList
await this.postTaskMetadataDelta({ currentFocusChainChecklist: markdownTodoList })
// Create a task_progress message to display the focus chain list in the UI
await this.say("task_progress", markdownTodoList)
+362 -82
View File
@@ -113,11 +113,27 @@ import { refreshWorkflowToggles } from "../context/instructions/user-instruction
import { Controller } from "../controller"
import { executeHook } from "../hooks/hook-executor"
import { StateManager } from "../storage/StateManager"
import { EphemeralMessageFlushScheduler } from "./EphemeralMessageFlushScheduler"
import { FocusChainManager } from "./focus-chain"
import {
getEnvironmentDetailsStaticCacheTtlMs,
getPresentationCadenceMs,
getRequestBoundaryCacheTtlMs,
getUsageUpdateCadenceMs,
isEphemeralMessagePersistenceDisabled,
isPresentationSchedulingDisabled,
isRemoteWorkspaceEnvironment,
shouldWaitForTerminalCooldown,
summarizeChunkToWebviewDelays,
type TaskLatencyTrigger,
} from "./latency"
import { MessageStateHandler } from "./message-state"
import { RequestBoundaryCache } from "./RequestBoundaryCache"
import { StreamChunkCoordinator } from "./StreamChunkCoordinator"
import { StreamResponseHandler } from "./StreamResponseHandler"
import { type PresentationPriority, TaskPresentationScheduler } from "./TaskPresentationScheduler"
import { TaskState } from "./TaskState"
import { TaskUsageUpdateScheduler } from "./TaskUsageUpdateScheduler"
import { ToolExecutor } from "./ToolExecutor"
import { detectAvailableCliTools, extractProviderDomainFromUrl, updateApiReqMsg } from "./utils"
import { buildUserFeedbackContent } from "./utils/buildUserFeedbackContent"
@@ -256,6 +272,20 @@ export class Task {
// Command executor for running shell commands (extracted from executeCommandTool)
private commandExecutor!: CommandExecutor
private isRemoteWorkspaceEnvironment = false
private readonly presentationScheduler: TaskPresentationScheduler
private readonly ephemeralMessageFlushScheduler: EphemeralMessageFlushScheduler
private readonly schedulerDebugLoggingEnabled = process.env.CLINE_DEBUG_LATENCY === "1"
private requestLatencyMetrics = this.createRequestLatencyMetrics()
private usageUpdateTimer: ReturnType<typeof setTimeout> | undefined
private readonly presentationSchedulingDisabled = isPresentationSchedulingDisabled()
private readonly ephemeralMessagePersistenceDisabled = isEphemeralMessagePersistenceDisabled()
private readonly openTabsCache: RequestBoundaryCache<string[]>
private readonly visibleTabsCache: RequestBoundaryCache<string[]>
private readonly existingOpenTabsCache: RequestBoundaryCache<string[]>
private readonly existingVisibleTabsCache: RequestBoundaryCache<string[]>
private readonly workspaceConfigCache: RequestBoundaryCache<string | null>
private readonly availableCliToolsCache: RequestBoundaryCache<string[]>
constructor(params: TaskParams) {
const {
@@ -283,6 +313,15 @@ export class Task {
this.taskInitializationStartTime = performance.now()
this.taskState = new TaskState()
void HostProvider.env
.getHostVersion({})
.then((hostVersion) => {
this.isRemoteWorkspaceEnvironment = isRemoteWorkspaceEnvironment(hostVersion)
this.taskState.isRemoteWorkspace = this.isRemoteWorkspaceEnvironment
})
.catch((error) => {
Logger.debug(`[Task ${taskId}] Failed to detect remote workspace state: ${error}`)
})
this.controller = controller
this.mcpHub = mcpHub
this.updateTaskHistory = updateTaskHistory
@@ -368,6 +407,7 @@ export class Task {
mode: this.stateManager.getGlobalSettingsKey("mode"),
stateManager: this.stateManager,
postStateToWebview: this.postStateToWebview,
postTaskMetadataDelta: (metadata) => this.controller.postTaskMetadataDelta(metadata, this.taskId),
say: this.say.bind(this),
focusChainSettings: focusChainSettings,
})
@@ -450,11 +490,6 @@ export class Task {
await this.messageStateHandler.updateClineMessage(lastApiReqStartedIndex, {
text: JSON.stringify(currentApiReqInfo),
})
// Post the updated state to the webview so the UI reflects the retry attempt
await this.postStateToWebview().catch((e) =>
Logger.error("Error posting state to webview in onRetryAttempt:", e),
)
} catch (e) {
Logger.error(`[Task ${this.taskId}] Error updating api_req_started with retryStatus:`, e)
}
@@ -531,6 +566,57 @@ export class Task {
this.commandExecutor = new CommandExecutor(commandExecutorConfig, commandExecutorCallbacks)
this.presentationScheduler = new TaskPresentationScheduler({
flush: async () => this.flushAssistantPresentation(),
getDelayMs: (priority) => getPresentationCadenceMs(this.isRemoteWorkspaceEnvironment, priority),
onFlushError: (error) => Logger.debug(`[Task] Failed scheduled presentation flush: ${error}`),
metrics: {
onFlushStarted: (priority) => {
if (this.schedulerDebugLoggingEnabled) {
Logger.debug(`[Task ${this.taskId}] presentation flush started (${priority})`)
}
},
onFlushCompleted: (durationMs, priority) => {
this.requestLatencyMetrics.presentationDurationMs += durationMs
this.requestLatencyMetrics.presentationTrigger = priority
if (this.schedulerDebugLoggingEnabled) {
Logger.debug(`[Task ${this.taskId}] presentation flush completed (${priority}) in ${durationMs}ms`)
}
},
},
})
this.ephemeralMessageFlushScheduler = new EphemeralMessageFlushScheduler({
flush: async () => this.messageStateHandler.flushClineMessagesAndUpdateHistory(),
getDelayMs: () => 1500,
onFlushError: (error) => Logger.debug(`[Task ${this.taskId}] Failed to flush ephemeral message state: ${error}`),
})
this.openTabsCache = new RequestBoundaryCache({
load: async () => (await HostProvider.window.getOpenTabs({})).paths || [],
getTtlMs: () => getRequestBoundaryCacheTtlMs(this.isRemoteWorkspaceEnvironment),
})
this.visibleTabsCache = new RequestBoundaryCache({
load: async () => (await HostProvider.window.getVisibleTabs({})).paths || [],
getTtlMs: () => getRequestBoundaryCacheTtlMs(this.isRemoteWorkspaceEnvironment),
})
this.existingOpenTabsCache = new RequestBoundaryCache({
load: async () => filterExistingFiles(await this.getCachedOpenTabPaths()),
getTtlMs: () => getRequestBoundaryCacheTtlMs(this.isRemoteWorkspaceEnvironment),
})
this.existingVisibleTabsCache = new RequestBoundaryCache({
load: async () => filterExistingFiles(await this.getCachedVisibleTabPaths()),
getTtlMs: () => getRequestBoundaryCacheTtlMs(this.isRemoteWorkspaceEnvironment),
})
this.workspaceConfigCache = new RequestBoundaryCache({
load: async () => (await this.workspaceManager?.buildWorkspacesJson()) ?? null,
getTtlMs: () => getEnvironmentDetailsStaticCacheTtlMs(this.isRemoteWorkspaceEnvironment),
})
this.availableCliToolsCache = new RequestBoundaryCache({
load: async () => detectAvailableCliTools(),
getTtlMs: () => getEnvironmentDetailsStaticCacheTtlMs(this.isRemoteWorkspaceEnvironment),
})
this.toolExecutor = new ToolExecutor(
this.taskState,
this.messageStateHandler,
@@ -569,6 +655,151 @@ export class Task {
)
}
private createRequestLatencyMetrics() {
return {
presentationInvocationCount: 0,
presentationDurationMs: 0,
presentationTrigger: undefined as string | undefined,
statePostCount: 0,
statePostBuildDurationMs: 0,
statePostSerializedBytes: 0,
statePostSendDurationMs: 0,
partialMessageCount: 0,
partialMessagePayloadBytes: 0,
partialMessageBroadcastDurationMs: 0,
chunkToWebviewDelaysMs: [] as number[],
}
}
public noteStateUpdateMetrics(metrics: { buildDurationMs: number; serializedBytes: number; sendDurationMs: number }) {
this.requestLatencyMetrics.statePostCount += 1
this.requestLatencyMetrics.statePostBuildDurationMs += metrics.buildDurationMs
this.requestLatencyMetrics.statePostSerializedBytes += metrics.serializedBytes
this.requestLatencyMetrics.statePostSendDurationMs += metrics.sendDurationMs
if (this.taskState.currentChunkReceivedAtMs) {
this.requestLatencyMetrics.chunkToWebviewDelaysMs.push(
Math.max(0, performance.now() - this.taskState.currentChunkReceivedAtMs),
)
}
}
private notePartialMessageEvent(message: ClineMessage, stats: { payloadBytes: number; broadcastDurationMs: number }) {
this.requestLatencyMetrics.partialMessageCount += 1
this.requestLatencyMetrics.partialMessagePayloadBytes += stats.payloadBytes
this.requestLatencyMetrics.partialMessageBroadcastDurationMs += stats.broadcastDurationMs
if (this.taskState.currentChunkReceivedAtMs) {
this.requestLatencyMetrics.chunkToWebviewDelaysMs.push(
Math.max(0, performance.now() - this.taskState.currentChunkReceivedAtMs),
)
}
}
private async emitPartialMessage(message: ClineMessage) {
const protoMessage = convertClineMessageToProto(message)
const stats = await sendPartialMessageEvent(protoMessage)
this.notePartialMessageEvent(message, stats)
}
private scheduleAssistantPresentation(trigger: TaskLatencyTrigger, priority: PresentationPriority = "normal") {
this.requestLatencyMetrics.presentationInvocationCount += 1
this.requestLatencyMetrics.presentationTrigger = trigger
if (this.presentationSchedulingDisabled) {
void this.flushAssistantPresentation().catch((error) =>
Logger.debug(`[Task] Failed immediate presentation flush: ${error}`),
)
return
}
this.presentationScheduler.requestFlush(priority)
}
private async flushAssistantPresentation() {
await this.presentAssistantMessage()
}
private getPresentationPriorityForChunk(args: {
chunkType: "text" | "reasoning" | "tool_calls"
hadVisibleAssistantContent: boolean
}): PresentationPriority {
if (!args.hadVisibleAssistantContent) {
return "immediate"
}
if (args.chunkType === "tool_calls") {
return "immediate"
}
return "normal"
}
private startEphemeralFlushTimer() {
if (this.ephemeralMessagePersistenceDisabled) {
return
}
this.ephemeralMessageFlushScheduler.start()
}
private stopEphemeralFlushTimer() {
this.ephemeralMessageFlushScheduler.stop()
}
private clearUsageUpdateTimer() {
if (this.usageUpdateTimer) {
clearTimeout(this.usageUpdateTimer)
this.usageUpdateTimer = undefined
}
}
private async getCachedOpenTabPaths(): Promise<string[]> {
return this.openTabsCache.get()
}
private async getCachedVisibleTabPaths(): Promise<string[]> {
return this.visibleTabsCache.get()
}
private async getCachedExistingOpenTabPaths(): Promise<string[]> {
return this.existingOpenTabsCache.get()
}
private async getCachedExistingVisibleTabPaths(): Promise<string[]> {
return this.existingVisibleTabsCache.get()
}
private async getCachedWorkspaceConfigurationJson(): Promise<string | null> {
return this.workspaceConfigCache.get()
}
private async getCachedAvailableCliTools(): Promise<string[]> {
return this.availableCliToolsCache.get()
}
private captureRequestLatencyMetrics() {
const persistence = this.messageStateHandler.consumeLatencyMetrics()
const chunkDelays = summarizeChunkToWebviewDelays(this.requestLatencyMetrics.chunkToWebviewDelaysMs)
telemetryService.captureTaskLatencyMetrics({
ulid: this.ulid,
requestIndex: this.taskState.apiRequestCount,
isRemoteWorkspace: this.isRemoteWorkspaceEnvironment,
presentationInvocationCount: this.requestLatencyMetrics.presentationInvocationCount,
presentationDurationMs: this.requestLatencyMetrics.presentationDurationMs,
presentationTrigger: this.requestLatencyMetrics.presentationTrigger,
statePostCount: this.requestLatencyMetrics.statePostCount,
statePostBuildDurationMs: this.requestLatencyMetrics.statePostBuildDurationMs,
statePostSerializedBytes: this.requestLatencyMetrics.statePostSerializedBytes,
statePostSendDurationMs: this.requestLatencyMetrics.statePostSendDurationMs,
partialMessageCount: this.requestLatencyMetrics.partialMessageCount,
partialMessagePayloadBytes: this.requestLatencyMetrics.partialMessagePayloadBytes,
partialMessageBroadcastDurationMs: this.requestLatencyMetrics.partialMessageBroadcastDurationMs,
persistenceFlushCount: persistence.persistenceFlushCount,
persistenceSaveMessagesDurationMs: persistence.saveMessagesDurationMs,
persistenceSaveConversationDurationMs: persistence.saveConversationDurationMs,
persistenceUpdateHistoryDurationMs: persistence.updateHistoryDurationMs,
chunkToWebviewMedianMs: chunkDelays.medianMs,
chunkToWebviewP95Ms: chunkDelays.p95Ms,
})
}
// Communicate with webview
// partial has three valid states true (partial message), false (completion of partial message), undefined (individual complete message)
@@ -598,15 +829,19 @@ export class Task {
if (partial) {
if (isUpdatingPreviousPartial) {
// existing partial message, so update it
await this.messageStateHandler.updateClineMessage(lastMessageIndex, {
text,
partial,
})
await (this.ephemeralMessagePersistenceDisabled
? this.messageStateHandler.updateClineMessage(lastMessageIndex, {
text,
partial,
})
: this.messageStateHandler.updateClineMessageEphemeral(lastMessageIndex, {
text,
partial,
}))
// todo be more efficient about saving and posting only new data or one whole message at a time so ignore partial for saves, and only post parts of partial message instead of whole array in new listener
// await this.saveClineMessagesAndUpdateHistory()
// await this.postStateToWebview()
const protoMessage = convertClineMessageToProto(lastMessage)
await sendPartialMessageEvent(protoMessage)
await this.emitPartialMessage(lastMessage)
throw new Error("Current ask promise was ignored 1")
}
// this is a new partial message, so add it with partial state
@@ -615,13 +850,21 @@ export class Task {
// this.askResponseImages = undefined
askTs = Date.now()
this.taskState.lastMessageTs = askTs
await this.messageStateHandler.addToClineMessages({
ts: askTs,
type: "ask",
ask: type,
text,
partial,
})
await (this.ephemeralMessagePersistenceDisabled
? this.messageStateHandler.addToClineMessages({
ts: askTs,
type: "ask",
ask: type,
text,
partial,
})
: this.messageStateHandler.addToClineMessagesEphemeral({
ts: askTs,
type: "ask",
ask: type,
text,
partial,
}))
await this.postStateToWebview()
throw new Error("Current ask promise was ignored 2")
}
@@ -647,8 +890,7 @@ export class Task {
partial: false,
})
// await this.postStateToWebview()
const protoMessage = convertClineMessageToProto(lastMessage)
await sendPartialMessageEvent(protoMessage)
await this.emitPartialMessage(lastMessage)
} else {
// this is a new partial=false message, so add it like normal
this.taskState.askResponse = undefined
@@ -779,30 +1021,47 @@ export class Task {
if (isUpdatingPreviousPartial) {
// existing partial message, so update it
const lastIndex = this.messageStateHandler.getClineMessages().length - 1
await this.messageStateHandler.updateClineMessage(lastIndex, {
text,
images,
files,
partial,
})
await (this.ephemeralMessagePersistenceDisabled
? this.messageStateHandler.updateClineMessage(lastIndex, {
text,
images,
files,
partial,
})
: this.messageStateHandler.updateClineMessageEphemeral(lastIndex, {
text,
images,
files,
partial,
}))
const protoMessage = convertClineMessageToProto(lastMessage)
await sendPartialMessageEvent(protoMessage)
await this.emitPartialMessage(lastMessage)
return undefined
}
// this is a new partial message, so add it with partial state
const sayTs = Date.now()
this.taskState.lastMessageTs = sayTs
await this.messageStateHandler.addToClineMessages({
ts: sayTs,
type: "say",
say: type,
text,
images,
files,
partial,
modelInfo,
})
await (this.ephemeralMessagePersistenceDisabled
? this.messageStateHandler.addToClineMessages({
ts: sayTs,
type: "say",
say: type,
text,
images,
files,
partial,
modelInfo,
})
: this.messageStateHandler.addToClineMessagesEphemeral({
ts: sayTs,
type: "say",
say: type,
text,
images,
files,
partial,
modelInfo,
}))
await this.postStateToWebview()
return sayTs
}
@@ -820,8 +1079,7 @@ export class Task {
})
// await this.postStateToWebview()
const protoMessage = convertClineMessageToProto(lastMessage)
await sendPartialMessageEvent(protoMessage) // more performant than an entire postStateToWebview
await this.emitPartialMessage(lastMessage) // more performant than an entire postStateToWebview
return undefined
}
// this is a new partial=false message, so add it like normal
@@ -971,6 +1229,7 @@ export class Task {
this.taskState.didFinishAbortingStream = true
// Save BOTH files so Controller.cancelTask() can find the task
await this.messageStateHandler.saveClineMessagesAndUpdateHistory()
this.stopEphemeralFlushTimer()
await this.messageStateHandler.overwriteApiConversationHistory(this.messageStateHandler.getApiConversationHistory())
await this.postStateToWebview()
}
@@ -1873,8 +2132,8 @@ export class Task {
// Snapshot editor tabs so prompt tools can decide whether to include
// filetype-specific instructions (e.g. notebooks) without adding bespoke flags.
const openTabPaths = (await HostProvider.window.getOpenTabs({})).paths || []
const visibleTabPaths = (await HostProvider.window.getVisibleTabs({})).paths || []
const openTabPaths = await this.getCachedExistingOpenTabPaths()
const visibleTabPaths = await this.getCachedExistingVisibleTabPaths()
const cap = 50
const editorTabs = {
open: openTabPaths.slice(0, cap),
@@ -2274,6 +2533,9 @@ export class Task {
// Increment API request counter for focus chain list management
this.taskState.apiRequestCount++
this.taskState.apiRequestsSinceLastTodoUpdate++
this.requestLatencyMetrics = this.createRequestLatencyMetrics()
this.startEphemeralFlushTimer()
this.clearUsageUpdateTimer()
// Used to know what models were used in the task if user wants to export metadata for error reporting purposes
const { model, providerId, customPrompt, mode } = this.getCurrentProviderInfo()
@@ -2555,7 +2817,13 @@ export class Task {
request: userContent.map((block) => formatContentBlockToMarkdown(block)).join("\n\n"),
} satisfies ClineApiReqInfo),
})
await this.postStateToWebview()
const usageUpdateScheduler = new TaskUsageUpdateScheduler({
getDelayMs: () => getUsageUpdateCadenceMs(this.isRemoteWorkspaceEnvironment),
onSideEffectError: (error) =>
Logger.debug(`[Task ${this.taskId}] Failed to process usage chunk side effects: ${error}`),
onUiFlushError: (error) => Logger.debug(`[Task ${this.taskId}] Failed to flush usage UI updates: ${error}`),
})
try {
const taskMetrics: {
@@ -2566,13 +2834,6 @@ export class Task {
totalCost: number | undefined
} = { cacheWriteTokens: 0, cacheReadTokens: 0, inputTokens: 0, outputTokens: 0, totalCost: undefined }
let didFinalizeApiReqMsg = false
let usageChunkSideEffectsQueue = Promise.resolve()
/*
Usage side effects run as soon as a usage chunk arrives.
queueUsageChunkSideEffects() appends work to this promise chain, and each appended step starts immediately
(once the previous step finishes). We only await usageChunkSideEffectsQueue at stream end to flush any in-flight
updates before finalizing api_req_started, not to start processing.
*/
const updateApiReqMsgFromMetrics = async (
cancelReason?: ClineApiReqCancelReason,
@@ -2597,15 +2858,11 @@ export class Task {
usageOutputTokens: number,
chunkOptions?: { cacheWriteTokens?: number; cacheReadTokens?: number; totalCost?: number },
) => {
usageChunkSideEffectsQueue = usageChunkSideEffectsQueue
// This executes immediately after enqueue (microtask if already resolved), not at stream end.
.then(async () => {
usageUpdateScheduler.enqueue({
sideEffect: async () => {
if (didFinalizeApiReqMsg || this.taskState.abort) {
return
}
await updateApiReqMsgFromMetrics()
await this.postStateToWebview()
await telemetryService.captureTokenUsage(
this.ulid,
usageInputTokens,
@@ -2614,16 +2871,23 @@ export class Task {
model.id,
chunkOptions,
)
})
.catch((error) => {
Logger.debug(`[Task ${this.taskId}] Failed to process usage chunk side effects: ${error}`)
})
},
flushUi: async () => {
if (didFinalizeApiReqMsg || this.taskState.abort) {
return
}
this.usageUpdateTimer = undefined
await updateApiReqMsgFromMetrics()
},
})
}
const finalizeApiReqMsg = async (cancelReason?: ClineApiReqCancelReason, streamingFailedMessage?: string) => {
didFinalizeApiReqMsg = true
await usageChunkSideEffectsQueue
await updateApiReqMsgFromMetrics(cancelReason, streamingFailedMessage)
this.clearUsageUpdateTimer()
await usageUpdateScheduler.flushFinal(async () =>
updateApiReqMsgFromMetrics(cancelReason, streamingFailedMessage),
)
}
const abortStream = async (cancelReason: ClineApiReqCancelReason, streamingFailedMessage?: string) => {
@@ -2764,6 +3028,9 @@ export class Task {
if (!chunk) {
break
}
this.taskState.currentChunkReceivedAtMs = performance.now()
const hadVisibleAssistantContent =
assistantMessage.length > 0 || this.taskState.assistantMessageContent.length > 0
if (!this.taskState.taskFirstTokenTimeMs) {
this.taskState.taskFirstTokenTimeMs = Math.max(0, Date.now() - this.taskState.taskStartTimeMs)
}
@@ -2790,6 +3057,10 @@ export class Task {
await this.say("reasoning", thinkingBlock.thinking, undefined, undefined, true)
}
}
this.scheduleAssistantPresentation(
"reasoning",
this.getPresentationPriorityForChunk({ chunkType: "reasoning", hadVisibleAssistantContent }),
)
break
}
@@ -2812,6 +3083,10 @@ export class Task {
}
await this.processNativeToolCalls(assistantTextOnly, toolUseHandler.getPartialToolUsesAsContent())
this.scheduleAssistantPresentation(
"tool",
this.getPresentationPriorityForChunk({ chunkType: "tool_calls", hadVisibleAssistantContent }),
)
break
}
case "text": {
@@ -2839,16 +3114,14 @@ export class Task {
if (this.taskState.assistantMessageContent.length > prevLength) {
this.taskState.userMessageContentReady = false // new content we need to present, reset to false in case previous content set this to true
}
this.scheduleAssistantPresentation(
"text",
this.getPresentationPriorityForChunk({ chunkType: "text", hadVisibleAssistantContent }),
)
break
}
}
// Present content once per chunk. Calling this from multiple case branches can
// race partial updates and duplicate text rows in the chat.
await this.presentAssistantMessage().catch((error) =>
Logger.debug("[Task] Failed to present message: " + error),
)
if (this.taskState.abort) {
this.api.abort?.()
if (!this.taskState.abandoned) {
@@ -2883,8 +3156,8 @@ export class Task {
} else {
await streamCoordinator.waitForCompletion()
}
// Flush any usage updates that were already executing/queued during streaming.
await usageChunkSideEffectsQueue
// Flush any usage side effects that were already executing/queued during streaming.
await Promise.resolve()
if (!this.taskState.abort && !didFinalizeReasoningForUi) {
const finalReasoning = reasonsHandler.getCurrentReasoning()
@@ -3075,10 +3348,7 @@ export class Task {
// in case there are native tool calls pending
const partialToolBlocks = toolUseHandler.getPartialToolUsesAsContent()?.map((block) => ({ ...block, partial: false }))
await this.processNativeToolCalls(assistantTextOnly, partialToolBlocks)
if (partialBlocks.length > 0) {
await this.presentAssistantMessage() // if there is content to update then it will complete and update this.userMessageContentReady to true, which we pwaitfor before making the next request. all this is really doing is presenting the last partial message that we just set to complete
}
await this.presentationScheduler.flushNow() // final drain so any coalesced content is fully presented before the next request
// now add to apiconversationhistory
// need to save assistant responses to file before proceeding to tool use since user can exit at any moment and we wouldn't be able to save the assistant's response
@@ -3205,8 +3475,16 @@ export class Task {
return true
}
this.captureRequestLatencyMetrics()
this.stopEphemeralFlushTimer()
this.clearUsageUpdateTimer()
usageUpdateScheduler.dispose()
return didEndLoop // will always be false for now
} catch (_error) {
this.captureRequestLatencyMetrics()
this.stopEphemeralFlushTimer()
this.clearUsageUpdateTimer()
usageUpdateScheduler.dispose()
// this should never happen since the only thing that can throw an error is the attemptApiRequest, which is wrapped in a try catch that sends an ask where if noButtonClicked, will clear current task and destroy this instance. However to avoid unhandled promise rejection, we will end this loop which will end execution of this instance (see startTask)
return true // needs to be true so parent loop knows to end task
}
@@ -3452,8 +3730,7 @@ export class Task {
// It could be useful for cline to know if the user went from one or no file to another between messages, so we always include this context
details += `\n\n# ${host.platform} Visible Files`
const rawVisiblePaths = (await HostProvider.window.getVisibleTabs({})).paths
const filteredVisiblePaths = await filterExistingFiles(rawVisiblePaths)
const filteredVisiblePaths = await this.getCachedExistingVisibleTabPaths()
const visibleFilePaths = filteredVisiblePaths.map((absolutePath) => path.relative(this.cwd, absolutePath))
// Filter paths through clineIgnoreController
@@ -3469,8 +3746,7 @@ export class Task {
}
details += `\n\n# ${host.platform} Open Tabs`
const rawOpenTabPaths = (await HostProvider.window.getOpenTabs({})).paths
const filteredOpenTabPaths = await filterExistingFiles(rawOpenTabPaths)
const filteredOpenTabPaths = await this.getCachedExistingOpenTabPaths()
const openTabPaths = filteredOpenTabPaths.map((absolutePath) => path.relative(this.cwd, absolutePath))
// Filter paths through clineIgnoreController
@@ -3493,8 +3769,12 @@ export class Task {
// || this.didEditFile
await setTimeoutPromise(300) // delay after saving file to let terminals catch up
}
// let terminalWasBusy = false
if (busyTerminals.length > 0) {
const shouldWaitForCooldown = shouldWaitForTerminalCooldown({
busyTerminalIds: busyTerminals.map((terminal) => terminal.id),
isProcessHot: (terminalId) => this.terminalManager.isProcessHot(terminalId),
didEditFile: this.taskState.didEditFile,
})
if (shouldWaitForCooldown) {
// wait for terminals to cool down
// terminalWasBusy = allTerminals.some((t) => this.terminalManager.isProcessHot(t.id))
await pWaitFor(() => busyTerminals.every((t) => !this.terminalManager.isProcessHot(t.id)), {
@@ -3585,14 +3865,14 @@ export class Task {
// Add workspace information in JSON format
if (this.workspaceManager) {
const workspacesJson = await this.workspaceManager.buildWorkspacesJson()
const workspacesJson = await this.getCachedWorkspaceConfigurationJson()
if (workspacesJson) {
details += `\n\n# Workspace Configuration\n${workspacesJson}`
}
}
// Add detected CLI tools
const availableCliTools = await detectAvailableCliTools()
const availableCliTools = await this.getCachedAvailableCliTools()
if (availableCliTools.length > 0) {
details += `\n\n# Detected CLI Tools\nThese are some of the tools on the user's machine, and may be useful if needed to accomplish the task: ${availableCliTools.join(", ")}. This list is not exhaustive, and other tools may be available.`
}
+164
View File
@@ -0,0 +1,164 @@
import type { PresentationPriority } from "./TaskPresentationScheduler"
export type TaskLatencyTrigger = "text" | "reasoning" | "tool" | "finalization" | "other"
function readBooleanEnv(envVarName: string): boolean {
const rawValue = process.env[envVarName]?.toLowerCase()
return rawValue === "1" || rawValue === "true" || rawValue === "yes"
}
function readCadenceOverride(envVarName: string): number | undefined {
const rawValue = process.env[envVarName]
if (!rawValue) {
return undefined
}
const parsed = Number.parseInt(rawValue, 10)
if (!Number.isFinite(parsed) || parsed < 0) {
return undefined
}
return parsed
}
function getCadenceOverride(args: { isRemoteWorkspace: boolean; localEnvVar: string; remoteEnvVar: string }): number | undefined {
return args.isRemoteWorkspace ? readCadenceOverride(args.remoteEnvVar) : readCadenceOverride(args.localEnvVar)
}
export function isRemoteWorkspaceEnvironment(host: { platform?: string; version?: string; remoteName?: string | null }): boolean {
if (host.remoteName) {
return true
}
const platform = host.platform?.toLowerCase() ?? ""
const version = host.version?.toLowerCase() ?? ""
return platform.includes("remote") || version.includes("remote")
}
export function isPresentationSchedulingDisabled(): boolean {
return readBooleanEnv("CLINE_DISABLE_PRESENTATION_SCHEDULER")
}
export function isEphemeralMessagePersistenceDisabled(): boolean {
return readBooleanEnv("CLINE_DISABLE_EPHEMERAL_MESSAGE_PERSISTENCE")
}
export function isTaskUiDeltaSyncDisabled(): boolean {
return readBooleanEnv("CLINE_DISABLE_TASK_UI_DELTA_SYNC")
}
export function getPresentationCadenceMs(isRemoteWorkspace: boolean, priority: PresentationPriority): number {
if (priority === "immediate") {
return 0
}
const override = getCadenceOverride({
isRemoteWorkspace,
localEnvVar: priority === "low" ? "CLINE_PRESENTATION_LOW_CADENCE_MS" : "CLINE_PRESENTATION_CADENCE_MS",
remoteEnvVar: priority === "low" ? "CLINE_REMOTE_PRESENTATION_LOW_CADENCE_MS" : "CLINE_REMOTE_PRESENTATION_CADENCE_MS",
})
if (override !== undefined) {
return override
}
if (priority === "low") {
return isRemoteWorkspace ? 125 : 50
}
return isRemoteWorkspace ? 90 : 40
}
export function getStateUpdateCadenceMs(isRemoteWorkspace: boolean, priority: PresentationPriority): number {
if (priority === "immediate") {
return 0
}
const override = getCadenceOverride({
isRemoteWorkspace,
localEnvVar: priority === "low" ? "CLINE_STATE_UPDATE_LOW_CADENCE_MS" : "CLINE_STATE_UPDATE_CADENCE_MS",
remoteEnvVar: priority === "low" ? "CLINE_REMOTE_STATE_UPDATE_LOW_CADENCE_MS" : "CLINE_REMOTE_STATE_UPDATE_CADENCE_MS",
})
if (override !== undefined) {
return override
}
if (priority === "low") {
return isRemoteWorkspace ? 150 : 40
}
return isRemoteWorkspace ? 110 : 16
}
export function getUsageUpdateCadenceMs(isRemoteWorkspace: boolean): number {
const override = getCadenceOverride({
isRemoteWorkspace,
localEnvVar: "CLINE_USAGE_UPDATE_CADENCE_MS",
remoteEnvVar: "CLINE_REMOTE_USAGE_UPDATE_CADENCE_MS",
})
if (override !== undefined) {
return override
}
return isRemoteWorkspace ? 400 : 250
}
export function getRequestBoundaryCacheTtlMs(isRemoteWorkspace: boolean): number {
const override = getCadenceOverride({
isRemoteWorkspace,
localEnvVar: "CLINE_REQUEST_BOUNDARY_CACHE_TTL_MS",
remoteEnvVar: "CLINE_REMOTE_REQUEST_BOUNDARY_CACHE_TTL_MS",
})
if (override !== undefined) {
return override
}
return isRemoteWorkspace ? 1000 : 500
}
export function getEnvironmentDetailsStaticCacheTtlMs(isRemoteWorkspace: boolean): number {
const override = getCadenceOverride({
isRemoteWorkspace,
localEnvVar: "CLINE_ENVIRONMENT_DETAILS_STATIC_CACHE_TTL_MS",
remoteEnvVar: "CLINE_REMOTE_ENVIRONMENT_DETAILS_STATIC_CACHE_TTL_MS",
})
if (override !== undefined) {
return override
}
return isRemoteWorkspace ? 60_000 : 30_000
}
export function shouldWaitForTerminalCooldown(args: {
busyTerminalIds: number[]
isProcessHot: (terminalId: number) => boolean
didEditFile: boolean
}): boolean {
if (args.busyTerminalIds.length === 0) {
return false
}
if (args.didEditFile) {
return true
}
return args.busyTerminalIds.some((terminalId) => {
try {
return args.isProcessHot(terminalId)
} catch {
return false
}
})
}
export function summarizeChunkToWebviewDelays(delaysMs: number[]): { medianMs: number; p95Ms: number } {
if (delaysMs.length === 0) {
return { medianMs: 0, p95Ms: 0 }
}
const sorted = [...delaysMs].sort((a, b) => a - b)
const percentile = (ratio: number) => sorted[Math.min(sorted.length - 1, Math.max(0, Math.ceil(sorted.length * ratio) - 1))]
return {
medianMs: percentile(0.5),
p95Ms: percentile(0.95),
}
}
+120
View File
@@ -11,7 +11,9 @@ import { HistoryItem } from "@/shared/HistoryItem"
import { ClineStorageMessage } from "@/shared/messages/content"
import { Logger } from "@/shared/services/Logger"
import { getCwd, getDesktopDir } from "@/utils/path"
import { sendTaskUiDelta } from "../controller/ui/subscribeToTaskUiDeltas"
import { ensureTaskDirectoryExists, saveApiConversationHistory, saveClineMessages } from "../storage/disk"
import { isTaskUiDeltaSyncDisabled } from "./latency"
import { TaskState } from "./TaskState"
// Event types for clineMessages changes
@@ -48,12 +50,20 @@ interface MessageStateHandlerParams {
export class MessageStateHandler extends EventEmitter<MessageStateHandlerEvents> {
private apiConversationHistory: ClineStorageMessage[] = []
private clineMessages: ClineMessage[] = []
private hasDirtyEphemeralChanges = false
private taskIsFavorited: boolean
private checkpointTracker: CheckpointTracker | undefined
private updateTaskHistory: (historyItem: HistoryItem) => Promise<HistoryItem[]>
private taskId: string
private ulid: string
private taskState: TaskState
private readonly latencyMetrics = {
persistenceFlushCount: 0,
saveMessagesDurationMs: 0,
saveConversationDurationMs: 0,
updateHistoryDurationMs: 0,
}
private readonly taskUiDeltaSyncDisabled = isTaskUiDeltaSyncDisabled()
// Mutex to prevent concurrent state modifications (RC-4)
// Protects against data loss from race conditions when multiple
@@ -75,6 +85,50 @@ export class MessageStateHandler extends EventEmitter<MessageStateHandlerEvents>
*/
private emitClineMessagesChanged(change: ClineMessageChange): void {
this.emit("clineMessagesChanged", change)
if (!this.taskUiDeltaSyncDisabled) {
void this.emitTaskUiDeltaForChange(change)
}
}
private async emitTaskUiDeltaForChange(change: ClineMessageChange): Promise<void> {
const sequence = ++this.taskState.taskUiDeltaSequence
if (change.type === "add" && change.message) {
await sendTaskUiDelta({
type: "message_added",
taskId: this.taskId,
sequence,
message: change.message,
})
return
}
if (change.type === "update" && change.message) {
await sendTaskUiDelta({
type: "message_updated",
taskId: this.taskId,
sequence,
message: change.message,
})
return
}
if (change.type === "delete" && change.previousMessage) {
await sendTaskUiDelta({
type: "message_deleted",
taskId: this.taskId,
sequence,
messageTs: change.previousMessage.ts,
})
return
}
if (change.type === "set") {
await sendTaskUiDelta({
type: "task_state_resynced",
taskId: this.taskId,
sequence,
})
}
}
setCheckpointTracker(tracker: CheckpointTracker | undefined) {
@@ -105,6 +159,7 @@ export class MessageStateHandler extends EventEmitter<MessageStateHandlerEvents>
setClineMessages(newMessages: ClineMessage[]) {
const previousMessages = this.clineMessages
this.clineMessages = newMessages
this.hasDirtyEphemeralChanges = true
this.emitClineMessagesChanged({
type: "set",
messages: this.clineMessages,
@@ -119,7 +174,10 @@ export class MessageStateHandler extends EventEmitter<MessageStateHandlerEvents>
*/
private async saveClineMessagesAndUpdateHistoryInternal(): Promise<void> {
try {
this.latencyMetrics.persistenceFlushCount += 1
const saveMessagesStartedAt = performance.now()
await saveClineMessages(this.taskId, this.clineMessages)
this.latencyMetrics.saveMessagesDurationMs += Math.max(0, performance.now() - saveMessagesStartedAt)
// combined as they are in ChatView
const apiMetrics = getApiMetrics(combineApiRequests(combineCommandSequences(this.clineMessages.slice(1))))
@@ -142,6 +200,7 @@ export class MessageStateHandler extends EventEmitter<MessageStateHandlerEvents>
Logger.error("Failed to get task directory size:", taskDir, error)
}
const cwd = await getCwd(getDesktopDir())
const updateHistoryStartedAt = performance.now()
await this.updateTaskHistory({
id: this.taskId,
ulid: this.ulid,
@@ -160,6 +219,8 @@ export class MessageStateHandler extends EventEmitter<MessageStateHandlerEvents>
checkpointManagerErrorMessage: this.taskState.checkpointManagerErrorMessage,
modelId: lastModelInfo?.modelInfo?.modelId,
})
this.latencyMetrics.updateHistoryDurationMs += Math.max(0, performance.now() - updateHistoryStartedAt)
this.hasDirtyEphemeralChanges = false
} catch (error) {
Logger.error("Failed to save cline messages:", error)
}
@@ -179,7 +240,9 @@ export class MessageStateHandler extends EventEmitter<MessageStateHandlerEvents>
// Protect with mutex to prevent concurrent modifications from corrupting data (RC-4)
return await this.withStateLock(async () => {
this.apiConversationHistory.push(message)
const saveConversationStartedAt = performance.now()
await saveApiConversationHistory(this.taskId, this.apiConversationHistory)
this.latencyMetrics.saveConversationDurationMs += Math.max(0, performance.now() - saveConversationStartedAt)
})
}
@@ -187,7 +250,25 @@ export class MessageStateHandler extends EventEmitter<MessageStateHandlerEvents>
// Protect with mutex to prevent concurrent modifications from corrupting data (RC-4)
return await this.withStateLock(async () => {
this.apiConversationHistory = newHistory
const saveConversationStartedAt = performance.now()
await saveApiConversationHistory(this.taskId, this.apiConversationHistory)
this.latencyMetrics.saveConversationDurationMs += Math.max(0, performance.now() - saveConversationStartedAt)
})
}
async addToClineMessagesEphemeral(message: ClineMessage) {
return await this.withStateLock(async () => {
message.conversationHistoryIndex = this.apiConversationHistory.length - 1
message.conversationHistoryDeletedRange = this.taskState.conversationHistoryDeletedRange
const index = this.clineMessages.length
this.clineMessages.push(message)
this.hasDirtyEphemeralChanges = true
this.emitClineMessagesChanged({
type: "add",
messages: this.clineMessages,
index,
message,
})
})
}
@@ -223,6 +304,7 @@ export class MessageStateHandler extends EventEmitter<MessageStateHandlerEvents>
return await this.withStateLock(async () => {
const previousMessages = this.clineMessages
this.clineMessages = newMessages
this.hasDirtyEphemeralChanges = true
this.emitClineMessagesChanged({
type: "set",
messages: this.clineMessages,
@@ -261,6 +343,26 @@ export class MessageStateHandler extends EventEmitter<MessageStateHandlerEvents>
})
}
async updateClineMessageEphemeral(index: number, updates: Partial<ClineMessage>): Promise<void> {
return await this.withStateLock(async () => {
if (index < 0 || index >= this.clineMessages.length) {
throw new Error(`Invalid message index: ${index}`)
}
const previousMessage = { ...this.clineMessages[index] }
Object.assign(this.clineMessages[index], updates)
this.hasDirtyEphemeralChanges = true
this.emitClineMessagesChanged({
type: "update",
messages: this.clineMessages,
index,
previousMessage,
message: this.clineMessages[index],
})
})
}
/**
* Delete a specific message from the clineMessages array
* The entire operation (validate, delete, save) is atomic to prevent races (RC-4)
@@ -288,4 +390,22 @@ export class MessageStateHandler extends EventEmitter<MessageStateHandlerEvents>
await this.saveClineMessagesAndUpdateHistoryInternal()
})
}
async flushClineMessagesAndUpdateHistory(): Promise<void> {
return await this.withStateLock(async () => {
if (!this.hasDirtyEphemeralChanges) {
return
}
await this.saveClineMessagesAndUpdateHistoryInternal()
})
}
consumeLatencyMetrics() {
const snapshot = { ...this.latencyMetrics }
this.latencyMetrics.persistenceFlushCount = 0
this.latencyMetrics.saveMessagesDurationMs = 0
this.latencyMetrics.saveConversationDurationMs = 0
this.latencyMetrics.updateHistoryDurationMs = 0
return snapshot
}
}
+1
View File
@@ -8,6 +8,7 @@ export async function getHostVersion(_: EmptyRequest): Promise<GetHostVersionRes
return {
platform: vscode.env.appName,
version: vscode.version,
remoteName: vscode.env.remoteName,
clineType: ClineClient.VSCode,
clineVersion: ExtensionRegistryInfo.version,
}
@@ -4,11 +4,18 @@ import { afterEach, beforeEach, describe, it } from "mocha"
import * as os from "os"
import pWaitFor from "p-wait-for"
import * as path from "path"
import sinon from "sinon"
import * as vscode from "vscode"
import { getOpenTabs } from "@/hosts/vscode/hostbridge/window/getOpenTabs"
import {
getOpenTabs,
resetOpenTabsCacheForTests,
setOpenTabsCacheTtlForTests,
} from "@/hosts/vscode/hostbridge/window/getOpenTabs"
import { GetOpenTabsRequest } from "@/shared/proto/host/window"
describe("Hostbridge - Window - getOpenTabs", () => {
let sandbox: sinon.SinonSandbox
async function createAndOpenTestDocument(name: string, column: vscode.ViewColumn): Promise<void> {
const content = `// Test file ${name}\nconsole.log('Hello from file ${name}');`
@@ -45,11 +52,17 @@ describe("Hostbridge - Window - getOpenTabs", () => {
}
beforeEach(async () => {
sandbox = sinon.createSandbox()
resetOpenTabsCacheForTests()
setOpenTabsCacheTtlForTests(500)
// Clean up any existing editors and wait for cleanup to complete
await waitForAllTabsClosed()
})
afterEach(async () => {
sandbox.restore()
resetOpenTabsCacheForTests()
setOpenTabsCacheTtlForTests(500)
// Clean up test documents and editors
await waitForAllTabsClosed()
})
@@ -180,4 +193,48 @@ describe("Hostbridge - Window - getOpenTabs", () => {
console.error(error)
}
})
it("reuses cached open-tab results within the TTL and refreshes after expiry", async () => {
setOpenTabsCacheTtlForTests(20)
await createAndOpenTestDocument("cache-1", vscode.ViewColumn.One)
await pWaitFor(async () => (await getOpenTabs(GetOpenTabsRequest.create({}))).paths.length === 1, {
timeout: 5000,
interval: 50,
})
const firstResponse = await getOpenTabs(GetOpenTabsRequest.create({}))
assert.strictEqual(firstResponse.paths.length, 1)
await createAndOpenTestDocument("cache-2", vscode.ViewColumn.Two)
const cachedResponse = await getOpenTabs(GetOpenTabsRequest.create({}))
assert.strictEqual(cachedResponse.paths.length, 1, "Expected cached result before TTL expiry")
await new Promise((resolve) => setTimeout(resolve, 25))
const refreshedResponse = await getOpenTabs(GetOpenTabsRequest.create({}))
assert.strictEqual(refreshedResponse.paths.length, 2, "Expected refreshed result after TTL expiry")
})
it("returns fresh open tabs immediately after cache reset", async () => {
setOpenTabsCacheTtlForTests(1_000)
await createAndOpenTestDocument("reset-1", vscode.ViewColumn.One)
await pWaitFor(async () => (await getOpenTabs(GetOpenTabsRequest.create({}))).paths.length === 1, {
timeout: 5_000,
interval: 50,
})
const firstResponse = await getOpenTabs(GetOpenTabsRequest.create({}))
assert.strictEqual(firstResponse.paths.length, 1)
await createAndOpenTestDocument("reset-2", vscode.ViewColumn.Two)
const cachedResponse = await getOpenTabs(GetOpenTabsRequest.create({}))
assert.strictEqual(cachedResponse.paths.length, 1, "Expected stale cached result before reset")
resetOpenTabsCacheForTests()
await pWaitFor(async () => (await getOpenTabs(GetOpenTabsRequest.create({}))).paths.length === 2, {
timeout: 5_000,
interval: 50,
})
const refreshedResponse = await getOpenTabs(GetOpenTabsRequest.create({}))
assert.strictEqual(refreshedResponse.paths.length, 2, "Expected cache reset to expose fresh open tabs")
})
})
@@ -1,11 +1,24 @@
import { TabInputText, window } from "vscode"
import { GetOpenTabsRequest, GetOpenTabsResponse } from "@/shared/proto/host/window"
import { createCachedTabQuery } from "./tabQueryCache"
const openTabsQuery = createCachedTabQuery(
async () =>
window.tabGroups.all
.flatMap((group) => group.tabs)
.map((tab) => (tab.input as TabInputText)?.uri?.fsPath)
.filter((path): path is string => Boolean(path)),
(paths) => GetOpenTabsResponse.create({ paths }),
)
export async function getOpenTabs(_: GetOpenTabsRequest): Promise<GetOpenTabsResponse> {
const openTabPaths = window.tabGroups.all
.flatMap((group) => group.tabs)
.map((tab) => (tab.input as TabInputText)?.uri?.fsPath)
.filter(Boolean)
return openTabsQuery.read()
}
return GetOpenTabsResponse.create({ paths: openTabPaths ?? [] })
export function resetOpenTabsCacheForTests(): void {
openTabsQuery.reset()
}
export function setOpenTabsCacheTtlForTests(ttlMs: number): void {
openTabsQuery.setTtlForTests(ttlMs)
}
@@ -3,11 +3,18 @@ import * as fs from "fs/promises"
import { afterEach, beforeEach, describe, it } from "mocha"
import * as os from "os"
import * as path from "path"
import sinon from "sinon"
import * as vscode from "vscode"
import { getVisibleTabs } from "@/hosts/vscode/hostbridge/window/getVisibleTabs"
import {
getVisibleTabs,
resetVisibleTabsCacheForTests,
setVisibleTabsCacheTtlForTests,
} from "@/hosts/vscode/hostbridge/window/getVisibleTabs"
import { GetVisibleTabsRequest } from "@/shared/proto/host/window"
describe("Hostbridge - Window - getVisibleTabs", () => {
let sandbox: sinon.SinonSandbox
/**
* Helper function to create and open a test document in a specific column
*/
@@ -31,11 +38,17 @@ describe("Hostbridge - Window - getVisibleTabs", () => {
}
beforeEach(async () => {
sandbox = sinon.createSandbox()
resetVisibleTabsCacheForTests()
setVisibleTabsCacheTtlForTests(500)
// Clean up any existing editors
await vscode.commands.executeCommand("workbench.action.closeAllEditors")
})
afterEach(async () => {
sandbox.restore()
resetVisibleTabsCacheForTests()
setVisibleTabsCacheTtlForTests(500)
// Clean up test documents and editors
await vscode.commands.executeCommand("workbench.action.closeAllEditors")
})
@@ -188,4 +201,41 @@ describe("Hostbridge - Window - getVisibleTabs", () => {
// Clean up temp directory
await fs.rm(tempDir, { recursive: true }).catch(() => {})
})
it("reuses cached visible-tab results within the TTL and refreshes after expiry", async () => {
setVisibleTabsCacheTtlForTests(20)
await createAndOpenTestDocument(1, vscode.ViewColumn.One)
await new Promise((resolve) => setTimeout(resolve, 100))
const firstResponse = await getVisibleTabs(GetVisibleTabsRequest.create({}))
assert.strictEqual(firstResponse.paths.length, 1)
await createAndOpenTestDocument(2, vscode.ViewColumn.Two)
await new Promise((resolve) => setTimeout(resolve, 100))
const cachedResponse = await getVisibleTabs(GetVisibleTabsRequest.create({}))
assert.strictEqual(cachedResponse.paths.length, 1, "Expected cached result before TTL expiry")
await new Promise((resolve) => setTimeout(resolve, 25))
const refreshedResponse = await getVisibleTabs(GetVisibleTabsRequest.create({}))
assert.strictEqual(refreshedResponse.paths.length, 2, "Expected refreshed result after TTL expiry")
})
it("returns fresh visible tabs immediately after cache reset", async () => {
setVisibleTabsCacheTtlForTests(1_000)
await createAndOpenTestDocument(1, vscode.ViewColumn.One)
await new Promise((resolve) => setTimeout(resolve, 100))
const firstResponse = await getVisibleTabs(GetVisibleTabsRequest.create({}))
assert.strictEqual(firstResponse.paths.length, 1)
await createAndOpenTestDocument(2, vscode.ViewColumn.Two)
await new Promise((resolve) => setTimeout(resolve, 100))
const cachedResponse = await getVisibleTabs(GetVisibleTabsRequest.create({}))
assert.strictEqual(cachedResponse.paths.length, 1, "Expected stale cached result before reset")
resetVisibleTabsCacheForTests()
const refreshedResponse = await getVisibleTabs(GetVisibleTabsRequest.create({}))
assert.strictEqual(refreshedResponse.paths.length, 2, "Expected cache reset to expose fresh visible tabs")
})
})
@@ -1,8 +1,23 @@
import { window } from "vscode"
import { GetVisibleTabsRequest, GetVisibleTabsResponse } from "@/shared/proto/host/window"
import { createCachedTabQuery } from "./tabQueryCache"
const visibleTabsQuery = createCachedTabQuery(
async () =>
window.visibleTextEditors
?.map((editor) => editor.document?.uri?.fsPath)
.filter((path): path is string => Boolean(path)) ?? [],
(paths) => GetVisibleTabsResponse.create({ paths }),
)
export async function getVisibleTabs(_: GetVisibleTabsRequest): Promise<GetVisibleTabsResponse> {
const visibleTabPaths = window.visibleTextEditors?.map((editor) => editor.document?.uri?.fsPath).filter(Boolean)
return visibleTabsQuery.read()
}
return GetVisibleTabsResponse.create({ paths: visibleTabPaths ?? [] })
export function resetVisibleTabsCacheForTests(): void {
visibleTabsQuery.reset()
}
export function setVisibleTabsCacheTtlForTests(ttlMs: number): void {
visibleTabsQuery.setTtlForTests(ttlMs)
}
@@ -0,0 +1,48 @@
import { strict as assert } from "assert"
import { createCachedTabQuery } from "@/hosts/vscode/hostbridge/window/tabQueryCache"
describe("tabQueryCache", () => {
it("reuses cached values within the TTL and refreshes after expiry", async () => {
let now = 0
let calls = 0
const query = createCachedTabQuery(
async () => {
calls += 1
return [`value-${calls}`]
},
(paths) => paths,
{
ttlMs: 50,
getNow: () => now,
},
)
assert.deepStrictEqual(await query.read(), ["value-1"])
assert.deepStrictEqual(await query.read(), ["value-1"])
assert.equal(calls, 1)
now = 49
assert.deepStrictEqual(await query.read(), ["value-1"])
assert.equal(calls, 1)
now = 50
assert.deepStrictEqual(await query.read(), ["value-2"])
assert.equal(calls, 2)
})
it("can be reset manually", async () => {
let calls = 0
const query = createCachedTabQuery(
async () => {
calls += 1
return [`value-${calls}`]
},
(paths) => paths,
)
assert.deepStrictEqual(await query.read(), ["value-1"])
query.reset()
assert.deepStrictEqual(await query.read(), ["value-2"])
assert.equal(calls, 2)
})
})
@@ -0,0 +1,49 @@
type CachedTabQueryOptions = {
ttlMs?: number
getNow?: () => number
}
type CacheEntry = {
value: string[]
expiresAt: number
}
export type CachedTabQuery<TResponse> = {
read: () => Promise<TResponse>
reset: () => void
setTtlForTests: (ttlMs: number) => void
}
export function createCachedTabQuery<TResponse>(
query: () => Promise<string[]>,
buildResponse: (paths: string[]) => TResponse,
options?: CachedTabQueryOptions,
): CachedTabQuery<TResponse> {
let ttlMs = options?.ttlMs ?? 500
const getNow = options?.getNow ?? (() => Date.now())
let cache: CacheEntry | undefined
return {
read: async () => {
const now = getNow()
if (cache && cache.expiresAt > now) {
return buildResponse(cache.value)
}
const paths = await query()
cache = {
value: paths,
expiresAt: now + ttlMs,
}
return buildResponse(paths)
},
reset: () => {
cache = undefined
},
setTtlForTests: (nextTtlMs: number) => {
ttlMs = nextTtlMs
cache = undefined
},
}
}
+147
View File
@@ -164,7 +164,20 @@ export class TelemetryService {
API: {
TTFT_SECONDS: "cline.api.ttft.seconds",
DURATION_SECONDS: "cline.api.duration.seconds",
TASK_INITIALIZATION_SECONDS: "cline.api.task_initialization.seconds",
THROUGHPUT_TOKENS_PER_SECOND: "cline.api.throughput.tokens_per_second",
PRESENTATION_INVOCATIONS_PER_REQUEST: "cline.api.presentation.invocations.per_request",
PRESENTATION_DURATION_SECONDS: "cline.api.presentation.duration.seconds",
STATE_POSTS_PER_REQUEST: "cline.api.state_posts.per_request",
STATE_BUILD_DURATION_SECONDS: "cline.api.state_build.duration.seconds",
STATE_PAYLOAD_BYTES: "cline.api.state_payload.bytes",
STATE_SEND_DURATION_SECONDS: "cline.api.state_send.duration.seconds",
PARTIAL_MESSAGES_PER_REQUEST: "cline.api.partial_messages.per_request",
PARTIAL_MESSAGE_PAYLOAD_BYTES: "cline.api.partial_message_payload.bytes",
PARTIAL_MESSAGE_BROADCAST_DURATION_SECONDS: "cline.api.partial_message_broadcast.duration.seconds",
PERSISTENCE_FLUSHES_PER_REQUEST: "cline.api.persistence.flushes.per_request",
PERSISTENCE_DURATION_SECONDS: "cline.api.persistence.duration.seconds",
CHUNK_TO_WEBVIEW_SECONDS: "cline.api.chunk_to_webview.seconds",
},
HOOKS: {
EXECUTIONS_TOTAL: "cline.hooks.executions.total",
@@ -310,6 +323,7 @@ export class TelemetryService {
SUBAGENT_COMPLETED: "task.subagent_completed",
// Skills telemetry events
SKILL_USED: "task.skill_used",
LATENCY_METRICS: "task.latency_metrics",
},
// UI interaction events for tracking user engagement
UI: {
@@ -1645,6 +1659,14 @@ export class TelemetryService {
hasCheckpoints,
},
})
if (Number.isFinite(durationMs)) {
this.recordHistogram(TelemetryService.METRICS.API.TASK_INITIALIZATION_SECONDS, durationMs / 1000, {
ulid,
taskId,
hasCheckpoints,
})
}
}
/**
@@ -2369,6 +2391,131 @@ export class TelemetryService {
}
}
public captureTaskLatencyMetrics(args: {
ulid: string
requestIndex: number
isRemoteWorkspace: boolean
presentationInvocationCount?: number
presentationDurationMs?: number
presentationTrigger?: string
statePostCount?: number
statePostBuildDurationMs?: number
statePostSerializedBytes?: number
statePostSendDurationMs?: number
partialMessageCount?: number
partialMessagePayloadBytes?: number
partialMessageBroadcastDurationMs?: number
persistenceFlushCount?: number
persistenceSaveMessagesDurationMs?: number
persistenceSaveConversationDurationMs?: number
persistenceUpdateHistoryDurationMs?: number
chunkToWebviewMedianMs?: number
chunkToWebviewP95Ms?: number
}): void {
this.capture({
event: TelemetryService.EVENTS.TASK.LATENCY_METRICS,
properties: args,
})
const attrs = {
ulid: args.ulid,
request_index: args.requestIndex,
is_remote_workspace: args.isRemoteWorkspace,
presentation_trigger: args.presentationTrigger,
}
if (Number.isFinite(args.presentationDurationMs)) {
this.recordHistogram(
TelemetryService.METRICS.API.PRESENTATION_DURATION_SECONDS,
(args.presentationDurationMs ?? 0) / 1000,
attrs,
)
}
if (Number.isFinite(args.presentationInvocationCount)) {
this.recordHistogram(
TelemetryService.METRICS.API.PRESENTATION_INVOCATIONS_PER_REQUEST,
args.presentationInvocationCount ?? 0,
attrs,
)
}
if (Number.isFinite(args.statePostCount)) {
this.recordHistogram(TelemetryService.METRICS.API.STATE_POSTS_PER_REQUEST, args.statePostCount ?? 0, attrs)
}
if (Number.isFinite(args.statePostBuildDurationMs) && (args.statePostBuildDurationMs ?? 0) > 0) {
this.recordHistogram(
TelemetryService.METRICS.API.STATE_BUILD_DURATION_SECONDS,
(args.statePostBuildDurationMs ?? 0) / 1000,
attrs,
)
}
if (Number.isFinite(args.statePostSendDurationMs) && (args.statePostSendDurationMs ?? 0) > 0) {
this.recordHistogram(
TelemetryService.METRICS.API.STATE_SEND_DURATION_SECONDS,
(args.statePostSendDurationMs ?? 0) / 1000,
attrs,
)
}
if (Number.isFinite(args.statePostSerializedBytes) && (args.statePostSerializedBytes ?? 0) > 0) {
this.recordHistogram(TelemetryService.METRICS.API.STATE_PAYLOAD_BYTES, args.statePostSerializedBytes ?? 0, attrs)
}
if (Number.isFinite(args.partialMessageCount)) {
this.recordHistogram(TelemetryService.METRICS.API.PARTIAL_MESSAGES_PER_REQUEST, args.partialMessageCount ?? 0, attrs)
}
if (Number.isFinite(args.partialMessagePayloadBytes) && (args.partialMessagePayloadBytes ?? 0) > 0) {
this.recordHistogram(
TelemetryService.METRICS.API.PARTIAL_MESSAGE_PAYLOAD_BYTES,
args.partialMessagePayloadBytes ?? 0,
attrs,
)
}
if (Number.isFinite(args.partialMessageBroadcastDurationMs) && (args.partialMessageBroadcastDurationMs ?? 0) > 0) {
this.recordHistogram(
TelemetryService.METRICS.API.PARTIAL_MESSAGE_BROADCAST_DURATION_SECONDS,
(args.partialMessageBroadcastDurationMs ?? 0) / 1000,
attrs,
)
}
if (Number.isFinite(args.persistenceFlushCount)) {
this.recordHistogram(
TelemetryService.METRICS.API.PERSISTENCE_FLUSHES_PER_REQUEST,
args.persistenceFlushCount ?? 0,
attrs,
)
}
const totalPersistenceMs =
(args.persistenceSaveMessagesDurationMs ?? 0) +
(args.persistenceSaveConversationDurationMs ?? 0) +
(args.persistenceUpdateHistoryDurationMs ?? 0)
if (Number.isFinite(totalPersistenceMs) && totalPersistenceMs > 0) {
this.recordHistogram(TelemetryService.METRICS.API.PERSISTENCE_DURATION_SECONDS, totalPersistenceMs / 1000, attrs)
}
if (Number.isFinite(args.chunkToWebviewMedianMs)) {
this.recordHistogram(
TelemetryService.METRICS.API.CHUNK_TO_WEBVIEW_SECONDS,
(args.chunkToWebviewMedianMs ?? 0) / 1000,
{ ...attrs, percentile: "p50" },
)
}
if (Number.isFinite(args.chunkToWebviewP95Ms)) {
this.recordHistogram(TelemetryService.METRICS.API.CHUNK_TO_WEBVIEW_SECONDS, (args.chunkToWebviewP95Ms ?? 0) / 1000, {
...attrs,
percentile: "p95",
})
}
}
/**
* Safely executes a telemetry call with error protection.
*
@@ -1,6 +1,7 @@
import { ApiFormat } from "@shared/proto/cline/models"
import * as assert from "assert"
import type { ITelemetryProvider, TelemetryProperties, TelemetrySettings } from "../providers/ITelemetryProvider"
import { NoOpTelemetryProvider } from "../TelemetryProviderFactory"
import { TelemetryMetadata, TelemetryService } from "../TelemetryService"
class FakeProvider implements ITelemetryProvider {
@@ -332,6 +333,29 @@ describe("TelemetryService metrics", () => {
assert.strictEqual(durationMetric?.attributes.scope, "task")
})
it("captureTaskInitialization records initialization event and histogram", () => {
const provider = new FakeProvider()
const service = createTelemetryService(provider)
service.captureTaskInitialization("task-init", "task-123", 875, true)
const initEvent = provider.logs.find((entry) => entry.event === "task.initialization")
assert.ok(initEvent)
assert.strictEqual(initEvent?.properties?.ulid, "task-init")
assert.strictEqual(initEvent?.properties?.taskId, "task-123")
assert.strictEqual(initEvent?.properties?.durationMs, 875)
assert.strictEqual(initEvent?.properties?.hasCheckpoints, true)
const initMetric = provider.histograms.find(
(entry) => entry.name === TelemetryService.METRICS.API.TASK_INITIALIZATION_SECONDS,
)
assert.ok(initMetric)
assert.strictEqual(initMetric?.value, 0.875)
assert.strictEqual(initMetric?.attributes.ulid, "task-init")
assert.strictEqual(initMetric?.attributes.taskId, "task-123")
assert.strictEqual(initMetric?.attributes.hasCheckpoints, true)
})
it("captureGrpcResponseSize records histogram with correct name, value, and attributes", () => {
const provider = new FakeProvider()
const service = createTelemetryService(provider)
@@ -370,4 +394,145 @@ describe("TelemetryService metrics", () => {
assert.strictEqual(entry.attributes.extension_version, "test")
assert.strictEqual(entry.attributes.platform, "test-platform")
})
it("captureTaskLatencyMetrics records presentation, persistence, and chunk-to-webview histograms", () => {
const provider = new FakeProvider()
const service = createTelemetryService(provider)
service.captureTaskLatencyMetrics({
ulid: "task-latency",
requestIndex: 2,
isRemoteWorkspace: true,
presentationInvocationCount: 4,
presentationDurationMs: 120,
presentationTrigger: "text",
statePostCount: 3,
statePostBuildDurationMs: 25,
statePostSerializedBytes: 4096,
statePostSendDurationMs: 35,
partialMessageCount: 7,
partialMessagePayloadBytes: 2048,
partialMessageBroadcastDurationMs: 15,
persistenceFlushCount: 2,
persistenceSaveMessagesDurationMs: 40,
persistenceSaveConversationDurationMs: 10,
persistenceUpdateHistoryDurationMs: 20,
chunkToWebviewMedianMs: 80,
chunkToWebviewP95Ms: 150,
})
const latencyEvent = provider.logs.find((entry) => entry.event === "task.latency_metrics")
assert.ok(latencyEvent)
assert.strictEqual(latencyEvent?.properties?.ulid, "task-latency")
assert.strictEqual(latencyEvent?.properties?.isRemoteWorkspace, true)
const presentationMetric = provider.histograms.find(
(entry) => entry.name === TelemetryService.METRICS.API.PRESENTATION_DURATION_SECONDS,
)
assert.ok(presentationMetric)
assert.strictEqual(presentationMetric?.value, 0.12)
const presentationCountMetric = provider.histograms.find(
(entry) => entry.name === TelemetryService.METRICS.API.PRESENTATION_INVOCATIONS_PER_REQUEST,
)
assert.ok(presentationCountMetric)
assert.strictEqual(presentationCountMetric?.value, 4)
const statePostCountMetric = provider.histograms.find(
(entry) => entry.name === TelemetryService.METRICS.API.STATE_POSTS_PER_REQUEST,
)
assert.ok(statePostCountMetric)
assert.strictEqual(statePostCountMetric?.value, 3)
const persistenceMetric = provider.histograms.find(
(entry) => entry.name === TelemetryService.METRICS.API.PERSISTENCE_DURATION_SECONDS,
)
assert.ok(persistenceMetric)
assert.strictEqual(persistenceMetric?.value, 0.07)
const stateBuildMetric = provider.histograms.find(
(entry) => entry.name === TelemetryService.METRICS.API.STATE_BUILD_DURATION_SECONDS,
)
assert.ok(stateBuildMetric)
assert.strictEqual(stateBuildMetric?.value, 0.025)
const statePayloadMetric = provider.histograms.find(
(entry) => entry.name === TelemetryService.METRICS.API.STATE_PAYLOAD_BYTES,
)
assert.ok(statePayloadMetric)
assert.strictEqual(statePayloadMetric?.value, 4096)
const stateSendMetric = provider.histograms.find(
(entry) => entry.name === TelemetryService.METRICS.API.STATE_SEND_DURATION_SECONDS,
)
assert.ok(stateSendMetric)
assert.strictEqual(stateSendMetric?.value, 0.035)
const partialMessageCountMetric = provider.histograms.find(
(entry) => entry.name === TelemetryService.METRICS.API.PARTIAL_MESSAGES_PER_REQUEST,
)
assert.ok(partialMessageCountMetric)
assert.strictEqual(partialMessageCountMetric?.value, 7)
const partialMessagePayloadMetric = provider.histograms.find(
(entry) => entry.name === TelemetryService.METRICS.API.PARTIAL_MESSAGE_PAYLOAD_BYTES,
)
assert.ok(partialMessagePayloadMetric)
assert.strictEqual(partialMessagePayloadMetric?.value, 2048)
const partialMessageBroadcastMetric = provider.histograms.find(
(entry) => entry.name === TelemetryService.METRICS.API.PARTIAL_MESSAGE_BROADCAST_DURATION_SECONDS,
)
assert.ok(partialMessageBroadcastMetric)
assert.strictEqual(partialMessageBroadcastMetric?.value, 0.015)
const persistenceFlushCountMetric = provider.histograms.find(
(entry) => entry.name === TelemetryService.METRICS.API.PERSISTENCE_FLUSHES_PER_REQUEST,
)
assert.ok(persistenceFlushCountMetric)
assert.strictEqual(persistenceFlushCountMetric?.value, 2)
const chunkMetrics = provider.histograms.filter(
(entry) => entry.name === TelemetryService.METRICS.API.CHUNK_TO_WEBVIEW_SECONDS,
)
assert.strictEqual(chunkMetrics.length, 2)
assert.deepStrictEqual(chunkMetrics.map((entry) => entry.attributes.percentile).sort(), ["p50", "p95"])
})
it("latency instrumentation helpers remain non-throwing when telemetry providers are disabled", () => {
const service = new TelemetryService([new NoOpTelemetryProvider()], {
extension_version: "test",
cline_type: "cline-unit-tests",
platform: "test-platform",
platform_version: "1.0.0",
os_type: "darwin",
os_version: "24",
is_dev: "true",
} as TelemetryMetadata)
assert.doesNotThrow(() => {
service.captureGrpcResponseSize(2048, "cline.UiService", "subscribeToPartialMessage", "req-disabled")
service.captureTaskLatencyMetrics({
ulid: "task-disabled",
requestIndex: 1,
isRemoteWorkspace: false,
presentationInvocationCount: 2,
presentationDurationMs: 32,
presentationTrigger: "text",
statePostCount: 1,
statePostBuildDurationMs: 4,
statePostSerializedBytes: 512,
statePostSendDurationMs: 6,
partialMessageCount: 3,
partialMessagePayloadBytes: 256,
partialMessageBroadcastDurationMs: 2,
persistenceFlushCount: 1,
persistenceSaveMessagesDurationMs: 5,
persistenceSaveConversationDurationMs: 2,
persistenceUpdateHistoryDurationMs: 3,
chunkToWebviewMedianMs: 12,
chunkToWebviewP95Ms: 20,
})
})
})
})
@@ -0,0 +1,90 @@
import { strict as assert } from "assert"
import { compareTaskLatencySummaries, summarizeTaskLatencyEvents } from "../taskLatencySummary"
describe("taskLatencySummary", () => {
it("summarizes averages and ranges across latency events", () => {
const summary = summarizeTaskLatencyEvents([
{
ulid: "task-1",
requestIndex: 1,
presentationInvocationCount: 2,
partialMessageCount: 4,
statePostCount: 1,
statePostSerializedBytes: 100,
persistenceFlushCount: 1,
chunkToWebviewMedianMs: 20,
chunkToWebviewP95Ms: 35,
taskInitializationDurationMs: 600,
},
{
ulid: "task-1",
requestIndex: 2,
presentationInvocationCount: 4,
partialMessageCount: 6,
statePostCount: 3,
statePostSerializedBytes: 300,
persistenceFlushCount: 2,
chunkToWebviewMedianMs: 30,
chunkToWebviewP95Ms: 45,
taskInitializationDurationMs: 900,
},
])
assert.equal(summary.eventCount, 2)
assert.equal(summary.requestCount, 2)
assert.deepStrictEqual(summary.metrics.presentationInvocationCount, { average: 3, min: 2, max: 4 })
assert.deepStrictEqual(summary.metrics.statePostSerializedBytes, { average: 200, min: 100, max: 300 })
assert.deepStrictEqual(summary.metrics.chunkToWebviewP95Ms, { average: 40, min: 35, max: 45 })
assert.deepStrictEqual(summary.metrics.taskInitializationDurationMs, { average: 750, min: 600, max: 900 })
})
it("returns zeroed summaries when events are empty or metrics are absent", () => {
const summary = summarizeTaskLatencyEvents([{ ulid: "task-1", requestIndex: 1 }])
assert.equal(summary.eventCount, 1)
assert.equal(summary.requestCount, 1)
assert.deepStrictEqual(summary.metrics.partialMessageCount, { average: 0, min: 0, max: 0 })
const empty = summarizeTaskLatencyEvents([])
assert.equal(empty.eventCount, 0)
assert.equal(empty.requestCount, 0)
assert.deepStrictEqual(empty.metrics.statePostCount, { average: 0, min: 0, max: 0 })
})
it("compares latency summaries for before/after analysis", () => {
const baseline = summarizeTaskLatencyEvents([
{ ulid: "task", requestIndex: 1, presentationInvocationCount: 5, statePostCount: 4 },
])
const candidate = summarizeTaskLatencyEvents([
{ ulid: "task", requestIndex: 1, presentationInvocationCount: 3, statePostCount: 2 },
])
const comparison = compareTaskLatencySummaries(baseline, candidate)
assert.equal(comparison.baselineEvents, 1)
assert.equal(comparison.candidateEvents, 1)
assert.deepStrictEqual(comparison.metricDiffs.presentationInvocationCount, {
averageDelta: -2,
minDelta: -2,
maxDelta: -2,
})
assert.deepStrictEqual(comparison.metricDiffs.statePostCount, {
averageDelta: -2,
minDelta: -2,
maxDelta: -2,
})
})
it("normalizes task initialization events into latency summaries", () => {
const summary = summarizeTaskLatencyEvents([
{ event: "task.initialization", ulid: "task-1", taskId: "task-a", durationMs: 500 },
{ event: "task.initialization", ulid: "task-2", taskId: "task-b", durationMs: 700 },
])
assert.equal(summary.eventCount, 2)
assert.equal(summary.requestCount, 0)
assert.deepStrictEqual(summary.metrics.taskInitializationDurationMs, {
average: 600,
min: 500,
max: 700,
})
})
})
@@ -0,0 +1,122 @@
type TaskLatencyEvent = {
presentationInvocationCount?: number
partialMessageCount?: number
statePostCount?: number
statePostSerializedBytes?: number
persistenceFlushCount?: number
chunkToWebviewMedianMs?: number
chunkToWebviewP95Ms?: number
taskInitializationDurationMs?: number
durationMs?: number
requestIndex?: number
ulid?: string
taskId?: string
event?: string
[key: string]: unknown
}
type NumericMetricKey =
| "presentationInvocationCount"
| "partialMessageCount"
| "statePostCount"
| "statePostSerializedBytes"
| "persistenceFlushCount"
| "chunkToWebviewMedianMs"
| "chunkToWebviewP95Ms"
| "taskInitializationDurationMs"
export type TaskLatencySummary = {
eventCount: number
requestCount: number
metrics: Record<NumericMetricKey, { average: number; min: number; max: number }>
}
export type TaskLatencySummaryComparison = {
baselineEvents: number
candidateEvents: number
metricDiffs: Record<NumericMetricKey, { averageDelta: number; minDelta: number; maxDelta: number }>
}
const METRIC_KEYS: NumericMetricKey[] = [
"presentationInvocationCount",
"partialMessageCount",
"statePostCount",
"statePostSerializedBytes",
"persistenceFlushCount",
"chunkToWebviewMedianMs",
"chunkToWebviewP95Ms",
"taskInitializationDurationMs",
]
function normalizeEvent(event: TaskLatencyEvent): TaskLatencyEvent {
if (Number.isFinite(event.taskInitializationDurationMs)) {
return event
}
if (event.event === "task.initialization" && Number.isFinite(event.durationMs)) {
return {
...event,
taskInitializationDurationMs: event.durationMs as number,
}
}
return event
}
function summarizeMetric(events: TaskLatencyEvent[], key: NumericMetricKey) {
const values = events.map((event) => event[key]).filter((value): value is number => Number.isFinite(value))
if (values.length === 0) {
return { average: 0, min: 0, max: 0 }
}
const total = values.reduce((sum, value) => sum + value, 0)
return {
average: total / values.length,
min: Math.min(...values),
max: Math.max(...values),
}
}
export function summarizeTaskLatencyEvents(events: TaskLatencyEvent[]): TaskLatencySummary {
const normalizedEvents = events.map(normalizeEvent)
const requestKeys = new Set(
normalizedEvents
.map((event) => {
if (event.ulid && Number.isFinite(event.requestIndex)) {
return `${event.ulid}:${event.requestIndex}`
}
return undefined
})
.filter((value): value is string => Boolean(value)),
)
return {
eventCount: normalizedEvents.length,
requestCount: requestKeys.size,
metrics: Object.fromEntries(
METRIC_KEYS.map((key) => [key, summarizeMetric(normalizedEvents, key)]),
) as TaskLatencySummary["metrics"],
}
}
export function compareTaskLatencySummaries(
baseline: TaskLatencySummary,
candidate: TaskLatencySummary,
): TaskLatencySummaryComparison {
return {
baselineEvents: baseline.eventCount,
candidateEvents: candidate.eventCount,
metricDiffs: Object.fromEntries(
METRIC_KEYS.map((key) => [
key,
{
averageDelta: candidate.metrics[key].average - baseline.metrics[key].average,
minDelta: candidate.metrics[key].min - baseline.metrics[key].min,
maxDelta: candidate.metrics[key].max - baseline.metrics[key].max,
},
]),
) as TaskLatencySummaryComparison["metricDiffs"],
}
}
export type { TaskLatencyEvent }
+1
View File
@@ -47,6 +47,7 @@ export interface ExtensionState {
preferredLanguage?: string
mode: Mode
checkpointManagerErrorMessage?: string
isRemoteWorkspace?: boolean
clineMessages: ClineMessage[]
currentTaskItem?: HistoryItem
currentFocusChainChecklist?: string | null
+42
View File
@@ -0,0 +1,42 @@
import type { ClineMessage, ExtensionState } from "@shared/ExtensionMessage"
export type TaskUiMetadataDelta = Partial<
Pick<ExtensionState, "currentFocusChainChecklist" | "backgroundCommandRunning" | "backgroundCommandTaskId">
>
export type TaskUiDelta =
| {
type: "message_added"
taskId: string
sequence: number
message: ClineMessage
}
| {
type: "message_updated"
taskId: string
sequence: number
message: ClineMessage
}
| {
type: "message_deleted"
taskId: string
sequence: number
messageTs: number
}
| {
type: "task_state_resynced"
taskId: string
sequence: number
}
| {
type: "task_metadata_updated"
taskId: string
sequence: number
metadata: TaskUiMetadataDelta
}
export function isTaskUiDeltaMessageMutation(
delta: TaskUiDelta,
): delta is Extract<TaskUiDelta, { type: "message_added" | "message_updated" | "message_deleted" }> {
return delta.type === "message_added" || delta.type === "message_updated" || delta.type === "message_deleted"
}
@@ -0,0 +1,180 @@
import { strict as assert } from "assert"
import { StateUpdateScheduler } from "@/core/controller/StateUpdateScheduler"
class FakeTimerController {
private now = 0
private nextId = 1
private timers = new Map<number, { time: number; callback: () => void }>()
setTimeout = (callback: () => void, delay: number) => {
const id = this.nextId++
this.timers.set(id, { time: this.now + delay, callback })
return id as unknown as ReturnType<typeof setTimeout>
}
clearTimeout = (handle: ReturnType<typeof setTimeout>) => {
this.timers.delete(handle as unknown as number)
}
advance(ms: number) {
this.now += ms
let ran = true
while (ran) {
ran = false
for (const [id, timer] of [...this.timers.entries()].sort((a, b) => a[1].time - b[1].time)) {
if (timer.time <= this.now) {
this.timers.delete(id)
timer.callback()
ran = true
}
}
}
}
getNow = () => this.now
}
describe("StateUpdateScheduler", () => {
it("coalesces repeated requests into one flush", async () => {
const timer = new FakeTimerController()
let flushCount = 0
const scheduler = new StateUpdateScheduler({
flush: async () => {
flushCount++
},
getDelayMs: () => 50,
setTimeoutFn: timer.setTimeout as typeof setTimeout,
clearTimeoutFn: timer.clearTimeout as typeof clearTimeout,
getNow: timer.getNow,
})
scheduler.requestFlush("normal")
scheduler.requestFlush("normal")
scheduler.requestFlush("low")
timer.advance(49)
assert.equal(flushCount, 0)
timer.advance(1)
await Promise.resolve()
assert.equal(flushCount, 1)
})
it("runs one follow-up flush when new work arrives during a flush", async () => {
const timer = new FakeTimerController()
let flushCount = 0
let resolveFlush: (() => void) | undefined
const scheduler = new StateUpdateScheduler({
flush: async () => {
flushCount++
await new Promise<void>((resolve) => {
resolveFlush = resolve
})
},
getDelayMs: () => 10,
setTimeoutFn: timer.setTimeout as typeof setTimeout,
clearTimeoutFn: timer.clearTimeout as typeof clearTimeout,
getNow: timer.getNow,
})
scheduler.requestFlush("normal")
timer.advance(10)
await Promise.resolve()
assert.equal(flushCount, 1)
scheduler.requestFlush("normal")
resolveFlush?.()
await Promise.resolve()
await Promise.resolve()
timer.advance(10)
await Promise.resolve()
assert.equal(flushCount, 2)
})
it("immediate flush preempts scheduled normal work", async () => {
const timer = new FakeTimerController()
let flushCount = 0
const scheduler = new StateUpdateScheduler({
flush: async () => {
flushCount++
},
getDelayMs: () => 50,
setTimeoutFn: timer.setTimeout as typeof setTimeout,
clearTimeoutFn: timer.clearTimeout as typeof clearTimeout,
getNow: timer.getNow,
})
scheduler.requestFlush("normal")
await scheduler.flushNow()
assert.equal(flushCount, 1)
timer.advance(100)
await Promise.resolve()
assert.equal(flushCount, 1)
})
it("does not schedule a follow-up flush after disposal during an active flush", async () => {
const timer = new FakeTimerController()
let flushCount = 0
let resolveFlush: (() => void) | undefined
const scheduler = new StateUpdateScheduler({
flush: async () => {
flushCount++
await new Promise<void>((resolve) => {
resolveFlush = resolve
})
},
getDelayMs: () => 10,
setTimeoutFn: timer.setTimeout as typeof setTimeout,
clearTimeoutFn: timer.clearTimeout as typeof clearTimeout,
getNow: timer.getNow,
})
scheduler.requestFlush("normal")
timer.advance(10)
await Promise.resolve()
assert.equal(flushCount, 1)
scheduler.requestFlush("normal")
await scheduler.dispose()
resolveFlush?.()
await Promise.resolve()
await Promise.resolve()
timer.advance(20)
await Promise.resolve()
assert.equal(flushCount, 1)
})
it("flushNow drains pending updates immediately after the current flush completes", async () => {
const timer = new FakeTimerController()
let flushCount = 0
let resolveFlush: (() => void) | undefined
const scheduler = new StateUpdateScheduler({
flush: async () => {
flushCount++
await new Promise<void>((resolve) => {
resolveFlush = resolve
})
},
getDelayMs: () => 25,
setTimeoutFn: timer.setTimeout as typeof setTimeout,
clearTimeoutFn: timer.clearTimeout as typeof clearTimeout,
getNow: timer.getNow,
})
scheduler.requestFlush("normal")
timer.advance(25)
await Promise.resolve()
assert.equal(flushCount, 1)
scheduler.requestFlush("normal")
const drainPromise = scheduler.flushNow()
resolveFlush?.()
await Promise.resolve()
await Promise.resolve()
assert.equal(flushCount, 2)
resolveFlush?.()
await drainPromise
timer.advance(50)
await Promise.resolve()
assert.equal(flushCount, 2)
})
})
@@ -0,0 +1,190 @@
import { afterEach, before, beforeEach, describe, it } from "mocha"
import "should"
import { Controller } from "@core/controller"
import * as sinon from "sinon"
import { ClineEndpoint } from "@/config"
import { HostProvider } from "@/hosts/host-provider"
describe("Controller postStateToWebview", () => {
let controller: Controller
let stateManagerStub: sinon.SinonStub
let mockStateManager: any
let hostProviderInitialized = false
let mockGetHostVersion: sinon.SinonStub
before(async () => {
if (!ClineEndpoint.isInitialized()) {
await ClineEndpoint.initialize("/test/extension")
}
})
beforeEach(async () => {
if (!HostProvider.isInitialized()) {
mockGetHostVersion = sinon.stub().resolves({
clineVersion: "1.0.0",
platform: "darwin",
clineType: "vscode",
})
const mockHostBridge: any = {
workspaceClient: {},
envClient: {
getHostVersion: mockGetHostVersion,
},
windowClient: {},
diffClient: {},
}
HostProvider.initialize(
() => null as any,
() => null as any,
() => null as any,
() => null as any,
mockHostBridge,
() => {},
async (path: string) => `http://localhost${path}`,
async () => "",
"/test/extension",
"/test/storage",
)
hostProviderInitialized = true
}
await require("@/registry").HostRegistryInfo.init()
mockStateManager = {
getRemoteConfigSettings: sinon.stub().returns({}),
getApiConfiguration: sinon.stub().returns({}),
getGlobalStateKey: sinon.stub().returns(undefined),
getGlobalSettingsKey: sinon.stub().returns(undefined),
getWorkspaceStateKey: sinon.stub().returns(undefined),
setGlobalState: sinon.stub(),
setApiConfiguration: sinon.stub(),
registerCallbacks: sinon.stub(),
}
const StateManager = require("@core/storage/StateManager").StateManager
stateManagerStub = sinon.stub(StateManager, "get").returns(mockStateManager)
controller = new Controller({
globalState: { get: sinon.stub(), update: sinon.stub().resolves() },
workspaceState: { get: sinon.stub(), update: sinon.stub().resolves() },
secrets: { get: sinon.stub().resolves(), store: sinon.stub().resolves(), delete: sinon.stub().resolves() },
subscriptions: [],
extensionPath: "/test/path",
globalStoragePath: "/test/storage",
globalStorageUri: { fsPath: "/test/storage" },
} as any)
})
afterEach(() => {
stateManagerStub.restore()
if (hostProviderInitialized) {
HostProvider.reset()
hostProviderInitialized = false
}
})
it("flushes immediately when there is no active task", async () => {
const flushNow = sinon.stub().resolves()
const requestFlush = sinon.stub()
;(controller as any).stateUpdateScheduler = {
flushNow,
requestFlush,
dispose: sinon.stub().resolves(),
}
await controller.postStateToWebview()
sinon.assert.calledOnce(flushNow)
sinon.assert.notCalled(requestFlush)
})
it("coalesces through the scheduler when the active task is streaming", async () => {
const flushNow = sinon.stub().resolves()
const requestFlush = sinon.stub()
;(controller as any).stateUpdateScheduler = {
flushNow,
requestFlush,
dispose: sinon.stub().resolves(),
}
;(controller as any).task = {
taskState: {
isStreaming: true,
},
}
await controller.postStateToWebview()
sinon.assert.calledOnceWithExactly(requestFlush, "normal")
sinon.assert.notCalled(flushNow)
})
it("honors explicit immediate priority even while streaming", async () => {
const flushNow = sinon.stub().resolves()
const requestFlush = sinon.stub()
;(controller as any).stateUpdateScheduler = {
flushNow,
requestFlush,
dispose: sinon.stub().resolves(),
}
;(controller as any).task = {
taskState: {
isStreaming: true,
},
}
await controller.postStateToWebview({ priority: "immediate" })
sinon.assert.calledOnce(flushNow)
sinon.assert.notCalled(requestFlush)
})
it("detects remote workspace host metadata during controller initialization", async () => {
HostProvider.reset()
hostProviderInitialized = false
mockGetHostVersion = sinon.stub().resolves({
clineVersion: "1.0.0",
platform: "darwin",
clineType: "vscode",
remoteName: "ssh-remote",
})
HostProvider.initialize(
() => null as any,
() => null as any,
() => null as any,
() => null as any,
{
workspaceClient: {},
envClient: {
getHostVersion: mockGetHostVersion,
},
windowClient: {},
diffClient: {},
} as any,
() => {},
async (path: string) => `http://localhost${path}`,
async () => "",
"/test/extension",
"/test/storage",
)
hostProviderInitialized = true
controller = new Controller({
globalState: { get: sinon.stub(), update: sinon.stub().resolves() },
workspaceState: { get: sinon.stub(), update: sinon.stub().resolves() },
secrets: { get: sinon.stub().resolves(), store: sinon.stub().resolves(), delete: sinon.stub().resolves() },
subscriptions: [],
extensionPath: "/test/path",
globalStoragePath: "/test/storage",
globalStorageUri: { fsPath: "/test/storage" },
} as any)
await Promise.resolve()
await Promise.resolve()
sinon.assert.called(mockGetHostVersion)
;(controller as any).isRemoteWorkspaceEnvironment.should.equal(true)
})
})
+11 -2
View File
@@ -53,7 +53,7 @@ The user wants me to replace the name "john" with "cline" in the test.ts file. I
export const name = "john"
\`\`\`
I need to change "john" to "cline". This is a simple targeted edit, so I should use the replace_in_file tool rather than write_to_file since I\'m only changing one small part of the file.
I need to change "john" to "cline". This is a simple targeted edit, so I should use the replace_in_file tool rather than write_to_file since I'm only changing one small part of the file.
I need to:
1. Use replace_in_file to change "john" to "cline" in the test.ts file
@@ -61,7 +61,7 @@ I need to:
3. The REPLACE block should be: \`export const name = "cline"\`
</thinking>
I\'ll replace "john" with "cline" in the test.ts file.
I'll replace "john" with "cline" in the test.ts file.
<replace_in_file>
<path>test.ts</path>
@@ -74,8 +74,17 @@ export const name = "cline"
</diff>
</replace_in_file>`
const latency_validation = `I verified the task and can complete it without using any additional tools.
<attempt_completion>
<result>
Latency validation mock completed successfully.
</result>
</attempt_completion>`
export const E2E_MOCK_API_RESPONSES = {
DEFAULT: "Hello! I'm a mock Cline API response.",
LATENCY_VALIDATION: latency_validation,
REPLACE_REQUEST: replace_in_file,
EDIT_REQUEST: edit_request,
}
+26 -24
View File
@@ -377,6 +377,9 @@ export class ClineApiServerMock {
const parsed = JSON.parse(body)
const { _messages, model = "claude-3-5-sonnet-20241022", stream = true } = parsed
let responseText = E2E_MOCK_API_RESPONSES.DEFAULT
if (body.includes("latency_validation")) {
responseText = E2E_MOCK_API_RESPONSES.LATENCY_VALIDATION
}
if (body.includes("[replace_in_file for 'test.ts'] Result:")) {
responseText = E2E_MOCK_API_RESPONSES.REPLACE_REQUEST
}
@@ -454,31 +457,30 @@ export class ClineApiServerMock {
sendChunk()
return
} else {
const response = {
id: generationId,
object: "chat.completion",
created: Math.floor(Date.now() / 1000),
model,
choices: [
{
index: 0,
message: {
role: "assistant",
content: "Hello! I'm a mock Cline API response.",
},
finish_reason: "stop",
},
],
usage: {
prompt_tokens: 140,
completion_tokens: responseText.length,
total_tokens: 140 + responseText.length,
cost: (140 + responseText.length) * 0.00015,
},
}
return sendJson(response)
}
const response = {
id: generationId,
object: "chat.completion",
created: Math.floor(Date.now() / 1000),
model,
choices: [
{
index: 0,
message: {
role: "assistant",
content: "Hello! I'm a mock Cline API response.",
},
finish_reason: "stop",
},
],
usage: {
prompt_tokens: 140,
completion_tokens: responseText.length,
total_tokens: 140 + responseText.length,
cost: (140 + responseText.length) * 0.00015,
},
}
return sendJson(response)
}
// Generation details endpoint
+158
View File
@@ -1,9 +1,14 @@
import { describe, it } from "mocha"
import "should"
import fs from "fs/promises"
import os from "os"
import path from "path"
import should from "should"
import { getSavedApiConversationHistory, getSavedClineMessages } from "../core/storage/disk"
import { MessageStateHandler } from "../core/task/message-state"
import { TaskState } from "../core/task/TaskState"
import { ClineMessage } from "../shared/ExtensionMessage"
import { setVscodeHostProviderMock } from "./host-provider-test-utils"
/**
* Unit tests for MessageStateHandler's mutex protection (RC-4)
@@ -11,6 +16,15 @@ import { ClineMessage } from "../shared/ExtensionMessage"
* to prevent race conditions, particularly the TOCTOU bug in addToClineMessages
*/
describe("MessageStateHandler Mutex Protection", () => {
let tempGlobalStorageDir: string | undefined
afterEach(async () => {
if (tempGlobalStorageDir) {
await fs.rm(tempGlobalStorageDir, { recursive: true, force: true })
tempGlobalStorageDir = undefined
}
})
/**
* Helper to create a minimal MessageStateHandler for testing
*/
@@ -264,4 +278,148 @@ describe("MessageStateHandler Mutex Protection", () => {
finalHistory[0].content.should.equal("new1")
finalHistory[1].content.should.equal("new2")
})
it("should update messages ephemerally without persisting until flush", async () => {
const handler = createTestHandler()
const changes: Array<{ type: string; text?: string; previousText?: string }> = []
handler.on("clineMessagesChanged", (change) => {
changes.push({
type: change.type,
text: change.message?.text,
previousText: change.previousMessage?.text,
})
})
await handler.addToClineMessagesEphemeral(createTestMessage("ephemeral-start"))
await handler.updateClineMessageEphemeral(0, { text: "ephemeral-updated", partial: true })
const messagesBeforeFlush = handler.getClineMessages()
messagesBeforeFlush.length.should.equal(1)
should.exist(messagesBeforeFlush[0])
const pendingMessage = messagesBeforeFlush[0]!
pendingMessage.text?.should.equal("ephemeral-updated")
should.exist(pendingMessage.partial)
pendingMessage.partial!.should.equal(true)
changes.length.should.equal(2)
should.exist(changes[0])
should.exist(changes[1])
const firstChange = changes[0]!
const secondChange = changes[1]!
firstChange.type.should.equal("add")
should.exist(firstChange.text)
firstChange.text!.should.equal("ephemeral-start")
secondChange.type.should.equal("update")
should.exist(secondChange.previousText)
should.exist(secondChange.text)
secondChange.previousText!.should.equal("ephemeral-start")
secondChange.text!.should.equal("ephemeral-updated")
const metricsBeforeFlush = handler.consumeLatencyMetrics()
metricsBeforeFlush.persistenceFlushCount.should.equal(0)
await handler.flushClineMessagesAndUpdateHistory()
const metricsAfterFlush = handler.consumeLatencyMetrics()
metricsAfterFlush.persistenceFlushCount.should.equal(1)
})
it("should not flush when there are no dirty ephemeral changes", async () => {
const handler = createTestHandler()
await handler.flushClineMessagesAndUpdateHistory()
const metrics = handler.consumeLatencyMetrics()
metrics.persistenceFlushCount.should.equal(0)
metrics.saveMessagesDurationMs.should.equal(0)
metrics.updateHistoryDurationMs.should.equal(0)
})
it("should persist when a partial message transitions to complete", async () => {
const handler = createTestHandler()
await handler.addToClineMessagesEphemeral({
...createTestMessage("partial-message"),
partial: true,
})
let metrics = handler.consumeLatencyMetrics()
metrics.persistenceFlushCount.should.equal(0)
await handler.updateClineMessage(0, { text: "completed-message", partial: false })
const completedMessage = handler.getClineMessages()[0]
should.exist(completedMessage)
completedMessage!.text?.should.equal("completed-message")
should.exist(completedMessage!.partial)
completedMessage!.partial!.should.equal(false)
metrics = handler.consumeLatencyMetrics()
metrics.persistenceFlushCount.should.equal(1)
})
it("should emit delete change metadata for removed messages", async () => {
const handler = createTestHandler()
const observedDeletes: Array<{ index?: number; previousText?: string }> = []
handler.on("clineMessagesChanged", (change) => {
if (change.type === "delete") {
observedDeletes.push({ index: change.index, previousText: change.previousMessage?.text })
}
})
await handler.addToClineMessages(createTestMessage("first"))
await handler.addToClineMessages(createTestMessage("second"))
await handler.deleteClineMessage(0)
observedDeletes.length.should.equal(1)
should.exist(observedDeletes[0])
const deletedChange = observedDeletes[0]!
should.exist(deletedChange.index)
should.exist(deletedChange.previousText)
deletedChange.index!.should.equal(0)
deletedChange.previousText!.should.equal("first")
handler.getClineMessages().length.should.equal(1)
should.exist(handler.getClineMessages()[0])
const remainingMessage = handler.getClineMessages()[0]!
remainingMessage.text?.should.equal("second")
})
it("persists flushed ephemeral messages to disk for task recovery", async () => {
tempGlobalStorageDir = await fs.mkdtemp(path.join(os.tmpdir(), "cline-message-state-"))
setVscodeHostProviderMock({ globalStorageFsPath: tempGlobalStorageDir })
const handler = createTestHandler()
await handler.addToClineMessagesEphemeral({
...createTestMessage("recoverable partial"),
partial: true,
})
await handler.flushClineMessagesAndUpdateHistory()
const savedMessages = await getSavedClineMessages("test-task-id")
savedMessages.length.should.equal(1)
should.exist(savedMessages[0])
savedMessages[0]!.text?.should.equal("recoverable partial")
should.exist(savedMessages[0]!.partial)
savedMessages[0]!.partial!.should.equal(true)
})
it("persists completed conversation history snapshots that resume can reload", async () => {
tempGlobalStorageDir = await fs.mkdtemp(path.join(os.tmpdir(), "cline-message-history-"))
setVscodeHostProviderMock({ globalStorageFsPath: tempGlobalStorageDir })
const handler = createTestHandler()
await handler.overwriteApiConversationHistory([
{ role: "user", content: "task request", ts: 1 },
{ role: "assistant", content: "task response", ts: 2 },
])
const savedHistory = await getSavedApiConversationHistory("test-task-id")
savedHistory.length.should.equal(2)
should.exist(savedHistory[0])
should.exist(savedHistory[1])
savedHistory[0]!.content.should.equal("task request")
savedHistory[1]!.content.should.equal("task response")
})
})
@@ -15,6 +15,8 @@ import AutoApproveBar from "./auto-approve-menu/AutoApproveBar"
// Import utilities and hooks from the new structure
import {
ActionButtons,
buildApiReqReasoningIndex,
buildPendingTextMessageIndex,
CHAT_CONSTANTS,
ChatLayout,
convertHtmlToMarkdown,
@@ -321,6 +323,9 @@ const ChatView = ({ isHidden, showAnnouncement, hideAnnouncement, showHistoryVie
return groupLowStakesTools(groupMessages(visibleMessages))
}, [visibleMessages])
const apiReqReasoningIndex = useMemo(() => buildApiReqReasoningIndex(modifiedMessages), [modifiedMessages])
const pendingTextMessageIndex = useMemo(() => buildPendingTextMessageIndex(modifiedMessages), [modifiedMessages])
// Use scroll behavior hook
const scrollBehavior = useScrollBehavior(messages, visibleMessages, groupedMessages, expandedRows, setExpandedRows)
@@ -359,10 +364,14 @@ const ChatView = ({ isHidden, showAnnouncement, hideAnnouncement, showHistoryVie
)}
{task && (
<MessagesArea
apiReqReasoningIndex={apiReqReasoningIndex}
chatState={chatState}
groupedMessages={groupedMessages}
messageHandlers={messageHandlers}
mode={mode}
modifiedMessages={modifiedMessages}
pendingTextMessageIndex={pendingTextMessageIndex}
rawMessages={messages}
scrollBehavior={scrollBehavior}
task={task}
/>
@@ -1,12 +1,10 @@
import type { ClineMessage, ClineSayTool } from "@shared/ExtensionMessage"
import type { Mode } from "@shared/storage/types"
import type { LucideIcon } from "lucide-react"
import type React from "react"
import { useMemo } from "react"
import { cleanPathPrefix } from "../common/CodeAccordian"
import { getIconByToolName } from "./chat-view"
import { isApiReqAbsorbable, isLowStakesTool } from "./chat-view/utils/messageUtils"
import ErrorRow from "./ErrorRow"
import { getRequestStartRowState } from "./requestStartRowState"
import { ThinkingRow } from "./ThinkingRow"
import { TypewriterText } from "./TypewriterText"
@@ -24,107 +22,6 @@ interface RequestStartRowProps {
handleToggle: () => void
}
// State type for api_req_started rendering
type ApiReqState = "pre" | "thinking" | "error" | "final"
// Helper to format search regex for display - show all terms separated by |
const formatSearchRegex = (regex: string, path: string, filePattern?: string): string => {
const cleanedPath = cleanPathPrefix(path)
const terms = regex
.split("|")
.map((t) => t.trim().replace(/\\b/g, "").replace(/\\s\?/g, " "))
.filter(Boolean)
.join(" | ")
return filePattern && filePattern !== "*" ? `"${terms}" in ${cleanedPath}/ (${filePattern})` : `"${terms}" in ${cleanedPath}/`
}
// Format activity text based on tool type
const getActivityText = (tool: ClineSayTool): string | null => {
const cleanedPath = cleanPathPrefix(tool.path || "")
switch (tool.tool) {
case "readFile":
return tool.path ? `Reading ${cleanedPath}...` : null
case "listFilesTopLevel":
case "listFilesRecursive":
return tool.path ? `Exploring ${cleanedPath}/...` : null
case "searchFiles":
return tool.regex && tool.path ? `Searching ${formatSearchRegex(tool.regex, tool.path, tool.filePattern)}...` : null
case "listCodeDefinitionNames":
return tool.path ? `Analyzing ${cleanedPath}/...` : null
default:
return null
}
}
// Collect tools in a given range, with optional stop condition
const collectToolsInRange = (
messages: ClineMessage[],
startIdx: number,
endIdx: number,
stopCondition?: (msg: ClineMessage) => boolean,
): { icon: LucideIcon; text: string }[] => {
const activities: { icon: LucideIcon; text: string }[] = []
for (let i = startIdx; i < endIdx; i++) {
const msg = messages[i]
if (stopCondition?.(msg)) {
break
}
// Only collect tools that are currently executing (ask === "tool")
// Skip completed tools (say === "tool") - they should be in the completed list
if (msg.say === "tool" || msg.ask !== "tool") {
continue
}
try {
const tool = JSON.parse(msg.text || "{}") as ClineSayTool
const activityText = getActivityText(tool)
if (activityText) {
const toolIcon = getIconByToolName(tool.tool)
activities.push({ icon: toolIcon, text: activityText })
}
} catch {
// ignore parse errors
}
}
return activities
}
// Find current api_req and determine if it has cost
const findCurrentApiReq = (messages: ClineMessage[]): { index: number; hasCost: boolean } | null => {
for (let i = messages.length - 1; i >= 0; i--) {
const msg = messages[i]
if (msg.say === "api_req_started" && msg.text) {
try {
const info = JSON.parse(msg.text)
return { index: i, hasCost: info.cost != null }
} catch {
return null
}
}
}
return null
}
// Find the most recent completed api_req before the given index
const findPrevCompletedApiReq = (messages: ClineMessage[], beforeIdx: number): number => {
for (let i = beforeIdx - 1; i >= 0; i--) {
const msg = messages[i]
if (msg.say === "api_req_started" && msg.text) {
try {
const info = JSON.parse(msg.text)
if (info.cost != null) {
return i
}
} catch {
// ignore parse errors
}
}
}
return -1
}
/**
* Displays the current state of an active tool operation,
*/
@@ -140,72 +37,21 @@ export const RequestStartRow: React.FC<RequestStartRowProps> = ({
isExpanded,
message,
}) => {
// Derive explicit state
const hasError = !!(apiRequestFailedMessage || apiReqStreamingFailedMessage)
const { apiReqState, currentActivities, shouldShowActivities } = useMemo(
() =>
getRequestStartRowState({
message,
clineMessages,
reasoningContent,
apiRequestFailedMessage,
apiReqStreamingFailedMessage,
cost,
responseStarted,
getIconByToolName: (toolName) => getIconByToolName(toolName as ClineSayTool["tool"]),
}),
[message, clineMessages, reasoningContent, apiRequestFailedMessage, apiReqStreamingFailedMessage, cost, responseStarted],
)
const hasCost = cost != null
const hasReasoning = !!reasoningContent
const hasCompletionResult = clineMessages.some(
(msg) => msg.ask === "completion_result" || msg.say === "completion_result" || msg.ask === "plan_mode_respond",
)
const apiReqState: ApiReqState = hasError ? "error" : hasCost ? "final" : hasReasoning ? "thinking" : "pre"
// While reasoning is streaming, keep the Brain ThinkingBlock exactly as-is.
// Once response content starts (any text/tool/command), collapse into a compact
// "🧠 Thinking" row that can be expanded to show the reasoning only.
const showStreamingThinking = useMemo(
() => hasReasoning && !hasError && !cost && !responseStarted,
[hasReasoning, hasError, cost, responseStarted],
)
// Check if this api_req will be absorbed into a tool group (reasoning will disappear)
const willBeAbsorbed = useMemo(() => {
return isApiReqAbsorbable(message.ts, clineMessages)
}, [message.ts, clineMessages])
// Find all exploratory tool activities that are currently in flight.
// Tools come AFTER the api_req_started message, so we look from currentApiReq forward.
const currentActivities = useMemo(() => {
const currentApiReq = findCurrentApiReq(clineMessages)
if (!currentApiReq) {
return []
}
if (!currentApiReq.hasCost) {
// CASE A: Current api_req is INCOMPLETE
// Look for ask === "tool" messages AFTER the current api_req_started
return collectToolsInRange(clineMessages, currentApiReq.index + 1, clineMessages.length)
}
// CASE B: Current api_req is COMPLETE - no activities to show
return []
}, [clineMessages])
// Check if there are any completed tools in the tool group
const hasCompletedTools = useMemo(() => {
// Look for any completed low-stakes tool messages that would be in a tool group
return clineMessages.some((msg, idx) => {
if (msg.say === "tool" && isLowStakesTool(msg)) {
// Check if this tool is from a completed API request
// (looking backwards for an api_req with cost)
for (let i = idx - 1; i >= 0; i--) {
const prevMsg = clineMessages[i]
if (prevMsg.say === "api_req_started" && prevMsg.text) {
try {
const info = JSON.parse(prevMsg.text)
return info.cost != null
} catch {
return false
}
}
}
}
return false
})
}, [clineMessages])
// Only show currentActivities if there are NO completed tools
// (otherwise they'll be shown in the unified ToolGroupRenderer list)
const shouldShowActivities = currentActivities.length > 0 && !hasCompletedTools
// Initial loading ("Thinking..." before any content) is injected as a synthetic in-list
// reasoning row in MessagesArea to avoid footer handoff flicker.
@@ -1,9 +1,9 @@
import type { ClineMessage } from "@shared/ExtensionMessage"
import type { Mode } from "@shared/storage/types"
import type React from "react"
import { useCallback, useMemo } from "react"
import { Virtuoso } from "react-virtuoso"
import { StickyUserMessage } from "@/components/chat/task-header/StickyUserMessage"
import { useExtensionState } from "@/context/ExtensionStateContext"
import { cn } from "@/lib/utils"
import type { ChatState, MessageHandlers, ScrollBehavior } from "../../types/chatTypes"
import { isToolGroup } from "../../utils/messageUtils"
@@ -11,11 +11,15 @@ import { createMessageRenderer } from "../messages/MessageRenderer"
interface MessagesAreaProps {
task: ClineMessage
rawMessages: ClineMessage[]
groupedMessages: (ClineMessage | ClineMessage[])[]
modifiedMessages: ClineMessage[]
mode: Mode
scrollBehavior: ScrollBehavior
chatState: ChatState
messageHandlers: MessageHandlers
apiReqReasoningIndex: Map<number, { reasoning: string | undefined; responseStarted: boolean }>
pendingTextMessageIndex: Set<number>
}
/**
@@ -24,14 +28,17 @@ interface MessagesAreaProps {
*/
export const MessagesArea: React.FC<MessagesAreaProps> = ({
task,
rawMessages,
groupedMessages,
modifiedMessages,
mode,
scrollBehavior,
chatState,
messageHandlers,
apiReqReasoningIndex,
pendingTextMessageIndex,
}) => {
const { clineMessages } = useExtensionState()
const lastRawMessage = useMemo(() => clineMessages.at(-1), [clineMessages])
const lastRawMessage = useMemo(() => rawMessages.at(-1), [rawMessages])
const {
virtuosoRef,
@@ -51,8 +58,8 @@ export const MessagesArea: React.FC<MessagesAreaProps> = ({
if (!scrolledPastUserMessage) {
return -1
}
return clineMessages.findIndex((msg) => msg.ts === scrolledPastUserMessage.ts)
}, [clineMessages, scrolledPastUserMessage])
return rawMessages.findIndex((msg) => msg.ts === scrolledPastUserMessage.ts)
}, [rawMessages, scrolledPastUserMessage])
// Handler to scroll to the scrolled past user message
const handleScrollToUserMessage = useCallback(() => {
@@ -172,23 +179,32 @@ export const MessagesArea: React.FC<MessagesAreaProps> = ({
createMessageRenderer(
displayedGroupedMessages,
modifiedMessages,
rawMessages,
apiReqReasoningIndex,
pendingTextMessageIndex,
mode,
expandedRows,
toggleRowExpansion,
handleRowHeightChange,
setActiveQuote,
inputValue,
messageHandlers,
false,
showThinkingLoaderRow,
),
[
displayedGroupedMessages,
modifiedMessages,
rawMessages,
apiReqReasoningIndex,
pendingTextMessageIndex,
mode,
expandedRows,
toggleRowExpansion,
handleRowHeightChange,
setActiveQuote,
inputValue,
messageHandlers,
showThinkingLoaderRow,
],
)
@@ -0,0 +1,83 @@
import { render } from "@testing-library/react"
import { beforeEach, describe, expect, it, vi } from "vitest"
import type { ClineMessage } from "../../../../../../src/shared/ExtensionMessage"
import { MemoizedMessageRenderer } from "./MessageRenderer"
const renderCounts = new Map<number, number>()
vi.mock("@/components/chat/ChatRow", () => ({
default: ({ message }: { message: ClineMessage }) => {
renderCounts.set(message.ts, (renderCounts.get(message.ts) ?? 0) + 1)
return <div data-testid={`chat-row-${message.ts}`}>{message.text}</div>
},
}))
vi.mock("@/components/chat/BrowserSessionRow", () => ({
default: () => <div data-testid="browser-session-row" />,
}))
vi.mock("./ToolGroupRenderer", () => ({
ToolGroupRenderer: () => <div data-testid="tool-group-row" />,
}))
describe("MemoizedMessageRenderer", () => {
beforeEach(() => {
renderCounts.clear()
})
it("avoids rerendering unrelated text rows when pending state changes for another row", () => {
const firstMessage = { ts: 1, type: "say", say: "text", text: "first" } as ClineMessage
const secondMessage = { ts: 2, type: "say", say: "text", text: "second" } as ClineMessage
const groupedMessages = [firstMessage, secondMessage]
const baseProps = {
groupedMessages,
modifiedMessages: [firstMessage, secondMessage],
rawMessages: [firstMessage, secondMessage],
mode: "act" as const,
expandedRows: {},
onToggleExpand: vi.fn(),
onHeightChange: vi.fn(),
onSetQuote: vi.fn(),
inputValue: "",
messageHandlers: {
executeButtonAction: vi.fn(),
handleSendMessage: vi.fn(),
handleTaskCloseButtonClick: vi.fn(),
startNewTask: vi.fn(),
},
footerActive: false,
apiReqReasoningIndex: new Map(),
pendingTextMessageIndex: new Set<number>(),
}
const { rerender } = render(
<>
<MemoizedMessageRenderer {...baseProps} index={0} messageOrGroup={firstMessage} />
<MemoizedMessageRenderer {...baseProps} index={1} messageOrGroup={secondMessage} />
</>,
)
expect(renderCounts.get(1)).toBe(1)
expect(renderCounts.get(2)).toBe(1)
rerender(
<>
<MemoizedMessageRenderer
{...baseProps}
index={0}
messageOrGroup={firstMessage}
pendingTextMessageIndex={new Set([2])}
/>
<MemoizedMessageRenderer
{...baseProps}
index={1}
messageOrGroup={secondMessage}
pendingTextMessageIndex={new Set([2])}
/>
</>,
)
expect(renderCounts.get(1)).toBe(1)
expect(renderCounts.get(2)).toBe(2)
})
})
@@ -1,12 +1,12 @@
import type { ClineMessage } from "@shared/ExtensionMessage"
import type { Mode } from "@shared/storage/types"
import type React from "react"
import { useMemo } from "react"
import { memo, useMemo } from "react"
import BrowserSessionRow from "@/components/chat/BrowserSessionRow"
import ChatRow from "@/components/chat/ChatRow"
import { useExtensionState } from "@/context/ExtensionStateContext"
import { cn } from "@/lib/utils"
import type { MessageHandlers } from "../../types/chatTypes"
import { findReasoningForApiReq, isTextMessagePendingToolCall, isToolGroup } from "../../utils/messageUtils"
import { isToolGroup } from "../../utils/messageUtils"
import { ToolGroupRenderer } from "./ToolGroupRenderer"
interface MessageRendererProps {
@@ -14,6 +14,8 @@ interface MessageRendererProps {
messageOrGroup: ClineMessage | ClineMessage[]
groupedMessages: (ClineMessage | ClineMessage[])[]
modifiedMessages: ClineMessage[]
rawMessages: ClineMessage[]
mode: Mode
expandedRows: Record<number, boolean>
onToggleExpand: (ts: number) => void
onHeightChange: (isTaller: boolean) => void
@@ -21,6 +23,8 @@ interface MessageRendererProps {
inputValue: string
messageHandlers: MessageHandlers
footerActive: boolean
apiReqReasoningIndex: Map<number, { reasoning: string | undefined; responseStarted: boolean }>
pendingTextMessageIndex: Set<number>
}
/**
@@ -32,6 +36,8 @@ export const MessageRenderer: React.FC<MessageRendererProps> = ({
messageOrGroup,
groupedMessages,
modifiedMessages,
rawMessages,
mode,
expandedRows,
onToggleExpand,
onHeightChange,
@@ -39,28 +45,24 @@ export const MessageRenderer: React.FC<MessageRendererProps> = ({
inputValue,
messageHandlers,
footerActive,
apiReqReasoningIndex,
pendingTextMessageIndex,
}) => {
const { mode } = useExtensionState()
const isLastMessage = useMemo(() => index === groupedMessages?.length - 1, [groupedMessages, index])
// Get reasoning content and response status for api_req_started messages
const reasoningData = useMemo(() => {
if (!Array.isArray(messageOrGroup) && messageOrGroup.say === "api_req_started") {
// Use the same message source-of-truth that `groupedMessages` is derived from.
return findReasoningForApiReq(messageOrGroup.ts, modifiedMessages)
return apiReqReasoningIndex.get(messageOrGroup.ts) ?? { reasoning: undefined, responseStarted: false }
}
return { reasoning: undefined, responseStarted: false }
}, [messageOrGroup, modifiedMessages])
}, [apiReqReasoningIndex, messageOrGroup])
// Check if a text message is waiting for tool call completion
const isRequestInProgress = useMemo(() => {
if (!Array.isArray(messageOrGroup) && messageOrGroup.say === "text") {
// Use modifiedMessages so this stays consistent with the rendered list.
return isTextMessagePendingToolCall(messageOrGroup.ts, modifiedMessages)
return pendingTextMessageIndex.has(messageOrGroup.ts)
}
return false
}, [messageOrGroup, modifiedMessages])
}, [messageOrGroup, pendingTextMessageIndex])
// Tool group (low-stakes tools grouped together)
// Determine if this is the last tool group to show active items
@@ -125,6 +127,49 @@ export const MessageRenderer: React.FC<MessageRendererProps> = ({
)
}
const areMessageRendererPropsEqual = (prev: MessageRendererProps, next: MessageRendererProps) => {
if (
prev.index !== next.index ||
prev.footerActive !== next.footerActive ||
prev.inputValue !== next.inputValue ||
prev.mode !== next.mode ||
prev.messageOrGroup !== next.messageOrGroup ||
prev.groupedMessages !== next.groupedMessages ||
prev.modifiedMessages !== next.modifiedMessages ||
prev.rawMessages !== next.rawMessages ||
prev.expandedRows !== next.expandedRows ||
prev.onToggleExpand !== next.onToggleExpand ||
prev.onHeightChange !== next.onHeightChange ||
prev.onSetQuote !== next.onSetQuote ||
prev.messageHandlers !== next.messageHandlers
) {
return false
}
if (Array.isArray(prev.messageOrGroup) || Array.isArray(next.messageOrGroup)) {
return true
}
if (prev.messageOrGroup.say === "api_req_started") {
const prevReasoning = prev.apiReqReasoningIndex.get(prev.messageOrGroup.ts)
const nextReasoning = next.apiReqReasoningIndex.get(next.messageOrGroup.ts)
return (
prevReasoning?.reasoning === nextReasoning?.reasoning &&
prevReasoning?.responseStarted === nextReasoning?.responseStarted
)
}
if (prev.messageOrGroup.say === "text") {
return (
prev.pendingTextMessageIndex.has(prev.messageOrGroup.ts) === next.pendingTextMessageIndex.has(next.messageOrGroup.ts)
)
}
return true
}
export const MemoizedMessageRenderer = memo(MessageRenderer, areMessageRendererPropsEqual)
/**
* Factory function to create the itemContent callback for Virtuoso
* This allows us to encapsulate the rendering logic while maintaining performance
@@ -132,6 +177,10 @@ export const MessageRenderer: React.FC<MessageRendererProps> = ({
export const createMessageRenderer = (
groupedMessages: (ClineMessage | ClineMessage[])[],
modifiedMessages: ClineMessage[],
rawMessages: ClineMessage[],
apiReqReasoningIndex: Map<number, { reasoning: string | undefined; responseStarted: boolean }>,
pendingTextMessageIndex: Set<number>,
mode: Mode,
expandedRows: Record<number, boolean>,
onToggleExpand: (ts: number) => void,
onHeightChange: (isTaller: boolean) => void,
@@ -141,7 +190,8 @@ export const createMessageRenderer = (
footerActive: boolean,
) => {
return (index: number, messageOrGroup: ClineMessage | ClineMessage[]) => (
<MessageRenderer
<MemoizedMessageRenderer
apiReqReasoningIndex={apiReqReasoningIndex}
expandedRows={expandedRows}
footerActive={footerActive}
groupedMessages={groupedMessages}
@@ -149,10 +199,13 @@ export const createMessageRenderer = (
inputValue={inputValue}
messageHandlers={messageHandlers}
messageOrGroup={messageOrGroup}
mode={mode}
modifiedMessages={modifiedMessages}
onHeightChange={onHeightChange}
onSetQuote={onSetQuote}
onToggleExpand={onToggleExpand}
pendingTextMessageIndex={pendingTextMessageIndex}
rawMessages={rawMessages}
/>
)
}
@@ -1,6 +1,6 @@
import type { ClineMessage } from "@shared/ExtensionMessage"
import { describe, expect, it } from "vitest"
import { groupLowStakesTools, isToolGroup } from "./messageUtils"
import type { ClineMessage } from "../../../../../../src/shared/ExtensionMessage"
import { buildApiReqReasoningIndex, buildPendingTextMessageIndex, groupLowStakesTools, isToolGroup } from "./messageUtils"
const createTextMessage = (ts: number, text: string): ClineMessage => ({
type: "say",
@@ -9,6 +9,35 @@ const createTextMessage = (ts: number, text: string): ClineMessage => ({
ts,
})
describe("message render state indexes", () => {
it("precomputes reasoning and response-start state per api request", () => {
const messages = [
{ type: "say", say: "api_req_started", text: JSON.stringify({ request: "one" }), ts: 1 },
{ type: "say", say: "reasoning", text: "thinking 1", ts: 2 },
{ type: "say", say: "text", text: "answer 1", ts: 3 },
{ type: "say", say: "api_req_started", text: JSON.stringify({ request: "two" }), ts: 4 },
{ type: "say", say: "reasoning", text: "thinking 2", ts: 5 },
] as ClineMessage[]
const index = buildApiReqReasoningIndex(messages)
expect(index.get(1)).toEqual({ reasoning: "thinking 1", responseStarted: true })
expect(index.get(4)).toEqual({ reasoning: "thinking 2", responseStarted: false })
})
it("precomputes pending text messages for incomplete api requests", () => {
const messages = [
{ type: "say", say: "api_req_started", text: JSON.stringify({ request: "one" }), ts: 1 },
{ type: "say", say: "text", text: "streaming", ts: 2 },
{ type: "say", say: "api_req_started", text: JSON.stringify({ request: "two", cost: 0.5 }), ts: 3 },
{ type: "say", say: "text", text: "complete", ts: 4 },
] as ClineMessage[]
const index = buildPendingTextMessageIndex(messages)
expect(index.has(2)).toBe(true)
expect(index.has(4)).toBe(false)
})
})
const createToolMessage = (ts: number, tool: string): ClineMessage => ({
type: "say",
say: "tool",
@@ -82,4 +111,29 @@ describe("groupLowStakesTools", () => {
expect(grouped[0]).toMatchObject({ type: "say", say: "reasoning", text: "Planning next read" })
expect(isToolGroup(grouped[1])).toBe(true)
})
it("keeps low-stakes tool rows grouped correctly while new coalesced updates append more tools", () => {
const firstPass = groupLowStakesTools([createTextMessage(1, "Starting analysis"), createToolMessage(2, "readFile")])
expect(firstPass).toHaveLength(2)
expect(isToolGroup(firstPass[1])).toBe(true)
if (isToolGroup(firstPass[1])) {
expect(firstPass[1].map((message) => message.ts)).toEqual([2])
}
const secondPass = groupLowStakesTools([
createTextMessage(1, "Starting analysis"),
createToolMessage(2, "readFile"),
createToolMessage(3, "searchFiles"),
createToolMessage(4, "listCodeDefinitionNames"),
])
expect(secondPass).toHaveLength(2)
expect(secondPass[0]).toMatchObject({ type: "say", say: "text", text: "Starting analysis" })
expect(isToolGroup(secondPass[1])).toBe(true)
if (isToolGroup(secondPass[1])) {
expect(secondPass[1].map((message) => message.ts)).toEqual([2, 3, 4])
expect(secondPass[1].every((message) => message.say === "tool")).toBe(true)
}
})
})
@@ -251,6 +251,58 @@ export function findReasoningForApiReq(
}
}
export function buildApiReqReasoningIndex(
allMessages: ClineMessage[],
): Map<number, { reasoning: string | undefined; responseStarted: boolean }> {
const result = new Map<number, { reasoning: string | undefined; responseStarted: boolean }>()
let currentApiReqTs: number | undefined
let reasoningParts: string[] = []
let responseStarted = false
const flushCurrent = () => {
if (currentApiReqTs === undefined) {
return
}
result.set(currentApiReqTs, {
reasoning: reasoningParts.length > 0 ? reasoningParts.join("\n\n") : undefined,
responseStarted,
})
currentApiReqTs = undefined
reasoningParts = []
responseStarted = false
}
for (const message of allMessages) {
if (message.say === "api_req_started") {
flushCurrent()
currentApiReqTs = message.ts
continue
}
if (currentApiReqTs === undefined) {
continue
}
if (message.say === "reasoning" && message.text) {
reasoningParts.push(message.text)
continue
}
if (
message.say === "text" ||
message.say === "tool" ||
message.ask === "tool" ||
message.ask === "command" ||
message.say === "command"
) {
responseStarted = true
}
}
flushCurrent()
return result
}
/**
* Find the API request info for a checkpoint message.
* Looks backwards from the checkpoint to find the preceding api_req_started.
@@ -401,6 +453,44 @@ export function isTextMessagePendingToolCall(textTs: number, allMessages: ClineM
return false
}
export function buildPendingTextMessageIndex(allMessages: ClineMessage[]): Set<number> {
const pendingTextMessages = new Set<number>()
let currentApiReqHasCost = false
let currentTextMessages: number[] = []
const flushCurrent = () => {
if (!currentApiReqHasCost) {
for (const ts of currentTextMessages) {
pendingTextMessages.add(ts)
}
}
currentApiReqHasCost = false
currentTextMessages = []
}
for (const message of allMessages) {
if (message.say === "api_req_started") {
flushCurrent()
if (message.text) {
try {
const info = JSON.parse(message.text)
currentApiReqHasCost = info.cost != null
} catch {
currentApiReqHasCost = false
}
}
continue
}
if (message.say === "text") {
currentTextMessages.push(message.ts)
}
}
flushCurrent()
return pendingTextMessages
}
/**
* Check if a tool group should be hidden because its tools are currently being
* displayed in the loading state animation.
@@ -0,0 +1,131 @@
import { FileCode2Icon, SearchIcon } from "lucide-react"
import { describe, expect, it } from "vitest"
import type { ClineMessage } from "../../../../src/shared/ExtensionMessage"
import { getRequestStartRowState } from "./requestStartRowState"
const createMessage = (overrides: Partial<ClineMessage>): ClineMessage =>
({
ts: Date.now(),
type: "say",
say: "text",
text: "",
...overrides,
}) as ClineMessage
describe("getRequestStartRowState", () => {
it("shows in-flight exploratory activities for an active api request without completed tools", () => {
const messages: ClineMessage[] = [
createMessage({ ts: 1, type: "say", say: "api_req_started", text: JSON.stringify({ request: "hello" }) }),
createMessage({
ts: 2,
type: "ask",
ask: "tool",
text: JSON.stringify({ tool: "readFile", path: "src/index.ts" }),
}),
createMessage({
ts: 3,
type: "ask",
ask: "tool",
text: JSON.stringify({ tool: "searchFiles", path: "src", regex: "latency|stream", filePattern: "*.ts" }),
}),
]
const state = getRequestStartRowState({
message: messages[0],
clineMessages: messages,
cost: undefined,
getIconByToolName: (toolName) => (toolName === "searchFiles" ? SearchIcon : FileCode2Icon),
})
expect(state.apiReqState).toBe("pre")
expect(state.shouldShowActivities).toBe(true)
expect(state.currentActivities.map((activity) => activity.text)).toEqual([
"Reading src/index.ts...",
'Searching "latency | stream" in src/ (*.ts)...',
])
})
it("hides transient activities after completed tools exist for a finished request", () => {
const messages: ClineMessage[] = [
createMessage({ ts: 1, type: "say", say: "api_req_started", text: JSON.stringify({ request: "hello", cost: 0.42 }) }),
createMessage({ ts: 2, type: "say", say: "tool", text: JSON.stringify({ tool: "readFile", path: "src/index.ts" }) }),
]
const state = getRequestStartRowState({
message: messages[0],
clineMessages: messages,
cost: 0.42,
getIconByToolName: () => FileCode2Icon,
})
expect(state.apiReqState).toBe("final")
expect(state.currentActivities).toEqual([])
expect(state.shouldShowActivities).toBe(false)
})
it("preserves streaming thinking state until response content starts", () => {
const message = createMessage({ ts: 1, type: "say", say: "api_req_started", text: JSON.stringify({ request: "hello" }) })
expect(
getRequestStartRowState({
message,
clineMessages: [message],
reasoningContent: "thinking",
responseStarted: false,
getIconByToolName: () => FileCode2Icon,
}).showStreamingThinking,
).toBe(true)
expect(
getRequestStartRowState({
message,
clineMessages: [message],
reasoningContent: "thinking",
responseStarted: true,
getIconByToolName: () => FileCode2Icon,
}).showStreamingThinking,
).toBe(false)
})
it("keeps final request metadata visible once cost arrives", () => {
const message = createMessage({
ts: 10,
type: "say",
say: "api_req_started",
text: JSON.stringify({ request: "hello", cost: 0.42, tokensIn: 120, tokensOut: 80 }),
})
const state = getRequestStartRowState({
message,
clineMessages: [message],
cost: 0.42,
responseStarted: true,
getIconByToolName: () => FileCode2Icon,
})
expect(state.apiReqState).toBe("final")
expect(state.showStreamingThinking).toBe(false)
expect(state.shouldShowActivities).toBe(false)
})
it("prioritizes error state over partial reasoning when streaming fails", () => {
const message = createMessage({
ts: 11,
type: "say",
say: "api_req_started",
text: JSON.stringify({ request: "hello", cancelReason: "streaming_failed" }),
})
const state = getRequestStartRowState({
message,
clineMessages: [message],
reasoningContent: "thinking",
apiReqStreamingFailedMessage: "network timeout",
responseStarted: false,
getIconByToolName: () => FileCode2Icon,
})
expect(state.apiReqState).toBe("error")
expect(state.showStreamingThinking).toBe(false)
})
})
@@ -0,0 +1,140 @@
import type { ClineMessage, ClineSayTool } from "@shared/ExtensionMessage"
import type { LucideIcon } from "lucide-react"
export type ApiReqState = "pre" | "thinking" | "error" | "final"
export type RequestStartRowActivity = {
icon: LucideIcon
text: string
}
type RequestStartRowStateArgs = {
message: ClineMessage
clineMessages: ClineMessage[]
reasoningContent?: string
apiRequestFailedMessage?: string
apiReqStreamingFailedMessage?: string
cost?: number
responseStarted?: boolean
getIconByToolName: (toolName: string) => LucideIcon
}
// Helper to format search regex for display - show all terms separated by |
const formatSearchRegex = (regex: string, path: string, filePattern?: string): string => {
const terms = regex
.split("|")
.map((t) => t.trim().replace(/\\b/g, "").replace(/\\s\?/g, " "))
.filter(Boolean)
.join(" | ")
return filePattern && filePattern !== "*" ? `"${terms}" in ${path}/ (${filePattern})` : `"${terms}" in ${path}/`
}
const getActivityText = (tool: ClineSayTool): string | null => {
switch (tool.tool) {
case "readFile":
return tool.path ? `Reading ${tool.path}...` : null
case "listFilesTopLevel":
case "listFilesRecursive":
return tool.path ? `Exploring ${tool.path}/...` : null
case "searchFiles":
return tool.regex && tool.path ? `Searching ${formatSearchRegex(tool.regex, tool.path, tool.filePattern)}...` : null
case "listCodeDefinitionNames":
return tool.path ? `Analyzing ${tool.path}/...` : null
default:
return null
}
}
const isCompletedApiReqMessage = (message: ClineMessage): boolean => {
if (message.say !== "api_req_started" || !message.text) {
return false
}
try {
const info = JSON.parse(message.text)
return info.cost != null
} catch {
return false
}
}
export function getRequestStartRowState(args: RequestStartRowStateArgs): {
apiReqState: ApiReqState
currentActivities: RequestStartRowActivity[]
shouldShowActivities: boolean
showStreamingThinking: boolean
} {
const {
message,
clineMessages,
reasoningContent,
apiRequestFailedMessage,
apiReqStreamingFailedMessage,
cost,
responseStarted,
getIconByToolName,
} = args
const hasError = !!(apiRequestFailedMessage || apiReqStreamingFailedMessage)
const hasCost = cost != null
const hasReasoning = !!reasoningContent
const apiReqState: ApiReqState = hasError ? "error" : hasCost ? "final" : hasReasoning ? "thinking" : "pre"
const showStreamingThinking = hasReasoning && !hasError && !hasCost && !responseStarted
let currentApiReqIndex = -1
let currentApiReqHasCost = false
let hasCompletedTools = false
for (let i = clineMessages.length - 1; i >= 0; i--) {
const currentMessage = clineMessages[i]
if (currentApiReqIndex === -1 && currentMessage.say === "api_req_started") {
currentApiReqIndex = i
currentApiReqHasCost = isCompletedApiReqMessage(currentMessage)
}
if (!hasCompletedTools && currentMessage.say === "tool" && currentMessage.ts !== message.ts) {
for (let j = i - 1; j >= 0; j--) {
if (isCompletedApiReqMessage(clineMessages[j])) {
hasCompletedTools = true
break
}
}
}
if (currentApiReqIndex !== -1 && hasCompletedTools) {
break
}
}
const currentActivities: RequestStartRowActivity[] = []
if (currentApiReqIndex !== -1 && !currentApiReqHasCost) {
for (let i = currentApiReqIndex + 1; i < clineMessages.length; i++) {
const currentMessage = clineMessages[i]
if (currentMessage.say === "tool" || currentMessage.ask !== "tool") {
continue
}
try {
const tool = JSON.parse(currentMessage.text || "{}") as ClineSayTool
const activityText = getActivityText(tool)
if (activityText) {
currentActivities.push({
icon: getIconByToolName(tool.tool),
text: activityText,
})
}
} catch {
// ignore parse errors
}
}
}
return {
apiReqState,
currentActivities,
shouldShowActivities: currentActivities.length > 0 && !hasCompletedTools,
showStreamingThinking,
}
}
@@ -0,0 +1,142 @@
import { render, screen, waitFor } from "@testing-library/react"
import { describe, expect, it, vi } from "vitest"
import { ExtensionStateContextProvider, useExtensionState } from "./ExtensionStateContext"
type StreamCallbacks<T> = {
onResponse?: (value: T) => void
onError?: (error: unknown) => void
onComplete?: () => void
}
const subscriptions = {
state: undefined as StreamCallbacks<{ stateJson?: string }> | undefined,
partial: undefined as StreamCallbacks<any> | undefined,
delta: undefined as StreamCallbacks<{ deltaJson?: string }> | undefined,
}
vi.mock("../services/grpc-client", () => ({
StateServiceClient: {
subscribeToState: (_request: unknown, callbacks: StreamCallbacks<{ stateJson?: string }>) => {
subscriptions.state = callbacks
return () => {
subscriptions.state = undefined
}
},
getLatestState: vi.fn().mockResolvedValue({
stateJson: JSON.stringify({
version: "resynced",
clineMessages: [{ ts: 99, type: "say", say: "text", text: "resynced" }],
currentTaskItem: { id: "task-1" },
}),
}),
getAvailableTerminalProfiles: vi.fn().mockResolvedValue({ profiles: [] }),
},
UiServiceClient: {
subscribeToMcpButtonClicked: vi.fn().mockReturnValue(() => {}),
subscribeToHistoryButtonClicked: vi.fn().mockReturnValue(() => {}),
subscribeToChatButtonClicked: vi.fn().mockReturnValue(() => {}),
subscribeToAccountButtonClicked: vi.fn().mockReturnValue(() => {}),
subscribeToSettingsButtonClicked: vi.fn().mockReturnValue(() => {}),
subscribeToWorktreesButtonClicked: vi.fn().mockReturnValue(() => {}),
subscribeToRelinquishControl: vi.fn().mockReturnValue(() => {}),
subscribeToPartialMessage: (_request: unknown, callbacks: StreamCallbacks<any>) => {
subscriptions.partial = callbacks
return () => {
subscriptions.partial = undefined
}
},
subscribeToTaskUiDeltas: (_request: unknown, callbacks: StreamCallbacks<{ deltaJson?: string }>) => {
subscriptions.delta = callbacks
return () => {
subscriptions.delta = undefined
}
},
initializeWebview: vi.fn().mockResolvedValue({}),
},
McpServiceClient: {
subscribeToMcpServers: vi.fn().mockReturnValue(() => {}),
subscribeToMcpMarketplaceCatalog: vi.fn().mockReturnValue(() => {}),
},
ModelsServiceClient: {
subscribeToOpenRouterModels: vi.fn().mockReturnValue(() => {}),
subscribeToLiteLlmModels: vi.fn().mockReturnValue(() => {}),
refreshOpenRouterModelsRpc: vi.fn().mockResolvedValue({ models: [] }),
refreshVercelAiGatewayModelsRpc: vi.fn().mockResolvedValue({ models: [] }),
refreshClineModelsRpc: vi.fn().mockResolvedValue({ models: [] }),
refreshBasetenModelsRpc: vi.fn().mockResolvedValue({ models: [] }),
refreshLiteLlmModelsRpc: vi.fn().mockResolvedValue({ models: [] }),
refreshHicapModels: vi.fn().mockResolvedValue({ models: [] }),
},
FileServiceClient: {},
}))
function ContextProbe() {
const state = useExtensionState() as any
return (
<>
<div data-testid="version">{state.version}</div>
<div data-testid="message-count">{state.clineMessages.length}</div>
<div data-testid="latest-message">{state.clineMessages.at(-1)?.text ?? ""}</div>
<div data-testid="background-command">{String(state.backgroundCommandRunning)}</div>
</>
)
}
describe("ExtensionStateContextProvider", () => {
it("hydrates from full state and applies streaming task UI deltas", async () => {
render(
<ExtensionStateContextProvider>
<ContextProbe />
</ExtensionStateContextProvider>,
)
subscriptions.state?.onResponse?.({
stateJson: JSON.stringify({
version: "initial",
clineMessages: [],
currentTaskItem: { id: "task-1" },
backgroundCommandRunning: false,
}),
})
await waitFor(() => {
expect(screen.getByTestId("version").textContent).toBe("initial")
})
subscriptions.delta?.onResponse?.({
deltaJson: JSON.stringify({
type: "message_added",
taskId: "task-1",
sequence: 1,
message: { ts: 1, type: "say", say: "text", text: "hello" },
}),
})
await waitFor(() => {
expect(screen.getByTestId("message-count").textContent).toBe("1")
expect(screen.getByTestId("latest-message").textContent).toBe("hello")
})
subscriptions.delta?.onResponse?.({
deltaJson: JSON.stringify({
type: "message_updated",
taskId: "task-1",
sequence: 2,
message: { ts: 1, type: "say", say: "text", text: "hello world" },
}),
})
subscriptions.delta?.onResponse?.({
deltaJson: JSON.stringify({
type: "task_metadata_updated",
taskId: "task-1",
sequence: 3,
metadata: { backgroundCommandRunning: true, backgroundCommandTaskId: "task-1" },
}),
})
await waitFor(() => {
expect(screen.getByTestId("latest-message").textContent).toBe("hello world")
expect(screen.getByTestId("background-command").textContent).toBe("true")
})
})
})
@@ -1,5 +1,4 @@
import { DEFAULT_AUTO_APPROVAL_SETTINGS } from "@shared/AutoApprovalSettings"
import { findLastIndex } from "@shared/array"
import { DEFAULT_BROWSER_SETTINGS } from "@shared/BrowserSettings"
import { DEFAULT_PLATFORM, type ExtensionState } from "@shared/ExtensionMessage"
import { DEFAULT_FOCUS_CHAIN_SETTINGS } from "@shared/FocusChainSettings"
@@ -26,7 +25,14 @@ import {
} from "../../../src/shared/api"
import { Environment } from "../../../src/shared/config-types"
import type { McpMarketplaceCatalog, McpServer, McpViewTab } from "../../../src/shared/mcp"
import type { TaskUiDelta } from "../../../src/shared/TaskUiDelta"
import { McpServiceClient, ModelsServiceClient, StateServiceClient, UiServiceClient } from "../services/grpc-client"
import { mergeExtensionStateSnapshot } from "./mergeExtensionState"
import { mergePartialMessage } from "./mergePartialMessage"
import { ensureDebugTaskUiCounters, incrementDebugTaskUiCounter } from "./taskUiDebugCounters"
import { applyTaskUiDeltaToState } from "./taskUiDeltaState"
const IS_DEV = process.env.IS_DEV === '"true"'
export interface ExtensionStateContextType extends ExtensionState {
didHydrateState: boolean
@@ -319,6 +325,7 @@ export const ExtensionStateContextProvider: React.FC<{
const [huggingFaceModels, setHuggingFaceModels] = useState<Record<string, ModelInfo>>({})
const [mcpServers, setMcpServers] = useState<McpServer[]>([])
const [mcpMarketplaceCatalog, setMcpMarketplaceCatalog] = useState<McpMarketplaceCatalog>({ items: [] })
const latestTaskUiDeltaSequenceRef = useRef<number>(0)
// References to store subscription cancellation functions
const stateSubscriptionRef = useRef<(() => void) | null>(null)
@@ -330,6 +337,7 @@ export const ExtensionStateContextProvider: React.FC<{
const settingsButtonClickedSubscriptionRef = useRef<(() => void) | null>(null)
const worktreesButtonClickedSubscriptionRef = useRef<(() => void) | null>(null)
const partialMessageUnsubscribeRef = useRef<(() => void) | null>(null)
const taskUiDeltaUnsubscribeRef = useRef<(() => void) | null>(null)
const mcpMarketplaceUnsubscribeRef = useRef<(() => void) | null>(null)
const openRouterModelsUnsubscribeRef = useRef<(() => void) | null>(null)
const liteLlmModelsUnsubscribeRef = useRef<(() => void) | null>(null)
@@ -348,6 +356,24 @@ export const ExtensionStateContextProvider: React.FC<{
}, [])
const mcpServersSubscriptionRef = useRef<(() => void) | null>(null)
const resyncCurrentTaskState = useCallback(async () => {
try {
const latestState = await StateServiceClient.getLatestState(EmptyRequest.create({}))
if (!latestState.stateJson) {
return
}
const stateData = JSON.parse(latestState.stateJson) as ExtensionState
setState((prevState) => ({
...prevState,
...stateData,
}))
latestTaskUiDeltaSequenceRef.current = 0
} catch (error) {
console.error("Failed to resync extension state after task delta gap:", error)
}
}, [])
// Subscribe to state updates and UI events using the gRPC streaming API
useEffect(() => {
// Set up state subscription
@@ -355,25 +381,14 @@ export const ExtensionStateContextProvider: React.FC<{
onResponse: (response) => {
if (response.stateJson) {
try {
incrementDebugTaskUiCounter(
IS_DEV,
typeof window === "undefined" ? undefined : window,
"fullStateApplications",
)
const stateData = JSON.parse(response.stateJson) as ExtensionState
setState((prevState) => {
// Versioning logic for autoApprovalSettings
const incomingVersion = stateData.autoApprovalSettings?.version ?? 1
const currentVersion = prevState.autoApprovalSettings?.version ?? 1
const shouldUpdateAutoApproval = incomingVersion > currentVersion
// HACK: Preserve clineMessages if currentTaskItem is the same
if (stateData.currentTaskItem?.id === prevState.currentTaskItem?.id) {
stateData.clineMessages = stateData.clineMessages?.length
? stateData.clineMessages
: prevState.clineMessages
}
const newState = {
...stateData,
autoApprovalSettings: shouldUpdateAutoApproval
? stateData.autoApprovalSettings
: prevState.autoApprovalSettings,
}
const newState = mergeExtensionStateSnapshot(prevState, stateData)
// Update welcome screen state based on API configuration if welcome view not in progress
if (!newState.welcomeViewCompleted && !showWelcome) {
@@ -385,6 +400,7 @@ export const ExtensionStateContextProvider: React.FC<{
}
setDidHydrateState(true)
latestTaskUiDeltaSequenceRef.current = 0
return newState
})
@@ -512,16 +528,12 @@ export const ExtensionStateContextProvider: React.FC<{
}
const partialMessage = convertProtoToClineMessage(protoMessage)
setState((prevState) => {
// worth noting it will never be possible for a more up-to-date message to be sent here or in normal messages post since the presentAssistantContent function uses lock
const lastIndex = findLastIndex(prevState.clineMessages, (msg) => msg.ts === partialMessage.ts)
if (lastIndex !== -1) {
const newClineMessages = [...prevState.clineMessages]
newClineMessages[lastIndex] = partialMessage
return { ...prevState, clineMessages: newClineMessages }
}
return prevState
})
incrementDebugTaskUiCounter(
IS_DEV,
typeof window === "undefined" ? undefined : window,
"partialMessageApplications",
)
setState((prevState) => mergePartialMessage(prevState, partialMessage))
} catch (error) {
console.error("Failed to process partial message:", error, protoMessage)
}
@@ -534,6 +546,47 @@ export const ExtensionStateContextProvider: React.FC<{
},
})
taskUiDeltaUnsubscribeRef.current = UiServiceClient.subscribeToTaskUiDeltas(EmptyRequest.create({}), {
onResponse: (response: { deltaJson?: string }) => {
if (!response.deltaJson) {
return
}
try {
const delta = JSON.parse(response.deltaJson) as TaskUiDelta
setState((prevState) => {
const result = applyTaskUiDeltaToState(prevState, delta, latestTaskUiDeltaSequenceRef.current)
const counters = ensureDebugTaskUiCounters(IS_DEV, typeof window === "undefined" ? undefined : window)
latestTaskUiDeltaSequenceRef.current = result.nextSequence
if (result.kind === "resync") {
if (counters) {
counters.taskUiDeltaResyncRequests += 1
}
void resyncCurrentTaskState()
return prevState
}
if (result.kind === "ignored") {
return prevState
}
if (counters) {
counters.taskUiDeltaApplications += 1
}
return result.state
})
} catch (error) {
console.error("Failed to process task UI delta:", error)
}
},
onError: (error: unknown) => {
const typedError = error as Error
console.error("Error in taskUiDelta subscription:", typedError)
},
onComplete: () => {
console.log("taskUiDelta subscription completed")
},
})
// Subscribe to MCP marketplace catalog updates
mcpMarketplaceUnsubscribeRef.current = McpServiceClient.subscribeToMcpMarketplaceCatalog(EmptyRequest.create({}), {
onResponse: (catalog) => {
@@ -660,6 +713,10 @@ export const ExtensionStateContextProvider: React.FC<{
partialMessageUnsubscribeRef.current()
partialMessageUnsubscribeRef.current = null
}
if (taskUiDeltaUnsubscribeRef.current) {
taskUiDeltaUnsubscribeRef.current()
taskUiDeltaUnsubscribeRef.current = null
}
if (mcpMarketplaceUnsubscribeRef.current) {
mcpMarketplaceUnsubscribeRef.current()
mcpMarketplaceUnsubscribeRef.current = null
@@ -685,7 +742,17 @@ export const ExtensionStateContextProvider: React.FC<{
mcpServersSubscriptionRef.current = null
}
}
}, [])
}, [
closeMcpView,
navigateToAccount,
navigateToChat,
navigateToHistory,
navigateToMcp,
navigateToSettings,
navigateToWorktrees,
resyncCurrentTaskState,
showWelcome,
])
const refreshOpenRouterModels = useCallback(() => {
ModelsServiceClient.refreshOpenRouterModelsRpc(EmptyRequest.create({}))
@@ -0,0 +1,123 @@
import { describe, expect, it } from "vitest"
import type { ExtensionState } from "../../../src/shared/ExtensionMessage"
import { mergeExtensionStateSnapshot } from "./mergeExtensionState"
const createState = (): ExtensionState =>
({
version: "test",
clineMessages: [],
taskHistory: [],
shouldShowAnnouncement: false,
autoApprovalSettings: { enabled: false, actions: {}, version: 1 },
browserSettings: { viewport: "desktop", screencast: true },
focusChainSettings: { enabled: false, reminderIntervalRequests: 5 },
preferredLanguage: "English",
mode: "act",
platform: "macOS",
environment: "production",
telemetrySetting: "unset",
distinctId: "distinct-id",
planActSeparateModelsSetting: true,
enableCheckpointsSetting: true,
mcpDisplayMode: "sidebar",
globalClineRulesToggles: {},
localClineRulesToggles: {},
localCursorRulesToggles: {},
localWindsurfRulesToggles: {},
localAgentsRulesToggles: {},
localWorkflowToggles: {},
globalWorkflowToggles: {},
shellIntegrationTimeout: 4_000,
terminalReuseEnabled: true,
vscodeTerminalExecutionMode: "vscodeTerminal",
terminalOutputLineLimit: 500,
maxConsecutiveMistakes: 3,
defaultTerminalProfile: "default",
isNewUser: false,
welcomeViewCompleted: true,
strictPlanModeEnabled: false,
yoloModeToggled: false,
useAutoCondense: false,
subagentsEnabled: false,
clineWebToolsEnabled: { user: true, featureFlag: false },
worktreesEnabled: { user: true, featureFlag: false },
favoritedModelIds: [],
lastDismissedInfoBannerVersion: 0,
lastDismissedModelBannerVersion: 0,
lastDismissedCliBannerVersion: 0,
remoteConfigSettings: {},
onboardingModels: undefined,
backgroundCommandRunning: false,
backgroundCommandTaskId: undefined,
backgroundEditEnabled: false,
doubleCheckCompletionEnabled: false,
globalSkillsToggles: {},
localSkillsToggles: {},
mcpResponsesCollapsed: false,
customPrompt: undefined,
workspaceRoots: [],
primaryRootIndex: 0,
isMultiRootWorkspace: false,
multiRootSetting: { user: false, featureFlag: false },
hooksEnabled: false,
nativeToolCallSetting: false,
enableParallelToolCalling: false,
currentTaskItem: {
id: "task-1",
ts: 1,
task: "demo",
tokensIn: 0,
tokensOut: 0,
cacheWrites: 0,
cacheReads: 0,
totalCost: 0,
size: 0,
cwdOnTaskInitialization: "/workspace",
isFavorited: false,
},
}) as unknown as ExtensionState
describe("mergeExtensionStateSnapshot", () => {
it("returns the previous object when the incoming snapshot is a no-op", () => {
const state = createState()
const merged = mergeExtensionStateSnapshot(state, { ...state, clineMessages: [] })
expect(merged).toBe(state)
})
it("preserves previous messages for the same task when snapshot omits them", () => {
const prev = createState()
prev.clineMessages = [{ ts: 10, type: "say", say: "text", text: "hello" } as any]
const incoming = { ...createState(), currentTaskItem: prev.currentTaskItem, clineMessages: [] }
const merged = mergeExtensionStateSnapshot(prev, incoming)
expect(merged.clineMessages).toBe(prev.clineMessages)
})
it("preserves previous auto-approval settings when the incoming version is older", () => {
const prev = createState()
prev.autoApprovalSettings = { enabled: true, actions: { readFiles: true }, version: 3 } as any
const incoming = createState()
incoming.autoApprovalSettings = { enabled: false, actions: {}, version: 1 } as any
const merged = mergeExtensionStateSnapshot(prev, incoming)
expect(merged.autoApprovalSettings).toBe(prev.autoApprovalSettings)
})
it("rehydrates cline messages from a full snapshot when reopening or resubscribing to a different snapshot payload", () => {
const prev = createState()
prev.clineMessages = [{ ts: 10, type: "say", say: "text", text: "stale local copy" } as any]
const incoming = {
...createState(),
currentTaskItem: prev.currentTaskItem,
clineMessages: [{ ts: 20, type: "say", say: "text", text: "fresh hydrated copy" } as any],
}
const merged = mergeExtensionStateSnapshot(prev, incoming)
expect(merged.clineMessages).toEqual(incoming.clineMessages)
expect(merged.clineMessages).not.toBe(prev.clineMessages)
})
})
@@ -0,0 +1,23 @@
import type { ExtensionState } from "@shared/ExtensionMessage"
import deepEqual from "fast-deep-equal"
export function mergeExtensionStateSnapshot(prevState: ExtensionState, incomingState: ExtensionState): ExtensionState {
const incomingVersion = incomingState.autoApprovalSettings?.version ?? 1
const currentVersion = prevState.autoApprovalSettings?.version ?? 1
const shouldUpdateAutoApproval = incomingVersion > currentVersion
const nextClineMessages =
incomingState.currentTaskItem?.id === prevState.currentTaskItem?.id
? incomingState.clineMessages?.length
? incomingState.clineMessages
: prevState.clineMessages
: incomingState.clineMessages
const newState = {
...incomingState,
clineMessages: nextClineMessages,
autoApprovalSettings: shouldUpdateAutoApproval ? incomingState.autoApprovalSettings : prevState.autoApprovalSettings,
}
return deepEqual(newState, prevState) ? prevState : newState
}
@@ -0,0 +1,150 @@
import { describe, expect, it } from "vitest"
import type { ExtensionState } from "../../../src/shared/ExtensionMessage"
import { mergePartialMessage } from "./mergePartialMessage"
const createState = (): ExtensionState =>
({
version: "test",
clineMessages: [],
taskHistory: [],
shouldShowAnnouncement: false,
autoApprovalSettings: { enabled: false, actions: {}, version: 1 },
browserSettings: { viewport: "desktop", screencast: true },
focusChainSettings: { enabled: false, reminderIntervalRequests: 5 },
preferredLanguage: "English",
mode: "act",
platform: "macOS",
environment: "production",
telemetrySetting: "unset",
distinctId: "distinct-id",
planActSeparateModelsSetting: true,
enableCheckpointsSetting: true,
mcpDisplayMode: "sidebar",
globalClineRulesToggles: {},
localClineRulesToggles: {},
localCursorRulesToggles: {},
localWindsurfRulesToggles: {},
localAgentsRulesToggles: {},
localWorkflowToggles: {},
globalWorkflowToggles: {},
shellIntegrationTimeout: 4_000,
terminalReuseEnabled: true,
vscodeTerminalExecutionMode: "vscodeTerminal",
terminalOutputLineLimit: 500,
maxConsecutiveMistakes: 3,
defaultTerminalProfile: "default",
isNewUser: false,
welcomeViewCompleted: true,
strictPlanModeEnabled: false,
yoloModeToggled: false,
useAutoCondense: false,
subagentsEnabled: false,
clineWebToolsEnabled: { user: true, featureFlag: false },
worktreesEnabled: { user: true, featureFlag: false },
favoritedModelIds: [],
lastDismissedInfoBannerVersion: 0,
lastDismissedModelBannerVersion: 0,
lastDismissedCliBannerVersion: 0,
remoteConfigSettings: {},
onboardingModels: undefined,
backgroundCommandRunning: false,
backgroundCommandTaskId: undefined,
backgroundEditEnabled: false,
doubleCheckCompletionEnabled: false,
globalSkillsToggles: {},
localSkillsToggles: {},
mcpResponsesCollapsed: false,
customPrompt: undefined,
workspaceRoots: [],
primaryRootIndex: 0,
isMultiRootWorkspace: false,
multiRootSetting: { user: false, featureFlag: false },
hooksEnabled: false,
nativeToolCallSetting: false,
enableParallelToolCalling: false,
}) as unknown as ExtensionState
describe("mergePartialMessage", () => {
it("returns previous state when the target message does not exist", () => {
const state = createState()
const merged = mergePartialMessage(state, { ts: 1, type: "say", say: "text", text: "hello" } as any)
expect(merged).toBe(state)
})
it("returns previous state when the partial message is unchanged", () => {
const state = createState()
state.clineMessages = [{ ts: 1, type: "say", say: "text", text: "hello", partial: true } as any]
const merged = mergePartialMessage(state, { ts: 1, type: "say", say: "text", text: "hello", partial: true } as any)
expect(merged).toBe(state)
})
it("replaces the matching message when the partial payload changes", () => {
const state = createState()
state.clineMessages = [{ ts: 1, type: "say", say: "text", text: "hello", partial: true } as any]
const merged = mergePartialMessage(state, { ts: 1, type: "say", say: "text", text: "hello world", partial: true } as any)
expect(merged).not.toBe(state)
expect(merged.clineMessages).not.toBe(state.clineMessages)
expect(merged.clineMessages[0]?.text).toBe("hello world")
})
it("patches only the matching active row and preserves unrelated message references", () => {
const state = createState()
const firstMessage = { ts: 1, type: "say", say: "text", text: "first" } as any
const activeMessage = { ts: 2, type: "say", say: "text", text: "streaming", partial: true } as any
state.clineMessages = [firstMessage, activeMessage]
const merged = mergePartialMessage(state, {
ts: 2,
type: "say",
say: "text",
text: "streaming updated",
partial: true,
} as any)
expect(merged).not.toBe(state)
expect(merged.clineMessages).toHaveLength(2)
expect(merged.clineMessages[0]).toBe(firstMessage)
expect(merged.clineMessages[1]).not.toBe(activeMessage)
expect(merged.clineMessages[1]?.text).toBe("streaming updated")
})
it("preserves message identity and timestamp semantics across partial-to-complete transitions", () => {
const state = createState()
state.clineMessages = [{ ts: 5, type: "say", say: "text", text: "partial", partial: true } as any]
const merged = mergePartialMessage(state, {
ts: 5,
type: "say",
say: "text",
text: "final",
partial: false,
} as any)
expect(merged.clineMessages).toHaveLength(1)
expect(merged.clineMessages[0]?.ts).toBe(5)
expect(merged.clineMessages[0]?.partial).toBe(false)
expect(merged.clineMessages[0]?.text).toBe("final")
})
it("keeps message ordering stable during partial-to-complete transitions to avoid chat flicker", () => {
const state = createState()
const before = { ts: 1, type: "say", say: "text", text: "before" } as any
const streaming = { ts: 2, type: "say", say: "text", text: "partial", partial: true } as any
const after = { ts: 3, type: "say", say: "text", text: "after" } as any
state.clineMessages = [before, streaming, after]
const merged = mergePartialMessage(state, {
ts: 2,
type: "say",
say: "text",
text: "complete",
partial: false,
} as any)
expect(merged.clineMessages).toHaveLength(3)
expect(merged.clineMessages[0]).toBe(before)
expect(merged.clineMessages[1]?.ts).toBe(2)
expect(merged.clineMessages[1]?.text).toBe("complete")
expect(merged.clineMessages[2]).toBe(after)
})
})
@@ -0,0 +1,18 @@
import { findLastIndex } from "@shared/array"
import type { ClineMessage, ExtensionState } from "@shared/ExtensionMessage"
import deepEqual from "fast-deep-equal"
export function mergePartialMessage(prevState: ExtensionState, partialMessage: ClineMessage): ExtensionState {
const lastIndex = findLastIndex(prevState.clineMessages, (msg) => msg.ts === partialMessage.ts)
if (lastIndex === -1) {
return prevState
}
if (deepEqual(prevState.clineMessages[lastIndex], partialMessage)) {
return prevState
}
const newClineMessages = [...prevState.clineMessages]
newClineMessages[lastIndex] = partialMessage
return { ...prevState, clineMessages: newClineMessages }
}
@@ -0,0 +1,38 @@
import { describe, expect, it } from "vitest"
import { ensureDebugTaskUiCounters, incrementDebugTaskUiCounter } from "./taskUiDebugCounters"
describe("taskUiDebugCounters", () => {
it("does not initialize counters outside development mode", () => {
const targetWindow = {} as Window
expect(ensureDebugTaskUiCounters(false, targetWindow)).toBeUndefined()
expect(targetWindow.__CLINE_DEBUG_TASK_UI_COUNTERS__).toBeUndefined()
})
it("initializes counters once in development mode", () => {
const targetWindow = {} as Window
const first = ensureDebugTaskUiCounters(true, targetWindow)
const second = ensureDebugTaskUiCounters(true, targetWindow)
expect(first).toEqual({
fullStateApplications: 0,
partialMessageApplications: 0,
taskUiDeltaApplications: 0,
taskUiDeltaResyncRequests: 0,
})
expect(second).toBe(first)
})
it("increments a named counter in development mode", () => {
const targetWindow = {} as Window
incrementDebugTaskUiCounter(true, targetWindow, "taskUiDeltaApplications")
incrementDebugTaskUiCounter(true, targetWindow, "taskUiDeltaApplications")
incrementDebugTaskUiCounter(true, targetWindow, "taskUiDeltaResyncRequests")
expect(targetWindow.__CLINE_DEBUG_TASK_UI_COUNTERS__).toEqual({
fullStateApplications: 0,
partialMessageApplications: 0,
taskUiDeltaApplications: 2,
taskUiDeltaResyncRequests: 1,
})
})
})
@@ -0,0 +1,43 @@
export type DebugTaskUiCounters = {
fullStateApplications: number
partialMessageApplications: number
taskUiDeltaApplications: number
taskUiDeltaResyncRequests: number
}
export type DebugTaskUiCounterKey = keyof DebugTaskUiCounters
declare global {
interface Window {
__CLINE_DEBUG_TASK_UI_COUNTERS__?: DebugTaskUiCounters
}
}
export function ensureDebugTaskUiCounters(isDev: boolean, targetWindow: Window | undefined): DebugTaskUiCounters | undefined {
if (!isDev || !targetWindow) {
return undefined
}
targetWindow.__CLINE_DEBUG_TASK_UI_COUNTERS__ ??= {
fullStateApplications: 0,
partialMessageApplications: 0,
taskUiDeltaApplications: 0,
taskUiDeltaResyncRequests: 0,
}
return targetWindow.__CLINE_DEBUG_TASK_UI_COUNTERS__
}
export function incrementDebugTaskUiCounter(
isDev: boolean,
targetWindow: Window | undefined,
key: DebugTaskUiCounterKey,
): DebugTaskUiCounters | undefined {
const counters = ensureDebugTaskUiCounters(isDev, targetWindow)
if (!counters) {
return undefined
}
counters[key] += 1
return counters
}
@@ -0,0 +1,359 @@
import { describe, expect, it } from "vitest"
import type { ExtensionState } from "../../../src/shared/ExtensionMessage"
import type { TaskUiDelta } from "../../../src/shared/TaskUiDelta"
import { applyTaskUiDeltaToState } from "./taskUiDeltaState"
const createState = (): ExtensionState =>
({
version: "test",
clineMessages: [],
taskHistory: [],
shouldShowAnnouncement: false,
autoApprovalSettings: { enabled: false, actions: {}, version: 1 },
browserSettings: { viewport: "desktop", screencast: true },
focusChainSettings: { enabled: false, reminderIntervalRequests: 5 },
preferredLanguage: "English",
mode: "act",
platform: "macOS",
environment: "production",
telemetrySetting: "unset",
distinctId: "distinct-id",
planActSeparateModelsSetting: true,
enableCheckpointsSetting: true,
mcpDisplayMode: "sidebar",
globalClineRulesToggles: {},
localClineRulesToggles: {},
localCursorRulesToggles: {},
localWindsurfRulesToggles: {},
localAgentsRulesToggles: {},
localWorkflowToggles: {},
globalWorkflowToggles: {},
shellIntegrationTimeout: 4_000,
terminalReuseEnabled: true,
vscodeTerminalExecutionMode: "vscodeTerminal",
terminalOutputLineLimit: 500,
maxConsecutiveMistakes: 3,
defaultTerminalProfile: "default",
isNewUser: false,
welcomeViewCompleted: true,
strictPlanModeEnabled: false,
yoloModeToggled: false,
useAutoCondense: false,
subagentsEnabled: false,
clineWebToolsEnabled: { user: true, featureFlag: false },
worktreesEnabled: { user: true, featureFlag: false },
favoritedModelIds: [],
lastDismissedInfoBannerVersion: 0,
lastDismissedModelBannerVersion: 0,
lastDismissedCliBannerVersion: 0,
remoteConfigSettings: {},
onboardingModels: undefined,
backgroundCommandRunning: false,
backgroundCommandTaskId: undefined,
backgroundEditEnabled: false,
doubleCheckCompletionEnabled: false,
globalSkillsToggles: {},
localSkillsToggles: {},
mcpResponsesCollapsed: false,
customPrompt: undefined,
workspaceRoots: [],
primaryRootIndex: 0,
isMultiRootWorkspace: false,
multiRootSetting: { user: false, featureFlag: false },
hooksEnabled: false,
nativeToolCallSetting: false,
enableParallelToolCalling: false,
currentTaskItem: {
id: "task-1",
ts: 1,
task: "demo",
tokensIn: 0,
tokensOut: 0,
cacheWrites: 0,
cacheReads: 0,
totalCost: 0,
size: 0,
cwdOnTaskInitialization: "/workspace",
isFavorited: false,
},
}) as unknown as ExtensionState
const createDelta = (overrides: Partial<TaskUiDelta>): TaskUiDelta =>
({
type: "task_state_resynced",
taskId: "task-1",
sequence: 1,
...overrides,
}) as TaskUiDelta
describe("applyTaskUiDeltaToState", () => {
it("applies added and updated message deltas", () => {
const state = createState()
const added = applyTaskUiDeltaToState(
state,
createDelta({
type: "message_added",
message: { ts: 10, type: "say", say: "text", text: "hello" },
}),
0,
)
expect(added.kind).toBe("applied")
if (added.kind !== "applied") {
throw new Error("expected applied result")
}
expect(added.state.clineMessages).toHaveLength(1)
const updated = applyTaskUiDeltaToState(
added.state,
createDelta({
sequence: 2,
type: "message_updated",
message: { ts: 10, type: "say", say: "text", text: "updated" },
}),
added.nextSequence,
)
expect(updated.kind).toBe("applied")
if (updated.kind !== "applied") {
throw new Error("expected applied result")
}
expect(updated.state.clineMessages[0].text).toBe("updated")
})
it("requests a resync when a sequence gap is detected", () => {
const state = createState()
const result = applyTaskUiDeltaToState(
state,
createDelta({
sequence: 3,
type: "message_added",
message: { ts: 10, type: "say", say: "text", text: "hello" },
}),
1,
)
expect(result).toEqual({ kind: "resync", nextSequence: 1 })
})
it("requests a full snapshot resync when the backend emits a task_state_resynced delta", () => {
const state = createState()
state.clineMessages = [{ ts: 10, type: "say", say: "text", text: "stale local state" } as any]
const result = applyTaskUiDeltaToState(
state,
createDelta({
sequence: 1,
type: "task_state_resynced",
}),
0,
)
expect(result).toEqual({ kind: "resync", nextSequence: 0 })
})
it("ignores deltas for other tasks", () => {
const state = createState()
const result = applyTaskUiDeltaToState(
state,
createDelta({
taskId: "task-2",
type: "message_added",
message: { ts: 10, type: "say", say: "text", text: "hello" },
}),
0,
)
expect(result).toEqual({ kind: "ignored", nextSequence: 1 })
})
it("applies task metadata deltas without replacing the message list", () => {
const state = createState()
state.clineMessages = [{ ts: 10, type: "say", say: "text", text: "hello" }]
const result = applyTaskUiDeltaToState(
state,
createDelta({
type: "task_metadata_updated",
metadata: {
currentFocusChainChecklist: "- [x] done",
backgroundCommandRunning: true,
backgroundCommandTaskId: "task-1",
},
}),
0,
)
expect(result.kind).toBe("applied")
if (result.kind !== "applied") {
throw new Error("expected applied result")
}
expect(result.state.currentFocusChainChecklist).toBe("- [x] done")
expect(result.state.backgroundCommandRunning).toBe(true)
expect(result.state.backgroundCommandTaskId).toBe("task-1")
expect(result.state.clineMessages).toEqual(state.clineMessages)
})
it("preserves state references when a metadata delta does not change values", () => {
const state = createState()
const result = applyTaskUiDeltaToState(
state,
createDelta({
type: "task_metadata_updated",
metadata: {
backgroundCommandRunning: false,
backgroundCommandTaskId: undefined,
},
}),
0,
)
expect(result.kind).toBe("applied")
if (result.kind !== "applied") {
throw new Error("expected applied result")
}
expect(result.state).toBe(state)
})
it("preserves message array reference when an update delta is identical to existing content", () => {
const state = createState()
const existingMessage = { ts: 10, type: "say", say: "text", text: "hello" } as const
state.clineMessages = [existingMessage as any]
const result = applyTaskUiDeltaToState(
state,
createDelta({
type: "message_updated",
message: { ...existingMessage },
}),
0,
)
expect(result.kind).toBe("applied")
if (result.kind !== "applied") {
throw new Error("expected applied result")
}
expect(result.state).toBe(state)
expect(result.state.clineMessages).toBe(state.clineMessages)
})
it("preserves state references when a delete delta targets a missing message", () => {
const state = createState()
state.clineMessages = [{ ts: 10, type: "say", say: "text", text: "hello" } as any]
const result = applyTaskUiDeltaToState(
state,
createDelta({
type: "message_deleted",
messageTs: 999,
}),
0,
)
expect(result.kind).toBe("applied")
if (result.kind !== "applied") {
throw new Error("expected applied result")
}
expect(result.state).toBe(state)
expect(result.state.clineMessages).toBe(state.clineMessages)
})
it("converges to the same final task state as an equivalent full snapshot", () => {
const initialState = createState()
const deltas: TaskUiDelta[] = [
createDelta({
sequence: 1,
type: "message_added",
message: { ts: 10, type: "say", say: "text", text: "hello" },
}),
createDelta({
sequence: 2,
type: "message_added",
message: { ts: 20, type: "say", say: "reasoning", text: "thinking", partial: true },
}),
createDelta({
sequence: 3,
type: "message_updated",
message: { ts: 20, type: "say", say: "reasoning", text: "thinking complete", partial: false },
}),
createDelta({
sequence: 4,
type: "task_metadata_updated",
metadata: {
backgroundCommandRunning: true,
backgroundCommandTaskId: "task-1",
currentFocusChainChecklist: "- [x] streamed",
},
}),
createDelta({
sequence: 5,
type: "message_deleted",
messageTs: 10,
}),
]
let state = initialState
let sequence = 0
for (const delta of deltas) {
const result = applyTaskUiDeltaToState(state, delta, sequence)
expect(result.kind).toBe("applied")
if (result.kind !== "applied") {
throw new Error("expected applied result")
}
state = result.state
sequence = result.nextSequence
}
const expectedSnapshot: ExtensionState = {
...createState(),
clineMessages: [{ ts: 20, type: "say", say: "reasoning", text: "thinking complete", partial: false } as any],
backgroundCommandRunning: true,
backgroundCommandTaskId: "task-1",
currentFocusChainChecklist: "- [x] streamed",
}
expect(state.clineMessages).toEqual(expectedSnapshot.clineMessages)
expect(state.backgroundCommandRunning).toBe(expectedSnapshot.backgroundCommandRunning)
expect(state.backgroundCommandTaskId).toBe(expectedSnapshot.backgroundCommandTaskId)
expect(state.currentFocusChainChecklist).toBe(expectedSnapshot.currentFocusChainChecklist)
})
it("applies ordered delta events sequentially while advancing the cursor", () => {
let state = createState()
let sequence = 0
const orderedDeltas: TaskUiDelta[] = [
createDelta({
sequence: 1,
type: "message_added",
message: { ts: 100, type: "say", say: "text", text: "first" },
}),
createDelta({
sequence: 2,
type: "message_updated",
message: { ts: 100, type: "say", say: "text", text: "first updated" },
}),
createDelta({
sequence: 3,
type: "task_metadata_updated",
metadata: { backgroundCommandRunning: true },
}),
]
for (const delta of orderedDeltas) {
const result = applyTaskUiDeltaToState(state, delta, sequence)
expect(result.kind).toBe("applied")
if (result.kind !== "applied") {
throw new Error("expected applied result")
}
state = result.state
sequence = result.nextSequence
}
expect(sequence).toBe(3)
expect(state.clineMessages).toEqual([{ ts: 100, type: "say", say: "text", text: "first updated" }])
expect(state.backgroundCommandRunning).toBe(true)
})
})
@@ -0,0 +1,97 @@
import { findLastIndex } from "@shared/array"
import type { ExtensionState } from "@shared/ExtensionMessage"
import type { TaskUiDelta } from "@shared/TaskUiDelta"
import deepEqual from "fast-deep-equal"
export type TaskUiDeltaApplicationResult =
| { kind: "ignored"; nextSequence: number }
| { kind: "resync"; nextSequence: number }
| { kind: "applied"; nextSequence: number; state: ExtensionState }
export function applyTaskUiDeltaToState(
state: ExtensionState,
delta: TaskUiDelta,
latestSequence: number,
): TaskUiDeltaApplicationResult {
const expectedSequence = latestSequence + 1
if (delta.sequence !== expectedSequence) {
return { kind: "resync", nextSequence: latestSequence }
}
if (delta.taskId !== state.currentTaskItem?.id) {
return { kind: "ignored", nextSequence: delta.sequence }
}
if (delta.type === "task_state_resynced") {
return { kind: "resync", nextSequence: 0 }
}
if (delta.type === "task_metadata_updated") {
const metadataChanged = Object.entries(delta.metadata).some(([key, value]) => {
return !deepEqual(state[key as keyof ExtensionState], value)
})
if (!metadataChanged) {
return { kind: "applied", nextSequence: delta.sequence, state }
}
return {
kind: "applied",
nextSequence: delta.sequence,
state: {
...state,
...delta.metadata,
},
}
}
if (delta.type === "message_deleted") {
const hasMessageToDelete = state.clineMessages.some((message) => message.ts === delta.messageTs)
if (!hasMessageToDelete) {
return {
kind: "applied",
nextSequence: delta.sequence,
state,
}
}
return {
kind: "applied",
nextSequence: delta.sequence,
state: {
...state,
clineMessages: state.clineMessages.filter((message) => message.ts !== delta.messageTs),
},
}
}
const existingIndex = findLastIndex(state.clineMessages, (message) => message.ts === delta.message.ts)
if (existingIndex === -1) {
return {
kind: "applied",
nextSequence: delta.sequence,
state: {
...state,
clineMessages: [...state.clineMessages, delta.message],
},
}
}
if (deepEqual(state.clineMessages[existingIndex], delta.message)) {
return {
kind: "applied",
nextSequence: delta.sequence,
state,
}
}
const clineMessages = [...state.clineMessages]
clineMessages[existingIndex] = delta.message
return {
kind: "applied",
nextSequence: delta.sequence,
state: {
...state,
clineMessages,
},
}
}