blog(agent-as-yjs-peer): "The Agent Is Just Another Peer" post + live demo (#6268)

* blog(agent-as-yjs-peer): draft post on streaming an agent into a collaborative doc as a Yjs peer

* blog(agent-as-yjs-peer): add cover image

* blog(agent-as-yjs-peer): add live two-peer editing demo component

* improvement(blog): match code blocks and inline code to the in-app editor tokens

* blog(agent-as-yjs-peer): finalize copy, streaming demo, and feature the post

* fix(blog): guard agent demo against missing/never-firing IntersectionObserver

* fix(blog): stream demo bullets and code chip per-character to match the typing cadence

* chore(blog): trim demo comments and apply cleanup (adopt cn, hoist languageMap, chip color)

* blog(agent-as-yjs-peer): swap cover image

* fix(blog): guard agent demo rAF loop against rescheduling after effect cleanup
This commit is contained in:
Waleed
2026-08-04 16:20:07 -07:00
committed by GitHub
parent b98dd8ba5e
commit bf5abc9a40
7 changed files with 646 additions and 50 deletions
@@ -0,0 +1 @@
export { AgentPeerDemo } from './components/agent-peer-demo'
@@ -0,0 +1,475 @@
'use client'
import { useEffect, useRef, useState } from 'react'
/**
* A small illustration for the post: two peers editing one document at once. A teammate ("Zoe")
* extends the intro line while the agent ("Sim") writes a short formatted block below (bold lead,
* code chip, bullet list). Neither overwrites the other. A scripted animation, but faithful to the
* idea: both carets advance independently and both edits land.
*
* Styled from the same design tokens as the real markdown editor, with carets replicating
* `CollaborationCaret` and identity colors from `USER_COLORS`. The finished document is always laid
* out; unrevealed text is kept `visibility: hidden` so the card never resizes as content streams in.
* Degrades to the finished document with JavaScript off or reduced motion.
*/
type Seg = { t: string } | { code: string } | { b: string }
interface Block {
kind: 'p' | 'li'
segs: Seg[]
}
const FONT_SANS = 'var(--font-inter, ui-sans-serif, system-ui, -apple-system, sans-serif)'
const FONT_MONO = 'var(--font-martian-mono, ui-monospace, SFMono-Regular, Menlo, monospace)'
// Colors from the real identity palette (USER_COLORS / getUserColor).
const AGENT = { name: 'Sim', color: '#60C5FF' }
const HUMAN = { name: 'Zoe', color: '#F472B6' }
// The intro line already exists; the teammate is still appending to it.
const HUMAN_BASE = 'Rollout is set for Friday.'
const HUMAN_ADD = ' Ops signed off this morning.'
// The agent's block, streamed one char at a time, arriving already formatted.
const AGENT_BLOCKS: Block[] = [
{
kind: 'p',
segs: [{ b: 'Load test: ' }, { t: 'p95 held at ' }, { code: '180ms' }, { t: ', errors flat.' }],
},
{ kind: 'li', segs: [{ t: 'Rollback path verified' }] },
{ kind: 'li', segs: [{ t: 'On-call paged and acked' }] },
]
const segText = (s: Seg): string => ('t' in s ? s.t : 'code' in s ? s.code : s.b)
const AGENT_FLAT = AGENT_BLOCKS.flatMap((b) => b.segs.map(segText)).join('')
const HUMAN_TOTAL = HUMAN_ADD.length
const AGENT_TOTAL = AGENT_FLAT.length
/**
* Per-character reveal timeline with a deterministic (no-RNG) speed wobble plus pauses at spaces and
* punctuation, so the server and client agree and it never jitters between renders. `times[i]` is the
* ms at which character `i` appears.
*/
function buildSchedule(
text: string,
o: { base: number; wobble: number; space: number; punct: number; startAt?: number }
): number[] {
const times: number[] = []
let t = o.startAt ?? 0
for (let i = 0; i < text.length; i++) {
const ch = text[i]
const wob = 1 + o.wobble * (0.6 * Math.sin(i * 1.7 + 0.5) + 0.4 * Math.sin(i * 0.53 + 1.1))
t += o.base * wob
if (ch === ' ') t += o.space
else if (ch === '.' || ch === ',' || ch === ':') t += o.punct
times.push(t)
}
return times
}
// Agent streams faster and steadier; the human is slower and starts a beat later, so the two never lock in sync.
const AGENT_TIMES = buildSchedule(AGENT_FLAT, { base: 34, wobble: 0.5, space: 45, punct: 150 })
const HUMAN_TIMES = buildSchedule(HUMAN_ADD, {
base: 58,
wobble: 0.45,
space: 65,
punct: 150,
startAt: 280,
})
const lastTime = (a: number[]) => (a.length ? a[a.length - 1] : 0)
const countReached = (times: number[], t: number): number => {
let n = 0
while (n < times.length && times[n] <= t) n++
return n
}
// RUN_MS: when the last character lands. The stream plays once, then clamps to the finished doc and stops.
const RUN_MS = Math.max(lastTime(AGENT_TIMES), lastTime(HUMAN_TIMES))
const proseStyle: React.CSSProperties = {
fontFamily: FONT_SANS,
fontSize: '15px',
fontWeight: 430,
lineHeight: '25px',
letterSpacing: 0,
color: 'var(--text-primary)',
overflowWrap: 'anywhere',
}
const codeStyle: React.CSSProperties = {
fontFamily: FONT_MONO,
fontSize: '0.875em',
background: 'var(--surface-5)',
borderRadius: '4px',
padding: '0.125rem 0.375rem',
}
/** Replicates CollaborationCaret (rich-markdown-editor.css): a 2px identity-colored bar with the
* notched name label above it. Zero inline width, so placing it at the write head never shifts text. */
function Caret({ who }: { who: typeof HUMAN }) {
return (
<span
style={{
position: 'relative',
display: 'inline-block',
width: 0,
height: '1em',
verticalAlign: 'text-bottom',
}}
>
<span
style={{
position: 'absolute',
top: '-0.15em',
height: '1.3em',
left: '-1px',
width: '2px',
background: who.color,
pointerEvents: 'none',
}}
/>
<span
style={{
position: 'absolute',
top: '-1.5em',
left: '-1px',
padding: '0.1rem 0.35rem',
borderRadius: '2px 2px 2px 0',
fontFamily: FONT_SANS,
fontSize: '11px',
fontWeight: 500,
lineHeight: 1.2,
whiteSpace: 'nowrap',
background: who.color,
color: 'var(--surface-1)',
userSelect: 'none',
pointerEvents: 'none',
}}
>
{who.name}
</span>
</span>
)
}
/** Renders a block's segments, revealing up to `revealed` chars and reserving the rest with
* `visibility: hidden` for stable layout. Places the caret at the write head if it falls in this
* block; returns the nodes and whether the caret was placed. */
function renderBlockNodes(
block: Block,
blockStart: number,
revealed: number,
placed: boolean,
isLast: boolean
) {
const nodes: React.ReactNode[] = []
let offset = blockStart
let didPlace = placed
for (let si = 0; si < block.segs.length; si++) {
const seg = block.segs[si]
const raw = segText(seg)
const start = offset
const end = offset + raw.length
const key = `${blockStart}-${si}`
if ('code' in seg) {
// Type the chip char by char with the caret moving through it; the box reserves full width up
// front (hidden tail) so the card never reflows. Revealing it as one chunk froze/jumped the caret.
const reached = revealed >= start
const shown = reached ? Math.min(raw.length, revealed - start) : 0
const atBoundary = !didPlace && reached && shown < raw.length
if (atBoundary) didPlace = true
nodes.push(
<code key={key} style={{ ...codeStyle, visibility: reached ? 'visible' : 'hidden' }}>
{reached ? raw.slice(0, shown) : raw}
{atBoundary && <Caret key={`${key}-c`} who={AGENT} />}
{reached && shown < raw.length && (
<span style={{ visibility: 'hidden' }}>{raw.slice(shown)}</span>
)}
</code>
)
} else {
const shown = Math.max(0, Math.min(raw.length, revealed - start))
const vis = raw.slice(0, shown)
const hid = raw.slice(shown)
const weight = 'b' in seg ? 600 : undefined
if (vis) {
nodes.push(
<span key={`${key}-v`} style={{ fontWeight: weight }}>
{vis}
</span>
)
}
if (!didPlace && revealed >= start && revealed <= end && shown < raw.length) {
nodes.push(<Caret key={`${key}-c`} who={AGENT} />)
didPlace = true
}
if (hid) {
nodes.push(
<span key={`${key}-h`} style={{ fontWeight: weight, visibility: 'hidden' }}>
{hid}
</span>
)
}
}
offset = end
}
// Caret rests at the end of the last block; the peer stays present.
if (isLast && !didPlace) {
nodes.push(<Caret key='end-caret' who={AGENT} />)
didPlace = true
}
return { nodes, placed: didPlace }
}
function AgentDoc({ revealed }: { revealed: number }) {
const out: React.ReactNode[] = []
let list: React.ReactNode[] = []
let offset = 0
let placed = false
const flushList = () => {
if (list.length === 0) return
out.push(
<ul
key={`ul-${out.length}`}
style={{ margin: '0.6em 0 0', paddingLeft: '1.25em', listStyle: 'disc' }}
>
{list}
</ul>
)
list = []
}
for (let bi = 0; bi < AGENT_BLOCKS.length; bi++) {
const block = AGENT_BLOCKS[bi]
const blockStart = offset
const res = renderBlockNodes(
block,
blockStart,
revealed,
placed,
bi === AGENT_BLOCKS.length - 1
)
placed = res.placed
offset = blockStart + block.segs.reduce((m, s) => m + segText(s).length, 0)
if (block.kind === 'li') {
// visibility:hidden on the text still leaves the <li> marker showing, so hide the whole item
// until the caret reaches it. Hidden items still reserve height, so the card never resizes.
list.push(
<li
key={`li-${bi}`}
style={{ marginTop: 0, visibility: revealed >= blockStart ? 'visible' : 'hidden' }}
>
{res.nodes}
</li>
)
} else {
flushList()
out.push(
<p key={`p-${bi}`} style={{ margin: '0.6em 0 0' }}>
{res.nodes}
</p>
)
}
}
flushList()
return <>{out}</>
}
function Avatar({ who, overlap }: { who: typeof HUMAN; overlap: boolean }) {
return (
<span
title={who.name}
style={{
display: 'inline-flex',
alignItems: 'center',
justifyContent: 'center',
width: '18px',
height: '18px',
marginLeft: overlap ? '-6px' : 0,
borderRadius: '999px',
background: who.color,
color: '#fff',
fontFamily: FONT_SANS,
fontSize: '8px',
fontWeight: 600,
border: '2px solid var(--bg)',
}}
>
{who.name[0]}
</span>
)
}
export function AgentPeerDemo() {
// Initial state (SSR / no-JS / reduced-motion) is the finished document; streaming resets it on scroll-in.
const [humanN, setHumanN] = useState(HUMAN_TOTAL)
const [agentN, setAgentN] = useState(AGENT_TOTAL)
const cardRef = useRef<HTMLDivElement>(null)
useEffect(() => {
const reduce =
typeof window !== 'undefined' &&
window.matchMedia?.('(prefers-reduced-motion: reduce)').matches
const el = cardRef.current
// No IntersectionObserver / no element / reduced motion: keep the finished document, never clear it.
if (reduce || !el || typeof IntersectionObserver === 'undefined') return
let raf = 0
let played = false
let cancelled = false
const play = () => {
// Clear to empty only once on-screen, so the doc is never left blank if the observer never fires.
setHumanN(0)
setAgentN(0)
const start = performance.now()
// Frames fire ~16ms but chars land every ~30-60ms; only setState when a count actually changes,
// so React re-renders ~once per character instead of every frame.
let lastH = 0
let lastA = 0
const tick = (now: number) => {
if (cancelled) return // cleanup ran (unmount / Strict Mode re-run); stop rescheduling
const t = now - start
if (t >= RUN_MS) {
if (lastH !== HUMAN_TOTAL) setHumanN(HUMAN_TOTAL)
if (lastA !== AGENT_TOTAL) setAgentN(AGENT_TOTAL)
return // rest on the finished document; no loop
}
const h = countReached(HUMAN_TIMES, t)
const a = countReached(AGENT_TIMES, t)
if (h !== lastH) {
setHumanN(h)
lastH = h
}
if (a !== lastA) {
setAgentN(a)
lastA = a
}
raf = requestAnimationFrame(tick)
}
raf = requestAnimationFrame(tick)
}
const io = new IntersectionObserver(
(entries) => {
if (entries[0]?.isIntersecting && !played) {
played = true
io.disconnect()
play()
}
},
{ threshold: 0.4 }
)
io.observe(el)
return () => {
cancelled = true
io.disconnect()
if (raf) cancelAnimationFrame(raf)
}
}, [])
return (
<div style={{ margin: '32px 0', WebkitFontSmoothing: 'antialiased' } as React.CSSProperties}>
<div
ref={cardRef}
style={{
maxWidth: '560px',
margin: '0 auto',
background: 'var(--bg)',
border: '1px solid var(--border)',
borderRadius: '8px',
overflow: 'hidden',
}}
>
{/* header: filename + presence, mirroring Resource.Header. */}
<div
style={{
display: 'flex',
alignItems: 'center',
justifyContent: 'space-between',
gap: '12px',
minHeight: '48px',
padding: '0 16px',
borderBottom: '1px solid var(--border)',
}}
>
<div style={{ display: 'flex', alignItems: 'center', gap: '8px', minWidth: 0 }}>
<svg
width='14'
height='14'
viewBox='0 0 24 24'
fill='none'
stroke='var(--text-icon)'
strokeWidth='2'
strokeLinecap='round'
strokeLinejoin='round'
aria-hidden='true'
>
<path d='M14 2H6a2 2 0 0 0-2 2v16a2 2 0 0 0 2 2h12a2 2 0 0 0 2-2V8z' />
<path d='M14 2v6h6' />
<path d='M16 13H8M16 17H8M10 9H8' />
</svg>
<span
style={{
fontFamily: FONT_SANS,
fontSize: '14px',
color: 'var(--text-body)',
whiteSpace: 'nowrap',
}}
>
launch-notes.md
</span>
</div>
<span style={{ display: 'inline-flex', alignItems: 'center' }}>
<Avatar who={AGENT} overlap={false} />
<Avatar who={HUMAN} overlap />
</span>
</div>
{/* document body: the real prose scale (see proseStyle). */}
<div style={{ ...proseStyle, padding: '18px 22px 22px' }}>
<div
style={{
margin: 0,
fontSize: '1.6em',
fontWeight: 600,
lineHeight: 1.3,
color: 'var(--text-primary)',
}}
>
Launch notes
</div>
<p style={{ margin: '0.6em 0 0' }}>
{HUMAN_BASE}
<span>{HUMAN_ADD.slice(0, humanN)}</span>
<Caret who={HUMAN} />
<span style={{ visibility: 'hidden' }}>{HUMAN_ADD.slice(humanN)}</span>
</p>
<AgentDoc revealed={agentN} />
</div>
</div>
<div
style={{
maxWidth: '560px',
margin: '10px auto 0',
textAlign: 'center',
fontFamily: FONT_SANS,
fontSize: '13px',
color: 'var(--text-muted)',
}}
>
Sim and a teammate editing the same file at once. Neither one overwrites the other.
</div>
</div>
)
}
@@ -0,0 +1,117 @@
---
slug: agent-as-yjs-peer
title: 'The Agent Is Just Another Peer'
description: How we let an AI agent write into a live collaborative document by making it an ordinary Yjs peer, and the things that turned out to be easy, hard, and surprising.
date: 2026-08-03
updated: 2026-08-03
authors:
- waleed
readingTime: 11
tags: [Yjs, CRDT, Collaboration, AI, ProseMirror, TipTap, Streaming, Architecture]
ogImage: /blog/agent-as-yjs-peer/cover.jpg
canonical: https://www.sim.ai/blog/agent-as-yjs-peer
draft: false
featured: true
faq:
- q: "How does Sim let an AI agent write into a document while people are editing it?"
a: "The agent is treated as an ordinary Yjs peer. Each streamed chunk is reconciled into a private copy of the document, the single resulting CRDT update is captured, and only that update is applied to the live document, the same way an edit from another user would arrive. Concurrent human edits merge with it automatically."
- q: "Does the agent overwrite what collaborators are typing?"
a: "No. The agent applies a minimal update rather than replacing the document, so its operations merge with everyone else's. An agent writing at the bottom leaves an edit at the top alone, and an agent inserting above where someone is typing does not move their text."
- q: "Do you need relative-position anchoring to keep the agent's edits in the right place?"
a: "No, and that was the surprising part. Yjs operations are defined by item identity rather than character offset, so they are already relative. A whole-document diff that produces Yjs operations is robust to concurrent inserts without any explicit anchoring. Anchoring is only needed by systems that insert at absolute offsets."
- q: "How does undo work when an agent and a person are both editing?"
a: "Agent operations are applied under a dedicated transaction origin that the collaboration undo manager does not track, so pressing undo only reverts your own edits and never steps through the agent's output. On other screens the agent's writing arrives as remote updates, which a local undo manager never captures."
- q: "Why is streaming an AI into a rich-text document harder than into a text box?"
a: "A collaborative rich-text document is a shared data structure, not a string, so you cannot append to it. And when the storage format is Markdown, a rich-text editor and a Markdown file disagree about what things like blank lines mean, which you have to reconcile."
---
Sim's files are documents you edit together. Several people can be in the same file at once, with everyone's cursors and changes showing up live, and we wanted the AI agent to be able to write into them too.
Our first attempt at that wiped out people's edits. The agent would generate a sentence, we'd set the document to the new text, and whatever you'd typed a moment earlier would just disappear. It worked fine when you were the only one editing, but the second someone else was typing, the agent kept clobbering their work.
The underlying issue is that a collaborative document isn't really a string you can write into. It's a shared data structure (a [CRDT](https://en.wikipedia.org/wiki/Conflict-free_replicated_data_type)), and everyone editing it is mutating that structure at the same time. So when the agent replaced the whole document on each chunk, it wasn't so much appending its text as overwriting the structure out from under everyone else.
What we needed was for the agent to change the document the way a person does, in small incremental edits that merge with everyone else's, instead of one big replace every time.
<AgentPeerDemo />
## Writing like a keystroke
The editor already knows how to do this. When you type a character, it doesn't ship the whole document to your collaborators; it works out the smallest change your keystroke made and sends only that. Sim's files use [Yjs](https://yjs.dev) for the shared state, and the function that turns an editor change into one of those minimal updates is `updateYFragment`. It's the same function that runs on every keypress.
So we wanted the agent to go through `updateYFragment` as well. The only real question was what to diff against. The live document keeps changing while the agent writes, and if the agent diffs against the live document, a collaborator's edit shows up as a difference the agent will try to reconcile away. The agent would end up quietly undoing the humans.
The way around that is to give the agent its own private copy of the document to work against. When a stream starts, we snapshot the live document into a throwaway replica that nothing else touches:
```ts
export function beginAgentStream(editor) {
const binding = ySyncPluginKey.getState(editor.state)?.binding
if (!binding) return null
const shadow = new Y.Doc()
Y.applyUpdate(shadow, Y.encodeStateAsUpdate(binding.doc)) // snapshot the live doc, once
return { shadow, fragment: shadow.getXmlFragment('default'), meta: null }
}
```
Then, for each chunk the model produces, the agent reconciles that private copy toward the new text, we capture the single update the reconciliation produced, and we apply just that update to the real document, exactly as if it had arrived over the network from another user:
```ts
// simplified; the real version cleans up the listener and handles the no-binding case
export function applyAgentStreamFrame(editor, session, body) {
const target = PMNode.fromJSON(editor.schema, parseMarkdownToDoc(body))
let delta = null
session.shadow.on('update', (update, origin) => {
if (origin === AGENT_STREAM_ORIGIN) delta = update
})
session.shadow.transact(() => {
updateYFragment(session.shadow, session.fragment, target, session.meta)
}, AGENT_STREAM_ORIGIN)
if (delta) Y.applyUpdate(binding.doc, delta, AGENT_STREAM_ORIGIN)
}
```
The private copy is the part that took a while to appreciate. Because it only ever sees the agent's own reconciliations and never anyone else's edits, the difference between one chunk and the next is precisely what the agent changed and nothing else. The agent never forms an opinion about what the whole document should look like. It produces a small, well-scoped change and hands it to the CRDT, which merges it with whatever everyone else is doing the same way it merges any two people's edits. Moment.dev [described arriving at a similar idea from a different direction](https://www.moment.dev/blog/collab-with-ai-is-hard): don't let the agent chase a moving target.
## What you get for free
Once the agent's output is just ordinary CRDT operations, a few things we'd braced for stopped being problems.
The first is concurrent editing, and this was the pleasant surprise. We assumed we would eventually need something like relative-position anchoring so the agent's edits landed correctly while people typed around them. We never did. Yjs operations are defined in terms of item identities rather than character offsets, so they are already relative. If a collaborator inserts a paragraph above where the agent is writing, the agent's operations still land in the right place, because they were never pinned to a numeric position to begin with. We tested this fairly hard: an agent appending at the bottom while someone edits the top, an agent inserting above where someone is typing, and an agent rewriting a paragraph someone else is editing. In every case the two documents converge to the same state instead of one write clobbering the other, which was the whole problem we started with. Systems that insert at absolute character offsets do need anchoring, and [Electric's server-side agent](https://electric.ax/blog/2026/04/08/ai-agents-as-crdt-peers-with-yjs) is a good example of one that uses it, but a whole-document diff never computes an offset, so the problem doesn't come up.
The second is undo. If the agent's output went onto your undo stack, a single undo would walk backward through the model's tokens, which is not what anyone wants. We apply every agent operation under a dedicated transaction origin (`AGENT_STREAM_ORIGIN`, which is just a `Symbol`), and the collaboration undo manager only tracks its own origin. Your undo history has your edits in it and not the agent's. On everyone else's screen the agent's writing arrives as remote updates, which a local undo manager never captures anyway.
The third is the streaming itself. Because each chunk is a real operation on the shared document, a collaborator watching the same file sees the same smooth, formatted stream that the person who triggered it sees, and we didn't write anything to make that happen. The channel that keeps two people in sync is the same channel that carries the stream, so the agent's output reaches everyone through the path a person's keystrokes already take. There is no separate streaming pipe to build or keep in step with it. The one server-side write left is the final save, which turns the finished document back into Markdown on disk.
## Where it got genuinely hard: Markdown
The concurrency turned out to be the easy part. The hard part was that our files are stored as Markdown, and a Markdown file and a rich-text editor don't agree on what a blank line means.
We found this the way you usually find these things. Someone opened a document, it looked fine for the first section, and then there was an enormous run of blank space with the rest of the content pushed so far down you'd never scroll to it. The file hadn't lost any data. It had picked up close to two thousand empty paragraphs in the middle.
The cause is a real impedance mismatch. In Markdown, a run of blank lines between two blocks is insignificant, and two blank lines mean the same thing as ten. But the editor rebuilds an empty paragraph node for every blank line it sees, to preserve spacing faithfully. So a document that at some point picked up a large run of blank lines, from a paste or from a model that emitted a lot of newlines, would turn that run into a couple thousand real nodes. Because the nodes are real, they get written back out to the Markdown file, parsed into nodes again on the next open, and never go away on their own. It was also behind a subtler problem where the document visibly reflowed about half a second after opening, because the static preview collapses empty paragraphs while the live editor gives each one a line of height.
The fix was to stop treating blank runs as meaningful when we parse. Markdown says they're insignificant, so we collapse them, which also makes the file render the same in our editor as it does on GitHub or anywhere else it gets opened. We left the serializer alone, since a blank line inside a fenced code block is significant and collapsing that would break code samples.
The part I find most interesting is why we hit this when Notion and Obsidian don't. Notion doesn't store Markdown; its source of truth is blocks, so it can keep empty blocks around without a file to answer to. Obsidian edits the Markdown text directly rather than through a rich-text node tree, so a blank line is just a blank line. We took the harder middle option, a rich-text tree on top of a Markdown file, because we want editing to feel like a document and the storage to be a plain file you can read, diff, and hand to an agent. It's the right tradeoff for us, but it means the normalization between the two representations is our problem to own, and blank lines are the first place that shows up.
## Keeping it fast and honest
A few smaller decisions are worth mentioning.
The server and the client convert Markdown with the exact same code. When a file opens, the server turns its Markdown into the initial CRDT document, and when it saves, it turns the CRDT document back into Markdown. Both directions go through the same parse and serialize functions the editor uses in the browser, and the CRDT step uses the same binding library. There is no second Markdown implementation to drift from the first, so the preview, the collaborative document, and the file on disk can't disagree.
There's also a performance detail on the hot path. Reconciling the shadow needs a mapping between the editor's nodes and the CRDT's, and building that mapping from scratch takes time proportional to the size of the document. But `updateYFragment` maintains the mapping in place as it runs, which is what the editor's own binding relies on for the whole life of a document. So we build it once, on the first chunk, and reuse it for every chunk after. That's only safe because nothing but the agent touches the shadow, and it's a small win that grows with document size.
On the screen that starts the stream, we make the editor read-only until the agent finishes. The agent is already writing there, and letting you type into the same place at the same moment is a fight over the cursor with nothing to gain. Everyone else stays fully editable, because on their screen the agent's writing is just remote updates arriving, the same as another person typing.
The last one is about who applies the stream when the file is open in more than one place. Only one client should write the agent's chunks into the shared document; if two did, every change would land twice. So the clients pick one to do it, over the same awareness channel they already use to show who's present, and everyone else receives the result as ordinary updates.
## How we convinced ourselves
We didn't want to trust any of this on reasoning alone, so we reproduced the two-writer situation with real peers: two editors wired together over Yjs, one running the actual streaming code and the other typing in between the agent's chunks. We ran it for the obvious cases and the awkward ones, including a full rewrite of the document while someone edits a paragraph the rewrite deletes.
That last case taught us something we had wrong. We assumed that if the agent rewrote the document and deleted the paragraph a collaborator was editing, that person's edit would be lost, and we wrote the test asserting exactly that. It failed. Yjs doesn't drop the edit. It keeps the inserted text and reattaches it to whatever survived nearby, so the edit moved instead of vanishing. Our intuition said someone loses in a conflict, and the CRDT quietly chose to lose nothing. We fixed the test to match what actually happens, which is the right order to do it in.
## Takeaways
If you're putting an AI agent into a document people edit together, the most useful move is to stop treating the agent as special. Have it produce the same small operations a keystroke produces, in the CRDT's own language, and let the CRDT do the merging. With Yjs in particular you probably don't need position anchoring, because the operations are already relative. The channel that syncs collaborators is the same one that streams the agent, so there's no reason to build two. And if your editor is rich text while your storage is Markdown, the merging is the easy part, and the real work is deciding, carefully, what the two representations are allowed to disagree about.
+1
View File
@@ -8,6 +8,7 @@ const AUTHORS_DIR = path.join(process.cwd(), 'content', 'authors')
const BLOG_COMPONENT_LOADERS = {
enterprise: () => import('@/content/blog/enterprise/components'),
'v0-5': () => import('@/content/blog/v0-5/components'),
'agent-as-yjs-peer': () => import('@/content/blog/agent-as-yjs-peer/components'),
}
const blogRegistry = createContentRegistry({
+12 -9
View File
@@ -1,21 +1,24 @@
'use client'
import { Code } from '@sim/emcn'
import { Code, chipFieldSurfaceClass, cn } from '@sim/emcn'
interface CodeBlockProps {
code: string
language: 'javascript' | 'json' | 'python'
}
/**
* Blog code block. Renders through emcn's read-only `Code.Viewer` on the same `chipFieldSurfaceClass`
* surface as the in-app custom-tools `CodeEditor`, so a fenced block in a post matches the real editor
* (Prism theme, gutter, font) on a theme-following surface rather than a forced-dark one.
*/
export function CodeBlock({ code, language }: CodeBlockProps) {
return (
<div className='dark w-full overflow-hidden rounded-md border border-[var(--border)] bg-[var(--code-bg)] text-sm'>
<Code.Viewer
code={code}
showGutter
language={language}
className='[&_pre]:!pb-0 m-0 rounded-none border-0 bg-transparent'
/>
</div>
<Code.Viewer
code={code}
showGutter
language={language}
className={cn(chipFieldSurfaceClass, 'w-full overflow-hidden text-sm')}
/>
)
}
+40 -41
View File
@@ -1,5 +1,5 @@
import type { ComponentPropsWithoutRef } from 'react'
import clsx from 'clsx'
import { cn } from '@sim/emcn'
import type { MDXRemoteProps } from 'next-mdx-remote/rsc'
import { CodeBlock } from '@/lib/content/code'
import { SITE_URL } from '@/lib/core/utils/urls'
@@ -28,6 +28,18 @@ function isExternalHref(href: string | undefined): boolean {
}
}
const LANGUAGE_MAP: Record<string, 'javascript' | 'json' | 'python'> = {
js: 'javascript',
jsx: 'javascript',
ts: 'javascript',
tsx: 'javascript',
typescript: 'javascript',
javascript: 'javascript',
json: 'json',
python: 'python',
py: 'python',
}
export const mdxComponents: MDXRemoteProps['components'] = {
img: (props: any) => (
<ContentImage
@@ -42,7 +54,7 @@ export const mdxComponents: MDXRemoteProps['components'] = {
<h2
{...props}
style={{ fontSize: '30px', marginTop: '3rem', marginBottom: '1.5rem' }}
className={clsx('font-medium text-[var(--text-primary)] leading-tight', className)}
className={cn('font-medium text-[var(--text-primary)] leading-tight', className)}
>
{children}
</h2>
@@ -51,7 +63,7 @@ export const mdxComponents: MDXRemoteProps['components'] = {
<h3
{...props}
style={{ fontSize: '24px', marginTop: '1.5rem', marginBottom: '0.75rem' }}
className={clsx('font-medium text-[var(--text-primary)] leading-tight', className)}
className={cn('font-medium text-[var(--text-primary)] leading-tight', className)}
>
{children}
</h3>
@@ -60,7 +72,7 @@ export const mdxComponents: MDXRemoteProps['components'] = {
<h4
{...props}
style={{ fontSize: '19px', marginTop: '1.5rem', marginBottom: '0.75rem' }}
className={clsx('font-medium text-[var(--text-primary)] leading-tight', className)}
className={cn('font-medium text-[var(--text-primary)] leading-tight', className)}
>
{children}
</h4>
@@ -69,14 +81,14 @@ export const mdxComponents: MDXRemoteProps['components'] = {
<p
{...props}
style={{ fontSize: '19px', marginBottom: '1.5rem', fontWeight: '400' }}
className={clsx('text-[var(--text-body)] leading-relaxed', props.className)}
className={cn('text-[var(--text-body)] leading-relaxed', props.className)}
/>
),
ul: (props: any) => (
<ul
{...props}
style={{ fontSize: '19px', marginBottom: '1rem', fontWeight: '400' }}
className={clsx(
className={cn(
'list-outside list-disc pl-6 text-[var(--text-body)] leading-relaxed',
props.className
)}
@@ -86,26 +98,26 @@ export const mdxComponents: MDXRemoteProps['components'] = {
<ol
{...props}
style={{ fontSize: '19px', marginBottom: '1rem', fontWeight: '400' }}
className={clsx(
className={cn(
'list-outside list-decimal pl-6 text-[var(--text-body)] leading-relaxed',
props.className
)}
/>
),
li: (props: any) => <li {...props} className={clsx('mb-1', props.className)} />,
li: (props: any) => <li {...props} className={cn('mb-1', props.className)} />,
strong: (props: any) => (
<strong
{...props}
className={clsx('font-semibold text-[var(--text-primary)]', props.className)}
className={cn('font-semibold text-[var(--text-primary)]', props.className)}
/>
),
em: (props: any) => (
<em {...props} className={clsx('text-[var(--text-muted)] italic', props.className)} />
<em {...props} className={cn('text-[var(--text-muted)] italic', props.className)} />
),
a: (props: any) => {
const isAnchorLink = props.className?.includes('anchor')
if (isAnchorLink) {
return <a {...props} className={clsx('text-inherit no-underline', props.className)} />
return <a {...props} className={cn('text-inherit no-underline', props.className)} />
}
/**
* Outbound citations in post bodies open in a new tab and carry
@@ -116,7 +128,7 @@ export const mdxComponents: MDXRemoteProps['components'] = {
<a
{...props}
{...(isExternal ? { target: '_blank', rel: 'noopener noreferrer' } : {})}
className={clsx(
className={cn(
'font-medium text-[var(--text-primary)] underline hover:text-[var(--text-primary)]',
props.className
)}
@@ -124,27 +136,22 @@ export const mdxComponents: MDXRemoteProps['components'] = {
)
},
/**
* GFM comparison tables run up to nine columns of prose, which no phone can
* fit. Without a scroll container the table sets the page's content width and
* the whole article scrolls sideways, clipping body copy at both edges. The
* wrapper confines that scrolling to the table's own axis.
*
* `min-w` keeps columns from collapsing to one word per line inside the
* scroll area — narrow enough that tables which already fit (two or three
* columns on a tablet, anything on desktop) never gain a scrollbar.
* Wide GFM tables would set the page's content width and scroll the whole article sideways on
* phones; the wrapper confines scrolling to the table's own axis. `min-w` stops columns from
* collapsing to one word per line, but is narrow enough that tables that already fit never scroll.
*/
table: ({ className, ...props }: ComponentPropsWithoutRef<'table'>) => (
<div className='my-6 w-full overflow-x-auto'>
<table {...props} className={clsx('my-0 min-w-[520px]', className)} />
<table {...props} className={cn('my-0 min-w-[520px]', className)} />
</div>
),
figure: (props: any) => (
<figure {...props} className={clsx('my-8 overflow-hidden rounded-lg', props.className)} />
<figure {...props} className={cn('my-8 overflow-hidden rounded-lg', props.className)} />
),
hr: (props: any) => (
<hr
{...props}
className={clsx('my-8 border-[var(--border)]', props.className)}
className={cn('my-8 border-[var(--border)]', props.className)}
style={{ marginBottom: '1.5rem' }}
/>
),
@@ -156,20 +163,7 @@ export const mdxComponents: MDXRemoteProps['components'] = {
const codeContent = child.props.children || ''
const className = child.props.className || ''
const language = className.replace('language-', '') || 'javascript'
const languageMap: Record<string, 'javascript' | 'json' | 'python'> = {
js: 'javascript',
jsx: 'javascript',
ts: 'javascript',
tsx: 'javascript',
typescript: 'javascript',
javascript: 'javascript',
json: 'json',
python: 'python',
py: 'python',
}
const mappedLanguage = languageMap[language.toLowerCase()] || 'javascript'
const mappedLanguage = LANGUAGE_MAP[language.toLowerCase()] || 'javascript'
return (
<div className='not-prose my-6'>
@@ -180,18 +174,23 @@ export const mdxComponents: MDXRemoteProps['components'] = {
</div>
)
}
return <pre {...props} className={clsx('my-4 overflow-x-auto rounded-lg', props.className)} />
return <pre {...props} className={cn('my-4 overflow-x-auto rounded-lg', props.className)} />
},
// Inline code chip, matching the rich markdown editor's `.rich-markdown-prose code`: no color set,
// so it composites over its surrounding context (e.g. a link's blue).
code: (props: any) => {
if (!props.className) {
return (
<code
{...props}
className={clsx(
'rounded bg-[var(--surface-hover)] px-1.5 py-0.5 font-mono font-normal text-[0.9em] text-[var(--text-primary)]',
className={cn(
'rounded-[4px] bg-[var(--surface-5)] px-[0.375rem] py-[0.125rem] font-normal text-[0.875em]',
props.className
)}
style={{ fontWeight: 400 }}
style={{
fontFamily: 'var(--font-martian-mono, ui-monospace, monospace)',
fontWeight: 400,
}}
/>
)
}
Binary file not shown.

After

Width:  |  Height:  |  Size: 63 KiB