diff --git a/.github/docs-sync/edit-prompt.md b/.github/docs-sync/edit-prompt.md index c9a2b86abed..0a3b7a98e1d 100644 --- a/.github/docs-sync/edit-prompt.md +++ b/.github/docs-sync/edit-prompt.md @@ -21,6 +21,7 @@ Hard rules: - Never remove or rename pages. Never document unreleased behavior. Never copy internal PR discussion into the docs; write user-facing documentation. - Do not run git commands and do not commit anything; automation handles git. - Keep the change small and precise. Do not rewrite sections that are already accurate. +- Never create, modify, or delete packages/kilo-docs/LEARNINGS.md. Automation owns that file. When finished, write the summary JSON file named in the batch specifics below: a JSON array with exactly one entry per batch PR, consumed by automation (this file is never committed). Use `action` values like `updated `, `created `, or `skipped`. Example: diff --git a/.github/docs-sync/edit.mjs b/.github/docs-sync/edit.mjs index 167656eda35..15b89734edc 100644 --- a/.github/docs-sync/edit.mjs +++ b/.github/docs-sync/edit.mjs @@ -19,6 +19,7 @@ import fs from "node:fs" import path from "node:path" import { fileURLToPath } from "node:url" import { backoffMsForAttempt, deadline, remainingMs, runKilo, sleepSync } from "./lib.mjs" +import { readLearningsBlock } from "./learn.mjs" const BATCH_SIZE = 5 const ATTEMPTS = 3 @@ -26,7 +27,7 @@ const OUT_DIR = "docs-sync-out" export const SUMMARY_FILE = ".docs-sync-summary.json" const HERE = path.dirname(fileURLToPath(import.meta.url)) -const basePrompt = fs.readFileSync(path.join(HERE, "edit-prompt.md"), "utf8") +const basePrompt = fs.readFileSync(path.join(HERE, "edit-prompt.md"), "utf8") + readLearningsBlock("edit") const model = process.env.EDIT_MODEL if (!model) throw new Error("EDIT_MODEL is required") @@ -58,14 +59,7 @@ function editBatch(batch, index, budgetDeadline) { const triageFile = `${OUT_DIR}/edit-batch-triage-${index}.json` const summaryFile = `${OUT_DIR}/edit-summary-${index}.json` fs.writeFileSync(batchFile, JSON.stringify(batch, null, 2)) - fs.writeFileSync( - triageFile, - JSON.stringify( - batch.map((d) => priority.get(d.url)).filter(Boolean), - null, - 2, - ), - ) + fs.writeFileSync(triageFile, JSON.stringify(batch.map((d) => priority.get(d.url)).filter(Boolean), null, 2)) const prompt = `${basePrompt} @@ -88,7 +82,21 @@ Batch specifics for this run: the PRs to handle are in the attached ${batchFile} // permission.bash map via KILO_CONFIG_CONTENT should replace --auto once the // required shell patterns are stable (see PR #12605 review thread). const result = runKilo({ - args: ["run", "--auto", prompt, "-m", model, "--variant", "high", "--dir", process.cwd(), "-f", batchFile, "-f", triageFile], + args: [ + "run", + "--auto", + prompt, + "-m", + model, + "--variant", + "high", + "--dir", + process.cwd(), + "-f", + batchFile, + "-f", + triageFile, + ], timeoutMs: Math.min(BATCH_TIMEOUT_MS, left), streamStdout: true, label: `edit batch ${index} attempt ${attempt}`, @@ -120,9 +128,7 @@ Batch specifics for this run: the PRs to handle are in the attached ${batchFile} console.warn(`batch ${index}: backing off ${wait / 1000}s before attempt ${attempt + 1}`) sleepSync(wait) } else if (wait > 0) { - console.warn( - `batch ${index}: skipping backoff — remaining budget cannot fit attempt ${attempt + 1} after wait`, - ) + console.warn(`batch ${index}: skipping backoff — remaining budget cannot fit attempt ${attempt + 1} after wait`) } } } diff --git a/.github/docs-sync/learn.mjs b/.github/docs-sync/learn.mjs new file mode 100644 index 00000000000..a6c0dad3d2e --- /dev/null +++ b/.github/docs-sync/learn.mjs @@ -0,0 +1,892 @@ +// kilocode_change - new file + +/** + * Learns general rules of thumb from maintainer corrections to the docs-sync + * bot's rolling pull request, and writes them into packages/kilo-docs/LEARNINGS.md + * so the triage and edit passes follow them on every subsequent run. + * + * Two modes: + * node learn.mjs — extraction: fetch corrections, call the model, validate + * node learn.mjs --apply — apply: write learnings.json into LEARNINGS.md + * + * Env: TRIAGE_MODEL (provider/model, reused), GH_TOKEN (or GITHUB_TOKEN). + * Budget: LEARNINGS_BUDGET_MINUTES (default 10). + * Test hook: DOCS_SYNC_FIXTURE. When set to a fixture JSON path, skips every + * GitHub API call and writes any marker PATCH to .patched instead of + * the network. The workflow never sets it — only selftests do. + * + * Test hook: DOCS_SYNC_BACKOFF_MS replaces wait between extraction retries, same as + * lib.mjs:138 documents for triage.mjs and edit.mjs. + * + * Patch suppression: DRY_RUN=true or LEARNINGS_NO_PATCH=1 suppress the marker PATCH. + */ + +import { execFileSync } from "node:child_process" +import fs from "node:fs" +import path from "node:path" +import { fileURLToPath, pathToFileURL } from "node:url" + +const LEARNINGS_FILE = "packages/kilo-docs/LEARNINGS.md" +const OUT_DIR = "docs-sync-out" +const ATTEMPTS = 2 +const LEARNINGS_BUDGET_MINUTES = Number(process.env.LEARNINGS_BUDGET_MINUTES) || 10 +const EXTRACTION_TIMEOUT_MS = LEARNINGS_BUDGET_MINUTES * 60 * 1000 +const COMMENT_BODY_CAP = 5000 + +const HERE = path.dirname(fileURLToPath(import.meta.url)) + +const LINE_RE = + /^- (?.+?) $/ + +const LEARNED_THROUGH_RE = // + +// Agent-generated strings land in the PR body next to machine-read markers. +// Identical to clean() at upsert-pr.mjs:37. +function clean(value) { + return String(value ?? "") + .replaceAll("", "") +} + +function warn(msg) { + console.warn(`::warning::${msg}`) +} + +function log(msg) { + console.log(msg) +} + +// --- pure exports --- + +/** + * Parse the LEARNINGS.md file text into an entry array. + * Drops lines inside the markers that do not match the format. + */ +export function parseLearnings(text) { + const m = String(text ?? "").match( + /([\s\S]*?)/, + ) + if (!m) return [] + const entries = [] + for (const line of m[1].split("\n")) { + const trimmed = line.trim() + if (!trimmed) continue + const parsed = trimmed.match(LINE_RE) + if (!parsed) { + warn(`LEARNINGS.md: dropping unparseable line: ${trimmed.slice(0, 80)}`) + continue + } + entries.push({ + id: parsed.groups.id, + rule: clean(parsed.groups.rule).replaceAll("\n", " "), + scope: parsed.groups.scope, + source: parsed.groups.source, + date: parsed.groups.date, + }) + } + return entries +} + +/** Render the full LEARNINGS.md file text from an entry array. Deterministic order. */ +export function renderLearnings(entries) { + const list = [...entries].sort((a, b) => { + if (a.date !== b.date) return a.date < b.date ? -1 : 1 + return a.id < b.id ? -1 : a.id > b.id ? 1 : 0 + }) + const lines = list.map( + (e) => + `- ${clean(e.rule).replaceAll("\n", " ")} `, + ) + return [ + "# docs-sync learnings", + "", + "Rules the docs-sync bot learned from maintainer corrections to its rolling pull request.", + "The bot reads this file at the start of every run and follows every rule below.", + "", + "To unlearn a rule, delete its line and commit. The next run reads this file from the", + "branch, so the rule is gone from its input, and the deletion itself is a correction the", + "extraction step is instructed not to undo.", + "", + "", + ...lines, + "", + "", + ].join("\n") +} + +/** Parse the learned-through watermark from a PR body. Returns { commit, comment } with nulls for absent/none. */ +export function parseLearnedThrough(body) { + const m = String(body ?? "").match(LEARNED_THROUGH_RE) + if (!m) return { commit: null, comment: null } + const commit = m[1] === "none" ? null : m[1] + const comment = m[2] === "none" ? null : m[2] + return { commit, comment } +} + +/** Render a single learned-through marker line. */ +export function renderLearnedThrough({ commit, comment }) { + const c = commit ?? "none" + const m = comment ?? "none" + return `` +} + +/** Replace or append the learned-through marker in a PR body. Pure — no API call. */ +export function patchMarkerIntoBody(body, marker) { + const b = String(body ?? "") + if (LEARNED_THROUGH_RE.test(b)) { + return b.replace(LEARNED_THROUGH_RE, marker) + } + return b + "\n" + marker + "\n" +} + +/** + * Extract { add, remove } from raw model stdout. + * Mirrors parseTriageEntries at extract-json.mjs:14-38, adapted for an object. + * `kilo run` prints the assistant message twice; the last copy wins. + * Walk "{" positions from right to left; return the first that parses to an object + * holding an array `add` or an array `remove`. + */ +export function parseDelta(raw) { + const r = String(raw ?? "") + const end = r.lastIndexOf("}") + if (end < 0) return null + + const starts = [] + for (let i = 0; i <= end; i++) { + if (r[i] === "{") starts.push(i) + } + + for (let s = starts.length - 1; s >= 0; s--) { + let parsed + try { + parsed = JSON.parse(r.slice(starts[s], end + 1)) + } catch { + continue + } + if (typeof parsed !== "object" || parsed === null || Array.isArray(parsed)) continue + if (Array.isArray(parsed.add) || Array.isArray(parsed.remove)) { + return { + add: Array.isArray(parsed.add) ? parsed.add : [], + remove: Array.isArray(parsed.remove) ? parsed.remove : [], + } + } + } + return null +} + +/** Cap a review comment body so a single long comment cannot dominate extraction input. */ +function capBody(body) { + const b = String(body ?? "") + if (b.length <= COMMENT_BODY_CAP) return b + return b.slice(0, COMMENT_BODY_CAP) + " [truncated]" +} + +/** Normalize rule text for duplicate comparison: lowercase, strip punctuation and whitespace runs. */ +function norm(text) { + return String(text ?? "") + .toLowerCase() + .replace(/[^\w\s]/g, "") + .replace(/\s+/g, " ") + .trim() +} + +/** + * Validate a delta against the existing entries and constraints. + * Returns { add, remove, rejected }. Never throws. + */ +export function validateDelta(delta, { existing, candidateSources, deletedInWindow }) { + const add = Array.isArray(delta.add) ? delta.add : [] + const remove = Array.isArray(delta.remove) ? delta.remove : [] + const ex = Array.isArray(existing) ? existing : [] + const candidates = Array.isArray(candidateSources) ? candidateSources : [] + const deleted = Array.isArray(deletedInWindow) ? deletedInWindow : [] + + const rejected = [] + const valid = [] + const toRemove = [] + const existingIds = new Set(ex.map((e) => e.id)) + + // Process remove first so toRemove is populated before the add loop checks + // for id collisions with entries listed in remove (criterion 8). + for (const id of remove) { + if (!existingIds.has(id)) { + rejected.push({ entry: { id, remove: id }, reason: `remove target ${id} not in existing entries` }) + } else { + toRemove.push(id) + } + } + + for (const a of add) { + let reason = null + + // Reject null, undefined, and non-object entries before any property access. + if (a === null || a === undefined || typeof a !== "object" || Array.isArray(a)) { + rejected.push({ entry: a, reason: "add entry is null, undefined, or not a plain object" }) + continue + } + + if (!a.rule || String(a.rule).length < 10 || String(a.rule).length > 300) { + reason = "rule text absent, shorter than 10 characters, or longer than 300" + } else if (!["triage", "edit", "both"].includes(a.scope)) { + reason = `invalid scope: ${a.scope}` + } else if (!/^commit:[0-9a-f]{7,40}$/.test(a.source) && !/^comment:\d+$/.test(a.source)) { + reason = `invalid source format: ${a.source}` + } else if (!candidates.includes(a.source)) { + reason = `source ${a.source} not in candidate sources` + } else if (!/^[a-z0-9][a-z0-9-]{2,48}$/.test(a.id)) { + reason = `invalid id format: ${a.id}` + } else if (existingIds.has(a.id) && !toRemove.includes(a.id)) { + reason = `id ${a.id} collides with an existing entry not listed in remove` + } else if (!/^\d{4}-\d{2}-\d{2}$/.test(a.date)) { + reason = `invalid date format: ${a.date}` + } else { + // Check that date is a real calendar date. + const d = new Date(a.date + "T00:00:00Z") + if (Number.isNaN(d.getTime()) || d.toISOString().slice(0, 10) !== a.date) { + reason = `invalid calendar date: ${a.date}` + } + } + + if (reason) { + rejected.push({ entry: a, reason }) + continue + } + + const n = norm(a.rule) + + // Duplicate of an existing entry not being removed. + if (ex.some((e) => norm(e.rule) === n && !remove.includes(e.id))) { + reason = `rule text is a duplicate of an existing entry not listed in remove` + rejected.push({ entry: a, reason }) + continue + } + + // Names a PR, URL, person, or docs page. The URL clause keeps docs-check-links.yml green. + if (String(a.rule).match(/#\d{2,}|https?:\/\/|@[A-Za-z0-9-]|packages\/kilo-docs|\.md\b/)) { + reason = "rule names a PR, URL, person, or docs page" + rejected.push({ entry: a, reason }) + continue + } + + // Duplicate of a rule deleted in this window. + if (deleted.some((d) => norm(d) === n)) { + reason = "rule text matches a line a maintainer deleted in this window" + rejected.push({ entry: a, reason }) + continue + } + + valid.push({ + id: a.id, + rule: clean(String(a.rule)).replaceAll("\n", " "), + scope: a.scope, + source: a.source, + date: a.date, + }) + } + + return { add: valid, remove: toRemove, rejected } +} + +/** Apply a validated delta to an existing entry array. Drops removed ids, appends adds. */ +export function applyDelta(existing, delta) { + const ex = Array.isArray(existing) ? existing : [] + const remove = new Set(Array.isArray(delta.remove) ? delta.remove : []) + const add = Array.isArray(delta.add) ? delta.add : [] + return [...ex.filter((e) => !remove.has(e.id)), ...add] +} + +/** Trust a review comment whose author_association is OWNER, MEMBER, or COLLABORATOR and is not a bot. */ +export function isTrustedComment(comment) { + if (!comment) return false + const login = String(comment.user?.login ?? "") + if (login.endsWith("[bot]")) return false + return ["OWNER", "MEMBER", "COLLABORATOR"].includes(comment.author_association) +} + +/** Render the prompt block for a given scope. Returns "" when no entry matches. */ +export function promptBlock(entries, scope) { + const matches = (Array.isArray(entries) ? entries : []).filter((e) => e.scope === scope || e.scope === "both") + if (matches.length === 0) return "" + return [ + "## Learnings from maintainer corrections", + "", + "Follow every rule below. Each was extracted from a correction a maintainer made to an", + "earlier run of this bot. A rule here outranks a general instruction above when they conflict.", + "", + ...matches.map((e) => `- ${e.rule}`), + ].join("\n") +} + +/** Read a prompt block artifact from docs-sync-out. Returns the content or "" when absent. */ +export function readLearningsBlock(scope) { + const file = `${OUT_DIR}/learnings-${scope}.md` + try { + return fs.readFileSync(file, "utf8") + } catch { + return "" + } +} + +// --- helpers for main --- + +function git(args) { + return execFileSync("git", args, { stdio: ["ignore", "pipe", "inherit"] }) + .toString() + .trim() +} + +// --- main --- + +async function main() { + // Step 0: ensure docs-sync-out exists. collect.mjs:139 is the only other unconditional + // mkdirSync of this directory, and it runs after the learn step. Without this line the + // empty-candidate path throws ENOENT on its first write, continue-on-error swallows it, + // and the feature silently never works. + fs.mkdirSync(OUT_DIR, { recursive: true }) + + if (process.argv.includes("--apply")) { + await apply() + return + } + + await extract() +} + +// --- apply mode --- + +async function apply() { + const learningsPath = `${OUT_DIR}/learnings.json` + if (!fs.existsSync(learningsPath)) { + log("learnings.json absent — extraction was skipped or failed; nothing to apply") + return + } + const entries = JSON.parse(fs.readFileSync(learningsPath, "utf8")) + const file = renderLearnings(entries) + fs.writeFileSync(LEARNINGS_FILE, file) + log(`wrote ${LEARNINGS_FILE} with ${entries.length} entries`) +} + +// --- extraction mode --- + +async function extract() { + // Load fixture when DOCS_SYNC_FIXTURE is set. + const fixturePath = process.env.DOCS_SYNC_FIXTURE + let fixture = null + let patchFile = null + if (fixturePath) { + fixture = JSON.parse(fs.readFileSync(fixturePath, "utf8")) + patchFile = fixturePath + ".patched" + } + + const { api, repo, searchIssues, appendOutput, appendSummary, backoffMsForAttempt, runKilo, sleepSync } = + await import("./lib.mjs") + + let prData + let prBody = "" + let prNumber = "" + let branch = "" + + if (fixture) { + // Fixture mode: skip all API calls. + prData = fixture.pr + prBody = prData.body ?? "" + prNumber = String(prData.number ?? 1) + branch = prData.head?.ref ?? "docs/auto-sync" + } else { + // Step 1: resolve the rolling PR. Use prepare-branch.mjs's selection rule so both + // target the same branch. searchIssues takes prs[0] with no author filter (like + // prepare-branch.mjs:69). But trust the body marker only when authored by + // github-actions[bot] (like watermark.mjs:35). The two rules differ on purpose: + // the branch must match what prepare-branch.mjs will check out, but a body is + // editable so its marker needs the author filter. + const r = repo() + const prs = await searchIssues(`repo:${r} is:pr is:open label:auto-docs sort:created-desc`, { maxPages: 1 }) + if (prs.length === 0) { + log("no open rolling pull request — nothing to learn from") + + // Read existing learnings from main for empty-state artifacts. + let existing = [] + try { + const existingText = git(["show", `origin/main:${LEARNINGS_FILE}`]) + existing = parseLearnings(existingText) + } catch { + existing = [] + } + log(`no-PR existing entries from main: ${existing.length}`) + writeEmptyStateArtifacts(existing) + appendOutput("count", String(existing.length)) + appendSummary("### docs-sync learnings\n\nNo open auto-docs pull request; extraction skipped.") + return + } + prData = await api(`/repos/${r}/pulls/${prs[0].number}`) + prBody = prData.body ?? "" + prNumber = String(prData.number) + branch = prData.head?.ref ?? "docs/auto-sync" + } + + // Step 2: read existing entries. + let existing = [] + let existingText = "" + if (fixture) { + try { + existingText = fs.readFileSync(LEARNINGS_FILE, "utf8") + } catch { + // file absent + } + existing = parseLearnings(existingText) + } else { + try { + existingText = git(["show", `origin/${branch}:${LEARNINGS_FILE}`]) + } catch { + // branch copy absent — fall back to main, then empty. + // Required for the first live run: the rolling branch predates the seeded file. + try { + existingText = git(["show", `origin/main:${LEARNINGS_FILE}`]) + } catch { + existingText = "" + } + } + existing = parseLearnings(existingText) + } + log(`existing entries: ${existing.length}`) + + // Step 3: parse marker. Trust only when authored by github-actions[bot] (like watermark.mjs:35). + let commitWm = null + let commentWm = null + const trusted = prData.user?.login === "github-actions[bot]" + if (trusted) { + ;({ commit: commitWm, comment: commentWm } = parseLearnedThrough(prBody)) + } else { + log("PR author is not github-actions[bot]; ignoring body marker") + } + log(`watermark: commit=${commitWm ?? "none"} comment=${commentWm ?? "none"}`) + + // Step 4: fetch and tip SHA. + let tipSha + if (fixture) { + tipSha = git(["rev-parse", "HEAD"]) + } else { + git(["fetch", "origin", "main", branch]) + tipSha = git(["rev-parse", `origin/${branch}`]) + } + + // Step 5: candidate commits. + let rangeArgs = [`origin/main..origin/${branch}`] + if (fixture) { + // In fixture mode, work from the local repo state. + try { + git(["rev-parse", "--verify", branch]) + rangeArgs = [`origin/main..${branch}`] + } catch { + rangeArgs = [`origin/main..HEAD`] + } + } + + if (commitWm) { + let wmExists = false + try { + git(["cat-file", "-e", `${commitWm}^{commit}`]) + wmExists = true + } catch { + wmExists = false + } + if (wmExists) { + rangeArgs.push(`^${commitWm}`) + } + // A missing watermark commit (force-push, rebase) drops the exclusion. + // The duplicate-rule-text rejection in validateDelta blocks the re-added duplicate. + } + + const logOut = git(["log", "--no-merges", "--format=%H|%ae|%cI|%s", ...rangeArgs]) + const rawCommits = logOut ? logOut.split("\n").filter(Boolean) : [] + + const botEmail = "41898282+github-actions[bot]@users.noreply.github.com" + const candidates = [] + const candidateSources = [] + const deletedInWindow = [] + + for (const line of rawCommits) { + const [sha, email, dateIso] = line.split("|") + // Drop commits authored by the sync job itself (criterion 5). + if (email === botEmail) continue + // Everything reachable from main is already excluded by the range (criterion 6). + + // Get the full file list. + let files = [] + try { + const out = git(["show", "--name-only", "--format=", sha]) + files = out + ? out + .split("\n") + .filter(Boolean) + .filter((f) => f) + : [] + } catch { + continue + } + + // Get the docs-scoped diff and message. + let message = "" + let docDiff = "" + try { + message = git(["show", "--format=%B", "--no-patch", sha]).trim() + docDiff = git(["show", "--format=", sha, "--", "packages/kilo-docs"]) + // Cap diff sizes. + if (docDiff.length > 20000) docDiff = docDiff.slice(0, 20000) + "\n[truncated]" + } catch { + // skip on error + } + + // Drop commits whose docs-scoped diff is empty. + if (!docDiff.trim()) continue + + // Collect deleted rule lines from LEARNINGS.md. + for (const dl of docDiff.split("\n")) { + if (!dl.startsWith("-")) continue + const stripped = dl.slice(1).trim() + const parsed = stripped.match(LINE_RE) + if (parsed) { + deletedInWindow.push(clean(parsed.groups.rule).replaceAll("\n", " ")) + } + } + + // Cap total diff data. + const totalDiff = candidates.reduce((n, c) => n + (c.diff ? c.diff.length : 0), 0) + if (totalDiff > 120000) { + log(`diff cap reached at commit ${sha.slice(0, 7)}; truncating`) + candidates.push({ + source: `commit:${sha.slice(0, 7)}`, + iso: dateIso, + date: dateIso.slice(0, 10), + message, + files, + diff: "[truncated]", + }) + candidateSources.push(`commit:${sha.slice(0, 7)}`) + break + } + + candidates.push({ + source: `commit:${sha.slice(0, 7)}`, + iso: dateIso, + date: dateIso.slice(0, 10), + message, + files, + diff: docDiff, + }) + candidateSources.push(`commit:${sha.slice(0, 7)}`) + } + + // Step 6: candidate comments. + let allComments = [] + let maxCommentAt = "none" + + if (fixture && fixture.comments) { + allComments = fixture.comments + } else if (prNumber) { + const pages = [] + for (let page = 1; page <= 5; page++) { + const batch = await api(`/repos/${repo()}/pulls/${prNumber}/comments?per_page=100&page=${page}`) + pages.push(...batch) + if (batch.length < 100) break + } + allComments = pages + } + + if (allComments.length > 0) { + let max = "" + for (const c of allComments) { + if (c.created_at && c.created_at > max) max = c.created_at + } + maxCommentAt = max || "none" + } + + // Filter trusted comments. + const trustedComments = allComments.filter((c) => { + if (!isTrustedComment(c)) return false + if (commentWm && c.created_at <= commentWm) return false + return true + }) + + // Step 7: correlate comments to commits. + // A comment is a commit's trigger when c.path is in that commit's full file list + // and c.created_at < commit date. The earliest such commit claims it. + // Compare parsed timestamps so different timezone offsets do not skew the ordering. + for (const c of trustedComments) { + let best = null + const cTime = Date.parse(c.created_at) + for (const cc of candidates) { + if (!Array.isArray(cc.files) || !cc.files.includes(c.path)) continue + const ccTime = Date.parse(cc.iso) + if (cTime < ccTime) { + if (!best || ccTime < Date.parse(best.iso)) { + best = cc + } + } + } + if (best) { + best.comment = { + author_association: c.author_association, + path: c.path, + body: capBody(c.body), + } + } else { + candidates.push({ + source: `comment:${c.id}`, + date: (c.created_at ?? "").slice(0, 10), + path: c.path, + body: capBody(c.body), + author_association: c.author_association, + }) + candidateSources.push(`comment:${c.id}`) + } + } + + // Step 8: no candidates → empty delta route. + const hasCandidates = candidates.length > 0 + + if (!hasCandidates) { + log("no candidate corrections; advancing marker with no model call") + writeEmptyStateArtifacts(existing) + appendOutput("count", String(existing.length)) + appendSummary( + `### docs-sync learnings\n\nNo new candidate corrections. Entries: ${existing.length}. Marker route: empty (no candidates).`, + ) + + const marker = renderLearnedThrough({ commit: tipSha, comment: maxCommentAt }) + await patchOrLogMarker({ prBody, prNumber, marker, fixture, patchFile }) + return + } + + // Step 9: write learnings input. + const input = { + existing: existing.map((e) => ({ id: e.id, rule: e.rule, scope: e.scope, source: e.source, date: e.date })), + deleted_in_window: deletedInWindow, + corrections: candidates, + } + const inputFile = `${OUT_DIR}/learnings-input.json` + fs.writeFileSync(inputFile, JSON.stringify(input, null, 2)) + log(`wrote ${inputFile} with ${candidates.length} candidates`) + + // Step 10: call the model. + // Deliberately no --auto. Every input is in the attached file and the output goes to + // stdout, so the agent needs no tool. Omitting --auto makes "the extraction step never + // writes outside LEARNINGS.md" structurally true instead of prompt-deep. triage.mjs:76 + // and edit.mjs:86 carry the opposite comment; do not copy them without updating the reason. + const prompt = fs.readFileSync(path.join(HERE, "learnings-prompt.md"), "utf8") + const model = process.env.TRIAGE_MODEL + if (!model) throw new Error("TRIAGE_MODEL is required") + + const budgetDeadline = Date.now() + EXTRACTION_TIMEOUT_MS + + let raw = null + let lastCause = "extraction failed" + + for (let attempt = 1; attempt <= ATTEMPTS; attempt++) { + const left = Math.max(0, budgetDeadline - Date.now()) + if (left <= 0) { + log("budget exhausted before extraction attempt") + break + } + + const result = runKilo({ + args: ["run", prompt, "-m", model, "--dir", process.cwd(), "-f", inputFile], + timeoutMs: Math.min(EXTRACTION_TIMEOUT_MS, left), + streamStdout: false, + label: "learnings extraction", + }) + + if (result.stdout) { + fs.writeFileSync(`${OUT_DIR}/learnings-raw.txt`, result.stdout) + raw = result.stdout + const delta = parseDelta(raw) + if (delta) break + lastCause = `parseDelta returned null (attempt ${attempt})` + } else { + lastCause = result.timedOut ? "timed out" : `exit ${result.exitCode}` + } + + if (attempt < ATTEMPTS) { + const wait = backoffMsForAttempt(1) // 60s, same as the sibling convention + if (wait > 0) { + log(`backing off ${wait / 1000}s before attempt ${attempt + 1}`) + sleepSync(wait) + } + } + } + + // Step 11: parse and validate. + const delta = raw ? parseDelta(raw) : null + + if (!delta) { + // parseDelta null after every try — retryable unhappy. + warn(`extraction failed: ${lastCause}. Leaving learnings untouched.`) + writeEmptyStateArtifacts(existing) + appendOutput("count", String(existing.length)) + appendSummary( + `### docs-sync learnings\n\nExtraction failed: ${lastCause}. Entries unchanged: ${existing.length}. No marker advance.`, + ) + return + } + + const validated = validateDelta(delta, { existing, candidateSources, deletedInWindow }) + + if (validated.rejected.length > 0) { + for (const r of validated.rejected) { + warn(`rejected: ${r.reason}` + (r.entry?.id ? ` (id=${r.entry.id})` : "")) + } + } + + const nonEmpty = validated.add.length > 0 || validated.remove.length > 0 + + // Step 12: route by outcome (G5 table, exact). + if (nonEmpty) { + // Non-empty validated delta. + const newEntries = applyDelta(existing, { add: validated.add, remove: validated.remove }) + fs.writeFileSync(`${OUT_DIR}/learnings.json`, JSON.stringify(newEntries, null, 2)) + const marker = renderLearnedThrough({ commit: tipSha, comment: maxCommentAt }) + const suppressed = process.env.DRY_RUN === "true" || process.env.LEARNINGS_NO_PATCH === "1" + if (!suppressed) appendOutput("learned_through", marker) + if (suppressed) log(`learned-through output suppressed: ${marker}`) + + const added = validated.add.length + const removed = validated.remove.length + const rejected = validated.rejected.length + log(`delta: +${added} -${removed} (${rejected} rejected)`) + appendSummary( + `### docs-sync learnings\n\n- added: ${added}\n- removed: ${removed}\n- rejected: ${rejected}\n- candidates: ${candidates.length}\n- marker route: upsert (non-empty delta)\n`, + ) + + writePromptArtifacts(newEntries) + appendOutput("count", String(newEntries.length)) + + // Marker rides through LEARNED_THROUGH into upsert-pr.mjs. No direct PATCH. + } else { + // Empty validated delta (nothing added, nothing removed, including every-add-rejected). + log("empty validated delta; advancing marker directly") + fs.writeFileSync(`${OUT_DIR}/learnings.json`, JSON.stringify(existing, null, 2)) + writePromptArtifacts(existing) + appendOutput("count", String(existing.length)) + + const marker = renderLearnedThrough({ commit: tipSha, comment: maxCommentAt }) + await patchOrLogMarker({ prBody, prNumber, marker, fixture, patchFile }) + + const rejected = validated.rejected.length + appendSummary( + `### docs-sync learnings\n\n- added: 0\n- removed: 0\n- rejected: ${rejected}\n- candidates: ${candidates.length}\n- marker route: direct PATCH (empty delta)\n`, + ) + } +} + +// --- shared helpers --- + +function writeEmptyStateArtifacts(entries) { + fs.writeFileSync(`${OUT_DIR}/learnings.json`, JSON.stringify(entries, null, 2)) + writePromptArtifacts(entries) +} + +function writePromptArtifacts(entries) { + const triage = promptBlock(entries, "triage") + if (triage) fs.writeFileSync(`${OUT_DIR}/learnings-triage.md`, triage) + const edit = promptBlock(entries, "edit") + if (edit) fs.writeFileSync(`${OUT_DIR}/learnings-edit.md`, edit) +} + +async function patchOrLogMarker({ prBody, prNumber, marker, fixture, patchFile }) { + const suppressed = process.env.DRY_RUN === "true" || process.env.LEARNINGS_NO_PATCH === "1" + + if (suppressed) { + log( + `marker PATCH suppressed (DRY_RUN=${process.env.DRY_RUN}, LEARNINGS_NO_PATCH=${process.env.LEARNINGS_NO_PATCH})`, + ) + log(`would have written marker: ${marker}`) + return + } + + if (fixture) { + // Write to the fixture patch file instead of the network. + fs.writeFileSync(patchFile, marker) + log(`wrote marker to ${patchFile}`) + return + } + + // Live PATCH: body-only, one line changed. The job already holds pull-requests: write. + const { api, repo } = await import("./lib.mjs") + const newBody = patchMarkerIntoBody(prBody, marker) + await api(`/repos/${repo()}/pulls/${prNumber}`, { + method: "PATCH", + body: { body: newBody }, + }) + log(`PATCHed learned-through marker on PR #${prNumber}`) +} + +// --- entry point --- + +const isMain = process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href + +// --- self-test harness (run: node .github/docs-sync/learn.mjs --self-test) --- +if (isMain && process.argv.includes("--self-test")) { + const failures = [] + const check = (label, fn) => { + try { + const ok = fn() + if (!ok) failures.push(label) + } catch (e) { + failures.push(label + " THREW: " + e.message) + } + } + + check("null in add does not throw", () => { + const r = validateDelta({ add: [null], remove: [] }, { existing: [], candidateSources: [], deletedInWindow: [] }) + return r.add.length === 0 && r.rejected.length === 1 && r.rejected[0].reason.includes("not a plain object") + }) + + check("undefined in add does not throw", () => { + const r = validateDelta( + { add: [undefined], remove: [] }, + { existing: [], candidateSources: [], deletedInWindow: [] }, + ) + return r.add.length === 0 && r.rejected.length === 1 && r.rejected[0].reason.includes("not a plain object") + }) + + check("mixed valid and null retains valid", () => { + const r = validateDelta( + { + add: [ + { + id: "valid-a", + rule: "Do not document experimental features", + scope: "both", + source: "commit:bbbbbbb", + date: "2026-08-03", + }, + null, + { + id: "valid-b", + rule: "Keep release notes concise", + scope: "edit", + source: "commit:bbbbbbb", + date: "2026-08-03", + }, + ], + remove: [], + }, + { existing: [], candidateSources: ["commit:bbbbbbb"], deletedInWindow: [] }, + ) + return r.add.length === 2 && r.rejected.length === 1 + }) + + if (failures.length) { + console.error("SELF-TEST FAILURES:", failures) + process.exit(1) + } + console.log("SELF-TEST PASSED (" + 3 + " checks)") + process.exit(0) +} + +if (isMain) { + main().catch((err) => { + console.error(err) + process.exit(1) + }) +} diff --git a/.github/docs-sync/learnings-prompt.md b/.github/docs-sync/learnings-prompt.md new file mode 100644 index 00000000000..cde193b8d4f --- /dev/null +++ b/.github/docs-sync/learnings-prompt.md @@ -0,0 +1,84 @@ +You are the extraction pass of an automated documentation pipeline for Kilo Code. Your only job: extract general rules of thumb from maintainer corrections to the docs-sync bot's rolling pull request. A correction is a commit or review comment a maintainer made to fix something the bot got wrong, and a learning is the general principle behind it that the bot should follow from now on. + +The attached `learnings-input.json` file contains: + +- `existing`: rules the bot already knows, each with `id`, `rule`, `scope`, `source`, and `date`. +- `deleted_in_window`: rule texts (not ids) a maintainer deleted from the learnings file in this extraction window. A maintainer deleted these on purpose — do not re-add them. +- `corrections`: the maintainer corrections to learn from. Each entry has a `source` (commit or comment id), `date`, and the relevant context. Commit entries have `message`, `files`, and `diff`. Comment entries have `path` and `body`. Some commits also carry an attached inline review `comment` that triggered them. + +Before writing anything: + +1. Read every correction in `corrections` and every rule in `existing`. +2. For each correction, decide whether it implies a general rule of thumb the bot should follow. Not every correction does — returning no new rules is a valid and expected answer. +3. When a correction implies a rule, write it as one imperative sentence stating the general principle, not what the specific correction did. + +Response format: a strict JSON object with no prose, no markdown fences, no comments: + +```json +{ + "add": [ + { + "id": "kebab-case-slug", + "rule": "One imperative sentence.", + "scope": "triage|edit|both", + "source": "commit:|comment:", + "date": "" + } + ], + "remove": [] +} +``` + +- `id`: a short kebab-case slug unique across this response. +- `rule`: one general imperative sentence. Never name a pull request, a number, a URL, a docs page, a file path, or a person. State the rule the correction implies, not what the correction changed. +- `scope`: `triage` when the rule changes which pull requests deserve documentation; `edit` when it changes how a page is written; `both` when it changes both. +- `source`: copied verbatim from the correction's `source` field. Never invent one. +- `date`: the correction's date, copied verbatim. + +The `remove` array lists `id` values of existing entries to drop. Remove an id only when a new rule contradicts or supersedes it. + +Hard rules: + +- The list of `add` entries may be empty. Returning `{"add": [], "remove": []}` is a valid and expected answer when no correction implies a general rule. +- Never re-add a rule listed in `deleted_in_window`, and never add a reworded near-duplicate of one. A maintainer deleted it. +- When a new rule is a near-duplicate of an existing one, merge them into one `add` and list the old id in `remove`. +- When a new rule contradicts an existing rule, `add` the new one and `remove` the contradicted id. +- Every `add` entry must have a `source` that appears in the input's `corrections` list. Never invent a source. +- Do not read files and do not run commands. Every input is already attached. + +Example. Input: + +```json +{ + "existing": [], + "deleted_in_window": [], + "corrections": [ + { + "source": "commit:9dd2c07", + "date": "2026-08-03", + "message": "docs: remove experimental features page", + "files": ["packages/kilo-docs/pages/code-with-ai/experimental-features.md"], + "diff": "- removed the entire experimental features page\n- the page documented features behind unreleased flags" + } + ] +} +``` + +Expected output: + +```json +{ + "add": [ + { + "id": "no-experimental-features", + "rule": "Do not document features that are behind unreleased flags.", + "scope": "both", + "source": "commit:9dd2c07", + "date": "2026-08-03" + } + ], + "remove": [] +} +``` + +The rule is `both` because documenting unreleased features is wrong at triage time (the feature is not docs-worthy yet) and at edit time (the page should not exist). diff --git a/.github/docs-sync/selftest.mjs b/.github/docs-sync/selftest.mjs index 9af35c58cdf..d6d552f5b32 100644 --- a/.github/docs-sync/selftest.mjs +++ b/.github/docs-sync/selftest.mjs @@ -21,6 +21,9 @@ import { routeRows, dropLegacySkipped, noDiffReport, + LEARNINGS_FILE, + nonContentFiles, + resolveLearnedThrough, renderBody, extractSectionRows, } from "./upsert-pr.mjs" @@ -31,11 +34,23 @@ import { applyRevertAnnotations, unannotatedRevertSignals, } from "./reverts.mjs" +import { + parseLearnings, + renderLearnings, + parseLearnedThrough, + patchMarkerIntoBody, + parseDelta, + validateDelta, + applyDelta, + isTrustedComment, + promptBlock, +} from "./learn.mjs" const HERE = path.dirname(fileURLToPath(import.meta.url)) const EDIT_SCRIPT = path.join(HERE, "edit.mjs") const TRIAGE_SCRIPT = path.join(HERE, "triage.mjs") const COLLECT_SCRIPT = path.join(HERE, "collect.mjs") +const LEARN_SCRIPT = path.join(HERE, "learn.mjs") const temps = [] @@ -154,6 +169,16 @@ if (mode === "triage-embed-env-secret") { process.stdout.write(JSON.stringify(entries) + "\\n"); process.exit(0); } +if (mode === "extraction-delta") { + const deltaFile = "docs-sync-out/extraction-delta.json"; + if (fs.existsSync(deltaFile)) { + const delta = JSON.parse(fs.readFileSync(deltaFile, "utf8")); + process.stdout.write(JSON.stringify(delta) + "\\n"); + } else { + process.stdout.write('{"add":[],"remove":[]}' + "\\n"); + } + process.exit(0); +} process.stderr.write("unknown stub mode\\n"); process.exit(1); ` @@ -321,9 +346,9 @@ function setupTriageCwd(digest) { return cwd } -function runNodeScript(scriptPath, { cwd, env = {}, kiloDir }) { +function runNodeScript(scriptPath, { cwd, env = {}, kiloDir, args = [] }) { const pathEnv = [kiloDir, process.env.PATH].filter(Boolean).join(path.delimiter) - const result = spawnSync(process.execPath, [scriptPath], { + const result = spawnSync(process.execPath, [scriptPath, ...args], { cwd, env: { ...process.env, @@ -1067,10 +1092,12 @@ function case4_routing() { // round-trip renderBody → extractSectionRows const through = "2026-07-20T09:59:59.999Z" + const learnedMarker = "" const body = renderBody({ date: "2026-07-27", since: "2026-07-17T00:00:00.000Z", through, + learnedThrough: learnedMarker, changesRows, pendingRows, skippedRows, @@ -1079,6 +1106,7 @@ function case4_routing() { note: "", }) assert.ok(body.includes(``)) + assert.ok(body.includes(learnedMarker), "learned-through marker must appear in rendered body") const extChanges = extractSectionRows(body, "changes") const extPending = extractSectionRows(body, "pending") const extSkipped = extractSectionRows(body, "skipped") @@ -1106,6 +1134,7 @@ function case4_routing() { date: "2026-07-27", since: "s", through: "t", + learnedThrough: "", changesRows: [], pendingRows: [], skippedRows: forgedRows.skippedRows, @@ -1417,7 +1446,7 @@ function case9_reverts() { console.log("case 9: revert interception") // --- revertTitleKind --- - assert.equal(revertTitleKind('revert(cli): restore opt-in stream idle timeouts'), "conventional") + assert.equal(revertTitleKind("revert(cli): restore opt-in stream idle timeouts"), "conventional") assert.equal(revertTitleKind('Revert "feat(cli): default stream watchdog"'), "github-native") assert.equal(revertTitleKind("REVERT: all of it"), "conventional") assert.equal(revertTitleKind("feat(cli): add x"), null) @@ -1824,6 +1853,1396 @@ Reverts #12249 and #12481. } } +// --------------------------------------------------------------------------- +// Case 10 — learnings extraction, validation, injection, upsert safety +// --------------------------------------------------------------------------- +function case10_learnings() { + console.log("case 10: learnings") + + // --- helpers for extraction runs --- + function writeFixture(cwd, data) { + const f = path.join(cwd, "fixture.json") + fs.writeFileSync(f, JSON.stringify(data, null, 2)) + return f + } + + function writeExtractionDelta(cwd, delta) { + fs.mkdirSync(path.join(cwd, "docs-sync-out"), { recursive: true }) + fs.writeFileSync(path.join(cwd, "docs-sync-out", "extraction-delta.json"), JSON.stringify(delta, null, 2)) + } + + // Prepare a git repo for learn.mjs tests: set origin refs, create docs-sync-out. + function setupLearnRepo(dir) { + fs.mkdirSync(path.join(dir, "docs-sync-out"), { recursive: true }) + gitIn(dir, ["update-ref", "refs/remotes/origin/main", "main"]) + gitIn(dir, ["update-ref", "refs/remotes/origin/docs/auto-sync", "docs/auto-sync"]) + return dir + } + + const githubBotEmail = "41898282+github-actions[bot]@users.noreply.github.com" + const kiloconnectBotEmail = "240665456+kiloconnect[bot]@users.noreply.github.com" + + // 10a — three commit classes + { + console.log(" 10a — three commit classes") + const dir = mktemp("docs-sync-learn-a-") + initRepoWithIdentity(dir) + fs.writeFileSync(path.join(dir, "base.txt"), "base\n") + gitIn(dir, ["add", "base.txt"]) + gitIn(dir, ["commit", "-m", "base"]) + + // Branch + gitIn(dir, ["checkout", "-b", "docs/auto-sync"]) + // 1. kiloconnect[bot] commit that touches packages/kilo-docs/pages/x.md + gitIn(dir, ["config", "user.email", kiloconnectBotEmail]) + fs.mkdirSync(path.join(dir, "packages", "kilo-docs", "pages"), { recursive: true }) + fs.writeFileSync(path.join(dir, "packages", "kilo-docs", "pages", "x.md"), "# x\n") + gitIn(dir, ["add", "packages/kilo-docs/pages/x.md"]) + gitIn(dir, ["commit", "-m", "docs: add x page"]) + const kiloconnectSha = gitIn(dir, ["rev-parse", "HEAD"]) + // 2. github-actions[bot] commit + gitIn(dir, ["config", "user.email", githubBotEmail]) + fs.writeFileSync(path.join(dir, "packages", "kilo-docs", "pages", "y.md"), "# y\n") + gitIn(dir, ["add", "packages/kilo-docs/pages/y.md"]) + gitIn(dir, ["commit", "-m", "docs: add y page"]) + // 3. Merge commit (non-merge filter) + gitIn(dir, ["config", "user.email", "someone@example.com"]) + gitIn(dir, ["checkout", "-b", "tmp-merge"]) + fs.writeFileSync(path.join(dir, "z.txt"), "z\n") + gitIn(dir, ["add", "z.txt"]) + gitIn(dir, ["commit", "-m", "tmp"]) + gitIn(dir, ["checkout", "docs/auto-sync"]) + gitIn(dir, ["merge", "--no-ff", "tmp-merge", "-m", "merge tmp"]) + // 4. commit reachable from main + gitIn(dir, ["checkout", "main"]) + fs.writeFileSync(path.join(dir, "main-only.txt"), "main only\n") + gitIn(dir, ["add", "main-only.txt"]) + gitIn(dir, ["commit", "-m", "main only"]) + + gitIn(dir, ["checkout", "docs/auto-sync"]) + const tip = gitIn(dir, ["rev-parse", "HEAD"]) + + // Setup origin refs (needed by learn.mjs git commands) + const cwd = setupLearnRepo(dir) + + const fixturePath = writeFixture(dir, { + pr: { number: 1, head: { ref: "docs/auto-sync" }, body: "", user: { login: "github-actions[bot]" } }, + comments: [], + }) + + const callLog = path.join(dir, "kilo-calls.log") + const kiloDir = makeStubKiloDir({ mode: "extraction-delta", callLog }) + writeExtractionDelta(dir, { add: [], remove: [] }) + + const result = runNodeScript(LEARN_SCRIPT, { + cwd: dir, + kiloDir, + env: { + TRIAGE_MODEL: "test/model", + DOCS_SYNC_FIXTURE: fixturePath, + LEARNINGS_BUDGET_MINUTES: "1", + DOCS_SYNC_BACKOFF_MS: "0", + }, + }) + + // Assert: input file written, exactly one correction — kiloconnect only. + // github-actions[bot] commit excluded (criterion 5), merge commit excluded + // via --no-merges, main-reachable commit excluded by origin/main range (criterion 6). + const inputFile = path.join(dir, "docs-sync-out", "learnings-input.json") + assert.ok(fs.existsSync(inputFile), `expected ${inputFile}`) + const input = JSON.parse(fs.readFileSync(inputFile, "utf8")) + assert.equal(input.corrections.length, 1, "exactly one correction (kiloconnect commit)") + assert.equal( + input.corrections[0].source, + `commit:${kiloconnectSha.slice(0, 7)}`, + "correction must be kiloconnect commit only", + ) + } + + // 10rz — timestamp correlation with different timezone offsets + // A comment must map to the chronologically earliest eligible commit even when + // timestamps use different timezone offsets (Z vs +05:00). + { + console.log(" 10rz — timestamp correlation") + const dir = mktemp("docs-sync-learn-rz-") + initRepoWithIdentity(dir) + fs.writeFileSync(path.join(dir, "base.txt"), "base\n") + gitIn(dir, ["add", "base.txt"]) + gitIn(dir, ["commit", "-m", "base"]) + + gitIn(dir, ["checkout", "-b", "docs/auto-sync"]) + + // Commit A: non-UTC offset, chronologically earliest (UTC 08:00) + // iso = 2026-08-03T13:00:00+05:00 + fs.mkdirSync(path.join(dir, "packages", "kilo-docs", "pages"), { recursive: true }) + fs.writeFileSync(path.join(dir, "packages", "kilo-docs", "pages", "x.md"), "# x\n") + gitIn(dir, ["add", "packages/kilo-docs/pages/x.md"]) + gitIn(dir, ["commit", "-m", "first edit x", "--author", `kiloconnect[bot] <${kiloconnectBotEmail}>`], { + GIT_COMMITTER_DATE: "2026-08-03T13:00:00+05:00", + }) + const shaA = gitIn(dir, ["rev-parse", "HEAD"]) + + // Commit B: UTC offset, chronologically later (UTC 09:00) + // iso = 2026-08-03T09:00:00Z — string comparison would pick this as "earlier" (09 < 13) + // but chronologically A is earlier (08:00 < 09:00) + fs.writeFileSync(path.join(dir, "packages", "kilo-docs", "pages", "x.md"), "# x\n## edit\n") + gitIn(dir, ["add", "packages/kilo-docs/pages/x.md"]) + gitIn(dir, ["commit", "-m", "second edit x", "--author", `kiloconnect[bot] <${kiloconnectBotEmail}>`], { + GIT_COMMITTER_DATE: "2026-08-03T09:00:00Z", + }) + const shaB = gitIn(dir, ["rev-parse", "HEAD"]) + + // Seed LEARNINGS.md so existing entries are non-empty but irrelevant + const existing = [ + { id: "pre", rule: "Pre-existing rule.", scope: "both", source: "commit:0000000", date: "2026-01-01" }, + ] + const learningsPath = path.join(dir, "packages", "kilo-docs", "LEARNINGS.md") + fs.mkdirSync(path.dirname(learningsPath), { recursive: true }) + fs.writeFileSync(learningsPath, renderLearnings(existing)) + gitIn(dir, ["add", "packages/kilo-docs/LEARNINGS.md"]) + gitIn(dir, ["commit", "-m", "seed learnings", "--author", `kiloconnect[bot] <${kiloconnectBotEmail}>`]) + + const cwd = setupLearnRepo(dir) + + // Comment at UTC 05:30 on x.md — before both commits chronologically. + // 05:30 < 08:00 (A) and 05:30 < 09:00 (B) → both eligible + // The chronologically earliest eligible commit is A (08:00). + const fixturePath = writeFixture(cwd, { + pr: { number: 1, head: { ref: "docs/auto-sync" }, body: "", user: { login: "github-actions[bot]" } }, + comments: [ + { + id: 101, + created_at: "2026-08-03T05:30:00Z", + path: "packages/kilo-docs/pages/x.md", + body: "Please fix the docs.", + author_association: "MEMBER", + user: { login: "maintainer" }, + }, + ], + }) + + const kiloDir = makeStubKiloDir({ mode: "extraction-delta", callLog: path.join(cwd, "kilo-calls.log") }) + writeExtractionDelta(cwd, { add: [], remove: [] }) + + const result = runNodeScript(LEARN_SCRIPT, { + cwd, + kiloDir, + env: { + TRIAGE_MODEL: "test/model", + DOCS_SYNC_FIXTURE: fixturePath, + LEARNINGS_BUDGET_MINUTES: "1", + DOCS_SYNC_BACKOFF_MS: "0", + }, + }) + + // Read learnings-input.json to inspect the correlation result + const inputFile = path.join(cwd, "docs-sync-out", "learnings-input.json") + assert.ok(fs.existsSync(inputFile), "learnings-input.json must exist") + const input = JSON.parse(fs.readFileSync(inputFile, "utf8")) + + // Both commits must appear as corrections + const commitA = input.corrections.find((c) => c.source === `commit:${shaA.slice(0, 7)}`) + const commitB = input.corrections.find((c) => c.source === `commit:${shaB.slice(0, 7)}`) + assert.ok(commitA, "commit A must be in corrections") + assert.ok(commitB, "commit B must be in corrections") + + // Comment must be associated with commit A (chronologically earliest) + assert.ok(commitA.comment, "commit A must have the comment associated") + assert.equal(commitA.comment.path, "packages/kilo-docs/pages/x.md") + assert.equal(commitB.comment, undefined, "commit B must not have the comment associated") + + // No standalone comment candidate — the comment was correlated, not orphaned + const standalone = input.corrections.filter((c) => c.source && c.source.startsWith("comment:")) + assert.equal(standalone.length, 0, "comment must be associated, not standalone") + } + + // 10b — watermark suppression (no model call when marker covers all) + { + console.log(" 10b — watermark suppression") + const dir = mktemp("docs-sync-learn-b-") + initRepoWithIdentity(dir) + fs.writeFileSync(path.join(dir, "base.txt"), "base\n") + gitIn(dir, ["add", "base.txt"]) + gitIn(dir, ["commit", "-m", "base"]) + + gitIn(dir, ["checkout", "-b", "docs/auto-sync"]) + // corrective commit + fs.mkdirSync(path.join(dir, "packages", "kilo-docs", "pages"), { recursive: true }) + fs.writeFileSync(path.join(dir, "packages", "kilo-docs", "pages", "x.md"), "# x\n") + gitIn(dir, ["add", "packages/kilo-docs/pages/x.md"]) + gitIn(dir, ["commit", "-m", "docs update", "--author", `kiloconnect[bot] <${kiloconnectBotEmail}>`]) + + // Write LEARNINGS.md on the branch first, so the marker can point to the tip after it + const existing = [ + { id: "test", rule: "existing rule", scope: "both", source: "commit:0000000", date: "2026-01-01" }, + ] + const learningsPath = path.join(dir, "packages", "kilo-docs", "LEARNINGS.md") + const learningsContent = renderLearnings(existing) + fs.mkdirSync(path.dirname(learningsPath), { recursive: true }) + fs.writeFileSync(learningsPath, learningsContent) + gitIn(dir, ["add", "packages/kilo-docs/LEARNINGS.md"]) + gitIn(dir, ["commit", "-m", "seed learnings"]) + + const tip = gitIn(dir, ["rev-parse", "HEAD"]) + const tipDate = new Date().toISOString() + + const cwd = setupLearnRepo(dir) + const body = `` + const fixturePath = writeFixture(cwd, { + pr: { number: 1, head: { ref: "docs/auto-sync" }, body, user: { login: "github-actions[bot]" } }, + comments: [], + }) + + const callLog = path.join(cwd, "kilo-calls.log") + const kiloDir = makeStubKiloDir({ mode: "extraction-delta", callLog }) + writeExtractionDelta(cwd, { add: [], remove: [] }) + + fs.writeFileSync(path.join(cwd, "docs-sync-out", "learnings.json"), JSON.stringify(existing, null, 2)) + + const result = runNodeScript(LEARN_SCRIPT, { + cwd, + kiloDir, + env: { + TRIAGE_MODEL: "test/model", + DOCS_SYNC_FIXTURE: fixturePath, + LEARNINGS_BUDGET_MINUTES: "1", + DOCS_SYNC_BACKOFF_MS: "0", + }, + }) + + // Assert: stub never invoked, learnings.json unchanged, no model call + const calls = fs.existsSync(callLog) ? fs.readFileSync(callLog, "utf8").trim() : "" + assert.equal(calls, "", `kilo must not be invoked when marker covers all; got ${calls}`) + const out = JSON.parse(fs.readFileSync(path.join(cwd, "docs-sync-out", "learnings.json"), "utf8")) + assert.deepEqual(out, existing, "learnings.json must equal existing entries") + + // Run apply and prove LEARNINGS.md is byte-unchanged + const lp = path.join(cwd, "packages", "kilo-docs", "LEARNINGS.md") + const before = fs.readFileSync(lp, "utf8") + runNodeScript(LEARN_SCRIPT, { + cwd, + kiloDir, + args: ["--apply"], + env: { DOCS_SYNC_BACKOFF_MS: "0" }, + }) + const after = fs.readFileSync(lp, "utf8") + assert.equal(after, before, "LEARNINGS.md must be byte-unchanged after apply") + } + + // 10c — rerun idempotency + { + console.log(" 10c — rerun idempotency") + const dir = mktemp("docs-sync-learn-c-") + initRepoWithIdentity(dir) + fs.writeFileSync(path.join(dir, "base.txt"), "base\n") + gitIn(dir, ["add", "base.txt"]) + gitIn(dir, ["commit", "-m", "base"]) + + gitIn(dir, ["checkout", "-b", "docs/auto-sync"]) + fs.mkdirSync(path.join(dir, "packages", "kilo-docs", "pages"), { recursive: true }) + fs.writeFileSync(path.join(dir, "packages", "kilo-docs", "pages", "x.md"), "# x\n") + gitIn(dir, ["add", "packages/kilo-docs/pages/x.md"]) + gitIn(dir, ["commit", "-m", "docs update", "--author", `kiloconnect[bot] <${kiloconnectBotEmail}>`]) + let tip = gitIn(dir, ["rev-parse", "HEAD"]) + + const add = [ + { + id: "new-rule", + rule: "A new rule from testing.", + scope: "both", + source: `commit:${tip.slice(0, 7)}`, + date: "2026-08-03", + }, + ] + let firstLEARNINGS + + // First run + { + const cwd = setupLearnRepo(dir) + const fixturePath = writeFixture(cwd, { + pr: { number: 1, head: { ref: "docs/auto-sync" }, body: "", user: { login: "github-actions[bot]" } }, + comments: [], + }) + const kiloDir = makeStubKiloDir({ mode: "extraction-delta", callLog: path.join(cwd, "kilo-calls.log") }) + writeExtractionDelta(cwd, { add, remove: [] }) + const result = runNodeScript(LEARN_SCRIPT, { + cwd, + kiloDir, + env: { + TRIAGE_MODEL: "test/model", + DOCS_SYNC_FIXTURE: fixturePath, + LEARNINGS_BUDGET_MINUTES: "1", + DOCS_SYNC_BACKOFF_MS: "0", + }, + }) + const out1 = JSON.parse(fs.readFileSync(path.join(cwd, "docs-sync-out", "learnings.json"), "utf8")) + assert.equal(out1.length, 1) + assert.equal(out1[0].id, "new-rule") + // Write LEARNINGS.md on the branch so second run sees existing entries + const learningsPath = path.join(dir, "packages", "kilo-docs", "LEARNINGS.md") + fs.mkdirSync(path.dirname(learningsPath), { recursive: true }) + fs.writeFileSync(learningsPath, renderLearnings(out1)) + gitIn(dir, ["add", "packages/kilo-docs/LEARNINGS.md"]) + gitIn(dir, ["commit", "-m", "seed learnings"]) + tip = gitIn(dir, ["rev-parse", "HEAD"]) + firstLEARNINGS = fs.readFileSync(path.join(dir, "packages", "kilo-docs", "LEARNINGS.md"), "utf8") + } + + // Second run with marker covering first run's result + { + const cwd = setupLearnRepo(dir) + const body = `` + const fixturePath = writeFixture(cwd, { + pr: { number: 1, head: { ref: "docs/auto-sync" }, body, user: { login: "github-actions[bot]" } }, + comments: [], + }) + const callLog2 = path.join(cwd, "kilo-calls-run2.log") + const kiloDir2 = makeStubKiloDir({ mode: "extraction-delta", callLog: callLog2 }) + writeExtractionDelta(cwd, { add, remove: [] }) + const result = runNodeScript(LEARN_SCRIPT, { + cwd, + kiloDir: kiloDir2, + env: { + TRIAGE_MODEL: "test/model", + DOCS_SYNC_FIXTURE: fixturePath, + LEARNINGS_BUDGET_MINUTES: "1", + DOCS_SYNC_BACKOFF_MS: "0", + }, + }) + const calls2 = fs.existsSync(callLog2) ? fs.readFileSync(callLog2, "utf8").trim() : "" + assert.equal(calls2, "", "second run must not invoke kilo") + const out2 = JSON.parse(fs.readFileSync(path.join(cwd, "docs-sync-out", "learnings.json"), "utf8")) + assert.equal(out2.length, 1) + assert.equal(out2[0].id, "new-rule") + // Run apply and prove LEARNINGS.md is byte-identical after idempotent rerun + runNodeScript(LEARN_SCRIPT, { + cwd, + kiloDir: kiloDir2, + args: ["--apply"], + env: { DOCS_SYNC_BACKOFF_MS: "0" }, + }) + const afterApply = fs.readFileSync(path.join(dir, "packages", "kilo-docs", "LEARNINGS.md"), "utf8") + assert.equal(afterApply.length, firstLEARNINGS.length, "LEARNINGS.md must be same length after apply") + assert.equal(afterApply, firstLEARNINGS, "LEARNINGS.md must be byte-identical after apply") + } + } + + // 10d — contradiction replacement (applyDelta) + { + console.log(" 10d — contradiction replacement") + const old = [ + { id: "old-rule", rule: "Old rule text.", scope: "both", source: "commit:aaaaaaa", date: "2026-01-01" }, + ] + const add = [ + { id: "new-rule", rule: "New rule text.", scope: "both", source: "commit:bbbbbbb", date: "2026-02-01" }, + ] + const delta = { add, remove: ["old-rule"] } + const result = applyDelta(old, delta) + assert.equal(result.length, 1) + assert.equal(result[0].id, "new-rule") + // Third delta touching neither does not bring old back + const again = applyDelta(result, { add: [], remove: [] }) + assert.equal(again.length, 1) + assert.equal(again[0].id, "new-rule") + } + + // 10e — prompt injection (triage/edit argv carry tagged rules) + { + console.log(" 10e — prompt injection") + const dir = mktemp("docs-sync-learn-e-") + initRepoWithIdentity(dir) + fs.writeFileSync(path.join(dir, "base.txt"), "base\n") + gitIn(dir, ["add", "base.txt"]) + gitIn(dir, ["commit", "-m", "base"]) + gitIn(dir, ["checkout", "-b", "docs/auto-sync"]) + // Add a corrective commit so extraction has a candidate + fs.mkdirSync(path.join(dir, "packages", "kilo-docs", "pages"), { recursive: true }) + fs.writeFileSync(path.join(dir, "packages", "kilo-docs", "pages", "x.md"), "# x\n") + gitIn(dir, ["add", "packages/kilo-docs/pages/x.md"]) + gitIn(dir, ["commit", "-m", "docs update", "--author", `kiloconnect[bot] <${kiloconnectBotEmail}>`]) + const commitSha = gitIn(dir, ["rev-parse", "HEAD"]) + + const cwd = setupLearnRepo(dir) + const fixturePath = writeFixture(cwd, { + pr: { number: 1, head: { ref: "docs/auto-sync" }, body: "", user: { login: "github-actions[bot]" } }, + comments: [], + }) + const callLog = path.join(cwd, "kilo-calls.log") + const kiloDir = makeStubKiloDir({ mode: "extraction-delta", callLog }) + // Three entries: triage, edit, both — all from the same candidate source + const src = `commit:${commitSha.slice(0, 7)}` + writeExtractionDelta(cwd, { + add: [ + { id: "triage-rule", rule: "Triage-only rule text.", scope: "triage", source: src, date: "2026-08-01" }, + { id: "edit-rule", rule: "Edit-only rule text.", scope: "edit", source: src, date: "2026-08-02" }, + { id: "both-rule", rule: "Both scope rule text.", scope: "both", source: src, date: "2026-08-03" }, + ], + remove: [], + }) + + // Run extraction so it writes learnings-.md blocks (extraction step 13). + // Only extraction writes these files; --apply writes only LEARNINGS.md. + runNodeScript(LEARN_SCRIPT, { + cwd, + kiloDir, + env: { + TRIAGE_MODEL: "test/model", + DOCS_SYNC_FIXTURE: fixturePath, + LEARNINGS_BUDGET_MINUTES: "1", + DOCS_SYNC_BACKOFF_MS: "0", + }, + }) + + const triageBlockPath = path.join(cwd, "docs-sync-out", "learnings-triage.md") + const editBlockPath = path.join(cwd, "docs-sync-out", "learnings-edit.md") + assert.ok(fs.existsSync(triageBlockPath), "learnings-triage.md must exist") + assert.ok(fs.existsSync(editBlockPath), "learnings-edit.md must exist") + + // Run triage.mjs against a recording stub. Copy the learnings block into + // its cwd so readLearningsBlock picks it up. + { + const triageCwd = setupTriageCwd([samplePr(1)]) + fs.copyFileSync(triageBlockPath, path.join(triageCwd, "docs-sync-out", "learnings-triage.md")) + const triageCallLog = path.join(triageCwd, "triage-calls.log") + const triageKiloDir = makeStubKiloDir({ mode: "record", callLog: triageCallLog }) + runNodeScript(TRIAGE_SCRIPT, { + cwd: triageCwd, + kiloDir: triageKiloDir, + env: { TRIAGE_MODEL: "test/model", DOCS_SYNC_BACKOFF_MS: "0" }, + }) + const logText = fs.readFileSync(triageCallLog, "utf8") + assert.ok(logText.includes("Triage-only rule text"), "triage argv must contain triage rule") + assert.ok(logText.includes("Both scope rule text"), "triage argv must contain both rule") + assert.ok(!logText.includes("Edit-only rule text"), "triage argv must not contain edit-only rule") + } + + // Run edit.mjs against a recording stub. + { + const triageEntry = { + pr: 1, + url: "https://github.com/Kilo-Org/cloud/pull/1", + docs_worthy: true, + reason: "needs docs", + target_sections: [], + priority: "medium", + } + const editCwd = setupEditCwd([samplePr(1)], [triageEntry]) + fs.copyFileSync(editBlockPath, path.join(editCwd, "docs-sync-out", "learnings-edit.md")) + const editCallLog = path.join(editCwd, "edit-calls.log") + const editKiloDir = makeStubKiloDir({ mode: "record", callLog: editCallLog }) + runNodeScript(EDIT_SCRIPT, { + cwd: editCwd, + kiloDir: editKiloDir, + env: { EDIT_MODEL: "test/model", DOCS_SYNC_BACKOFF_MS: "0" }, + }) + const logText = fs.readFileSync(editCallLog, "utf8") + assert.ok(logText.includes("Edit-only rule text"), "edit argv must contain edit rule") + assert.ok(logText.includes("Both scope rule text"), "edit argv must contain both rule") + assert.ok(!logText.includes("Triage-only rule text"), "edit argv must not contain triage-only rule") + } + } + + // 10f — failure path + { + console.log(" 10f — failure path") + const dir = mktemp("docs-sync-learn-f-") + initRepoWithIdentity(dir) + fs.writeFileSync(path.join(dir, "base.txt"), "base\n") + gitIn(dir, ["add", "base.txt"]) + gitIn(dir, ["commit", "-m", "base"]) + gitIn(dir, ["checkout", "-b", "docs/auto-sync"]) + fs.mkdirSync(path.join(dir, "packages", "kilo-docs", "pages"), { recursive: true }) + fs.writeFileSync(path.join(dir, "packages", "kilo-docs", "pages", "x.md"), "# x\n") + gitIn(dir, ["add", "packages/kilo-docs/pages/x.md"]) + gitIn(dir, ["commit", "-m", "docs update", "--author", `kiloconnect[bot] <${kiloconnectBotEmail}>`]) + + const existing = [ + { id: "test", rule: "existing rule", scope: "both", source: "commit:0000000", date: "2026-01-01" }, + ] + // Write LEARNINGS.md on the branch so learn.mjs reads it as existing entries + const learningsPath = path.join(dir, "packages", "kilo-docs", "LEARNINGS.md") + fs.mkdirSync(path.dirname(learningsPath), { recursive: true }) + fs.writeFileSync(learningsPath, renderLearnings(existing)) + gitIn(dir, ["add", "packages/kilo-docs/LEARNINGS.md"]) + gitIn(dir, ["commit", "-m", "seed learnings"]) + + const cwd = setupLearnRepo(dir) + const fixturePath = writeFixture(cwd, { + pr: { number: 1, head: { ref: "docs/auto-sync" }, body: "", user: { login: "github-actions[bot]" } }, + comments: [], + }) + + // Stub exits 0 with garbage stdout (stderr-exit0 mode) + const stderrText = "some fake error stream" + const kiloDir = makeStubKiloDir({ mode: "stderr-exit0", stderrText }) + + const outputFile = path.join(cwd, "gh-output-f") + const result = runNodeScript(LEARN_SCRIPT, { + cwd, + kiloDir, + env: { + TRIAGE_MODEL: "test/model", + DOCS_SYNC_FIXTURE: fixturePath, + LEARNINGS_BUDGET_MINUTES: "1", + DOCS_SYNC_BACKOFF_MS: "0", + GITHUB_OUTPUT: outputFile, + }, + }) + + assert.equal(result.status, 0, "learn.mjs must exit 0 on failure") + const out = JSON.parse(fs.readFileSync(path.join(cwd, "docs-sync-out", "learnings.json"), "utf8")) + assert.deepEqual(out, existing, "learnings.json must equal existing entries on failure") + // No marker PATCH file + assert.ok(!fs.existsSync(`${fixturePath}.patched`), "no marker PATCH on failure") + // No learned_through output + if (fs.existsSync(outputFile)) { + const ghOut = fs.readFileSync(outputFile, "utf8") + assert.ok(!ghOut.includes("learned_through="), "GITHUB_OUTPUT must not contain learned_through on failure") + } + + // Run apply and prove LEARNINGS.md is byte-unchanged + const lp = path.join(cwd, "packages", "kilo-docs", "LEARNINGS.md") + const before = fs.readFileSync(lp, "utf8") + runNodeScript(LEARN_SCRIPT, { + cwd, + kiloDir, + args: ["--apply"], + env: { DOCS_SYNC_BACKOFF_MS: "0" }, + }) + const after = fs.readFileSync(lp, "utf8") + assert.equal(after, before, "LEARNINGS.md must be byte-unchanged after apply on failure") + } + + // 10g — general-rule check (validateDelta rejections) + { + console.log(" 10g — general-rule check") + const existing = [] + const sources = ["commit:aaaaaaa"] + const entryWithPR = { + id: "bad-pr", + rule: "See #12716 for details", + scope: "both", + source: "commit:aaaaaaa", + date: "2026-08-03", + } + const entryWithURL = { + id: "bad-url", + rule: "Check https://example.com", + scope: "both", + source: "commit:aaaaaaa", + date: "2026-08-03", + } + const entryWithPerson = { + id: "bad-person", + rule: "Ask @emilieschario", + scope: "both", + source: "commit:aaaaaaa", + date: "2026-08-03", + } + const entryWithPage = { + id: "bad-page", + rule: "Edit packages/kilo-docs/pages/x.md", + scope: "both", + source: "commit:aaaaaaa", + date: "2026-08-03", + } + const entryBadSource = { + id: "bad-source", + rule: "A valid sentence.", + scope: "both", + source: "invented", + date: "2026-08-03", + } + const entryBadScope = { + id: "bad-scope", + rule: "A valid sentence.", + scope: "wrong", + source: "commit:aaaaaaa", + date: "2026-08-03", + } + const entryCollide = { + id: "existing-id", + rule: "A valid sentence.", + scope: "both", + source: "commit:aaaaaaa", + date: "2026-08-03", + } + const entryBadDate = { + id: "bad-date", + rule: "A valid sentence.", + scope: "both", + source: "commit:aaaaaaa", + date: "not-a-date", + } + const entryRemoveUnknown = { + id: "valid", + rule: "A valid sentence.", + scope: "both", + source: "commit:aaaaaaa", + date: "2026-08-03", + } + + const existingWithId = [ + { id: "existing-id", rule: "Existing rule.", scope: "both", source: "commit:aaaaaaa", date: "2026-01-01" }, + ] + + // PR number + { + const { rejected } = validateDelta( + { add: [entryWithPR], remove: [] }, + { existing, candidateSources: sources, deletedInWindow: [] }, + ) + assert.equal(rejected.length, 1, "PR number must be rejected") + assert.ok(rejected[0].reason, "rejection must carry a reason") + } + + // URL — docs-check-links.yml link-checks LEARNINGS.md with fail:true, so + // a URL in a rule would break CI on every bot commit. + { + const { rejected } = validateDelta( + { add: [entryWithURL], remove: [] }, + { existing, candidateSources: sources, deletedInWindow: [] }, + ) + assert.equal(rejected.length, 1, "URL must be rejected") + } + + // Person + { + const { rejected } = validateDelta( + { add: [entryWithPerson], remove: [] }, + { existing, candidateSources: sources, deletedInWindow: [] }, + ) + assert.equal(rejected.length, 1, "person reference must be rejected") + } + + // Docs page + { + const { rejected } = validateDelta( + { add: [entryWithPage], remove: [] }, + { existing, candidateSources: sources, deletedInWindow: [] }, + ) + assert.equal(rejected.length, 1, "docs page reference must be rejected") + } + + // Bad source + { + const { rejected } = validateDelta( + { add: [entryBadSource], remove: [] }, + { existing, candidateSources: sources, deletedInWindow: [] }, + ) + assert.equal(rejected.length, 1, "invented source must be rejected") + } + + // Bad scope + { + const { rejected } = validateDelta( + { add: [entryBadScope], remove: [] }, + { existing, candidateSources: sources, deletedInWindow: [] }, + ) + assert.equal(rejected.length, 1, "bad scope must be rejected") + } + + // Colliding id + { + const { rejected } = validateDelta( + { add: [entryCollide], remove: [] }, + { existing: existingWithId, candidateSources: sources, deletedInWindow: [] }, + ) + assert.equal(rejected.length, 1, "colliding id must be rejected") + } + + // Remove unknown id + { + const { rejected } = validateDelta( + { add: [entryRemoveUnknown], remove: ["unknown-id"] }, + { existing, candidateSources: sources, deletedInWindow: [] }, + ) + assert.equal(rejected.length, 1, "remove of unknown id must be rejected") + } + + // Bad date + { + const { rejected } = validateDelta( + { add: [entryBadDate], remove: [] }, + { existing, candidateSources: sources, deletedInWindow: [] }, + ) + assert.equal(rejected.length, 1, "bad date must be rejected") + } + } + + // 10h — comment trust (isTrustedComment) + { + console.log(" 10h — comment trust") + assert.equal(isTrustedComment({ author_association: "OWNER", user: { login: "owner-user" } }), true) + assert.equal(isTrustedComment({ author_association: "MEMBER", user: { login: "emilieschario" } }), true) + assert.equal(isTrustedComment({ author_association: "COLLABORATOR", user: { login: "collab-user" } }), true) + assert.equal(isTrustedComment({ author_association: "CONTRIBUTOR", user: { login: "kilo-code-bot[bot]" } }), false) + assert.equal(isTrustedComment({ author_association: "NONE", user: { login: "rando" } }), false) + // MEMBER whose login ends in [bot] + assert.equal(isTrustedComment({ author_association: "MEMBER", user: { login: "some-bot[bot]" } }), false) + } + + // 10i — draft gate (nonContentFiles) + { + console.log(" 10i — draft gate") + const files = ["packages/kilo-docs/LEARNINGS.md", "packages/kilo-docs/pages/a.md"] + const result = nonContentFiles(files) + assert.equal(result.length, 0, "LEARNINGS.md and pages must not trigger the draft gate") + // Still flags non-content + const withConfig = ["packages/kilo-docs/next.config.js"] + const flagged = nonContentFiles(withConfig) + assert.equal(flagged.length, 1, "next.config.js must still trigger the gate") + assert.equal(flagged[0], "packages/kilo-docs/next.config.js") + } + + // 10j — no --auto on extraction call + { + console.log(" 10j — no --auto on extraction call") + const src = fs.readFileSync(LEARN_SCRIPT, "utf8") + + // Find the extraction-mode runKilo args array + const argsStart = src.indexOf("runKilo({") + assert.ok(argsStart >= 0, "runKilo call must exist in learn.mjs") + const argsBlock = src.slice(argsStart, src.indexOf("})", argsStart) + 2) + assert.ok(!argsBlock.includes("--auto"), "extraction runKilo must not include --auto") + assert.ok(argsBlock.includes("-f"), "extraction runKilo must include -f") + } + + // 10k — hand-mangled file (parseLearnings) + { + console.log(" 10k — hand-mangled file") + const text = `# header + +- Valid rule. +- Broken meta. +Just prose, not a rule line. +` + const entries = parseLearnings(text) + assert.equal(entries.length, 1, "only valid entry must parse") + assert.equal(entries[0].id, "valid-rule") + } + + // 10l — first-run fallback (main when branch has none) + // Prove the git commands learn.mjs relies on: when the branch file is absent, + // git show origin/:packages/kilo-docs/LEARNINGS.md fails, and + // git show origin/main:packages/kilo-docs/LEARNINGS.md returns the main's entries. + { + console.log(" 10l — empty file fallback") + const dir = mktemp("docs-sync-learn-l-") + initRepoWithIdentity(dir) + + // Write LEARNINGS.md on main + const entries = [ + { id: "test", rule: "Test rule text.", scope: "both", source: "commit:aaaaaaa", date: "2026-01-01" }, + ] + const lp = path.join(dir, "packages", "kilo-docs", "LEARNINGS.md") + fs.mkdirSync(path.dirname(lp), { recursive: true }) + fs.writeFileSync(lp, renderLearnings(entries)) + gitIn(dir, ["add", "packages/kilo-docs/LEARNINGS.md"]) + gitIn(dir, ["commit", "-m", "main learnings"]) + + // Branch from main, then remove LEARNINGS.md + gitIn(dir, ["checkout", "-b", "docs/auto-sync"]) + fs.rmSync(lp) + gitIn(dir, ["add", "packages/kilo-docs/LEARNINGS.md"]) + gitIn(dir, ["commit", "-m", "remove learnings on branch"]) + + // Set up origin refs so git show origin/ resolves + setupLearnRepo(dir) + + // git show on branch must fail — file absent at that ref + let branchFailed = false + try { + gitIn(dir, ["show", "origin/docs/auto-sync:packages/kilo-docs/LEARNINGS.md"]) + } catch { + branchFailed = true + } + assert.ok(branchFailed, "git show on branch must fail when LEARNINGS.md absent") + + // git show on main must succeed with the main's entries + const mainContent = gitIn(dir, ["show", "origin/main:packages/kilo-docs/LEARNINGS.md"]) + const parsed = parseLearnings(mainContent) + assert.equal(parsed.length, 1, "main fallback must return the main's entries") + assert.equal(parsed[0].id, "test") + + // Also verify: empty parse and render (unit coverage of the empty case) + const empty = parseLearnings("") + assert.equal(empty.length, 0) + assert.deepEqual(empty, []) + const rendered = renderLearnings([]) + assert.ok(rendered.includes("")) + assert.ok(rendered.includes("")) + } + + // 10m — empty delta advances marker with no file change + { + console.log(" 10m — empty delta marker advance") + const dir = mktemp("docs-sync-learn-m-") + initRepoWithIdentity(dir) + fs.writeFileSync(path.join(dir, "base.txt"), "base\n") + gitIn(dir, ["add", "base.txt"]) + gitIn(dir, ["commit", "-m", "base"]) + gitIn(dir, ["checkout", "-b", "docs/auto-sync"]) + fs.mkdirSync(path.join(dir, "packages", "kilo-docs", "pages"), { recursive: true }) + fs.writeFileSync(path.join(dir, "packages", "kilo-docs", "pages", "x.md"), "# x\n") + gitIn(dir, ["add", "packages/kilo-docs/pages/x.md"]) + gitIn(dir, ["commit", "-m", "docs update", "--author", `kiloconnect[bot] <${kiloconnectBotEmail}>`]) + const tip = gitIn(dir, ["rev-parse", "HEAD"]) + + const cwd = setupLearnRepo(dir) + const fixturePath = writeFixture(cwd, { + pr: { number: 1, head: { ref: "docs/auto-sync" }, body: "", user: { login: "github-actions[bot]" } }, + comments: [], + }) + + const kiloDir = makeStubKiloDir({ mode: "extraction-delta", callLog: path.join(cwd, "kilo-calls.log") }) + writeExtractionDelta(cwd, { add: [], remove: [] }) + + const existing = [] + fs.writeFileSync(path.join(cwd, "docs-sync-out", "learnings.json"), JSON.stringify(existing, null, 2)) + const learningsPath = path.join(cwd, "packages", "kilo-docs", "LEARNINGS.md") + fs.mkdirSync(path.dirname(learningsPath), { recursive: true }) + fs.writeFileSync(learningsPath, renderLearnings(existing)) + const before = fs.readFileSync(learningsPath, "utf8") + + const outputFile = path.join(cwd, "gh-output-m") + const result = runNodeScript(LEARN_SCRIPT, { + cwd, + kiloDir, + env: { + TRIAGE_MODEL: "test/model", + DOCS_SYNC_FIXTURE: fixturePath, + LEARNINGS_BUDGET_MINUTES: "1", + DOCS_SYNC_BACKOFF_MS: "0", + GITHUB_OUTPUT: outputFile, + }, + }) + + // learnings.json unchanged + const out = JSON.parse(fs.readFileSync(path.join(cwd, "docs-sync-out", "learnings.json"), "utf8")) + assert.deepEqual(out, existing) + + // No learned_through output (empty delta route omits it per G5 table) + if (fs.existsSync(outputFile)) { + const ghOut = fs.readFileSync(outputFile, "utf8") + assert.ok(!ghOut.includes("learned_through="), "GITHUB_OUTPUT must not contain learned_through on empty delta") + } + const patched = `${fixturePath}.patched` + assert.ok(fs.existsSync(patched), "marker PATCH file must be written for empty delta") + const markerText = fs.readFileSync(patched, "utf8") + assert.ok(markerText.includes(tip), "marker PATCH must contain tip SHA") + + runNodeScript(LEARN_SCRIPT, { + cwd, + kiloDir, + args: ["--apply"], + env: { DOCS_SYNC_BACKOFF_MS: "0" }, + }) + assert.equal(fs.readFileSync(learningsPath, "utf8"), before, "LEARNINGS.md must be byte-unchanged after apply") + + // patchMarkerIntoBody: existing marker → replaced in place + { + const oldBody = "some text\n\nmore text\n" + const newLine = "" + const patched = patchMarkerIntoBody(oldBody, newLine) + assert.ok(patched.includes(newLine), "new marker must be in body") + assert.ok(!patched.includes("commit=old"), "old marker must be gone") + assert.equal( + (patched.match(/" + const patched = patchMarkerIntoBody(oldBody, newLine) + assert.ok(patched.includes(newLine)) + } + } + + // 10n — non-empty delta routes through upsert (not learn.mjs PATCH) + { + console.log(" 10n — non-empty delta marker route") + const dir = mktemp("docs-sync-learn-n-") + initRepoWithIdentity(dir) + fs.writeFileSync(path.join(dir, "base.txt"), "base\n") + gitIn(dir, ["add", "base.txt"]) + gitIn(dir, ["commit", "-m", "base"]) + gitIn(dir, ["checkout", "-b", "docs/auto-sync"]) + fs.mkdirSync(path.join(dir, "packages", "kilo-docs", "pages"), { recursive: true }) + fs.writeFileSync(path.join(dir, "packages", "kilo-docs", "pages", "x.md"), "# x\n") + gitIn(dir, ["add", "packages/kilo-docs/pages/x.md"]) + gitIn(dir, ["commit", "-m", "docs update", "--author", `kiloconnect[bot] <${kiloconnectBotEmail}>`]) + const source = `commit:${gitIn(dir, ["rev-parse", "HEAD"]).slice(0, 7)}` + const tip = gitIn(dir, ["rev-parse", "HEAD"]) + + const cwd = setupLearnRepo(dir) + const fixturePath = writeFixture(cwd, { + pr: { number: 1, head: { ref: "docs/auto-sync" }, body: "", user: { login: "github-actions[bot]" } }, + comments: [], + }) + + const kiloDir = makeStubKiloDir({ mode: "extraction-delta", callLog: path.join(cwd, "kilo-calls.log") }) + writeExtractionDelta(cwd, { + add: [ + { id: "new-rule", rule: "A new rule.", scope: "both", source: `commit:${tip.slice(0, 7)}`, date: "2026-08-03" }, + ], + remove: [], + }) + + const summaryFile = path.join(cwd, "step-summary.md") + fs.writeFileSync(summaryFile, "") + const outputFile = path.join(cwd, "gh-output") + const result = runNodeScript(LEARN_SCRIPT, { + cwd, + kiloDir, + env: { + TRIAGE_MODEL: "test/model", + DOCS_SYNC_FIXTURE: fixturePath, + LEARNINGS_BUDGET_MINUTES: "1", + DOCS_SYNC_BACKOFF_MS: "0", + GITHUB_STEP_SUMMARY: summaryFile, + GITHUB_OUTPUT: outputFile, + }, + }) + + // No marker PATCH file + assert.ok(!fs.existsSync(`${fixturePath}.patched`), "non-empty delta must not PATCH marker") + + // Assert learned_through was written to GITHUB_OUTPUT (non-empty delta route) + const ghOut = fs.readFileSync(outputFile, "utf8") + assert.ok(ghOut.includes("learned_through="), "GITHUB_OUTPUT must contain learned_through on non-empty delta") + assert.ok(ghOut.includes(`commit=${tip}`), "learned_through must contain tip SHA") + + // renderBody with learnedThrough marker + const marker = "" + const body1 = renderBody({ + date: "2026-08-03", + since: "2026-07-01T00:00:00.000Z", + through: "2026-08-03T00:00:00.000Z", + learnedThrough: marker, + changesRows: [], + pendingRows: [], + skippedRows: [], + verified: true, + draftReasons: [], + note: "", + }) + assert.ok(body1.includes(marker), "renderBody must emit marker when given") + + // renderBody with parameter omitted → no marker + const body2 = renderBody({ + date: "2026-08-03", + since: "2026-07-01T00:00:00.000Z", + through: "2026-08-03T00:00:00.000Z", + changesRows: [], + pendingRows: [], + skippedRows: [], + verified: true, + draftReasons: [], + note: "", + }) + assert.ok(!body2.includes("learned-through"), "renderBody without learnedThrough must emit no marker") + // Still round-trips + const extChanges = extractSectionRows(body2, "changes") + assert.deepEqual(extChanges, []) + } + + // 10o — deleted-in-window rejection + { + console.log(" 10o — deleted-in-window rejection") + // Normalized match catches near-identical wording + { + const delta = { + add: [ + { + id: "dup", + rule: "do not document experimental features!", + scope: "both", + source: "commit:aaaaaaa", + date: "2026-08-03", + }, + ], + remove: [], + } + const { rejected } = validateDelta(delta, { + existing: [], + candidateSources: ["commit:aaaaaaa"], + deletedInWindow: ["Do not document experimental features."], + }) + assert.equal(rejected.length, 1, "normalized match must reject deleted rule") + } + + // Different meaning on same topic — NOT rejected (accepted limit) + { + const delta = { + add: [ + { + id: "good", + rule: "Document experimental features in a separate section.", + scope: "both", + source: "commit:aaaaaaa", + date: "2026-08-03", + }, + ], + remove: [], + } + const { rejected } = validateDelta(delta, { + existing: [], + candidateSources: ["commit:aaaaaaa"], + deletedInWindow: ["Do not document experimental features."], + }) + assert.equal(rejected.length, 0, "different rule on same topic must not be rejected") + } + } + + // 10p — resolveLearnedThrough pure function and anti-drift assertions + { + console.log(" 10p — resolveLearnedThrough + anti-drift") + // Unit tests on the pure export + // env value set wins over body marker + assert.equal( + resolveLearnedThrough({ + envValue: "", + prBody: "", + }), + "", + ) + // env unset, body marker present → body wins + assert.equal( + resolveLearnedThrough({ envValue: "", prBody: "" }), + "", + ) + // both absent → empty string + assert.equal(resolveLearnedThrough({ envValue: "", prBody: "" }), "") + // env set to whitespace → treated as unset + assert.equal( + resolveLearnedThrough({ + envValue: " ", + prBody: "", + }), + "", + ) + + // renderBody emits no marker for "" + const bodyEmpty = renderBody({ + date: "d", + since: "s", + through: "t", + learnedThrough: "", + changesRows: [], + pendingRows: [], + skippedRows: [], + verified: true, + draftReasons: [], + note: "", + }) + assert.ok(!bodyEmpty.includes("learned-through"), "empty learnedThrough must emit no marker") + + // Anti-drift: static-source assertions on upsert-pr.mjs + const upsertSrc = fs.readFileSync(path.join(HERE, "upsert-pr.mjs"), "utf8") + + // (a) exactly one resolveLearnedThrough( call passing process.env.LEARNED_THROUGH + const calls = upsertSrc.match(/resolveLearnedThrough\(/g) || [] + // One in the export definition, one in the call site + assert.ok(calls.length >= 2, `expected at least 2 resolveLearnedThrough( occurrences; got ${calls.length}`) + assert.ok( + upsertSrc.includes("process.env.LEARNED_THROUGH"), + "resolveLearnedThrough must receive process.env.LEARNED_THROUGH", + ) + assert.ok(upsertSrc.includes("prBody"), "resolveLearnedThrough must receive prBody") + + // (b) prBody is at function scope (let prBody before the if block) + const prBodyIdx = upsertSrc.indexOf('let prBody = ""') + assert.ok(prBodyIdx >= 0, 'prBody must be declared at function scope with let prBody = ""') + const ifIdx = upsertSrc.indexOf('if (mode === "update"') + assert.ok(prBodyIdx < ifIdx, 'let prBody must appear before if (mode === "update"...)') + + // (c) renderBody({ argument object contains learnedThrough + const renderBodyIdx = upsertSrc.indexOf("const body = renderBody({") + assert.ok(renderBodyIdx >= 0, "renderBody call must exist") + const afterRenderBody = upsertSrc.slice(renderBodyIdx) + const renderBodyArgsEnd = afterRenderBody.indexOf("})") + const renderBodyArgs = afterRenderBody.slice(0, renderBodyArgsEnd) + assert.ok(renderBodyArgs.includes("learnedThrough"), "renderBody call in main() must pass learnedThrough") + } + + // 10q — dry run makes no live write + { + console.log(" 10q — dry run no live write") + const dir = mktemp("docs-sync-learn-q-") + initRepoWithIdentity(dir) + fs.writeFileSync(path.join(dir, "base.txt"), "base\n") + gitIn(dir, ["add", "base.txt"]) + gitIn(dir, ["commit", "-m", "base"]) + gitIn(dir, ["checkout", "-b", "docs/auto-sync"]) + fs.mkdirSync(path.join(dir, "packages", "kilo-docs", "pages"), { recursive: true }) + fs.writeFileSync(path.join(dir, "packages", "kilo-docs", "pages", "x.md"), "# x\n") + gitIn(dir, ["add", "packages/kilo-docs/pages/x.md"]) + gitIn(dir, ["commit", "-m", "docs update", "--author", `kiloconnect[bot] <${kiloconnectBotEmail}>`]) + + // DRY_RUN=true + { + const cwd = setupLearnRepo(dir) + const fixturePath = writeFixture(cwd, { + pr: { number: 1, head: { ref: "docs/auto-sync" }, body: "", user: { login: "github-actions[bot]" } }, + comments: [], + }) + + const kiloDir = makeStubKiloDir({ mode: "extraction-delta", callLog: path.join(cwd, "kilo-calls.log") }) + writeExtractionDelta(cwd, { add: [], remove: [] }) + + const outputFile = path.join(cwd, "gh-output-q-dry") + const result = runNodeScript(LEARN_SCRIPT, { + cwd, + kiloDir, + env: { + TRIAGE_MODEL: "test/model", + DOCS_SYNC_FIXTURE: fixturePath, + LEARNINGS_BUDGET_MINUTES: "1", + DOCS_SYNC_BACKOFF_MS: "0", + DRY_RUN: "true", + GITHUB_OUTPUT: outputFile, + }, + }) + + assert.ok(!fs.existsSync(`${fixturePath}.patched`), "DRY_RUN must suppress marker PATCH") + if (fs.existsSync(outputFile)) { + const ghOut = fs.readFileSync(outputFile, "utf8") + assert.ok(!ghOut.includes("learned_through="), "GITHUB_OUTPUT must not contain learned_through on DRY_RUN") + } + assert.ok(result.stdout.includes("marker PATCH suppressed"), "stdout must log marker suppression for DRY_RUN") + assert.ok( + result.stdout.includes("would have written marker"), + "stdout must log the suppressed marker for DRY_RUN", + ) + } + + // LEARNINGS_NO_PATCH=1 (same mechanism) + { + const cwd = setupLearnRepo(dir) + const fixturePath = writeFixture(cwd, { + pr: { number: 1, head: { ref: "docs/auto-sync" }, body: "", user: { login: "github-actions[bot]" } }, + comments: [], + }) + + const kiloDir = makeStubKiloDir({ mode: "extraction-delta", callLog: path.join(cwd, "kilo-calls.log") }) + writeExtractionDelta(cwd, { add: [], remove: [] }) + + const outputFile = path.join(cwd, "gh-output-q-nopatch") + const result = runNodeScript(LEARN_SCRIPT, { + cwd, + kiloDir, + env: { + TRIAGE_MODEL: "test/model", + DOCS_SYNC_FIXTURE: fixturePath, + LEARNINGS_BUDGET_MINUTES: "1", + DOCS_SYNC_BACKOFF_MS: "0", + LEARNINGS_NO_PATCH: "1", + GITHUB_OUTPUT: outputFile, + }, + }) + + assert.ok(!fs.existsSync(`${fixturePath}.patched`), "LEARNINGS_NO_PATCH must suppress marker PATCH") + if (fs.existsSync(outputFile)) { + const ghOut = fs.readFileSync(outputFile, "utf8") + assert.ok( + !ghOut.includes("learned_through="), + "GITHUB_OUTPUT must not contain learned_through on LEARNINGS_NO_PATCH", + ) + } + assert.ok( + result.stdout.includes("marker PATCH suppressed"), + "stdout must log marker suppression for LEARNINGS_NO_PATCH", + ) + assert.ok( + result.stdout.includes("would have written marker"), + "stdout must log the suppressed marker for LEARNINGS_NO_PATCH", + ) + } + + // DRY_RUN=true with non-empty delta (suppresses learned_through output) + { + const cwd = setupLearnRepo(dir) + const tipSource = `commit:${gitIn(dir, ["rev-parse", "HEAD"]).slice(0, 7)}` + const fixturePath = writeFixture(cwd, { + pr: { number: 1, head: { ref: "docs/auto-sync" }, body: "", user: { login: "github-actions[bot]" } }, + comments: [], + }) + + const kiloDir = makeStubKiloDir({ mode: "extraction-delta", callLog: path.join(cwd, "kilo-calls.log") }) + writeExtractionDelta(cwd, { + add: [ + { id: "dry-suppress", rule: "A rule suppressed under dry run.", scope: "both", source: tipSource, date: "2026-08-03" }, + ], + remove: [], + }) + + const outputFile = path.join(cwd, "gh-output-q-dry-nonempty") + const result = runNodeScript(LEARN_SCRIPT, { + cwd, + kiloDir, + env: { + TRIAGE_MODEL: "test/model", + DOCS_SYNC_FIXTURE: fixturePath, + LEARNINGS_BUDGET_MINUTES: "1", + DOCS_SYNC_BACKOFF_MS: "0", + DRY_RUN: "true", + GITHUB_OUTPUT: outputFile, + }, + }) + + if (fs.existsSync(outputFile)) { + const ghOut = fs.readFileSync(outputFile, "utf8") + assert.ok( + !ghOut.includes("learned_through="), + "GITHUB_OUTPUT must not contain learned_through on DRY_RUN non-empty delta", + ) + } + assert.ok( + result.stdout.includes("learned-through output suppressed"), + "stdout must log learned-through output suppression for DRY_RUN non-empty delta", + ) + } + + // LEARNINGS_NO_PATCH=1 with non-empty delta + { + const cwd = setupLearnRepo(dir) + const tipSource = `commit:${gitIn(dir, ["rev-parse", "HEAD"]).slice(0, 7)}` + const fixturePath = writeFixture(cwd, { + pr: { number: 1, head: { ref: "docs/auto-sync" }, body: "", user: { login: "github-actions[bot]" } }, + comments: [], + }) + + const kiloDir = makeStubKiloDir({ mode: "extraction-delta", callLog: path.join(cwd, "kilo-calls.log") }) + writeExtractionDelta(cwd, { + add: [ + { id: "nopatch-suppress", rule: "A rule suppressed under no-patch.", scope: "both", source: tipSource, date: "2026-08-03" }, + ], + remove: [], + }) + + const outputFile = path.join(cwd, "gh-output-q-nopatch-nonempty") + const result = runNodeScript(LEARN_SCRIPT, { + cwd, + kiloDir, + env: { + TRIAGE_MODEL: "test/model", + DOCS_SYNC_FIXTURE: fixturePath, + LEARNINGS_BUDGET_MINUTES: "1", + DOCS_SYNC_BACKOFF_MS: "0", + LEARNINGS_NO_PATCH: "1", + GITHUB_OUTPUT: outputFile, + }, + }) + + if (fs.existsSync(outputFile)) { + const ghOut = fs.readFileSync(outputFile, "utf8") + assert.ok( + !ghOut.includes("learned_through="), + "GITHUB_OUTPUT must not contain learned_through on LEARNINGS_NO_PATCH non-empty delta", + ) + } + assert.ok( + result.stdout.includes("learned-through output suppressed"), + "stdout must log learned-through output suppression for LEARNINGS_NO_PATCH non-empty delta", + ) + } + } + + // Prompt block format + { + console.log(" 10 — promptBlock format") + const entries = [ + { id: "r1", rule: "First rule.", scope: "triage", source: "commit:aaaaaaa", date: "2026-08-01" }, + { id: "r2", rule: "Second rule.", scope: "edit", source: "commit:bbbbbbb", date: "2026-08-02" }, + { id: "r3", rule: "Both rule.", scope: "both", source: "commit:ccccccc", date: "2026-08-03" }, + ] + + const triageBlock = promptBlock(entries, "triage") + assert.ok(triageBlock.includes("First rule"), "triage block must include triage-scoped rule") + assert.ok(triageBlock.includes("Both rule"), "triage block must include both-scoped rule") + assert.ok(!triageBlock.includes("Second rule"), "triage block must not include edit-only rule") + assert.ok(triageBlock.includes("## Learnings from maintainer corrections")) + + const editBlock = promptBlock(entries, "edit") + assert.ok(editBlock.includes("Second rule"), "edit block must include edit-scoped rule") + assert.ok(editBlock.includes("Both rule"), "edit block must include both-scoped rule") + assert.ok(!editBlock.includes("First rule"), "edit block must not include triage-only rule") + + // Empty: no matching entries + const emptyBlock = promptBlock( + [{ id: "r1", rule: "R.", scope: "edit", source: "commit:aa", date: "2026-01-01" }], + "triage", + ) + assert.equal(emptyBlock, "") + } + + // parseLearnedThrough + { + console.log(" 10 — parseLearnedThrough") + const body = "stuff\n\nmore" + const parsed = parseLearnedThrough(body) + assert.equal(parsed.commit, "abc1234") + assert.equal(parsed.comment, "2026-08-03T12:00:00Z") + + // none values → null + const noneBody = "" + const noneParsed = parseLearnedThrough(noneBody) + assert.equal(noneParsed.commit, null) + assert.equal(noneParsed.comment, null) + + // absent → both null + const absent = parseLearnedThrough("no marker") + assert.equal(absent.commit, null) + assert.equal(absent.comment, null) + } + + // renderLearnings deterministic order + { + console.log(" 10 — renderLearnings deterministic order") + const entries = [ + { id: "b", rule: "B rule.", scope: "both", source: "commit:bb", date: "2026-08-02" }, + { id: "a", rule: "A rule.", scope: "both", source: "commit:aa", date: "2026-08-01" }, + { id: "c", rule: "C rule.", scope: "both", source: "commit:cc", date: "2026-08-01" }, + ] + const r1 = renderLearnings(entries) + const r2 = renderLearnings(entries) + assert.equal(r1, r2, "renderLearnings must be deterministic") + // Order: date ascending, then id ascending. So a before b before c (a.date < c.date, both before b) + const aPos = r1.indexOf("A rule") + const cPos = r1.indexOf("C rule") + const bPos = r1.indexOf("B rule") + assert.ok(aPos < cPos, "a (earlier date) must come before c") + assert.ok(cPos < bPos, "c (same date as a but later id) must come before b (later date)") + } +} + +// Case 11 — the created rolling PR gets an assignee and a review request +function case11_prOwner() { + console.log("case 11 — created PR assignee and reviewer") + const src = fs.readFileSync(path.join(HERE, "upsert-pr.mjs"), "utf8") + + assert.ok(/const DOCS_OWNER = "\S+"/.test(src), "DOCS_OWNER must be a module constant") + assert.ok(src.includes("/assignees`, {"), "created PR must POST assignees") + assert.ok(src.includes("/requested_reviewers`, {"), "created PR must POST requested_reviewers") + + // Both calls belong to the create arm, after the PR exists. + const createIdx = src.indexOf("const pr = await api(`/repos/${repo()}/pulls`") + assert.ok(createIdx >= 0, "create-PR call must exist") + assert.ok(src.indexOf("/assignees`, {") > createIdx, "assignees POST must follow PR creation") + assert.ok(src.indexOf("/requested_reviewers`, {") > createIdx, "reviewer POST must follow PR creation") + + // A failure here must not fail the run — the PR is already open. + const ownerIdx = src.indexOf("/assignees`, {") + const tryIdx = src.lastIndexOf("try {", ownerIdx) + const catchIdx = src.indexOf("} catch", ownerIdx) + assert.ok(tryIdx >= 0 && catchIdx > src.indexOf("/requested_reviewers`, {"), "both POSTs must sit in one try/catch") +} + // --------------------------------------------------------------------------- // main // --------------------------------------------------------------------------- @@ -1846,6 +3265,8 @@ function main() { case7_cap, case8_triage, case9_reverts, + case10_learnings, + case11_prOwner, ] let failed = 0 for (const fn of cases) { diff --git a/.github/docs-sync/triage.mjs b/.github/docs-sync/triage.mjs index a01eef0a998..eb680846510 100644 --- a/.github/docs-sync/triage.mjs +++ b/.github/docs-sync/triage.mjs @@ -21,6 +21,7 @@ import path from "node:path" import { fileURLToPath } from "node:url" import { parseTriageEntries } from "./extract-json.mjs" import { appendSummary, backoffMsForAttempt, deadline, remainingMs, runKilo, sleepSync } from "./lib.mjs" +import { readLearningsBlock } from "./learn.mjs" const CHUNK_SIZE = 25 const ATTEMPTS = 3 @@ -28,7 +29,7 @@ const OUT_DIR = "docs-sync-out" const CHUNK_TIMEOUT_MS = 10 * 60 * 1000 const HERE = path.dirname(fileURLToPath(import.meta.url)) -const prompt = fs.readFileSync(path.join(HERE, "triage-prompt.md"), "utf8") +const prompt = fs.readFileSync(path.join(HERE, "triage-prompt.md"), "utf8") + readLearningsBlock("triage") const model = process.env.TRIAGE_MODEL if (!model) throw new Error("TRIAGE_MODEL is required") @@ -115,9 +116,7 @@ function triageChunk(chunk, index, budgetDeadline) { console.warn(`chunk ${index}: backing off ${wait / 1000}s before attempt ${attempt + 1}`) sleepSync(wait) } else if (wait > 0) { - console.warn( - `chunk ${index}: skipping backoff — remaining budget cannot fit attempt ${attempt + 1} after wait`, - ) + console.warn(`chunk ${index}: skipping backoff — remaining budget cannot fit attempt ${attempt + 1} after wait`) } } } @@ -125,7 +124,12 @@ function triageChunk(chunk, index, budgetDeadline) { console.warn( `::warning::chunk ${index} failed triage after up to ${ATTEMPTS} attempts; marking ${chunk.length} PRs pending`, ) - return chunk.map((d) => pendingEntry(d, lastCause.includes("triage failed") ? lastCause : `triage failed to classify this PR (${lastCause})`)) + return chunk.map((d) => + pendingEntry( + d, + lastCause.includes("triage failed") ? lastCause : `triage failed to classify this PR (${lastCause})`, + ), + ) } const chunks = [] diff --git a/.github/docs-sync/upsert-pr.mjs b/.github/docs-sync/upsert-pr.mjs index 1ef25b75964..d4b7d2a21fc 100644 --- a/.github/docs-sync/upsert-pr.mjs +++ b/.github/docs-sync/upsert-pr.mjs @@ -27,15 +27,23 @@ const FILE_CAP = 15 const ROW_CAP = 150 const PENDING_DISPLAY_CAP = 60 const SUMMARY_FILE = ".docs-sync-summary.json" +// Owner of the rolling docs PR: assigned and asked for review on creation. +const DOCS_OWNER = "emilieschario" const DOCS_PATH = "packages/kilo-docs" +export const LEARNINGS_FILE = "packages/kilo-docs/LEARNINGS.md" -const git = (args) => execFileSync("git", args, { stdio: ["ignore", "pipe", "inherit"] }).toString().trim() +const git = (args) => + execFileSync("git", args, { stdio: ["ignore", "pipe", "inherit"] }) + .toString() + .trim() // Agent-generated strings land in the PR body next to machine-read markers. // Strip HTML-comment sequences so a crafted/adversarial value cannot forge // section boundaries or the processed-through watermark. function clean(value) { - return String(value ?? "").replaceAll("", "") + return String(value ?? "") + .replaceAll("", "") } function shortRef(url) { @@ -52,7 +60,9 @@ function skippedRow(e) { } function pendingRow(e) { - const reason = clean(e.reason ?? e.cause ?? "").replaceAll("|", "\\|").replaceAll("\n", " ") + const reason = clean(e.reason ?? e.cause ?? "") + .replaceAll("|", "\\|") + .replaceAll("\n", " ") return `| [${shortRef(e.url)}](${clean(e.url)}) | ${reason} |` } @@ -79,7 +89,18 @@ function section(name, header, rows) { return `\n${body}\n` } -export function renderBody({ date, since, through, changesRows, pendingRows, skippedRows, verified, draftReasons, note }) { +export function renderBody({ + date, + since, + through, + learnedThrough = "", + changesRows, + pendingRows, + skippedRows, + verified, + draftReasons, + note, +}) { const pendingDisplay = pendingRows.length > PENDING_DISPLAY_CAP ? [...pendingRows.slice(0, PENDING_DISPLAY_CAP), `| +${pendingRows.length - PENDING_DISPLAY_CAP} more | |`] @@ -108,7 +129,7 @@ ${section("skipped", "| PR | Reason |", skippedRows)} (bot) Generated by the docs-sync workflow. Humans review and merge; while this PR stays open, the next daily run appends new changes here. Branch: \`${BRANCH}\`. -` +${learnedThrough ? learnedThrough + "\n" : ""}` } function mergeRows(oldRows, newRows) { @@ -242,6 +263,23 @@ export function computeProcessedThrough({ uncovered, digest, now, fallback }) { return new Date(earliest - 1).toISOString() } +/** + * Content gate: legitimate bot edits are docs pages and nav files. Only + * those are built and tested during verify (content-integrity.test.ts + * walks pages/ only), so anything else in the docs package forces human + * review. LEARNINGS.md is a root-level .md file, like the three sibling + * .md files already at that level, so it is not built or tested and is + * safe to exclude from the gate. + */ +export function nonContentFiles(changedFiles) { + return (Array.isArray(changedFiles) ? changedFiles : []).filter( + (f) => + f !== LEARNINGS_FILE && + !f.startsWith("packages/kilo-docs/pages/") && + !f.startsWith("packages/kilo-docs/lib/nav/"), + ) +} + /** * Route summary + triage into the three body sections. * changesRows = action neither skipped nor pending @@ -280,6 +318,22 @@ export function dropLegacySkipped(rows) { }) } +/** + * Resolve the learned-through marker for renderBody. + * + * Order: env LEARNED_THROUGH when set and non-empty; else the marker + * parsed out of the existing PR body; else "". + * The fallback is load-bearing: a run where extraction was skipped, + * failed, or already PATCHed the marker itself must not clobber a good + * marker. + */ +export function resolveLearnedThrough({ envValue, prBody }) { + const fromEnv = String(envValue ?? "").trim() + if (fromEnv) return fromEnv + const m = String(prBody ?? "").match(//) + return m ? m[0] : "" +} + /** * No-diff early-return report. Returns summary markdown and an optional * replay warning. Warns IFF sinceOverride && uncovered non-empty (no commit @@ -347,18 +401,14 @@ async function main() { const through = computeProcessedThrough({ uncovered, digest, now, fallback: since }) // The draft cap bounds the cumulative PR diff, not just this run's commit. - const changedFiles = git(["diff", "--name-only", "origin/main...HEAD", "--", DOCS_PATH]) - .split("\n") - .filter(Boolean) + const changedFiles = git(["diff", "--name-only", "origin/main...HEAD", "--", DOCS_PATH]).split("\n").filter(Boolean) const draftReasons = [] if (changedFiles.length > FILE_CAP) draftReasons.push(`diff exceeds ${FILE_CAP} files (${changedFiles.length})`) if (!verified) draftReasons.push("docs build/tests not passing") // Content gate: legitimate bot edits are docs pages and nav files. Anything // else in the docs package (build config, components, tests) executes // during the verify build, so force human review before merge. - const nonContent = changedFiles.filter( - (f) => !f.startsWith("packages/kilo-docs/pages/") && !f.startsWith("packages/kilo-docs/lib/nav/"), - ) + const nonContent = nonContentFiles(changedFiles) if (nonContent.length > 0) { // File paths are agent-chosen; sanitize before they land in the PR body. const listed = nonContent @@ -369,9 +419,17 @@ async function main() { } const draft = draftReasons.length > 0 - git(mode === "update" ? ["push", "origin", `HEAD:${BRANCH}`] : ["push", "--force-with-lease", "origin", `HEAD:${BRANCH}`]) + git( + mode === "update" + ? ["push", "origin", `HEAD:${BRANCH}`] + : ["push", "--force-with-lease", "origin", `HEAD:${BRANCH}`], + ) - const { changesRows: changesNew, pendingRows: pendingNew, skippedRows: skippedNew } = routeRows({ + const { + changesRows: changesNew, + pendingRows: pendingNew, + skippedRows: skippedNew, + } = routeRows({ summary: agentSummary, triage, uncovered, @@ -380,8 +438,10 @@ async function main() { let oldChanges = [] let oldSkipped = [] let oldPending = [] + let prBody = "" if (mode === "update" && existingPr) { const pr = await api(`/repos/${repo()}/pulls/${existingPr}`) + prBody = pr.body ?? "" oldChanges = extractSectionRows(pr.body, "changes") oldSkipped = dropLegacySkipped(extractSectionRows(pr.body, "skipped")) oldPending = extractSectionRows(pr.body, "pending") @@ -392,10 +452,13 @@ async function main() { // extractSectionRows stays exercised; discarded deliberately. void oldPending + const learnedThrough = resolveLearnedThrough({ envValue: process.env.LEARNED_THROUGH, prBody }) + const body = renderBody({ date, since, through, + learnedThrough, changesRows: mergeRows(oldChanges, changesNew), pendingRows: pendingNew, skippedRows: mergeRows(oldSkipped, skippedNew), @@ -445,6 +508,20 @@ async function main() { prNumber = pr.number prUrl = pr.html_url await api(`/repos/${repo()}/issues/${prNumber}/labels`, { method: "POST", body: { labels: ["auto-docs"] } }) + // Best effort: the PR already exists here, so a non-collaborator or a + // revoked account must not fail the run. + try { + await api(`/repos/${repo()}/issues/${prNumber}/assignees`, { + method: "POST", + body: { assignees: [DOCS_OWNER] }, + }) + await api(`/repos/${repo()}/pulls/${prNumber}/requested_reviewers`, { + method: "POST", + body: { reviewers: [DOCS_OWNER] }, + }) + } catch (err) { + console.warn(`::warning::docs-sync: could not assign or request review from ${DOCS_OWNER}: ${err.message}`) + } if (mode === "conflict" && existingPr) { await api(`/repos/${repo()}/issues/${existingPr}/comments`, { method: "POST", @@ -459,7 +536,9 @@ async function main() { appendSummary( `### docs-sync PR\n\n- ${prUrl}\n- changed files: ${changedFiles.length}\n- draft: ${draft}\n- uncovered: ${uncovered.length}\n- processed-through: ${through}\n`, ) - console.log(`PR ${prNumber}: ${prUrl} (draft=${draft}, files=${changedFiles.length}, uncovered=${uncovered.length}, through=${through})`) + console.log( + `PR ${prNumber}: ${prUrl} (draft=${draft}, files=${changedFiles.length}, uncovered=${uncovered.length}, through=${through})`, + ) } const isMain = process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href diff --git a/.github/workflows/docs-sync.yml b/.github/workflows/docs-sync.yml index 147cde8a474..1ae78f3edb4 100644 --- a/.github/workflows/docs-sync.yml +++ b/.github/workflows/docs-sync.yml @@ -66,13 +66,13 @@ jobs: sync: if: github.repository == 'Kilo-Org/kilocode' && github.event_name != 'pull_request' runs-on: blacksmith-4vcpu-ubuntu-2404 - # Budget: 4 setup/collect + 90 triage + 120 edit + 2 verify + 10 fix + 2 upsert = 228 min, 12-minute reserve. + # Budget: 4 setup/collect + 10 learn + 90 triage + 120 edit + 2 verify + 10 fix + 2 upsert = 238 min, 12-minute reserve. # These are ceilings, not costs: a caught-up run triages ~2 chunks and edits # ~1 batch and finishes in ~25 min. The old 35/50 pair was the binding # constraint on backlog drain — run 30306629290 deferred 54 PRs untriaged and # 31 unedited purely on budget, with no attempt made. See the throughput note # in the PR description for the arithmetic. - timeout-minutes: 240 + timeout-minutes: 250 env: # Both are required: without KILO_ORG_ID the gateway bills the key # owner's personal balance (402 "Add credits") instead of the org. @@ -110,6 +110,15 @@ jobs: INPUT_SINCE: ${{ inputs.since }} run: node .github/docs-sync/watermark.mjs + - name: Learn from maintainer corrections + id: learn + continue-on-error: true + env: + GH_TOKEN: ${{ github.token }} + LEARNINGS_BUDGET_MINUTES: "10" + DRY_RUN: ${{ inputs.dry_run }} + run: node .github/docs-sync/learn.mjs + - name: Collect merged PRs id: collect env: @@ -215,10 +224,15 @@ jobs: echo "ok=false" >> "$GITHUB_OUTPUT" fi + - name: Write the learnings file + if: (steps.worthy.outputs.count || '0') != '0' && inputs.dry_run != true + run: node .github/docs-sync/learn.mjs --apply + - name: Upsert rolling PR if: (steps.worthy.outputs.count || '0') != '0' && inputs.dry_run != true env: GH_TOKEN: ${{ github.token }} + LEARNED_THROUGH: ${{ steps.learn.outputs.learned_through }} PROCESSED_THROUGH: ${{ steps.wm.outputs.now }} SINCE: ${{ steps.wm.outputs.since }} SINCE_OVERRIDE: ${{ steps.wm.outputs.since_override }} diff --git a/packages/kilo-docs/LEARNINGS.md b/packages/kilo-docs/LEARNINGS.md new file mode 100644 index 00000000000..44bb6a77430 --- /dev/null +++ b/packages/kilo-docs/LEARNINGS.md @@ -0,0 +1,11 @@ +# docs-sync learnings + +Rules the docs-sync bot learned from maintainer corrections to its rolling pull request. +The bot reads this file at the start of every run and follows every rule below. + +To unlearn a rule, delete its line and commit. The next run reads this file from the +branch, so the rule is gone from its input, and the deletion itself is a correction the +extraction step is instructed not to undo. + + +