feat(cli): support XLSX text extraction in read tool

This commit is contained in:
marius-kilocode
2026-05-29 16:49:08 +02:00
parent ec3d7a3f67
commit 2081af2b33
6 changed files with 316 additions and 4 deletions
+1
View File
@@ -191,6 +191,7 @@
"web-tree-sitter": "0.25.10",
"which": "6.0.1",
"xdg-basedir": "5.1.0",
"xlsx": "https://cdn.sheetjs.com/xlsx-0.20.3/xlsx-0.20.3.tgz",
"yargs": "18.0.0",
"zod": "catalog:",
"zod-to-json-schema": "3.24.5"
@@ -0,0 +1,72 @@
import path from "path"
import { Readable } from "stream"
import { read, utils, type CellObject, type WorkBook } from "xlsx"
const ROW_LIMIT = 50_000
export function is(filepath: string) {
return path.extname(filepath).toLowerCase() === ".xlsx"
}
export async function open(filepath: string) {
const bytes = new Uint8Array(await Bun.file(filepath).arrayBuffer())
if (bytes[0] !== 0x50 || bytes[1] !== 0x4b) {
throw new Error(`Cannot read spreadsheet file: ${filepath} is not a valid XLSX workbook`)
}
try {
const book = read(bytes, { type: "array", cellDates: true })
return Readable.from(lines(book))
} catch (err) {
const message = err instanceof Error ? err.message : String(err)
throw new Error(`Cannot read spreadsheet file: ${filepath}: ${message}`, { cause: err })
}
}
function cell(value: CellObject | undefined) {
if (!value) return ""
if (value.f) {
if (value.w !== undefined && value.w !== null) return value.w
if (value.v !== undefined && value.v !== null) return String(value.v)
return `[Formula: ${value.f}]`
}
if (value.v === undefined || value.v === null) return ""
if (value.t === "e") return `[Error: ${value.w ?? String(value.v)}]`
if (value.t === "d") return value.v instanceof Date ? value.v.toISOString().slice(0, 10) : String(value.v)
if (value.l?.Target) return `${value.w ?? String(value.v)} (${value.l.Target})`
return value.w ?? String(value.v)
}
function* lines(book: WorkBook) {
const sheets = book.SheetNames.filter((_, index) => {
const hidden = book.Workbook?.Sheets?.[index]?.Hidden
return hidden !== 1 && hidden !== 2
})
for (const [index, name] of sheets.entries()) {
if (index > 0) yield "\n"
yield `--- Sheet: ${name} ---\n`
const sheet = book.Sheets[name]
if (!sheet?.["!ref"]) continue
const range = utils.decode_range(sheet["!ref"])
const end = Math.min(range.e.r, ROW_LIMIT - 1)
const rows = new Map<number, Map<number, string>>()
for (const key of Object.keys(sheet)) {
if (key.startsWith("!")) continue
const pos = utils.decode_cell(key)
if (pos.r < range.s.r || pos.r > end || pos.c < range.s.c || pos.c > range.e.c) continue
const value = cell(sheet[key])
if (!value.trim()) continue
const row = rows.get(pos.r) ?? new Map<number, string>()
row.set(pos.c, value)
rows.set(pos.r, row)
}
for (const values of [...rows.entries()].sort((a, b) => a[0] - b[0]).map((entry) => entry[1])) {
const last = Math.max(...values.keys())
const row = Array.from({ length: last - range.s.c + 1 }, (_, col) => values.get(col + range.s.c) ?? "")
yield row.join("\t") + "\n"
}
if (range.e.r > end) yield `[... truncated at row ${ROW_LIMIT} ...]\n`
}
}
+10 -4
View File
@@ -15,6 +15,7 @@ import { isPdfAttachment, sniffAttachmentMime } from "@/util/media"
// kilocode_change start
import * as Encoding from "../kilocode/encoding"
import * as TextStream from "../kilocode/text-stream"
import * as Xlsx from "../kilocode/tool/xlsx"
// kilocode_change end
const DEFAULT_READ_LIMIT = 2000
@@ -300,13 +301,18 @@ export const ReadTool = Tool.define(
}
}
if (isBinaryFile(filepath, sample)) {
// kilocode_change start - extract XLSX text before generic binary rejection
const xlsx = Xlsx.is(filepath)
const opts = { limit: params.limit ?? DEFAULT_READ_LIMIT, offset: params.offset || 1 }
const read = xlsx
? () => Xlsx.open(filepath).then((stream) => readLines(stream, opts))
: () => lines(filepath, opts)
if (!xlsx && isBinaryFile(filepath, sample)) {
return yield* Effect.fail(new Error(`Cannot read binary file: ${filepath}`))
}
const file = yield* Effect.promise(() =>
lines(filepath, { limit: params.limit ?? DEFAULT_READ_LIMIT, offset: params.offset || 1 }),
)
const file = yield* Effect.promise(read)
// kilocode_change end
if (file.count < file.offset && !(file.count === 0 && file.offset === 1)) {
return yield* Effect.fail(
new Error(`Offset ${file.offset} is out of range for this file (${file.count} lines)`),
@@ -0,0 +1,225 @@
import { Cause, Effect, Exit, Layer } from "effect"
import { describe, expect } from "bun:test"
import path from "path"
import { write, utils, type WorkBook, type WorkSheet } from "xlsx"
import { TextReader, TextWriter, Uint8ArrayReader, Uint8ArrayWriter, ZipReader, ZipWriter } from "@zip.js/zip.js"
import { Agent } from "../../src/agent/agent"
import * as CrossSpawnSpawner from "@opencode-ai/core/cross-spawn-spawner"
import { AppFileSystem } from "@opencode-ai/core/filesystem"
import { LSP } from "../../src/lsp/lsp"
import { Instruction } from "../../src/session/instruction"
import { MessageID, SessionID } from "../../src/session/schema"
import { ReadTool } from "../../src/tool/read"
import { Tool } from "../../src/tool/tool"
import { Truncate } from "../../src/tool/truncate"
import { provideInstance, tmpdirScoped } from "../fixture/fixture"
import { testEffect } from "../lib/effect"
const ctx = {
sessionID: SessionID.make("ses_test"),
messageID: MessageID.make(""),
callID: "",
agent: "code",
abort: AbortSignal.any([]),
messages: [],
metadata: () => Effect.void,
ask: () => Effect.void,
}
const it = testEffect(
Layer.mergeAll(
Agent.defaultLayer,
AppFileSystem.defaultLayer,
CrossSpawnSpawner.defaultLayer,
Instruction.defaultLayer,
LSP.defaultLayer,
Truncate.defaultLayer,
),
)
const run = Effect.fn("XlsxReadTest.run")(function* (
dir: string,
file: string,
opts: { limit?: number; offset?: number } = {},
) {
const info = yield* ReadTool
const tool = yield* Tool.init(info)
return yield* provideInstance(dir)(tool.execute({ filePath: file, ...opts }, ctx))
})
const fail = Effect.fn("XlsxReadTest.fail")(function* (dir: string, file: string) {
const exit = yield* run(dir, file).pipe(Effect.exit)
if (Exit.isFailure(exit)) {
const err = Cause.squash(exit.cause)
return err instanceof Error ? err : new Error(String(err))
}
throw new Error("expected read to fail")
})
const put = Effect.fn("XlsxReadTest.put")(function* (file: string, bytes: Uint8Array | string) {
const fs = yield* AppFileSystem.Service
yield* fs.writeWithDirs(file, bytes)
})
function bytes(book: WorkBook) {
return new Uint8Array(write(book, { bookType: "xlsx", type: "buffer" }) as Uint8Array)
}
async function range(bytes: Uint8Array) {
const reader = new ZipReader(new Uint8ArrayReader(bytes))
const output = new ZipWriter(new Uint8ArrayWriter())
for (const entry of await reader.getEntries()) {
if (entry.directory) {
await output.add(entry.filename)
continue
}
if (entry.filename === "xl/worksheets/sheet1.xml") {
const xml = await entry.getData!(new TextWriter())
const sheet = xml
.replace(/ref="A1"/, 'ref="A1:XFD50001"')
.replace("</sheetData>", '<row r="50001"><c r="A50001" t="str"><v>last</v></c></row></sheetData>')
await output.add(entry.filename, new TextReader(sheet))
continue
}
await output.add(entry.filename, new Uint8ArrayReader(await entry.getData!(new Uint8ArrayWriter())))
}
await reader.close()
return output.close()
}
function book(sheet: WorkSheet, name = "Visible") {
const value = utils.book_new()
utils.book_append_sheet(value, sheet, name)
return value
}
describe("kilocode XLSX reads", () => {
it.live("extracts labelled formatted content from case-variant XLSX files", () =>
Effect.gen(function* () {
const dir = yield* tmpdirScoped()
const sheet: WorkSheet = {
A1: { t: "s", v: "Link", l: { Target: "https://kilo.ai" } },
B1: { t: "d", v: new Date("2026-05-29T00:00:00.000Z") },
C1: { t: "n", v: 42, f: "SUM(40,2)" },
D1: { t: "e", v: 0x07, w: "#DIV/0!" },
C2: { t: "n", f: "SUM(C1:C1)" },
A4: { t: "s", v: "After blank row" },
"!ref": "A1:D4",
}
const file = path.join(dir, "report.XLSX")
yield* put(file, bytes(book(sheet)))
const result = yield* run(dir, file)
expect(result.output).toContain("--- Sheet: Visible ---")
expect(result.output).toContain("Link (https://kilo.ai)")
expect(result.output).toContain("2026-05-29")
expect(result.output).toContain("42")
expect(result.output).toContain("[Formula: SUM(C1:C1)]")
expect(result.output).toContain("[Error: #DIV/0!]")
expect(result.output).toContain("After blank row")
expect(result.attachments).toBeUndefined()
}),
)
it.live("omits hidden and very-hidden worksheets", () =>
Effect.gen(function* () {
const dir = yield* tmpdirScoped()
const value = book(utils.aoa_to_sheet([["Visible content"]]))
utils.book_append_sheet(value, utils.aoa_to_sheet([["Hidden content"]]), "Hidden")
utils.book_append_sheet(value, utils.aoa_to_sheet([["Secret content"]]), "Secret")
value.Workbook = { Sheets: [{ Hidden: 0 }, { Hidden: 1 }, { Hidden: 2 }] }
const file = path.join(dir, "sheets.xlsx")
yield* put(file, bytes(value))
const result = yield* run(dir, file)
expect(result.output).toContain("Visible content")
expect(result.output).not.toContain("Hidden content")
expect(result.output).not.toContain("Secret content")
}),
)
it.live("caps worksheet extraction rows before ordinary read limits", () =>
Effect.gen(function* () {
const dir = yield* tmpdirScoped()
const sheet = utils.aoa_to_sheet(Array.from({ length: 50_001 }, (_, row) => [`row-${row + 1}`]))
const file = path.join(dir, "large.xlsx")
yield* put(file, bytes(book(sheet)))
const result = yield* run(dir, file, { offset: 49_999, limit: 4 })
expect(result.output).toContain("row-50000")
expect(result.output).not.toContain("row-50001")
expect(result.output).toContain("[... truncated at row 50000 ...]")
}),
)
it.live("applies ordinary read line limits to spreadsheet text", () =>
Effect.gen(function* () {
const dir = yield* tmpdirScoped()
const file = path.join(dir, "limited.xlsx")
yield* put(file, bytes(book(utils.aoa_to_sheet([["one"], ["two"], ["three"]]))))
const result = yield* run(dir, file, { limit: 2 })
expect(result.metadata.truncated).toBe(true)
expect(result.output).toContain("1: --- Sheet: Visible ---")
expect(result.output).toContain("2: one")
expect(result.output).not.toContain("3: two")
}),
)
it.live("does not traverse every blank cell in a sparse wide range", () =>
Effect.gen(function* () {
const dir = yield* tmpdirScoped()
const file = path.join(dir, "sparse.xlsx")
const source = bytes(book(utils.aoa_to_sheet([["first"]])))
yield* put(file, yield* Effect.promise(() => range(source)))
const result = yield* run(dir, file)
expect(result.output).toContain("--- Sheet: Visible ---")
expect(result.output).toContain("first")
expect(result.output).toContain("[... truncated at row 50000 ...]")
}),
)
it.live("fails clearly for invalid spreadsheet input", () =>
Effect.gen(function* () {
const dir = yield* tmpdirScoped()
const file = path.join(dir, "invalid.xlsx")
yield* put(file, "not an xlsx workbook")
const err = yield* fail(dir, file)
expect(err.message).toContain("Cannot read spreadsheet file")
expect(err.message).toContain("not a valid XLSX workbook")
}),
)
it.live("continues rejecting unsupported workbook formats as binary", () =>
Effect.gen(function* () {
const dir = yield* tmpdirScoped()
const file = path.join(dir, "legacy.xls")
yield* put(file, bytes(book(utils.aoa_to_sheet([["ignored"]]))))
const err = yield* fail(dir, file)
expect(err.message).toContain("Cannot read binary file")
}),
)
it.live("continues returning PDF files as native attachments", () =>
Effect.gen(function* () {
const dir = yield* tmpdirScoped()
const file = path.join(dir, "native.pdf")
yield* put(file, "%PDF-1.7\n")
const result = yield* run(dir, file)
expect(result.output).toBe("PDF read successfully")
expect(result.attachments?.[0].mime).toBe("application/pdf")
}),
)
})