diff --git a/CLAUDE.md b/CLAUDE.md index f1c29cd0..280f003f 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -292,7 +292,7 @@ Codeman is a Claude Code session manager with web interface and autonomous Ralph **Attachments** (live external document references; all wiring in `file-routes.ts`): a **registry** maps a stable `attachmentId` to a realpath-resolved, extension-allowlisted absolute path, so browser requests never carry arbitrary absolute paths. ⚠️ The **magic-link scanner** (`codeman://attach?...` in terminal output) is **prompt-injectable**, so its scan path is force-confined to the session workspace; a hostile prompt could otherwise exfiltrate arbitrary host files over SSE. The security gate is an extension **allowlist**, not a blocklist. `document-conversion-limiter.ts` caps converter spawns globally: without it, N large docs detected at once fork N multi-minute processes, which is a resource-exhaustion vector. → [architecture-invariants#attachments](docs/architecture-invariants.md#attachments) -**File-path links (terminal + chat)**: a path an agent prints is clickable on BOTH surfaces and opens the file-preview overlay. ⚠️ ONE pattern (`FILE_PATH_LINK_PATTERN` / `absoluteFilePathPattern()` in constants.js) feeds the xterm link provider AND `_linkifyFilePaths()`, a fresh instance per call (`lastIndex`). The chat linkifier walks TEXT NODES with DOM APIs, never rebuilds sanitized markup as a string. ⚠️ An out-of-workspace path goes through the ATTACHMENT routes (`POST /api/sessions/:id/attachments` with `notify: false`), never by widening `file-content`/`file-raw` or `file-stream-manager`'s `tail -f` allowlist. ⚠️ `TEXT_ATTACHMENT_EXTENSIONS` IS `EDITABLE_EXTENSIONS` (never a second list), and widening READ must never widen RUN: `html`/`htm`/`svg` stay download-only, other text is inert `text/plain`+`nosniff`. Media extensions are single-sourced in `attachment-registry.ts`. ⚠️ **XLSX previews parse in the BROWSER**, never on the server: `spreadsheet-preview.js` fetches the raw route with `?preview=true` (413 above `MAX_XLSX_BROWSER_PREVIEW_BYTES`, 10 MB) and hands the bytes to `spreadsheet-preview-worker.js`, the only place the pinned `exceljs`/`fflate` vendor bundles load (never on page load). `admitXlsx()` caps the ZIP before ExcelJS runs, counting what ExcelJS will EXPAND as well as what it reads (a merge costs its area, a `
` notice, and the refusal messages never echo the code. ⚠️ Cell text reaching the page is bounded in the worker: `formatCellValue` caps every display string at `LIMITS.maxCellTextChars` (1,000; the last kept character is an ellipsis and a cut never splits a surrogate pair) for every value shape, strings, rich text (runs are joined only until the cap is passed), hyperlink text, formula source and results, and errors. Without it each tile cell carried its whole string and structured clone copied it once per cell, after the load timeout had already been cleared by the metadata reply: one 1 MB shared string over a 60 x 20 block left the page unresponsive for 150 s at 9.6 GB. A cell is one `nowrap` line with an ellipsis, so nothing past the column width was ever shown. A worker test asserts every tile cell's `text` is within the cap for a long shared string, rich text and a hyperlink. The worker is a stable URL cache-busted by `SPREADSHEET_ASSET_VERSION` in `spreadsheet-preview.js`, the content hash of the worker, the core and both vendor bundles; `npm run check:public-assets` fails when it drifts, so editing any of those (or bumping either package) means updating the token, or a deploy pairs a new worker with a year-cached old core.
### Filesystem path picker
diff --git a/src/web/public/spreadsheet-preview.js b/src/web/public/spreadsheet-preview.js
index ff630d83..eee04d92 100644
--- a/src/web/public/spreadsheet-preview.js
+++ b/src/web/public/spreadsheet-preview.js
@@ -20,7 +20,7 @@
(function initSpreadsheetPreview(global) {
'use strict';
- const SPREADSHEET_ASSET_VERSION = '6e843e269018';
+ const SPREADSHEET_ASSET_VERSION = 'd83f632d5c69';
const MAX_PREVIEW_BYTES = 10 * 1024 * 1024;
const DEFAULT_TIMEOUT_MS = 20000;
const MAX_SCROLL_PX = 8000000;
diff --git a/src/web/public/spreadsheet-xlsx-core.js b/src/web/public/spreadsheet-xlsx-core.js
index fb71f1da..15c2d994 100644
--- a/src/web/public/spreadsheet-xlsx-core.js
+++ b/src/web/public/spreadsheet-xlsx-core.js
@@ -27,12 +27,24 @@
maxCellsPerSheet: 100000,
maxRows: 250000,
maxRowsPerSheet: 100000,
- maxMergesPerSheet: 5000,
+ // ExcelJS's `_mergeCellsInternal` checks each new merge against every
+ // earlier one on its sheet, so a sheet costs the SQUARE of its merge count
+ // (5,000 on one sheet took 1.6 s). Both caps keep the worst case near 1.3 s.
+ maxMergesPerSheet: 2000,
+ maxMerges: 10000,
maxStyles: 5000,
// ExcelJS stores each sheet at `_worksheets[sheetId]`, so the id is an array
// length. Excel numbers sheets from 1 and never reuses an id, so real ids
// stay small; 65535 leaves room for heavy editing at a negligible cost.
maxSheetId: 65535,
+ // Excel's own limit on a number format. ExcelJS runs `isDateFmt` on the code
+ // once per numeric cell, and the code is echoed into the notice bar.
+ maxNumFmtChars: 255,
+ // Display text per cell. A cell is one `nowrap` line ending in an ellipsis,
+ // so nothing past the column width shows; uncapped, every tile cell carried
+ // its whole string to the page (a 1 MB shared string over a 60 x 20 block
+ // froze the page's main thread at 9.6 GB).
+ maxCellTextChars: 1000,
});
const MAX_ROW = 1048576;
const MAX_COL = 16384;
@@ -240,6 +252,35 @@
}
}
+ // Attribute values as the XML parser inside ExcelJS (saxes) hands them over:
+ // the predefined entities and character references decoded. Checking the raw
+ // text would let `[` hide a `[` and would overcount `"`.
+ function decodeXmlAttribute(value) {
+ return value.replace(/&(?:#x([0-9a-fA-F]+)|#([0-9]+)|(amp|lt|gt|quot|apos));/g, (entity, hex, dec, name) => {
+ if (name) return { amp: '&', lt: '<', gt: '>', quot: '"', apos: "'" }[name];
+ const code = hex ? Number.parseInt(hex, 16) : Number(dec);
+ return code <= 0x10ffff ? String.fromCodePoint(code) : entity;
+ });
+ }
+
+ // ExcelJS's `isDateFmt` strips `/\[[^\]]*]/g` from the code once per numeric
+ // cell, and that pattern rescans to the end of the code for every `[` with no
+ // later `]` (a 60,000-character code of `[` cost 2.2 s a cell; even 255
+ // characters added up at the cell cap). Once nothing follows the last `]`
+ // but text without `[`, each `[` stops at the next `]` and the scan is linear.
+ // The messages never echo the code.
+ function checkNumberFormat(attributes, limits) {
+ const raw = attributes.get('formatCode');
+ if (raw === undefined) return;
+ const code = decodeXmlAttribute(raw);
+ if (code.length > limits.maxNumFmtChars) {
+ fail('number-format', `Workbook has a number format longer than ${limits.maxNumFmtChars} characters`);
+ }
+ if (code.lastIndexOf('[') > code.lastIndexOf(']')) {
+ fail('number-format', 'Workbook has a number format with an unclosed bracket');
+ }
+ }
+
function createXmlCounter(name, counts, limits) {
let tail = '';
const decoder = new TextDecoder();
@@ -286,8 +327,8 @@
}
sheetMerges += 1;
counts.merges += 1;
- if (sheetMerges > limits.maxMergesPerSheet)
- fail('merge-limit', 'Worksheet exceeds the merged ranges limit');
+ if (sheetMerges > limits.maxMergesPerSheet || counts.merges > limits.maxMerges)
+ fail('merge-limit', 'Workbook exceeds the merged ranges limit');
addCells(mergeArea(attributes));
}
}
@@ -304,6 +345,12 @@
// sits in can be desynced by a closing tag inside an XML comment.
counts.styles += (scan.match(/ ])/g) || []).length;
if (counts.styles > limits.maxStyles) fail('style-limit', 'Workbook exceeds the cell styles limit');
+ // Every , cellXfs' and dxfs' alike; the lookahead skips .
+ for (const match of scan.matchAll(/ ])/g)) {
+ const attributes = readTagAttributes(scan, match.index + match[0].length);
+ if (!attributes) fail('malformed', 'Workbook has a whose attributes do not parse');
+ checkNumberFormat(attributes, limits);
+ }
}
tail = text.slice(safeEnd);
if (tail.length > MAX_CARRIED_TAG) fail('malformed', 'Workbook XML has an oversized tag');
@@ -529,33 +576,68 @@
return 'formula' in value || 'sharedFormula' in value;
}
+ // Cut to LIMITS.maxCellTextChars, ending in an ellipsis, never between the
+ // halves of a surrogate pair.
+ function capCellText(text) {
+ const max = LIMITS.maxCellTextChars;
+ if (text.length <= max) return text;
+ let end = max - 1;
+ const last = text.charCodeAt(end - 1);
+ if (last >= 0xd800 && last <= 0xdbff) end -= 1;
+ return `${text.slice(0, end)}…`;
+ }
+
+ // Joins rich-text runs only until the cap is passed, so a long run is never
+ // copied whole once per cell.
+ function richTextPrefix(runs) {
+ let text = '';
+ for (const run of runs) {
+ if (text.length > LIMITS.maxCellTextChars) break;
+ if (typeof run?.text === 'string') text += run.text.slice(0, LIMITS.maxCellTextChars + 1);
+ }
+ return text;
+ }
+
+ /**
+ * A cell's display text and any warning. The text is ALWAYS at most
+ * `LIMITS.maxCellTextChars` characters, whatever shape the value has.
+ */
+ function formatCellValue(value, format, date1904) {
+ const formatted = formatCellValueUncapped(value, format, date1904);
+ return formatted.text.length > LIMITS.maxCellTextChars
+ ? Object.assign({}, formatted, { text: capCellText(formatted.text) })
+ : formatted;
+ }
+
// Every non-scalar shape ExcelJS loads a cell value as. Anything not handled
// here would otherwise reach String() and render as "[object Object]".
- function formatCellValue(value, format, date1904) {
+ function formatCellValueUncapped(value, format, date1904) {
if (value === null || value === undefined) return { text: '' };
if (isDateValue(value)) {
const code = String(format || 'General');
const known = DATE_FORMAT.test(code) || TIME_FORMAT.test(code) || DATE_TIME_FORMAT.test(code);
const serial = dateToSerial(value, date1904);
- if (known) return formatCellValue(serial, code, date1904);
+ if (known) return formatCellValueUncapped(serial, code, date1904);
const fallback = serial % 1 === 0 ? 'yyyy-mm-dd' : 'yyyy-mm-dd hh:mm';
- const formatted = formatCellValue(serial, fallback, date1904);
+ const formatted = formatCellValueUncapped(serial, fallback, date1904);
return /^General$/i.test(code)
? formatted
: { text: formatted.text, warning: `Unsupported number format: ${code}` };
}
if (typeof value === 'object') {
if (isFormulaValue(value)) {
- if (value.result !== undefined && value.result !== null) return formatCellValue(value.result, format, date1904);
+ if (value.result !== undefined && value.result !== null) {
+ return formatCellValueUncapped(value.result, format, date1904);
+ }
const source = typeof value.formula === 'string' ? `=${value.formula}` : '';
return { text: source, warning: 'Formula has no cached result' };
}
if (typeof value.error === 'string') return { text: value.error };
if (Array.isArray(value.richText)) {
- return { text: value.richText.map((run) => (typeof run?.text === 'string' ? run.text : '')).join('') };
+ return { text: richTextPrefix(value.richText) };
}
// Hyperlink: the display text, never the target. The text may be rich.
- if ('text' in value) return formatCellValue(value.text, 'General', date1904);
+ if ('text' in value) return formatCellValueUncapped(value.text, 'General', date1904);
return { text: '', warning: 'Unsupported cell value' };
}
const code = String(format || 'General');
diff --git a/test/spreadsheet-preview-worker.test.ts b/test/spreadsheet-preview-worker.test.ts
index 1c89f1ea..cbf3f9f0 100644
--- a/test/spreadsheet-preview-worker.test.ts
+++ b/test/spreadsheet-preview-worker.test.ts
@@ -856,3 +856,133 @@ describe('spreadsheet preview worker: rows and cells are indexed by their presen
}
}, 60_000);
});
+
+/** Deterministic, poorly compressible text, so the ratio cap does not refuse it first. */
+function noisyText(length: number, seed = 1): string {
+ let state = seed;
+ let text = '';
+ for (let i = 0; i < length; i += 1) {
+ state = (state * 1103515245 + 12345) & 0x7fffffff;
+ text += String.fromCharCode(97 + (state % 26));
+ }
+ return text;
+}
+
+describe('spreadsheet preview worker: cell text reaching the page is bounded', () => {
+ // Every tile cell carried its whole string and structured clone copied it
+ // once per cell: one 1 MB shared string over a 60 x 20 block froze the page.
+ it('caps every tile cell text, shared string, rich text and hyperlink alike', async () => {
+ const long = noisyText(200_000);
+ const workbook = new ExcelJS.Workbook();
+ const sheet = workbook.addWorksheet('Long');
+ for (let row = 1; row <= 20; row += 1) {
+ for (let col = 1; col <= 10; col += 1) sheet.getCell(row, col).value = long;
+ }
+ sheet.getCell(21, 1).value = { richText: [{ text: long }, { font: { bold: true }, text: long }] };
+ sheet.getCell(21, 2).value = { text: long, hyperlink: 'https://example.invalid/' };
+ sheet.getCell(21, 3).value = 'short';
+ const bytes = await writeWorkbook(workbook);
+ const entries = fflate.unzipSync(new Uint8Array(bytes));
+ // One shared string (index 0), referenced by every cell of the block.
+ expect(fflate.strFromU8(entries['xl/sharedStrings.xml'])).toContain(`${long} `);
+ expect(
+ fflate.strFromU8(entries['xl/worksheets/sheet1.xml']).match(/t="s">0<\/v>/g)?.length
+ ).toBeGreaterThanOrEqual(200);
+
+ const harness = createHarness();
+ const metadata = await loadMetadata(harness, bytes);
+ const tile = await requestTile(harness, metadata.sheets[0].id, { r1: 1, c1: 1, r2: 21, c2: 10 });
+ expect(tile.type).toBe('tile');
+ const cells = tile.cells as Array<{ row: number; col: number; text: string }>;
+ expect(cells).toHaveLength(203);
+ for (const cell of cells) expect(cell.text.length, `${cell.row}:${cell.col}`).toBeLessThanOrEqual(1000);
+ expect(cells.find((cell) => cell.row === 1 && cell.col === 1)?.text).toBe(`${long.slice(0, 999)}…`);
+ expect(cells.find((cell) => cell.row === 21 && cell.col === 1)?.text.length).toBe(1000);
+ expect(cells.find((cell) => cell.row === 21 && cell.col === 2)?.text.length).toBe(1000);
+ expect(cells.find((cell) => cell.row === 21 && cell.col === 3)?.text).toBe('short');
+ }, 60_000);
+});
+
+/** A workbook of `sheets` one-cell sheets, each carrying `mergesPerSheet` one-row merges. */
+async function mergeHeavyWorkbook(sheets: number, mergesPerSheet: number): Promise {
+ const workbook = new ExcelJS.Workbook();
+ for (let i = 1; i <= sheets; i += 1) workbook.addWorksheet(`S${i}`).getCell('A1').value = 'one';
+ const entries = fflate.unzipSync(new Uint8Array(await workbook.xlsx.writeBuffer()));
+ const merges = Array.from({ length: mergesPerSheet }, (_, i) => ` `).join('');
+ for (let i = 1; i <= sheets; i += 1) {
+ const name = `xl/worksheets/sheet${i}.xml`;
+ const sheet = fflate.strFromU8(entries[name]);
+ const patched = sheet.replace(
+ '',
+ `${merges} `
+ );
+ expect(patched).not.toBe(sheet);
+ entries[name] = fflate.strToU8(patched);
+ }
+ return toArrayBuffer(fflate.zipSync(entries));
+}
+
+describe('spreadsheet preview worker: merges are bounded per sheet and workbook-wide', () => {
+ // ExcelJS checks each new merge against every earlier one on its sheet, so
+ // a sheet's load cost grows with the square of its merge count.
+ it('refuses one sheet over the per-sheet merge cap before ExcelJS loads', async () => {
+ const harness = createHarness();
+ await harness.send({ type: 'load', bytes: await mergeHeavyWorkbook(1, 2_001) });
+ expect(harness.messages.at(-1)).toMatchObject({ type: 'error', code: 'merge-limit' });
+ expect(harness.imports.some((url) => url.includes('exceljs'))).toBe(false);
+ }, 60_000);
+
+ it('refuses sheets each under the per-sheet cap once the workbook total passes 10,000', async () => {
+ const harness = createHarness();
+ await harness.send({ type: 'load', bytes: await mergeHeavyWorkbook(6, 2_000) });
+ expect(harness.messages.at(-1)).toMatchObject({ type: 'error', code: 'merge-limit' });
+ expect(harness.imports.some((url) => url.includes('exceljs'))).toBe(false);
+ }, 60_000);
+
+ it('still previews a sheet at the per-sheet merge cap', async () => {
+ const harness = createHarness();
+ const metadata = await loadMetadata(harness, await mergeHeavyWorkbook(1, 2_000));
+ expect(metadata.sheets[0].merges).toHaveLength(2_000);
+ }, 60_000);
+});
+
+/** A one-cell numeric workbook whose styles.xml declares `formatCode` for the cell. */
+async function numberFormatWorkbook(formatCode: string): Promise {
+ const workbook = new ExcelJS.Workbook();
+ const sheet = workbook.addWorksheet('Data');
+ sheet.getCell('A1').value = 1.5;
+ sheet.getCell('A1').numFmt = '0.000';
+ const entries = fflate.unzipSync(new Uint8Array(await workbook.xlsx.writeBuffer()));
+ const styles = fflate.strFromU8(entries['xl/styles.xml']);
+ const patched = styles.replace('formatCode="0.000"', `formatCode="${formatCode}"`);
+ expect(patched).not.toBe(styles);
+ entries['xl/styles.xml'] = fflate.strToU8(patched);
+ return toArrayBuffer(fflate.zipSync(entries));
+}
+
+describe('spreadsheet preview worker: number formats ExcelJS would rescan', () => {
+ // `isDateFmt` runs `/\[[^\]]*]/g` per numeric cell; a 60,000-character code of
+ // `[` cost 2.2 s a cell, and the code is echoed into the notice bar.
+ it.each([
+ ['an unclosed [', `0${'['.repeat(200)}`],
+ ['a code over 255 characters', `${'0'.repeat(300)}.00`],
+ ])(
+ 'refuses %s before ExcelJS loads',
+ async (_label, code) => {
+ const harness = createHarness();
+ await harness.send({ type: 'load', bytes: await numberFormatWorkbook(code) });
+ expect(harness.messages.at(-1)).toMatchObject({ type: 'error', code: 'number-format' });
+ expect(harness.imports.some((url) => url.includes('exceljs'))).toBe(false);
+ },
+ 60_000
+ );
+
+ it('previews a closed bracketed format and keeps its warning bounded', async () => {
+ const code = `[Red]${'0'.repeat(200)}`;
+ const harness = createHarness();
+ const metadata = await loadMetadata(harness, await numberFormatWorkbook(code));
+ const tile = await requestTile(harness, metadata.sheets[0].id, { r1: 1, c1: 1, r2: 1, c2: 1 });
+ expect(tile.cells).toEqual([expect.objectContaining({ row: 1, col: 1, text: '1.5' })]);
+ expect(tile.warnings).toEqual([`Unsupported number format: ${code}`]);
+ }, 60_000);
+});
diff --git a/test/spreadsheet-xlsx-core.test.ts b/test/spreadsheet-xlsx-core.test.ts
index c33af908..427a1f70 100644
--- a/test/spreadsheet-xlsx-core.test.ts
+++ b/test/spreadsheet-xlsx-core.test.ts
@@ -333,6 +333,101 @@ describe('spreadsheet XLSX core', () => {
expect(() => counter.push(fflate.strToU8(xml), true)).toThrowError(/styles limit/i);
});
+ // ExcelJS's `_mergeCellsInternal` checks every new merge against every earlier
+ // one on its sheet, so a sheet costs the SQUARE of its merge count: one sheet
+ // at the old 5,000 cap took 1.6 s, and 20 such sheets ran past the page timeout.
+ it('caps merges per sheet and across the workbook', () => {
+ expect(core.LIMITS.maxMergesPerSheet).toBe(2000);
+ expect(core.LIMITS.maxMerges).toBe(10000);
+ const merges = (n: number, row = 1) =>
+ Array.from({ length: n }, (_, i) => ` `).join('');
+ const sheetXml = (n: number) => `${merges(n)} `;
+ expect(() => core.admitXlsx(workbookZip(sheetXml(4)), fflate, { maxMergesPerSheet: 3 })).toThrowError(
+ /merged ranges limit/i
+ );
+ expect(() => core.admitXlsx(workbookZip(sheetXml(3)), fflate, { maxMergesPerSheet: 3 })).not.toThrow();
+ // Two sheets each under the per-sheet cap still trip the workbook-wide one.
+ const twoSheets = fflate.zipSync({
+ ...fflate.unzipSync(workbookZip(sheetXml(3))),
+ 'xl/worksheets/sheet2.xml': fflate.strToU8(sheetXml(3)),
+ });
+ expect(() => core.admitXlsx(twoSheets, fflate, { maxMergesPerSheet: 3, maxMerges: 5 })).toThrowError(
+ /merged ranges limit/i
+ );
+ expect(() => core.admitXlsx(twoSheets, fflate, { maxMergesPerSheet: 3, maxMerges: 6 })).not.toThrow();
+ expect(() => core.admitXlsx(twoSheets, fflate, { maxMergesPerSheet: 3, maxMerges: 5 })).toThrowError(
+ expect.objectContaining({ code: 'merge-limit' })
+ );
+ });
+
+ // ExcelJS runs `isDateFmt` once per numeric cell, and its first step,
+ // `fmt.replace(/\[[^\]]*]/g, '')`, rescans to the end of the code for every
+ // `[` with no later `]`. The code is also echoed into the notice bar.
+ it('refuses a formatCode over 255 characters or with a [ after its last ]', () => {
+ const styles = (numFmts: string) => {
+ const counts = { cells: 0, merges: 0, styles: 0, rows: 0 };
+ const counter = core.createXmlCounter('xl/styles.xml', counts as never, core.LIMITS);
+ const xml = `${numFmts} `;
+ return () => counter.push(fflate.strToU8(xml), true);
+ };
+ const fmt = (code: string) => ` `;
+ expect(styles(fmt('0'.repeat(256)))).toThrowError(/longer than 255/i);
+ expect(styles(fmt('0'.repeat(255)))).not.toThrow();
+ for (const code of ['[', '0[', '[Red]0[', '[[[[', '[Red]0.00;[']) {
+ expect(styles(fmt(code)), code).toThrowError(/unclosed bracket/i);
+ }
+ for (const code of ['[Red]0.00', '[$-409]mmm d, yyyy', '[h]:mm:ss', '0.00', '#,##0;[Red]-#,##0', ']', '[[]']) {
+ expect(styles(fmt(code)), code).not.toThrow();
+ }
+ // The code is checked as ExcelJS decodes it: an entity cannot hide a `[`,
+ // and an entity-heavy code is measured by its decoded length.
+ expect(styles(fmt('0['))).toThrowError(/unclosed bracket/i);
+ expect(styles(fmt('0['))).toThrowError(/unclosed bracket/i);
+ expect(styles(fmt('"x"'.repeat(60)))).not.toThrow();
+ expect(styles(fmt('&'.repeat(256)))).toThrowError(/longer than 255/i);
+ // Attributes are read in order, so a quoted fake formatCode cannot mask the real one.
+ expect(styles(` `)).toThrowError(/unclosed bracket/i);
+ expect(styles(' ')).toThrowError(/do not parse/i);
+ // `` is the container, not a format.
+ expect(styles('')).not.toThrow();
+ // The refusal never echoes the code itself.
+ for (const code of ['QZJX[', `${'QZJX'.repeat(64)}0`]) {
+ expect(styles(fmt(code))).toThrowError(
+ expect.objectContaining({ code: 'number-format', message: expect.not.stringContaining('QZJX') })
+ );
+ }
+ // Every in the file is read, wherever it sits (dxfs carry them too).
+ const counts = { cells: 0, merges: 0, styles: 0, rows: 0 };
+ const counter = core.createXmlCounter('xl/styles.xml', counts as never, core.LIMITS);
+ const dxf = `${fmt('0[')} `;
+ expect(() => counter.push(fflate.strToU8(dxf), true)).toThrowError(/unclosed bracket/i);
+ });
+
+ it('caps cell display text at 1,000 characters for every value shape', () => {
+ const long = 'x'.repeat(50_000);
+ const shapes: Array<[string, unknown]> = [
+ ['string', long],
+ ['rich text', { richText: [{ text: long }, { text: long }] }],
+ ['hyperlink', { text: long, hyperlink: 'https://example.invalid/' }],
+ ['rich hyperlink', { text: { richText: [{ text: long }] }, hyperlink: 'https://example.invalid/' }],
+ ['formula source', { formula: long }],
+ ['formula result', { formula: 'A1', result: long }],
+ ['error', { error: long }],
+ ];
+ for (const [label, value] of shapes) {
+ const { text } = core.formatCellValue(value, 'General');
+ expect(text.length, label).toBeLessThanOrEqual(1000);
+ expect(text.endsWith('…'), label).toBe(true);
+ }
+ expect(core.formatCellValue('y'.repeat(1000), 'General').text).toBe('y'.repeat(1000));
+ expect(core.formatCellValue({ richText: [{ text: 'a' }, { text: 'b' }] }, 'General').text).toBe('ab');
+ // A cut never leaves half of a surrogate pair.
+ // 'aa' puts a high surrogate at index 998, exactly where a naive cut lands.
+ const emoji = core.formatCellValue('aa' + '😀'.repeat(600), 'General').text;
+ expect(emoji.length).toBeLessThanOrEqual(1000);
+ expect(/[\uD800-\uDBFF](?![\uDC00-\uDFFF])/.test(emoji)).toBe(false);
+ });
+
it('refuses an entry whose declared compressed size runs past the file', () => {
const zip = workbookZip();
const view = new DataView(zip.buffer, zip.byteOffset, zip.byteLength);