fix(preview): bound what ExcelJS expands during XLSX admission

ExcelJS 4.4.0 expands three constructs into one object per cell or column
at load time, so a few KB admitted as one cell could cost a gigabyte:

- a <mergeCell> now costs its full area against the per-sheet and total
  cell caps, and a ref that does not parse is refused
- a <col> whose min or max is past 16384 is refused
- the worker loads with ignoreNodes: ['dataValidations']; the preview
  never shows validations, and a whole-column dropdown took 5 s

The XML counter now scans up to the last complete tag and carries the
rest, so a merge or col tag cut by an inflate-chunk edge is read whole.
A central-directory compressedSize that runs past the file is refused,
since the ratio cap divides by it.

The renderer and core axis offsets use prefix sums with a binary search
instead of walking every override per call.
This commit is contained in:
Aamer Akhter
2026-10-01 09:18:35 -04:00
parent 4edb7b8f80
commit bfab172608
7 changed files with 259 additions and 27 deletions
+65
View File
@@ -132,6 +132,8 @@ function createHarness() {
messages,
imports,
self,
/** Evaluate an expression against the worker's own top-level bindings. */
peek: (expression: string): unknown => vm.runInContext(expression, context),
send: async (data: unknown) => {
await (self.onmessage as (event: { data: unknown }) => Promise<void>)({ data });
},
@@ -508,3 +510,66 @@ describe('spreadsheet preview worker: ExcelJS sees only what admission checked',
else expect(result.type).toBe('error');
}, 60_000);
});
/** A one-cell ExcelJS workbook whose sheet1.xml gets `xml` spliced in at `where`. */
async function sheetWithInjectedXml(where: 'before-sheetData' | 'after-sheetData', xml: string): Promise<ArrayBuffer> {
const workbook = new ExcelJS.Workbook();
workbook.addWorksheet('Data').getCell('A1').value = 'one';
const entries = fflate.unzipSync(new Uint8Array(await workbook.xlsx.writeBuffer()));
const sheet = fflate.strFromU8(entries['xl/worksheets/sheet1.xml']);
const patched =
where === 'before-sheetData'
? sheet.replace('<sheetData>', `${xml}<sheetData>`)
: sheet.replace('</sheetData>', `</sheetData>${xml}`);
expect(patched).not.toBe(sheet);
entries['xl/worksheets/sheet1.xml'] = fflate.strToU8(patched);
return toArrayBuffer(fflate.zipSync(entries));
}
describe('spreadsheet preview worker: admission bounds what ExcelJS expands', () => {
// ExcelJS creates one cell object per covered cell of a merge, so a single
// `<mergeCell>` tag over 3M cells took 10 s and a gigabyte of heap.
it('refuses a merge whose area exceeds the cell caps before ExcelJS loads', async () => {
const harness = createHarness();
await harness.send({
type: 'load',
bytes: await sheetWithInjectedXml(
'after-sheetData',
'<mergeCells count="1"><mergeCell ref="A1:CV30000"/></mergeCells>'
),
});
expect(harness.messages.at(-1)).toMatchObject({ type: 'error', code: 'cell-limit' });
expect(harness.imports.some((url) => url.includes('exceljs'))).toBe(false);
});
// `Column.fromModel` builds every column up to `<col max>` with no clamp.
it('refuses a <col> range past column 16384 before ExcelJS loads', async () => {
const harness = createHarness();
await harness.send({
type: 'load',
bytes: await sheetWithInjectedXml('before-sheetData', '<cols><col min="1" max="3000000" width="9"/></cols>'),
});
expect(harness.messages.at(-1)).toMatchObject({ type: 'error', code: 'malformed' });
expect(harness.imports.some((url) => url.includes('exceljs'))).toBe(false);
});
// A whole-column dropdown is a few bytes of XML that ExcelJS expands into one
// object per address (5 s here; a whole-sheet range was still running after
// 60 s). The preview never shows validations, so the worker does not parse
// them, and the file still previews.
it('previews a sheet with a whole-column data validation without expanding it', async () => {
const harness = createHarness();
const metadata = await loadMetadata(
harness,
await sheetWithInjectedXml(
'after-sheetData',
'<dataValidations count="1"><dataValidation type="list" allowBlank="1" sqref="B2:B1048576">' +
'<formula1>"a,b"</formula1></dataValidation></dataValidations>'
)
);
expect(harness.peek('Object.keys(workbook.worksheets[0].dataValidations.model).length')).toBe(0);
expect(metadata.sheets[0]).toMatchObject({ rows: 1, cols: 1 });
const tile = await requestTile(harness, metadata.sheets[0].id, { r1: 1, c1: 1, r2: 1, c2: 1 });
expect(tile.cells).toEqual([expect.objectContaining({ row: 1, col: 1, text: 'one' })]);
}, 30_000);
});
+76 -1
View File
@@ -11,6 +11,11 @@ type Core = {
XlsxPreviewError: new (code: string, message: string) => Error & { code: string };
inspectZipDirectory(bytes: Uint8Array, limits?: Record<string, number>): { entries: Array<{ name: string }> };
admitXlsx(bytes: Uint8Array, zip: typeof fflate, limits?: Record<string, number>): unknown;
createXmlCounter(
name: string,
counts: { cells: number; merges: number; styles: number },
limits: Record<string, number>
): { push(chunk: Uint8Array, final: boolean): void };
buildAdmittedArchive(admission: unknown, zip: typeof fflate): Uint8Array;
parseCellRef(ref: string): { row: number; col: number } | null;
deriveExtent(cells: string[], merges: string[]): { rows: number; cols: number };
@@ -107,7 +112,8 @@ describe('spreadsheet XLSX core', () => {
counts: { worksheets: number; cells: number; merges: number; styles: number };
features: string[];
};
expect(result.counts).toEqual({ worksheets: 1, cells: 1, merges: 1, styles: 1 });
// The 2x2 merge costs its four covered cells on top of the one real cell.
expect(result.counts).toEqual({ worksheets: 1, cells: 5, merges: 1, styles: 1 });
expect(result.features).toEqual(expect.arrayContaining(['charts', 'externalLinks']));
});
@@ -146,6 +152,63 @@ describe('spreadsheet XLSX core', () => {
expect(() => core.admitXlsx(zip, duplicate)).toThrowError(/duplicate/i);
});
it('charges a merged range its full area and refuses one that does not parse', () => {
const merged = (ref: string) =>
workbookZip(
`<worksheet><sheetData><row><c r="A1"/></row></sheetData><mergeCells><mergeCell ref="${ref}"/></mergeCells></worksheet>`
);
expect(() => core.admitXlsx(merged('A1:CV30000'), fflate)).toThrowError(/cells limit/i);
// Reversed corners describe the same area.
expect(() => core.admitXlsx(merged('CV30000:A1'), fflate)).toThrowError(/cells limit/i);
expect(() => core.admitXlsx(merged('A1:XFE2'), fflate)).toThrowError(/merged range/i);
expect(() => core.admitXlsx(merged('not-a-range'), fflate)).toThrowError(/merged range/i);
const admitted = core.admitXlsx(merged('A1:J10'), fflate) as { counts: { cells: number } };
expect(admitted.counts.cells).toBe(101);
});
it('refuses a <col> whose min or max is past the last Excel column', () => {
const cols = (attrs: string) =>
workbookZip(
`<worksheet><cols><col ${attrs} width="9"/></cols><sheetData><row><c r="A1"/></row></sheetData></worksheet>`
);
expect(() => core.admitXlsx(cols('min="1" max="3000000"'), fflate)).toThrowError(/column max/i);
expect(() => core.admitXlsx(cols('min="16385" max="16385"'), fflate)).toThrowError(/column min/i);
expect(() => core.admitXlsx(cols('min="1" max="1e9"'), fflate)).toThrowError(/column max/i);
expect(() => core.admitXlsx(cols('min="1" max="16384"'), fflate)).not.toThrow();
});
it('reads a merge or col tag whole even when a stream chunk boundary cuts through it', () => {
// Markup on both sides, so a cut can land 128+ bytes past the tag's start.
const pad = '<sheetView workbookViewId="0"/>'.repeat(12);
const cases: Array<[string, RegExp]> = [
[`<worksheet>${pad}<cols><col min="1" max="99999"/></cols>${pad}</worksheet>`, /column max/i],
[`<worksheet>${pad}<mergeCells><mergeCell ref="A1:CV30000"/></mergeCells>${pad}</worksheet>`, /cells limit/i],
];
for (const [xml, pattern] of cases) {
const bytes = fflate.strToU8(xml);
for (let cut = 1; cut < bytes.length; cut += 1) {
const counter = core.createXmlCounter(
'xl/worksheets/sheet1.xml',
{ cells: 0, merges: 0, styles: 0 },
core.LIMITS
);
expect(() => {
counter.push(bytes.subarray(0, cut), false);
counter.push(bytes.subarray(cut), true);
}, `cut at ${cut}`).toThrowError(pattern);
}
}
});
it('refuses an entry whose declared compressed size runs past the file', () => {
const zip = workbookZip();
const view = new DataView(zip.buffer, zip.byteOffset, zip.byteLength);
const eocd = zip.length - 22;
const centralStart = view.getUint32(eocd + 16, true);
view.setUint32(centralStart + 20, 0x7fffffff, true);
expect(() => core.inspectZipDirectory(zip)).toThrowError(/compressed bytes/i);
});
it('derives bounded extents from real cells and merges', () => {
expect(core.parseCellRef('XFD1048576')).toEqual({ row: 1_048_576, col: 16_384 });
expect(core.parseCellRef('XFE1')).toBeNull();
@@ -164,6 +227,18 @@ describe('spreadsheet XLSX core', () => {
expect(end - start).toBeLessThan(10);
});
it('matches a plain walk over the overrides at every index', () => {
const overrides: Array<[number, number]> = [];
for (let index = 3; index <= 400; index += 7) overrides.push([index, (index * 13) % 50]);
const axis = core.createSparseAxis(500, 20, overrides);
for (let index = 0; index <= 502; index += 1) {
const bounded = Math.max(1, Math.min(501, index));
let expected = (bounded - 1) * 20;
for (const [at, size] of overrides) if (at < bounded) expected += size - 20;
expect(core.axisOffset(axis, index), `index ${index}`).toBe(expected);
}
});
it('returns intersecting merges even when their anchor is offscreen', () => {
expect(core.intersectingMerges(['A1:D4', 'Z1:Z2'], { r1: 3, c1: 3, r2: 6, c2: 6 })).toEqual(['A1:D4']);
});