fix(preview): parse admission tag attributes in order and count empty rows

XML allows a raw `>` and the other quote character inside an attribute
value, so a first-match search for `ref=`/`max=` could be fed a fake
value from an earlier attribute while saxes read the real one:

- <mergeCell>/<col> attributes are now read in order from the tag name
  with a sticky regex that consumes each quoted value whole. A tag whose
  attributes do not parse up to `>`, or that repeats a name, is refused.
- The chunk carry keeps everything from the last `<`, which can never
  appear inside an attribute value, instead of comparing against the
  last `>`.

ExcelJS keeps a Row object for every <row>, cells or not, so <row> tags
now count against per-sheet (100k) and total (250k) caps with their own
row-limit code, and the worker passes maxRows as a per-sheet backstop.

styles.xml counts every <xf> without tracking which list it sits in,
since a </cellXfs> inside a comment desynced that state.
This commit is contained in:
Aamer Akhter
2026-10-01 21:39:26 -04:00
parent e85b4f34dd
commit d65ee4f89d
6 changed files with 160 additions and 31 deletions
+52
View File
@@ -573,3 +573,55 @@ describe('spreadsheet preview worker: admission bounds what ExcelJS expands', ()
expect(tile.cells).toEqual([expect.objectContaining({ row: 1, col: 1, text: 'one' })]);
}, 30_000);
});
describe('spreadsheet preview worker: quoted attribute values and empty rows', () => {
// XML allows a raw `>` and the other quote character inside an attribute
// value. Each of these was admitted before and made ExcelJS build millions of
// cells or columns.
it.each([
[
'a quoted fake ref before the real merge ref',
'after-sheetData',
`<mergeCells count="1"><mergeCell x=' ref="A1"' ref="A1:CV30000"/></mergeCells>`,
'cell-limit',
],
[
'a quoted > before the real col max',
'before-sheetData',
'<cols><col x=">" min="1" max="3000000"/></cols>',
'malformed',
],
[
'a quoted fake max before the real col max',
'before-sheetData',
`<cols><col x=' max="1"' min="1" max="3000000"/></cols>`,
'malformed',
],
] as const)('refuses %s before ExcelJS loads', async (_label, where, xml, code) => {
const harness = createHarness();
await harness.send({ type: 'load', bytes: await sheetWithInjectedXml(where, xml) });
expect(harness.messages.at(-1)).toMatchObject({ type: 'error', code });
expect(harness.imports.some((url) => url.includes('exceljs'))).toBe(false);
});
// ExcelJS keeps a Row object for every <row>, so cell-less rows cost memory
// too: three sheets of a million empty rows sat inside every cell cap.
it('refuses a sheet of empty rows past the row cap before ExcelJS loads', async () => {
const harness = createHarness();
const rows = Array.from({ length: 100_001 }, (_, i) => `<row r="${i + 2}"/>`).join('');
await harness.send({ type: 'load', bytes: await emptyRowsWorkbook(rows) });
expect(harness.messages.at(-1)).toMatchObject({ type: 'error', code: 'row-limit' });
expect(harness.imports.some((url) => url.includes('exceljs'))).toBe(false);
}, 30_000);
});
async function emptyRowsWorkbook(extraRows: string): Promise<ArrayBuffer> {
const workbook = new ExcelJS.Workbook();
workbook.addWorksheet('Data').getCell('A1').value = 'one';
const entries = fflate.unzipSync(new Uint8Array(await workbook.xlsx.writeBuffer()));
const sheet = fflate.strFromU8(entries['xl/worksheets/sheet1.xml']);
const patched = sheet.replace('</sheetData>', `${extraRows}</sheetData>`);
expect(patched).not.toBe(sheet);
entries['xl/worksheets/sheet1.xml'] = fflate.strToU8(patched);
return toArrayBuffer(fflate.zipSync(entries));
}
+42 -1
View File
@@ -113,7 +113,7 @@ describe('spreadsheet XLSX core', () => {
features: string[];
};
// The 2x2 merge costs its four covered cells on top of the one real cell.
expect(result.counts).toEqual({ worksheets: 1, cells: 5, merges: 1, styles: 1 });
expect(result.counts).toEqual({ worksheets: 1, cells: 5, rows: 0, merges: 1, styles: 1 });
expect(result.features).toEqual(expect.arrayContaining(['charts', 'externalLinks']));
});
@@ -182,6 +182,7 @@ describe('spreadsheet XLSX core', () => {
const pad = '<sheetView workbookViewId="0"/>'.repeat(12);
const cases: Array<[string, RegExp]> = [
[`<worksheet>${pad}<cols><col min="1" max="99999"/></cols>${pad}</worksheet>`, /column max/i],
[`<worksheet>${pad}<cols><col x=">" min="1" max="99999"/></cols>${pad}</worksheet>`, /column max/i],
[`<worksheet>${pad}<mergeCells><mergeCell ref="A1:CV30000"/></mergeCells>${pad}</worksheet>`, /cells limit/i],
];
for (const [xml, pattern] of cases) {
@@ -200,6 +201,46 @@ describe('spreadsheet XLSX core', () => {
}
});
it('reads attributes in order, so a quoted value cannot hide or fake one', () => {
const counter = () =>
core.createXmlCounter(
'xl/worksheets/sheet1.xml',
{ cells: 0, merges: 0, styles: 0, rows: 0 } as never,
core.LIMITS
);
const push = (xml: string) => counter().push(fflate.strToU8(xml), true);
// A raw `>` or the other quote character is legal inside a value.
expect(() => push(`<mergeCell x=' ref="A1"' ref="A1:CV30000"/>`)).toThrowError(/cells limit/i);
expect(() => push('<col x=">" min="1" max="3000000"/>')).toThrowError(/column max/i);
expect(() => push(`<col x=' max="1"' min="1" max="3000000"/>`)).toThrowError(/column max/i);
// Anything the walk cannot read up to `>` is refused, as is a repeated name.
expect(() => push('<mergeCell ref="A1:B2" junk/>')).toThrowError(/do not parse/i);
expect(() => push('<mergeCell ref="A1" ref="A1:CV30000"/>')).toThrowError(/do not parse/i);
expect(() => push('<cols><col min="1" max="3" width="9"></col></cols><mergeCell ref="A1:B2" />')).not.toThrow();
});
it('counts every <row>, empty or not, against per-sheet and total caps', () => {
const rows = (n: number) => '<worksheet><sheetData>' + '<row r="1"/>'.repeat(n) + '</sheetData></worksheet>';
expect(() => core.admitXlsx(workbookZip(rows(4)), fflate, { maxRowsPerSheet: 3 })).toThrowError(/rows limit/i);
expect(() => core.admitXlsx(workbookZip(rows(4)), fflate, { maxRows: 3 })).toThrowError(/rows limit/i);
const admitted = core.admitXlsx(workbookZip(rows(3)), fflate, { maxRowsPerSheet: 3 }) as {
counts: { rows: number };
};
expect(admitted.counts.rows).toBe(3);
// <rowBreaks>/<rows...> style names are not rows.
expect(
(core.admitXlsx(workbookZip('<worksheet><rowBreaks/></worksheet>'), fflate) as { counts: { rows: number } })
.counts.rows
).toBe(0);
});
it('counts every <xf> in styles.xml, so a </cellXfs> inside a comment cannot hide styles', () => {
const counts = { cells: 0, merges: 0, styles: 0, rows: 0 };
const counter = core.createXmlCounter('xl/styles.xml', counts as never, { ...core.LIMITS, maxStyles: 100 });
const xml = '<styleSheet><cellXfs><!-- </cellXfs> -->' + '<xf/>'.repeat(200) + '</cellXfs></styleSheet>';
expect(() => counter.push(fflate.strToU8(xml), true)).toThrowError(/styles limit/i);
});
it('refuses an entry whose declared compressed size runs past the file', () => {
const zip = workbookZip();
const view = new DataView(zip.buffer, zip.byteOffset, zip.byteLength);