mirror of
https://github.com/Ark0N/Codeman.git
synced 2026-10-10 01:09:43 +02:00
fix(preview): budget every start tag ExcelJS will parse
Admission counted cells, rows, merges and styles only in worksheets, styles.xml and workbook.xml, so the objects ExcelJS builds per element elsewhere (shared-string runs, fonts, fills, borders, comments, drawings, VML, tables) were bounded only by the inflated-byte caps, and empty stored deflate blocks pad a stream past the ratio cap. createXmlCounter now counts every start tag in every part except the pure-bytes xl/media/<name>.<ext> entries into counts.elements and refuses above LIMITS.maxElements (2,000,000) as element-limit. Parts that read no attributes carry only a trailing '<' between chunks, so long text or binary is never taken for an oversized tag. Very tall sheets now scale only the scroll position: the scroll range maps onto the sheet's whole range and the tile is laid out at real row heights and column widths, with spans clipped at the spacer, instead of dividing every cell and heading by the scale.
This commit is contained in:
@@ -1008,3 +1008,65 @@ describe('spreadsheet preview worker: the notice bar stays bounded', () => {
|
||||
expect(tile.warnings).toEqual(['800 unsupported number formats']);
|
||||
}, 60_000);
|
||||
});
|
||||
|
||||
/**
|
||||
* A one-cell workbook plus an `xl/sharedStrings.xml` whose one string holds
|
||||
* `runs` empty `<r/>` runs. Its deflate stream is prefixed with empty stored
|
||||
* blocks (5 bytes each, inflating to nothing) until the compressed size clears
|
||||
* the 100:1 ratio cap, so only an element budget can refuse it.
|
||||
*/
|
||||
async function emptyRunsWorkbook(runs: number): Promise<Uint8Array> {
|
||||
const workbook = new ExcelJS.Workbook();
|
||||
workbook.addWorksheet('Data').getCell('A1').value = 1;
|
||||
const entries = fflate.unzipSync(new Uint8Array(await workbook.xlsx.writeBuffer()));
|
||||
delete entries['xl/sharedStrings.xml'];
|
||||
const strings = fflate.strToU8(`<sst count="1" uniqueCount="1"><si>${'<r/>'.repeat(runs)}</si></sst>`);
|
||||
const deflated = fflate.deflateSync(strings, { level: 9 });
|
||||
const emptyStoredBlock = Uint8Array.from([0x00, 0x00, 0x00, 0xff, 0xff]);
|
||||
const blocks = Math.ceil((strings.length / 50 - deflated.length) / emptyStoredBlock.length);
|
||||
const padded = concatBytes([...Array.from({ length: blocks }, () => emptyStoredBlock), deflated]);
|
||||
expect(fflate.inflateSync(padded)).toEqual(strings);
|
||||
const parts = Object.keys(entries).map((name) => zipPart(name, entries[name], 8));
|
||||
parts.push({
|
||||
name: 'xl/sharedStrings.xml',
|
||||
data: padded,
|
||||
method: 8,
|
||||
size: strings.length,
|
||||
crc: crc32(strings) >>> 0,
|
||||
});
|
||||
const locals: Uint8Array[] = [];
|
||||
const central: Uint8Array[] = [];
|
||||
let offset = 0;
|
||||
for (const part of parts) {
|
||||
central.push(centralHeader(part, offset));
|
||||
const chunk = concatBytes([localHeader(part), part.data]);
|
||||
locals.push(chunk);
|
||||
offset += chunk.length;
|
||||
}
|
||||
const directory = concatBytes(central);
|
||||
const eocd = new Uint8Array(22);
|
||||
const view = new DataView(eocd.buffer);
|
||||
view.setUint32(0, 0x06054b50, true);
|
||||
view.setUint16(8, central.length, true);
|
||||
view.setUint16(10, central.length, true);
|
||||
view.setUint32(12, directory.length, true);
|
||||
view.setUint32(16, offset, true);
|
||||
return concatBytes([...locals, directory, eocd]);
|
||||
}
|
||||
|
||||
describe('spreadsheet preview worker: element budget', () => {
|
||||
// ExcelJS builds one object per <r> run in sharedStrings.xml, which no
|
||||
// worksheet counter sees; 8.3M empty runs took 570 MB of heap to load.
|
||||
it('refuses a shared string of 2,000,001 empty runs before ExcelJS loads', async () => {
|
||||
const crafted = await emptyRunsWorkbook(2_000_001);
|
||||
const harness = createHarness();
|
||||
const core = harness.self.CodemanSpreadsheetXlsxCore as {
|
||||
admitXlsx(bytes: Uint8Array, zip: typeof fflate, overrides?: Record<string, number>): unknown;
|
||||
};
|
||||
// Every other cap admits it: only the element budget is in the way.
|
||||
expect(() => core.admitXlsx(crafted, fflate, { maxElements: Infinity })).not.toThrow();
|
||||
await harness.send({ type: 'load', bytes: toArrayBuffer(crafted) });
|
||||
expect(harness.messages.at(-1)).toMatchObject({ type: 'error', code: 'element-limit' });
|
||||
expect(harness.imports.some((url) => url.includes('exceljs'))).toBe(false);
|
||||
}, 60_000);
|
||||
});
|
||||
|
||||
@@ -291,6 +291,81 @@ describe('spreadsheet preview renderer', () => {
|
||||
expect(heading('.spreadsheet-row-heading', '3')).toBeUndefined();
|
||||
});
|
||||
|
||||
// Past MAX_SCROLL_PX only the scroll position may be scaled: dividing every
|
||||
// cell and heading by the scale drew a 1,048,576-row sheet 7.6 px a row.
|
||||
it('lays rows out at their real height on a sheet taller than the scroll cap', async () => {
|
||||
const fetchMock = vi.fn(async () => ({ ok: true, arrayBuffer: async () => new ArrayBuffer(8) }));
|
||||
const renderer = loadRenderer(fetchMock);
|
||||
renderer.open({ container: document.querySelector('#preview'), url: '/book.xlsx', size: 8 });
|
||||
const worker = WorkerMock.instances[0];
|
||||
worker.emit({ type: 'ready' });
|
||||
await vi.waitFor(() => expect(worker.postMessage).toHaveBeenCalled());
|
||||
const sheet = { id: '1', name: 'Tall', rows: 1_048_576, cols: 1, defaultRowHeight: 20, defaultColumnWidth: 64 };
|
||||
worker.emit({ type: 'metadata', styles: [], sheets: [{ ...sheet, rowOverrides: [], columnOverrides: [] }] });
|
||||
const cellAt = (row: number) => document.querySelector(`.spreadsheet-cell[data-row="${row}"]`) as HTMLElement;
|
||||
const rowHeading = (row: number) =>
|
||||
[...document.querySelectorAll('.spreadsheet-row-heading')].find(
|
||||
(element) => element.textContent === String(row)
|
||||
) as HTMLElement;
|
||||
const px = (value: string) => Number.parseFloat(value);
|
||||
|
||||
const first = worker.postMessage.mock.calls.at(-1)?.[0];
|
||||
expect(first.range.r1).toBe(1);
|
||||
const cell = (row: number) => ({ row, col: 1, text: `A${row}`, styleId: 0 });
|
||||
worker.emit({ type: 'tile', requestId: first.requestId, sheetId: '1', cells: [cell(1), cell(2)], warnings: [] });
|
||||
expect(cellAt(1).style.top).toBe('20px');
|
||||
expect(cellAt(1).style.height).toBe('20px');
|
||||
expect(px(cellAt(2).style.top) - px(cellAt(1).style.top)).toBe(20);
|
||||
expect(rowHeading(2).style.height).toBe('20px');
|
||||
// At scale 1 the first tile asks for about a viewport of rows, not a scaled one.
|
||||
expect(first.range.r2).toBeLessThan(40);
|
||||
|
||||
// Scrolled to the end, the last row is requested, drawn at its real height,
|
||||
// and ends exactly at the bottom of the scroll area.
|
||||
const grid = document.querySelector('.spreadsheet-grid') as HTMLElement;
|
||||
const spacerHeight = px((document.querySelector('.spreadsheet-grid-spacer') as HTMLElement).style.height);
|
||||
grid.scrollTop = spacerHeight - 500;
|
||||
grid.dispatchEvent(new window.Event('scroll'));
|
||||
await vi.waitFor(() =>
|
||||
expect(worker.postMessage.mock.calls.at(-1)?.[0].requestId).toBeGreaterThan(first.requestId)
|
||||
);
|
||||
const last = worker.postMessage.mock.calls.at(-1)?.[0];
|
||||
expect(last.range.r2).toBe(1_048_576);
|
||||
expect(last.range.r2 - last.range.r1).toBeLessThan(40);
|
||||
worker.emit({ type: 'tile', requestId: last.requestId, sheetId: '1', cells: [cell(1_048_576)], warnings: [] });
|
||||
expect(cellAt(1_048_576).style.height).toBe('20px');
|
||||
expect(rowHeading(1_048_576).style.height).toBe('20px');
|
||||
expect(px(cellAt(1_048_576).style.top) + 20).toBeCloseTo(spacerHeight, 6);
|
||||
expect(px(rowHeading(1_048_575).style.top)).toBeCloseTo(px(cellAt(1_048_576).style.top) - 20, 6);
|
||||
});
|
||||
|
||||
it('keeps a merge spanning a too-tall sheet inside the scroll area', async () => {
|
||||
const fetchMock = vi.fn(async () => ({ ok: true, arrayBuffer: async () => new ArrayBuffer(8) }));
|
||||
const renderer = loadRenderer(fetchMock);
|
||||
renderer.open({ container: document.querySelector('#preview'), url: '/book.xlsx', size: 8 });
|
||||
const worker = WorkerMock.instances[0];
|
||||
worker.emit({ type: 'ready' });
|
||||
await vi.waitFor(() => expect(worker.postMessage).toHaveBeenCalled());
|
||||
const sheet = { id: '1', name: 'Tall', rows: 1_048_576, cols: 1, defaultRowHeight: 20, defaultColumnWidth: 64 };
|
||||
worker.emit({ type: 'metadata', styles: [], sheets: [{ ...sheet, rowOverrides: [], columnOverrides: [] }] });
|
||||
const request = worker.postMessage.mock.calls.at(-1)?.[0];
|
||||
worker.emit({
|
||||
type: 'tile',
|
||||
requestId: request.requestId,
|
||||
sheetId: '1',
|
||||
cells: [{ row: 1, col: 1, text: 'whole column', styleId: 0 }],
|
||||
merges: ['A1:A1048576'],
|
||||
warnings: [],
|
||||
});
|
||||
const merged = document.querySelector('.spreadsheet-cell') as HTMLElement;
|
||||
const spacerHeight = Number.parseFloat(
|
||||
(document.querySelector('.spreadsheet-grid-spacer') as HTMLElement).style.height
|
||||
);
|
||||
expect(Number.parseFloat(merged.style.top) + Number.parseFloat(merged.style.height)).toBeLessThanOrEqual(
|
||||
spacerHeight
|
||||
);
|
||||
});
|
||||
|
||||
it('emits colour and background together or not at all', async () => {
|
||||
const fetchMock = vi.fn(async () => ({ ok: true, arrayBuffer: async () => new ArrayBuffer(8) }));
|
||||
const renderer = loadRenderer(fetchMock);
|
||||
|
||||
@@ -115,7 +115,8 @@ describe('spreadsheet XLSX core', () => {
|
||||
features: string[];
|
||||
};
|
||||
// The 2x2 merge costs its four covered cells on top of the one real cell.
|
||||
expect(result.counts).toEqual({ worksheets: 1, cells: 5, rows: 0, merges: 1, styles: 1 });
|
||||
// `elements` is every start tag across all six parts: 1 + 3 + 3 + 4 + 1 + 1.
|
||||
expect(result.counts).toEqual({ worksheets: 1, cells: 5, rows: 0, merges: 1, styles: 1, elements: 13 });
|
||||
expect(result.features).toEqual(expect.arrayContaining(['charts', 'externalLinks']));
|
||||
});
|
||||
|
||||
@@ -327,6 +328,91 @@ describe('spreadsheet XLSX core', () => {
|
||||
expect(() => other.push(fflate.strToU8(one('30000000')), true)).not.toThrow();
|
||||
});
|
||||
|
||||
// ExcelJS builds an object per element in every part it parses, not only
|
||||
// worksheets: each <r> run and <si> in sharedStrings.xml, each <font>, <fill>
|
||||
// and <border> in styles, comments, drawings, VML and tables.
|
||||
it('budgets every start tag outside xl/media, whatever part it sits in', () => {
|
||||
expect(core.LIMITS.maxElements).toBe(2_000_000);
|
||||
const withPart = (name: string, xml: string) =>
|
||||
fflate.zipSync({ ...fflate.unzipSync(workbookZip()), [name]: fflate.strToU8(xml) });
|
||||
const elementsOf = (zip: Uint8Array) =>
|
||||
(core.admitXlsx(zip, fflate) as { counts: { elements: number } }).counts.elements;
|
||||
const base = elementsOf(workbookZip());
|
||||
const runs = '<sst><si>' + '<r/>'.repeat(50) + '</si></sst>';
|
||||
const fonts = '<styleSheet><fonts>' + '<font/>'.repeat(50) + '</fonts></styleSheet>';
|
||||
const cases: Array<[string, string, number]> = [
|
||||
['xl/sharedStrings.xml', runs, 52],
|
||||
['xl/comments1.xml', '<comments>' + '<comment/>'.repeat(50) + '</comments>', 51],
|
||||
['xl/drawings/vmlDrawing1.vml', '<xml>' + '<v:shape/>'.repeat(50) + '</xml>', 51],
|
||||
// The name ExcelJS sees, so `/xl/./` cannot slip a part past the budget.
|
||||
['/xl/./sharedStrings.xml', runs, 52],
|
||||
];
|
||||
for (const [name, xml, added] of cases) {
|
||||
expect(elementsOf(withPart(name, xml)), name).toBe(base + added);
|
||||
let thrown: unknown;
|
||||
try {
|
||||
core.admitXlsx(withPart(name, xml), fflate, { maxElements: base + added - 1 });
|
||||
} catch (error) {
|
||||
thrown = error;
|
||||
}
|
||||
expect((thrown as { code?: string })?.code, name).toBe('element-limit');
|
||||
}
|
||||
// styles.xml is counted on top of its own <xf>/<numFmt> checks.
|
||||
const styled = fflate.unzipSync(workbookZip());
|
||||
styled['xl/styles.xml'] = fflate.strToU8(fonts);
|
||||
expect(() => core.admitXlsx(fflate.zipSync(styled), fflate, { maxElements: 50 })).toThrowError(/element/i);
|
||||
// Processing instructions, comments and end tags are not elements; a tag
|
||||
// INSIDE a comment is still counted, which errs toward refusing.
|
||||
expect(elementsOf(withPart('xl/other.xml', '<?xml version="1.0"?><!-- c --><a></a><_b/>'))).toBe(base + 2);
|
||||
expect(elementsOf(withPart('xl/other.xml', '<!-- <a> --><a/>'))).toBe(base + 2);
|
||||
});
|
||||
|
||||
it('never budgets an xl/media part, which ExcelJS keeps as bytes', () => {
|
||||
// Stored, so the compression-ratio cap stays out of the way.
|
||||
const tags = fflate.strToU8('<r/>'.repeat(500));
|
||||
for (const name of ['xl/media/image1.png', '/xl/./media/image1.jpeg']) {
|
||||
const zip = fflate.zipSync({ ...fflate.unzipSync(workbookZip()), [name]: [tags, { level: 0 }] });
|
||||
expect(() => core.admitXlsx(zip, fflate, { maxElements: 100 }), name).not.toThrow();
|
||||
}
|
||||
// A name under xl/media/ that ExcelJS parses as XML (its patterns are
|
||||
// unanchored) is budgeted like any other part.
|
||||
for (const name of ['xl/media/xl/drawings/drawing1.xml', 'xl/media/xl/worksheets/sheet2.xml']) {
|
||||
const zip = fflate.zipSync({ ...fflate.unzipSync(workbookZip()), [name]: [tags, { level: 0 }] });
|
||||
expect(() => core.admitXlsx(zip, fflate, { maxElements: 100 }), name).toThrowError(/element/i);
|
||||
}
|
||||
});
|
||||
|
||||
it('counts start tags exactly when a chunk boundary cuts through one', () => {
|
||||
const xml = '<sst><si>' + '<r/><t>x</t>'.repeat(20) + '</si></sst>';
|
||||
const bytes = fflate.strToU8(xml);
|
||||
for (const name of ['xl/sharedStrings.xml', 'xl/worksheets/sheet1.xml', 'xl/styles.xml']) {
|
||||
for (let cut = 1; cut < bytes.length; cut += 1) {
|
||||
const counts = { worksheets: 0, cells: 0, rows: 0, merges: 0, styles: 0, elements: 0 };
|
||||
const counter = core.createXmlCounter(name, counts as never, core.LIMITS);
|
||||
counter.push(bytes.subarray(0, cut), false);
|
||||
counter.push(bytes.subarray(cut), true);
|
||||
expect(counts.elements, `${name} cut at ${cut}`).toBe(42);
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
it('does not carry long text or binary in a non-worksheet part as an oversized tag', () => {
|
||||
const counts = { worksheets: 0, cells: 0, rows: 0, merges: 0, styles: 0, elements: 0 };
|
||||
const text = fflate.strToU8('<sst><si><t>' + 'x'.repeat(600_000) + '</t></si></sst>');
|
||||
const strings = core.createXmlCounter('xl/sharedStrings.xml', counts as never, core.LIMITS);
|
||||
const binary = core.createXmlCounter('xl/embeddings/oleObject1.bin', counts as never, core.LIMITS);
|
||||
const zeros = new Uint8Array(600_000);
|
||||
expect(() => {
|
||||
for (let at = 0; at < text.length; at += 65_536) {
|
||||
strings.push(text.subarray(at, at + 65_536), at + 65_536 >= text.length);
|
||||
}
|
||||
for (let at = 0; at < zeros.length; at += 65_536) {
|
||||
binary.push(zeros.subarray(at, at + 65_536), at + 65_536 >= zeros.length);
|
||||
}
|
||||
}).not.toThrow();
|
||||
expect(counts.elements).toBe(3);
|
||||
});
|
||||
|
||||
it('counts every <xf> in styles.xml, so a </cellXfs> inside a comment cannot hide styles', () => {
|
||||
const counts = { cells: 0, merges: 0, styles: 0, rows: 0 };
|
||||
const counter = core.createXmlCounter('xl/styles.xml', counts as never, { ...core.LIMITS, maxStyles: 100 });
|
||||
|
||||
Reference in New Issue
Block a user