fix(preview): parse admission tag attributes in order and count empty rows

XML allows a raw `>` and the other quote character inside an attribute
value, so a first-match search for `ref=`/`max=` could be fed a fake
value from an earlier attribute while saxes read the real one:

- <mergeCell>/<col> attributes are now read in order from the tag name
  with a sticky regex that consumes each quoted value whole. A tag whose
  attributes do not parse up to `>`, or that repeats a name, is refused.
- The chunk carry keeps everything from the last `<`, which can never
  appear inside an attribute value, instead of comparing against the
  last `>`.

ExcelJS keeps a Row object for every <row>, cells or not, so <row> tags
now count against per-sheet (100k) and total (250k) caps with their own
row-limit code, and the worker passes maxRows as a per-sheet backstop.

styles.xml counts every <xf> without tracking which list it sits in,
since a </cellXfs> inside a comment desynced that state.
This commit is contained in:
Aamer Akhter
2026-10-01 21:39:26 -04:00
parent e85b4f34dd
commit d65ee4f89d
6 changed files with 160 additions and 31 deletions
+58 -26
View File
@@ -23,6 +23,8 @@
maxWorksheets: 50,
maxCells: 250000,
maxCellsPerSheet: 100000,
maxRows: 250000,
maxRowsPerSheet: 100000,
maxMergesPerSheet: 5000,
maxStyles: 5000,
});
@@ -125,24 +127,49 @@
// from turning the carried tail into quadratic re-scanning.
const MAX_CARRIED_TAG = 256 * 1024;
function attribute(tag, name) {
const match = new RegExp(`\\s${name}\\s*=\\s*(?:"([^"]*)"|'([^']*)')`).exec(tag);
return match ? (match[1] ?? match[2]) : null;
// XML attribute syntax, read in order from just after the tag name. A quoted
// value may hold a raw `>` or the other quote character, so a value is always
// consumed whole; a tag is only understood if this walk reaches its `>`.
const XML_SPACE = '[ \\t\\r\\n]';
const ATTRIBUTE = new RegExp(
`${XML_SPACE}+([^ \\t\\r\\n=/>]+)${XML_SPACE}*=${XML_SPACE}*(?:"([^"]*)"|'([^']*)')`,
'y'
);
const TAG_END = new RegExp(`${XML_SPACE}*/?>`, 'y');
/**
* Parse the attributes of the tag whose name ends at `from` in `text`.
* Returns the attributes, or null when they do not parse cleanly up to the
* tag's closing `/>` or `>` (or a name repeats, which XML forbids).
*/
function readTagAttributes(text, from) {
const attributes = new Map();
let at = from;
for (;;) {
ATTRIBUTE.lastIndex = at;
const match = ATTRIBUTE.exec(text);
if (!match) break;
if (attributes.has(match[1])) return null;
attributes.set(match[1], match[2] ?? match[3]);
at = ATTRIBUTE.lastIndex;
}
TAG_END.lastIndex = at;
return TAG_END.test(text) ? attributes : null;
}
// ExcelJS expands a merge into one cell object per covered cell at load time,
// so a merge costs its AREA, not one tag.
function mergeArea(tag) {
const range = parseRange(attribute(tag, 'ref'));
function mergeArea(attributes) {
const range = parseRange(attributes.get('ref'));
if (!range) fail('malformed', 'Worksheet has a merged range that does not parse');
return (Math.abs(range.r2 - range.r1) + 1) * (Math.abs(range.c2 - range.c1) + 1);
}
// ExcelJS builds one column object for every index up to `<col max>`, unclamped.
function checkColumnSpan(tag) {
function checkColumnSpan(attributes) {
for (const name of ['min', 'max']) {
const value = attribute(tag, name);
if (value === null) continue;
const value = attributes.get(name);
if (value === undefined) continue;
const index = Number(value);
if (!Number.isInteger(index) || index < 1 || index > MAX_COL) {
fail('malformed', `Worksheet column ${name} is outside 1-${MAX_COL}`);
@@ -155,7 +182,7 @@
const decoder = new TextDecoder();
let sheetCells = 0;
let sheetMerges = 0;
let inCellXfs = false;
let sheetRows = 0;
const worksheet = /^xl\/worksheets\/[^/]+\.xml$/i.test(name);
const styles = name === 'xl/styles.xml';
const addCells = (cells) => {
@@ -168,32 +195,37 @@
push(chunk, final) {
if (!worksheet && !styles) return;
const text = tail + decoder.decode(chunk, { stream: !final });
// Scan only up to a tag boundary: a tag cut by a chunk edge is carried
// whole into the next scan, so its attributes are read in one piece.
let safeEnd = text.length;
if (!final) {
const open = text.lastIndexOf('<');
if (open > text.lastIndexOf('>')) safeEnd = open;
}
// `<` can never appear inside an attribute value, so every tag before the
// last `<` is complete. Carry everything from that `<` into the next scan
// so a tag cut by a chunk edge is always read in one piece.
const safeEnd = final ? text.length : Math.max(0, text.lastIndexOf('<'));
const scan = text.slice(0, safeEnd);
if (worksheet) {
addCells((scan.match(/<c(?:\s|>)/g) || []).length);
for (const [tag] of scan.matchAll(/<mergeCell(?:\s[^>]*)?>/g)) {
// ExcelJS keeps a Row object for every <row>, with or without cells.
const rows = (scan.match(/<row(?=[\s/>])/g) || []).length;
sheetRows += rows;
counts.rows += rows;
if (sheetRows > limits.maxRowsPerSheet || counts.rows > limits.maxRows)
fail('row-limit', 'Workbook exceeds the rows limit');
for (const match of scan.matchAll(/<(mergeCell|col)(?=[\s/>])/g)) {
const attributes = readTagAttributes(scan, match.index + match[0].length);
if (!attributes) fail('malformed', `Worksheet has a <${match[1]}> whose attributes do not parse`);
if (match[1] === 'col') {
checkColumnSpan(attributes);
continue;
}
sheetMerges += 1;
counts.merges += 1;
if (sheetMerges > limits.maxMergesPerSheet)
fail('merge-limit', 'Worksheet exceeds the merged ranges limit');
addCells(mergeArea(tag));
addCells(mergeArea(attributes));
}
for (const [tag] of scan.matchAll(/<col(?:\s[^>]*)?>/g)) checkColumnSpan(tag);
}
if (styles) {
const tokens = scan.match(/<cellXfs(?:\s|>)|<\/cellXfs\s*>|<xf(?:\s|\/?>)/g) || [];
for (const token of tokens) {
if (token.startsWith('<cellXfs')) inCellXfs = true;
else if (token.startsWith('</cellXfs')) inCellXfs = false;
else if (inCellXfs) counts.styles += 1;
}
// Every <xf> counts, cellStyleXfs included: tracking which list a tag
// sits in can be desynced by a closing tag inside an XML comment.
counts.styles += (scan.match(/<xf(?=[\s/>])/g) || []).length;
if (counts.styles > limits.maxStyles) fail('style-limit', 'Workbook exceeds the cell styles limit');
}
tail = text.slice(safeEnd);
@@ -210,7 +242,7 @@
for (const entry of directory.entries) expectedEntries.set(entry.name, (expectedEntries.get(entry.name) || 0) + 1);
const streamedEntries = new Map();
const inflatedEntries = Object.create(null);
const counts = { worksheets: 0, cells: 0, merges: 0, styles: 0 };
const counts = { worksheets: 0, cells: 0, rows: 0, merges: 0, styles: 0 };
const features = new Set();
let totalInflated = 0;
let seenEntries = 0;