mirror of
https://github.com/Ark0N/Codeman.git
synced 2026-10-10 09:19:42 +02:00
fix(preview): bound what ExcelJS expands during XLSX admission
ExcelJS 4.4.0 expands three constructs into one object per cell or column at load time, so a few KB admitted as one cell could cost a gigabyte: - a <mergeCell> now costs its full area against the per-sheet and total cell caps, and a ref that does not parse is refused - a <col> whose min or max is past 16384 is refused - the worker loads with ignoreNodes: ['dataValidations']; the preview never shows validations, and a whole-column dropdown took 5 s The XML counter now scans up to the last complete tag and carries the rest, so a merge or col tag cut by an inflate-chunk edge is read whole. A central-directory compressedSize that runs past the file is refused, since the ratio cap divides by it. The renderer and core axis offsets use prefix sums with a binary search instead of walking every override per call.
This commit is contained in:
@@ -284,7 +284,7 @@ Codeman is a Claude Code session manager with web interface and autonomous Ralph
|
|||||||
|
|
||||||
**Attachments** (live external document references; all wiring in `file-routes.ts`): a **registry** maps a stable `attachmentId` to a realpath-resolved, extension-allowlisted absolute path, so browser requests never carry arbitrary absolute paths. ⚠️ The **magic-link scanner** (`codeman://attach?...` in terminal output) is **prompt-injectable**, so its scan path is force-confined to the session workspace; a hostile prompt could otherwise exfiltrate arbitrary host files over SSE. The security gate is an extension **allowlist**, not a blocklist. `document-conversion-limiter.ts` caps converter spawns globally: without it, N large docs detected at once fork N multi-minute processes, which is a resource-exhaustion vector. → [architecture-invariants#attachments](docs/architecture-invariants.md#attachments)
|
**Attachments** (live external document references; all wiring in `file-routes.ts`): a **registry** maps a stable `attachmentId` to a realpath-resolved, extension-allowlisted absolute path, so browser requests never carry arbitrary absolute paths. ⚠️ The **magic-link scanner** (`codeman://attach?...` in terminal output) is **prompt-injectable**, so its scan path is force-confined to the session workspace; a hostile prompt could otherwise exfiltrate arbitrary host files over SSE. The security gate is an extension **allowlist**, not a blocklist. `document-conversion-limiter.ts` caps converter spawns globally: without it, N large docs detected at once fork N multi-minute processes, which is a resource-exhaustion vector. → [architecture-invariants#attachments](docs/architecture-invariants.md#attachments)
|
||||||
|
|
||||||
**File-path links (terminal + chat)**: a path an agent prints is clickable on BOTH surfaces and opens the file-preview overlay. ⚠️ ONE pattern (`FILE_PATH_LINK_PATTERN` / `absoluteFilePathPattern()` in constants.js) feeds the xterm link provider AND `_linkifyFilePaths()`, a fresh instance per call (`lastIndex`). The chat linkifier walks TEXT NODES with DOM APIs, never rebuilds sanitized markup as a string. ⚠️ An out-of-workspace path goes through the ATTACHMENT routes (`POST /api/sessions/:id/attachments` with `notify: false`), never by widening `file-content`/`file-raw` or `file-stream-manager`'s `tail -f` allowlist. ⚠️ `TEXT_ATTACHMENT_EXTENSIONS` IS `EDITABLE_EXTENSIONS` (never a second list), and widening READ must never widen RUN: `html`/`htm`/`svg` stay download-only, other text is inert `text/plain`+`nosniff`. Media extensions are single-sourced in `attachment-registry.ts`. ⚠️ **XLSX previews parse in the BROWSER**, never on the server: `spreadsheet-preview.js` fetches the raw route with `?preview=true` (413 above `MAX_XLSX_BROWSER_PREVIEW_BYTES`, 10 MB) and hands the bytes to `spreadsheet-preview-worker.js`, the only place the pinned `exceljs`/`fflate` vendor bundles load (never on page load). `admitXlsx()` caps the ZIP before ExcelJS runs, and ExcelJS parses only a STORE-only archive rebuilt from the entries admission inflated (`buildAdmittedArchive()`), never the fetched bytes; cell text goes through `textContent`, formulas are never evaluated. Bumping either package or editing the worker/core changes `SPREADSHEET_ASSET_VERSION`, which `npm run check:public-assets` pins. xls/ods stay download-only. → [architecture-invariants#file-path-links-terminal--response-viewer](docs/architecture-invariants.md#file-path-links-terminal--response-viewer)
|
**File-path links (terminal + chat)**: a path an agent prints is clickable on BOTH surfaces and opens the file-preview overlay. ⚠️ ONE pattern (`FILE_PATH_LINK_PATTERN` / `absoluteFilePathPattern()` in constants.js) feeds the xterm link provider AND `_linkifyFilePaths()`, a fresh instance per call (`lastIndex`). The chat linkifier walks TEXT NODES with DOM APIs, never rebuilds sanitized markup as a string. ⚠️ An out-of-workspace path goes through the ATTACHMENT routes (`POST /api/sessions/:id/attachments` with `notify: false`), never by widening `file-content`/`file-raw` or `file-stream-manager`'s `tail -f` allowlist. ⚠️ `TEXT_ATTACHMENT_EXTENSIONS` IS `EDITABLE_EXTENSIONS` (never a second list), and widening READ must never widen RUN: `html`/`htm`/`svg` stay download-only, other text is inert `text/plain`+`nosniff`. Media extensions are single-sourced in `attachment-registry.ts`. ⚠️ **XLSX previews parse in the BROWSER**, never on the server: `spreadsheet-preview.js` fetches the raw route with `?preview=true` (413 above `MAX_XLSX_BROWSER_PREVIEW_BYTES`, 10 MB) and hands the bytes to `spreadsheet-preview-worker.js`, the only place the pinned `exceljs`/`fflate` vendor bundles load (never on page load). `admitXlsx()` caps the ZIP before ExcelJS runs, counting what ExcelJS will EXPAND as well as what it reads (a merge costs its area, a `<col>` past 16384 is refused, validations are never parsed), and ExcelJS parses only a STORE-only archive rebuilt from the entries admission inflated (`buildAdmittedArchive()`), never the fetched bytes; cell text goes through `textContent`, formulas are never evaluated. Bumping either package or editing the worker/core changes `SPREADSHEET_ASSET_VERSION`, which `npm run check:public-assets` pins. xls/ods stay download-only. → [architecture-invariants#file-path-links-terminal--response-viewer](docs/architecture-invariants.md#file-path-links-terminal--response-viewer)
|
||||||
|
|
||||||
**Filesystem path picker** (Link Existing "Browse" + the mobile keyboard's `📁 Path` key): lazy one-directory browsing via `GET /api/filesystem/browse`, with `GET /api/filesystem/preview` for the tapped file. Inserts the path **without** Enter, so the prompt is never submitted; the sibling `⌫ All` key clears only the unsent prompt and must never send the agent's `/clear`. ⚠️ This is a **second file-serving surface and inherits neither the attachment confinement nor its ownership scoping** — it allowlists Home, `CASES_DIR`, `/mnt/d` and `CODEMAN_FILE_PICKER_ROOTS`, blocks sensitive trees, and rejects symlink escapes **after** `realpath`. ⚠️ The optional `sessionId` is an ownership boundary that must be `canAccessOwned`-checked by hand (it does not go through `findSessionOrFail`), and in multi-user mode a non-admin gets only their own `userSpacePath` as a root: per-user spaces live INSIDE `homedir()`, so a `Home` root exposes every other user's workspace. Previews go through the same global conversion limiter, and Markdown/TXT/JSON are served as inert `text/plain`. → [architecture-invariants#filesystem-path-picker](docs/architecture-invariants.md#filesystem-path-picker)
|
**Filesystem path picker** (Link Existing "Browse" + the mobile keyboard's `📁 Path` key): lazy one-directory browsing via `GET /api/filesystem/browse`, with `GET /api/filesystem/preview` for the tapped file. Inserts the path **without** Enter, so the prompt is never submitted; the sibling `⌫ All` key clears only the unsent prompt and must never send the agent's `/clear`. ⚠️ This is a **second file-serving surface and inherits neither the attachment confinement nor its ownership scoping** — it allowlists Home, `CASES_DIR`, `/mnt/d` and `CODEMAN_FILE_PICKER_ROOTS`, blocks sensitive trees, and rejects symlink escapes **after** `realpath`. ⚠️ The optional `sessionId` is an ownership boundary that must be `canAccessOwned`-checked by hand (it does not go through `findSessionOrFail`), and in multi-user mode a non-admin gets only their own `userSpacePath` as a root: per-user spaces live INSIDE `homedir()`, so a `Home` root exposes every other user's workspace. Previews go through the same global conversion limiter, and Markdown/TXT/JSON are served as inert `text/plain`. → [architecture-invariants#filesystem-path-picker](docs/architecture-invariants.md#filesystem-path-picker)
|
||||||
|
|
||||||
|
|||||||
@@ -358,7 +358,7 @@ A file path an agent prints is a link on both surfaces it can appear on, and cli
|
|||||||
|
|
||||||
⚠️ **The preview overlay must outrank the panel that launched it.** `.file-preview-overlay` sits at `z-index: 5100`, above the response viewer (5000) and its backdrop (4999); at its historical 2000 a path clicked in the chat opened the overlay *behind* the chat, which reads as a dead link. It stays below the toast/picker band (10000+) so a "Saved" toast still lands on top.
|
⚠️ **The preview overlay must outrank the panel that launched it.** `.file-preview-overlay` sits at `z-index: 5100`, above the response viewer (5000) and its backdrop (4999); at its historical 2000 a path clicked in the chat opened the overlay *behind* the chat, which reads as a dead link. It stays below the toast/picker band (10000+) so a "Saved" toast still lands on top.
|
||||||
|
|
||||||
**XLSX previews parse in the browser worker, and ExcelJS only ever sees what admission checked.** An `.xlsx` joins the raw routes like any other file (`?preview=true` caps it at 10 MB with a 413); the server never parses it. `spreadsheet-preview.js` hands the bytes to `spreadsheet-preview-worker.js`, the ONLY place the pinned `exceljs`/`fflate` vendor bundles load: fflate and `spreadsheet-xlsx-core.js` at worker start, ExcelJS only after `admitXlsx()` passes, and the page itself never loads either. `admitXlsx()` streams every entry through fflate and enforces the entry count, per-entry and total inflated bytes (64 MB), compression ratio, worksheet, cell, merge and style caps on the actual inflated output, refusing (never truncating) a workbook that trips one. ⚠️ Admission walks LOCAL headers while ExcelJS (JSZip) reads the CENTRAL directory, so overlapping entries (a stored entry hiding a whole `sheet1.xml`, a one-cell decoy later in the stream) once let a file be admitted as 1 cell and parsed as 300k. The worker therefore never hands ExcelJS the fetched bytes: it gets `buildAdmittedArchive()`, a STORE-only `fflate.zipSync(entries, { level: 0 })` of exactly the entries admission inflated, and a name streamed twice is refused. That costs one transient copy of the inflated entries (bounded by the same 64 MB cap) and is pinned by the overlapping-entry fixture in `test/spreadsheet-preview-worker.test.ts`. The worker is a stable URL cache-busted by `SPREADSHEET_ASSET_VERSION` in `spreadsheet-preview.js`, the content hash of the worker, the core and both vendor bundles; `npm run check:public-assets` fails when it drifts, so editing any of those (or bumping either package) means updating the token, or a deploy pairs a new worker with a year-cached old core.
|
**XLSX previews parse in the browser worker, and ExcelJS only ever sees what admission checked.** An `.xlsx` joins the raw routes like any other file (`?preview=true` caps it at 10 MB with a 413); the server never parses it. `spreadsheet-preview.js` hands the bytes to `spreadsheet-preview-worker.js`, the ONLY place the pinned `exceljs`/`fflate` vendor bundles load: fflate and `spreadsheet-xlsx-core.js` at worker start, ExcelJS only after `admitXlsx()` passes, and the page itself never loads either. `admitXlsx()` streams every entry through fflate and enforces the entry count, per-entry and total inflated bytes (64 MB), compression ratio, worksheet, cell, merge and style caps on the actual inflated output, refusing (never truncating) a workbook that trips one. ⚠️ Admission walks LOCAL headers while ExcelJS (JSZip) reads the CENTRAL directory, so overlapping entries (a stored entry hiding a whole `sheet1.xml`, a one-cell decoy later in the stream) once let a file be admitted as 1 cell and parsed as 300k. The worker therefore never hands ExcelJS the fetched bytes: it gets `buildAdmittedArchive()`, a STORE-only `fflate.zipSync(entries, { level: 0 })` of exactly the entries admission inflated, and a name streamed twice is refused. That costs one transient copy of the inflated entries (bounded by the same 64 MB cap) and is pinned by the overlapping-entry fixture in `test/spreadsheet-preview-worker.test.ts`. ⚠️ Admission also has to bound what ExcelJS EXPANDS, not just what it reads: ExcelJS 4.4.0 builds one object per covered cell of a `<mergeCell>`, per address of a `<dataValidation sqref>` and per index up to `<col max>`, so a 6.5 KB file admitted as one cell once cost a gigabyte. The counter therefore charges each merge its full AREA against the cell caps (an unparseable `ref` is refused), refuses a `<col>` whose `min`/`max` is past 16384, and the worker loads with `ignoreNodes: ['dataValidations']` (the preview never shows validations). The counter reads tags whole across inflate-chunk edges (it scans only up to the last complete tag and carries the rest), and a central-directory `compressedSize` that runs past the file is refused, since the ratio cap divides by it. Fixtures for each are in `test/spreadsheet-preview-worker.test.ts` and `test/spreadsheet-xlsx-core.test.ts`. The worker is a stable URL cache-busted by `SPREADSHEET_ASSET_VERSION` in `spreadsheet-preview.js`, the content hash of the worker, the core and both vendor bundles; `npm run check:public-assets` fails when it drifts, so editing any of those (or bumping either package) means updating the token, or a deploy pairs a new worker with a year-cached old core.
|
||||||
|
|
||||||
### Filesystem path picker
|
### Filesystem path picker
|
||||||
|
|
||||||
|
|||||||
@@ -125,7 +125,10 @@ async function loadWorkbook(bytes) {
|
|||||||
const admitted = core.buildAdmittedArchive(admission, self.fflate);
|
const admitted = core.buildAdmittedArchive(admission, self.fflate);
|
||||||
if (!self.ExcelJS) importScripts(`vendor/exceljs.min.js${spreadsheetAssetQuery}`);
|
if (!self.ExcelJS) importScripts(`vendor/exceljs.min.js${spreadsheetAssetQuery}`);
|
||||||
const nextWorkbook = new self.ExcelJS.Workbook();
|
const nextWorkbook = new self.ExcelJS.Workbook();
|
||||||
await nextWorkbook.xlsx.load(admitted);
|
// ExcelJS expands every address of a `<dataValidation sqref>` into its own
|
||||||
|
// object (a whole-column dropdown is a million), and the preview never shows
|
||||||
|
// validations, so they are not parsed at all.
|
||||||
|
await nextWorkbook.xlsx.load(admitted, { ignoreNodes: ['dataValidations'] });
|
||||||
const nextSheets = new Map();
|
const nextSheets = new Map();
|
||||||
const nextRows = new Map();
|
const nextRows = new Map();
|
||||||
normalizedStyles = [];
|
normalizedStyles = [];
|
||||||
|
|||||||
@@ -20,7 +20,7 @@
|
|||||||
(function initSpreadsheetPreview(global) {
|
(function initSpreadsheetPreview(global) {
|
||||||
'use strict';
|
'use strict';
|
||||||
|
|
||||||
const SPREADSHEET_ASSET_VERSION = '91b615278d5b';
|
const SPREADSHEET_ASSET_VERSION = '078d54b1055a';
|
||||||
const MAX_PREVIEW_BYTES = 10 * 1024 * 1024;
|
const MAX_PREVIEW_BYTES = 10 * 1024 * 1024;
|
||||||
const DEFAULT_TIMEOUT_MS = 20000;
|
const DEFAULT_TIMEOUT_MS = 20000;
|
||||||
const MAX_SCROLL_PX = 8000000;
|
const MAX_SCROLL_PX = 8000000;
|
||||||
@@ -82,14 +82,37 @@
|
|||||||
return metadata?.sheets.find((sheet) => String(sheet.id) === String(activeSheetId));
|
return metadata?.sheets.find((sheet) => String(sheet.id) === String(activeSheetId));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Prefix sums per override list, built once per sheet's axis: renderTile
|
||||||
|
// asks for several offsets per cell, so a linear walk over every override
|
||||||
|
// (one per row on a sheet with explicit heights) made each tile O(n) per cell.
|
||||||
|
const axisDeltas = new WeakMap();
|
||||||
|
|
||||||
|
function overrideDeltas(overrides, defaultSize) {
|
||||||
|
let entry = axisDeltas.get(overrides);
|
||||||
|
if (!entry || entry.defaultSize !== defaultSize) {
|
||||||
|
const deltas = [];
|
||||||
|
let delta = 0;
|
||||||
|
for (const [, size] of overrides) {
|
||||||
|
delta += size - defaultSize;
|
||||||
|
deltas.push(delta);
|
||||||
|
}
|
||||||
|
entry = { defaultSize, deltas };
|
||||||
|
axisDeltas.set(overrides, entry);
|
||||||
|
}
|
||||||
|
return entry.deltas;
|
||||||
|
}
|
||||||
|
|
||||||
function axisOffset(count, defaultSize, overrides, index) {
|
function axisOffset(count, defaultSize, overrides, index) {
|
||||||
const bounded = Math.max(1, Math.min(count + 1, index));
|
const bounded = Math.max(1, Math.min(count + 1, index));
|
||||||
let value = (bounded - 1) * defaultSize;
|
const list = overrides || [];
|
||||||
for (const [overrideIndex, size] of overrides || []) {
|
let low = 0;
|
||||||
if (overrideIndex >= bounded) break;
|
let high = list.length;
|
||||||
value += size - defaultSize;
|
while (low < high) {
|
||||||
|
const mid = (low + high) >> 1;
|
||||||
|
if (list[mid][0] < bounded) low = mid + 1;
|
||||||
|
else high = mid;
|
||||||
}
|
}
|
||||||
return value;
|
return (bounded - 1) * defaultSize + (low ? overrideDeltas(list, defaultSize)[low - 1] : 0);
|
||||||
}
|
}
|
||||||
|
|
||||||
function axisIndex(count, defaultSize, overrides, offset) {
|
function axisIndex(count, defaultSize, overrides, offset) {
|
||||||
|
|||||||
@@ -98,6 +98,11 @@
|
|||||||
const localExtraLength = u16(bytes, localHeaderOffset + 28);
|
const localExtraLength = u16(bytes, localHeaderOffset + 28);
|
||||||
const localNameEnd = localHeaderOffset + 30 + localNameLength;
|
const localNameEnd = localHeaderOffset + 30 + localNameLength;
|
||||||
if (localNameEnd + localExtraLength > directoryOffset) fail('malformed', 'Malformed XLSX local entry bounds');
|
if (localNameEnd + localExtraLength > directoryOffset) fail('malformed', 'Malformed XLSX local entry bounds');
|
||||||
|
// The compression-ratio cap divides by this declared size, so it must
|
||||||
|
// describe bytes that actually exist before the central directory.
|
||||||
|
if (localNameEnd + localExtraLength + compressedSize > directoryOffset) {
|
||||||
|
fail('malformed', 'XLSX entry declares more compressed bytes than the file holds');
|
||||||
|
}
|
||||||
const localName = new TextDecoder().decode(bytes.subarray(localHeaderOffset + 30, localNameEnd));
|
const localName = new TextDecoder().decode(bytes.subarray(localHeaderOffset + 30, localNameEnd));
|
||||||
if (localName !== name) fail('malformed', 'XLSX local and central directory names do not match');
|
if (localName !== name) fail('malformed', 'XLSX local and central directory names do not match');
|
||||||
entries.push({ name, compressedSize, declaredSize, localHeaderOffset });
|
entries.push({ name, compressedSize, declaredSize, localHeaderOffset });
|
||||||
@@ -115,6 +120,36 @@
|
|||||||
return null;
|
return null;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Longest single XML tag the streaming counter will carry across chunks. Real
|
||||||
|
// worksheet tags are a few hundred bytes; this only stops a pathological tag
|
||||||
|
// from turning the carried tail into quadratic re-scanning.
|
||||||
|
const MAX_CARRIED_TAG = 256 * 1024;
|
||||||
|
|
||||||
|
function attribute(tag, name) {
|
||||||
|
const match = new RegExp(`\\s${name}\\s*=\\s*(?:"([^"]*)"|'([^']*)')`).exec(tag);
|
||||||
|
return match ? (match[1] ?? match[2]) : null;
|
||||||
|
}
|
||||||
|
|
||||||
|
// ExcelJS expands a merge into one cell object per covered cell at load time,
|
||||||
|
// so a merge costs its AREA, not one tag.
|
||||||
|
function mergeArea(tag) {
|
||||||
|
const range = parseRange(attribute(tag, 'ref'));
|
||||||
|
if (!range) fail('malformed', 'Worksheet has a merged range that does not parse');
|
||||||
|
return (Math.abs(range.r2 - range.r1) + 1) * (Math.abs(range.c2 - range.c1) + 1);
|
||||||
|
}
|
||||||
|
|
||||||
|
// ExcelJS builds one column object for every index up to `<col max>`, unclamped.
|
||||||
|
function checkColumnSpan(tag) {
|
||||||
|
for (const name of ['min', 'max']) {
|
||||||
|
const value = attribute(tag, name);
|
||||||
|
if (value === null) continue;
|
||||||
|
const index = Number(value);
|
||||||
|
if (!Number.isInteger(index) || index < 1 || index > MAX_COL) {
|
||||||
|
fail('malformed', `Worksheet column ${name} is outside 1-${MAX_COL}`);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
function createXmlCounter(name, counts, limits) {
|
function createXmlCounter(name, counts, limits) {
|
||||||
let tail = '';
|
let tail = '';
|
||||||
const decoder = new TextDecoder();
|
const decoder = new TextDecoder();
|
||||||
@@ -123,22 +158,34 @@
|
|||||||
let inCellXfs = false;
|
let inCellXfs = false;
|
||||||
const worksheet = /^xl\/worksheets\/[^/]+\.xml$/i.test(name);
|
const worksheet = /^xl\/worksheets\/[^/]+\.xml$/i.test(name);
|
||||||
const styles = name === 'xl/styles.xml';
|
const styles = name === 'xl/styles.xml';
|
||||||
|
const addCells = (cells) => {
|
||||||
|
sheetCells += cells;
|
||||||
|
counts.cells += cells;
|
||||||
|
if (sheetCells > limits.maxCellsPerSheet || counts.cells > limits.maxCells)
|
||||||
|
fail('cell-limit', 'Workbook exceeds the cells limit');
|
||||||
|
};
|
||||||
return {
|
return {
|
||||||
push(chunk, final) {
|
push(chunk, final) {
|
||||||
if (!worksheet && !styles) return;
|
if (!worksheet && !styles) return;
|
||||||
const text = tail + decoder.decode(chunk, { stream: !final });
|
const text = tail + decoder.decode(chunk, { stream: !final });
|
||||||
const safeEnd = final ? text.length : Math.max(0, text.length - 128);
|
// Scan only up to a tag boundary: a tag cut by a chunk edge is carried
|
||||||
|
// whole into the next scan, so its attributes are read in one piece.
|
||||||
|
let safeEnd = text.length;
|
||||||
|
if (!final) {
|
||||||
|
const open = text.lastIndexOf('<');
|
||||||
|
if (open > text.lastIndexOf('>')) safeEnd = open;
|
||||||
|
}
|
||||||
const scan = text.slice(0, safeEnd);
|
const scan = text.slice(0, safeEnd);
|
||||||
if (worksheet) {
|
if (worksheet) {
|
||||||
const cells = (scan.match(/<c(?:\s|>)/g) || []).length;
|
addCells((scan.match(/<c(?:\s|>)/g) || []).length);
|
||||||
const merges = (scan.match(/<mergeCell(?:\s|>)/g) || []).length;
|
for (const [tag] of scan.matchAll(/<mergeCell(?:\s[^>]*)?>/g)) {
|
||||||
sheetCells += cells;
|
sheetMerges += 1;
|
||||||
sheetMerges += merges;
|
counts.merges += 1;
|
||||||
counts.cells += cells;
|
if (sheetMerges > limits.maxMergesPerSheet)
|
||||||
counts.merges += merges;
|
fail('merge-limit', 'Worksheet exceeds the merged ranges limit');
|
||||||
if (sheetCells > limits.maxCellsPerSheet || counts.cells > limits.maxCells)
|
addCells(mergeArea(tag));
|
||||||
fail('cell-limit', 'Workbook exceeds the cells limit');
|
}
|
||||||
if (sheetMerges > limits.maxMergesPerSheet) fail('merge-limit', 'Worksheet exceeds the merged ranges limit');
|
for (const [tag] of scan.matchAll(/<col(?:\s[^>]*)?>/g)) checkColumnSpan(tag);
|
||||||
}
|
}
|
||||||
if (styles) {
|
if (styles) {
|
||||||
const tokens = scan.match(/<cellXfs(?:\s|>)|<\/cellXfs\s*>|<xf(?:\s|\/?>)/g) || [];
|
const tokens = scan.match(/<cellXfs(?:\s|>)|<\/cellXfs\s*>|<xf(?:\s|\/?>)/g) || [];
|
||||||
@@ -150,6 +197,7 @@
|
|||||||
if (counts.styles > limits.maxStyles) fail('style-limit', 'Workbook exceeds the cell styles limit');
|
if (counts.styles > limits.maxStyles) fail('style-limit', 'Workbook exceeds the cell styles limit');
|
||||||
}
|
}
|
||||||
tail = text.slice(safeEnd);
|
tail = text.slice(safeEnd);
|
||||||
|
if (tail.length > MAX_CARRIED_TAG) fail('malformed', 'Workbook XML has an oversized tag');
|
||||||
},
|
},
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
@@ -277,17 +325,34 @@
|
|||||||
.filter(([index, size]) => index >= 1 && index <= count && Number.isFinite(size))
|
.filter(([index, size]) => index >= 1 && index <= count && Number.isFinite(size))
|
||||||
.map(([index, size]) => [index, Math.max(0, size)])
|
.map(([index, size]) => [index, Math.max(0, size)])
|
||||||
.sort((a, b) => a[0] - b[0]);
|
.sort((a, b) => a[0] - b[0]);
|
||||||
return { count: Math.max(0, count), defaultSize: Math.max(0, defaultSize), overrides: sorted };
|
const base = Math.max(0, defaultSize);
|
||||||
|
// deltas[i] is the summed size change of overrides[0..i], so an offset is
|
||||||
|
// one binary search instead of a walk over every override.
|
||||||
|
const deltas = [];
|
||||||
|
let delta = 0;
|
||||||
|
for (const [, size] of sorted) {
|
||||||
|
delta += size - base;
|
||||||
|
deltas.push(delta);
|
||||||
|
}
|
||||||
|
return { count: Math.max(0, count), defaultSize: base, overrides: sorted, deltas };
|
||||||
|
}
|
||||||
|
|
||||||
|
// Number of overrides whose index is below `bounded`.
|
||||||
|
function overridesBefore(overrides, bounded) {
|
||||||
|
let low = 0;
|
||||||
|
let high = overrides.length;
|
||||||
|
while (low < high) {
|
||||||
|
const mid = (low + high) >> 1;
|
||||||
|
if (overrides[mid][0] < bounded) low = mid + 1;
|
||||||
|
else high = mid;
|
||||||
|
}
|
||||||
|
return low;
|
||||||
}
|
}
|
||||||
|
|
||||||
function axisOffset(axis, index) {
|
function axisOffset(axis, index) {
|
||||||
const bounded = Math.max(1, Math.min(axis.count + 1, index));
|
const bounded = Math.max(1, Math.min(axis.count + 1, index));
|
||||||
let offset = (bounded - 1) * axis.defaultSize;
|
const before = overridesBefore(axis.overrides, bounded);
|
||||||
for (const [overrideIndex, size] of axis.overrides) {
|
return (bounded - 1) * axis.defaultSize + (before ? axis.deltas[before - 1] : 0);
|
||||||
if (overrideIndex >= bounded) break;
|
|
||||||
offset += size - axis.defaultSize;
|
|
||||||
}
|
|
||||||
return offset;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
function axisIndexAt(axis, offset) {
|
function axisIndexAt(axis, offset) {
|
||||||
@@ -641,6 +706,7 @@
|
|||||||
MAX_COL,
|
MAX_COL,
|
||||||
XlsxPreviewError,
|
XlsxPreviewError,
|
||||||
inspectZipDirectory,
|
inspectZipDirectory,
|
||||||
|
createXmlCounter,
|
||||||
admitXlsx,
|
admitXlsx,
|
||||||
buildAdmittedArchive,
|
buildAdmittedArchive,
|
||||||
parseCellRef,
|
parseCellRef,
|
||||||
|
|||||||
@@ -132,6 +132,8 @@ function createHarness() {
|
|||||||
messages,
|
messages,
|
||||||
imports,
|
imports,
|
||||||
self,
|
self,
|
||||||
|
/** Evaluate an expression against the worker's own top-level bindings. */
|
||||||
|
peek: (expression: string): unknown => vm.runInContext(expression, context),
|
||||||
send: async (data: unknown) => {
|
send: async (data: unknown) => {
|
||||||
await (self.onmessage as (event: { data: unknown }) => Promise<void>)({ data });
|
await (self.onmessage as (event: { data: unknown }) => Promise<void>)({ data });
|
||||||
},
|
},
|
||||||
@@ -508,3 +510,66 @@ describe('spreadsheet preview worker: ExcelJS sees only what admission checked',
|
|||||||
else expect(result.type).toBe('error');
|
else expect(result.type).toBe('error');
|
||||||
}, 60_000);
|
}, 60_000);
|
||||||
});
|
});
|
||||||
|
|
||||||
|
/** A one-cell ExcelJS workbook whose sheet1.xml gets `xml` spliced in at `where`. */
|
||||||
|
async function sheetWithInjectedXml(where: 'before-sheetData' | 'after-sheetData', xml: string): Promise<ArrayBuffer> {
|
||||||
|
const workbook = new ExcelJS.Workbook();
|
||||||
|
workbook.addWorksheet('Data').getCell('A1').value = 'one';
|
||||||
|
const entries = fflate.unzipSync(new Uint8Array(await workbook.xlsx.writeBuffer()));
|
||||||
|
const sheet = fflate.strFromU8(entries['xl/worksheets/sheet1.xml']);
|
||||||
|
const patched =
|
||||||
|
where === 'before-sheetData'
|
||||||
|
? sheet.replace('<sheetData>', `${xml}<sheetData>`)
|
||||||
|
: sheet.replace('</sheetData>', `</sheetData>${xml}`);
|
||||||
|
expect(patched).not.toBe(sheet);
|
||||||
|
entries['xl/worksheets/sheet1.xml'] = fflate.strToU8(patched);
|
||||||
|
return toArrayBuffer(fflate.zipSync(entries));
|
||||||
|
}
|
||||||
|
|
||||||
|
describe('spreadsheet preview worker: admission bounds what ExcelJS expands', () => {
|
||||||
|
// ExcelJS creates one cell object per covered cell of a merge, so a single
|
||||||
|
// `<mergeCell>` tag over 3M cells took 10 s and a gigabyte of heap.
|
||||||
|
it('refuses a merge whose area exceeds the cell caps before ExcelJS loads', async () => {
|
||||||
|
const harness = createHarness();
|
||||||
|
await harness.send({
|
||||||
|
type: 'load',
|
||||||
|
bytes: await sheetWithInjectedXml(
|
||||||
|
'after-sheetData',
|
||||||
|
'<mergeCells count="1"><mergeCell ref="A1:CV30000"/></mergeCells>'
|
||||||
|
),
|
||||||
|
});
|
||||||
|
expect(harness.messages.at(-1)).toMatchObject({ type: 'error', code: 'cell-limit' });
|
||||||
|
expect(harness.imports.some((url) => url.includes('exceljs'))).toBe(false);
|
||||||
|
});
|
||||||
|
|
||||||
|
// `Column.fromModel` builds every column up to `<col max>` with no clamp.
|
||||||
|
it('refuses a <col> range past column 16384 before ExcelJS loads', async () => {
|
||||||
|
const harness = createHarness();
|
||||||
|
await harness.send({
|
||||||
|
type: 'load',
|
||||||
|
bytes: await sheetWithInjectedXml('before-sheetData', '<cols><col min="1" max="3000000" width="9"/></cols>'),
|
||||||
|
});
|
||||||
|
expect(harness.messages.at(-1)).toMatchObject({ type: 'error', code: 'malformed' });
|
||||||
|
expect(harness.imports.some((url) => url.includes('exceljs'))).toBe(false);
|
||||||
|
});
|
||||||
|
|
||||||
|
// A whole-column dropdown is a few bytes of XML that ExcelJS expands into one
|
||||||
|
// object per address (5 s here; a whole-sheet range was still running after
|
||||||
|
// 60 s). The preview never shows validations, so the worker does not parse
|
||||||
|
// them, and the file still previews.
|
||||||
|
it('previews a sheet with a whole-column data validation without expanding it', async () => {
|
||||||
|
const harness = createHarness();
|
||||||
|
const metadata = await loadMetadata(
|
||||||
|
harness,
|
||||||
|
await sheetWithInjectedXml(
|
||||||
|
'after-sheetData',
|
||||||
|
'<dataValidations count="1"><dataValidation type="list" allowBlank="1" sqref="B2:B1048576">' +
|
||||||
|
'<formula1>"a,b"</formula1></dataValidation></dataValidations>'
|
||||||
|
)
|
||||||
|
);
|
||||||
|
expect(harness.peek('Object.keys(workbook.worksheets[0].dataValidations.model).length')).toBe(0);
|
||||||
|
expect(metadata.sheets[0]).toMatchObject({ rows: 1, cols: 1 });
|
||||||
|
const tile = await requestTile(harness, metadata.sheets[0].id, { r1: 1, c1: 1, r2: 1, c2: 1 });
|
||||||
|
expect(tile.cells).toEqual([expect.objectContaining({ row: 1, col: 1, text: 'one' })]);
|
||||||
|
}, 30_000);
|
||||||
|
});
|
||||||
|
|||||||
@@ -11,6 +11,11 @@ type Core = {
|
|||||||
XlsxPreviewError: new (code: string, message: string) => Error & { code: string };
|
XlsxPreviewError: new (code: string, message: string) => Error & { code: string };
|
||||||
inspectZipDirectory(bytes: Uint8Array, limits?: Record<string, number>): { entries: Array<{ name: string }> };
|
inspectZipDirectory(bytes: Uint8Array, limits?: Record<string, number>): { entries: Array<{ name: string }> };
|
||||||
admitXlsx(bytes: Uint8Array, zip: typeof fflate, limits?: Record<string, number>): unknown;
|
admitXlsx(bytes: Uint8Array, zip: typeof fflate, limits?: Record<string, number>): unknown;
|
||||||
|
createXmlCounter(
|
||||||
|
name: string,
|
||||||
|
counts: { cells: number; merges: number; styles: number },
|
||||||
|
limits: Record<string, number>
|
||||||
|
): { push(chunk: Uint8Array, final: boolean): void };
|
||||||
buildAdmittedArchive(admission: unknown, zip: typeof fflate): Uint8Array;
|
buildAdmittedArchive(admission: unknown, zip: typeof fflate): Uint8Array;
|
||||||
parseCellRef(ref: string): { row: number; col: number } | null;
|
parseCellRef(ref: string): { row: number; col: number } | null;
|
||||||
deriveExtent(cells: string[], merges: string[]): { rows: number; cols: number };
|
deriveExtent(cells: string[], merges: string[]): { rows: number; cols: number };
|
||||||
@@ -107,7 +112,8 @@ describe('spreadsheet XLSX core', () => {
|
|||||||
counts: { worksheets: number; cells: number; merges: number; styles: number };
|
counts: { worksheets: number; cells: number; merges: number; styles: number };
|
||||||
features: string[];
|
features: string[];
|
||||||
};
|
};
|
||||||
expect(result.counts).toEqual({ worksheets: 1, cells: 1, merges: 1, styles: 1 });
|
// The 2x2 merge costs its four covered cells on top of the one real cell.
|
||||||
|
expect(result.counts).toEqual({ worksheets: 1, cells: 5, merges: 1, styles: 1 });
|
||||||
expect(result.features).toEqual(expect.arrayContaining(['charts', 'externalLinks']));
|
expect(result.features).toEqual(expect.arrayContaining(['charts', 'externalLinks']));
|
||||||
});
|
});
|
||||||
|
|
||||||
@@ -146,6 +152,63 @@ describe('spreadsheet XLSX core', () => {
|
|||||||
expect(() => core.admitXlsx(zip, duplicate)).toThrowError(/duplicate/i);
|
expect(() => core.admitXlsx(zip, duplicate)).toThrowError(/duplicate/i);
|
||||||
});
|
});
|
||||||
|
|
||||||
|
it('charges a merged range its full area and refuses one that does not parse', () => {
|
||||||
|
const merged = (ref: string) =>
|
||||||
|
workbookZip(
|
||||||
|
`<worksheet><sheetData><row><c r="A1"/></row></sheetData><mergeCells><mergeCell ref="${ref}"/></mergeCells></worksheet>`
|
||||||
|
);
|
||||||
|
expect(() => core.admitXlsx(merged('A1:CV30000'), fflate)).toThrowError(/cells limit/i);
|
||||||
|
// Reversed corners describe the same area.
|
||||||
|
expect(() => core.admitXlsx(merged('CV30000:A1'), fflate)).toThrowError(/cells limit/i);
|
||||||
|
expect(() => core.admitXlsx(merged('A1:XFE2'), fflate)).toThrowError(/merged range/i);
|
||||||
|
expect(() => core.admitXlsx(merged('not-a-range'), fflate)).toThrowError(/merged range/i);
|
||||||
|
const admitted = core.admitXlsx(merged('A1:J10'), fflate) as { counts: { cells: number } };
|
||||||
|
expect(admitted.counts.cells).toBe(101);
|
||||||
|
});
|
||||||
|
|
||||||
|
it('refuses a <col> whose min or max is past the last Excel column', () => {
|
||||||
|
const cols = (attrs: string) =>
|
||||||
|
workbookZip(
|
||||||
|
`<worksheet><cols><col ${attrs} width="9"/></cols><sheetData><row><c r="A1"/></row></sheetData></worksheet>`
|
||||||
|
);
|
||||||
|
expect(() => core.admitXlsx(cols('min="1" max="3000000"'), fflate)).toThrowError(/column max/i);
|
||||||
|
expect(() => core.admitXlsx(cols('min="16385" max="16385"'), fflate)).toThrowError(/column min/i);
|
||||||
|
expect(() => core.admitXlsx(cols('min="1" max="1e9"'), fflate)).toThrowError(/column max/i);
|
||||||
|
expect(() => core.admitXlsx(cols('min="1" max="16384"'), fflate)).not.toThrow();
|
||||||
|
});
|
||||||
|
|
||||||
|
it('reads a merge or col tag whole even when a stream chunk boundary cuts through it', () => {
|
||||||
|
// Markup on both sides, so a cut can land 128+ bytes past the tag's start.
|
||||||
|
const pad = '<sheetView workbookViewId="0"/>'.repeat(12);
|
||||||
|
const cases: Array<[string, RegExp]> = [
|
||||||
|
[`<worksheet>${pad}<cols><col min="1" max="99999"/></cols>${pad}</worksheet>`, /column max/i],
|
||||||
|
[`<worksheet>${pad}<mergeCells><mergeCell ref="A1:CV30000"/></mergeCells>${pad}</worksheet>`, /cells limit/i],
|
||||||
|
];
|
||||||
|
for (const [xml, pattern] of cases) {
|
||||||
|
const bytes = fflate.strToU8(xml);
|
||||||
|
for (let cut = 1; cut < bytes.length; cut += 1) {
|
||||||
|
const counter = core.createXmlCounter(
|
||||||
|
'xl/worksheets/sheet1.xml',
|
||||||
|
{ cells: 0, merges: 0, styles: 0 },
|
||||||
|
core.LIMITS
|
||||||
|
);
|
||||||
|
expect(() => {
|
||||||
|
counter.push(bytes.subarray(0, cut), false);
|
||||||
|
counter.push(bytes.subarray(cut), true);
|
||||||
|
}, `cut at ${cut}`).toThrowError(pattern);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
});
|
||||||
|
|
||||||
|
it('refuses an entry whose declared compressed size runs past the file', () => {
|
||||||
|
const zip = workbookZip();
|
||||||
|
const view = new DataView(zip.buffer, zip.byteOffset, zip.byteLength);
|
||||||
|
const eocd = zip.length - 22;
|
||||||
|
const centralStart = view.getUint32(eocd + 16, true);
|
||||||
|
view.setUint32(centralStart + 20, 0x7fffffff, true);
|
||||||
|
expect(() => core.inspectZipDirectory(zip)).toThrowError(/compressed bytes/i);
|
||||||
|
});
|
||||||
|
|
||||||
it('derives bounded extents from real cells and merges', () => {
|
it('derives bounded extents from real cells and merges', () => {
|
||||||
expect(core.parseCellRef('XFD1048576')).toEqual({ row: 1_048_576, col: 16_384 });
|
expect(core.parseCellRef('XFD1048576')).toEqual({ row: 1_048_576, col: 16_384 });
|
||||||
expect(core.parseCellRef('XFE1')).toBeNull();
|
expect(core.parseCellRef('XFE1')).toBeNull();
|
||||||
@@ -164,6 +227,18 @@ describe('spreadsheet XLSX core', () => {
|
|||||||
expect(end - start).toBeLessThan(10);
|
expect(end - start).toBeLessThan(10);
|
||||||
});
|
});
|
||||||
|
|
||||||
|
it('matches a plain walk over the overrides at every index', () => {
|
||||||
|
const overrides: Array<[number, number]> = [];
|
||||||
|
for (let index = 3; index <= 400; index += 7) overrides.push([index, (index * 13) % 50]);
|
||||||
|
const axis = core.createSparseAxis(500, 20, overrides);
|
||||||
|
for (let index = 0; index <= 502; index += 1) {
|
||||||
|
const bounded = Math.max(1, Math.min(501, index));
|
||||||
|
let expected = (bounded - 1) * 20;
|
||||||
|
for (const [at, size] of overrides) if (at < bounded) expected += size - 20;
|
||||||
|
expect(core.axisOffset(axis, index), `index ${index}`).toBe(expected);
|
||||||
|
}
|
||||||
|
});
|
||||||
|
|
||||||
it('returns intersecting merges even when their anchor is offscreen', () => {
|
it('returns intersecting merges even when their anchor is offscreen', () => {
|
||||||
expect(core.intersectingMerges(['A1:D4', 'Z1:Z2'], { r1: 3, c1: 3, r2: 6, c2: 6 })).toEqual(['A1:D4']);
|
expect(core.intersectingMerges(['A1:D4', 'Z1:Z2'], { r1: 3, c1: 3, r2: 6, c2: 6 })).toEqual(['A1:D4']);
|
||||||
});
|
});
|
||||||
|
|||||||
Reference in New Issue
Block a user