From 0ae5ce017ac170f0be3dd067f39c9097903140b5 Mon Sep 17 00:00:00 2001 From: Aamer Akhter Date: Sun, 4 Oct 2026 20:15:09 -0400 Subject: [PATCH] fix(preview): cap cell text, bound merges workbook-wide, refuse runaway number formats --- CLAUDE.md | 2 +- docs/architecture-invariants.md | 2 +- src/web/public/spreadsheet-preview.js | 2 +- src/web/public/spreadsheet-xlsx-core.js | 100 ++++++++++++++++-- test/spreadsheet-preview-worker.test.ts | 130 ++++++++++++++++++++++++ test/spreadsheet-xlsx-core.test.ts | 95 +++++++++++++++++ 6 files changed, 319 insertions(+), 12 deletions(-) diff --git a/CLAUDE.md b/CLAUDE.md index f1c29cd0..280f003f 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -292,7 +292,7 @@ Codeman is a Claude Code session manager with web interface and autonomous Ralph **Attachments** (live external document references; all wiring in `file-routes.ts`): a **registry** maps a stable `attachmentId` to a realpath-resolved, extension-allowlisted absolute path, so browser requests never carry arbitrary absolute paths. ⚠️ The **magic-link scanner** (`codeman://attach?...` in terminal output) is **prompt-injectable**, so its scan path is force-confined to the session workspace; a hostile prompt could otherwise exfiltrate arbitrary host files over SSE. The security gate is an extension **allowlist**, not a blocklist. `document-conversion-limiter.ts` caps converter spawns globally: without it, N large docs detected at once fork N multi-minute processes, which is a resource-exhaustion vector. → [architecture-invariants#attachments](docs/architecture-invariants.md#attachments) -**File-path links (terminal + chat)**: a path an agent prints is clickable on BOTH surfaces and opens the file-preview overlay. ⚠️ ONE pattern (`FILE_PATH_LINK_PATTERN` / `absoluteFilePathPattern()` in constants.js) feeds the xterm link provider AND `_linkifyFilePaths()`, a fresh instance per call (`lastIndex`). The chat linkifier walks TEXT NODES with DOM APIs, never rebuilds sanitized markup as a string. ⚠️ An out-of-workspace path goes through the ATTACHMENT routes (`POST /api/sessions/:id/attachments` with `notify: false`), never by widening `file-content`/`file-raw` or `file-stream-manager`'s `tail -f` allowlist. ⚠️ `TEXT_ATTACHMENT_EXTENSIONS` IS `EDITABLE_EXTENSIONS` (never a second list), and widening READ must never widen RUN: `html`/`htm`/`svg` stay download-only, other text is inert `text/plain`+`nosniff`. Media extensions are single-sourced in `attachment-registry.ts`. ⚠️ **XLSX previews parse in the BROWSER**, never on the server: `spreadsheet-preview.js` fetches the raw route with `?preview=true` (413 above `MAX_XLSX_BROWSER_PREVIEW_BYTES`, 10 MB) and hands the bytes to `spreadsheet-preview-worker.js`, the only place the pinned `exceljs`/`fflate` vendor bundles load (never on page load). `admitXlsx()` caps the ZIP before ExcelJS runs, counting what ExcelJS will EXPAND as well as what it reads (a merge costs its area, a `` past 16384 is refused, validations are never parsed via `ignoreNodes`, and defined names are never expanded: the worker stubs `_definedNames.model`). ⚠️ The INDEX a row or sheet claims is bounded too, since ExcelJS allocates and walks up to it: a `` outside 1-1048576 and an `xl/workbook.xml` `` above `LIMITS.maxSheetId` (65535) are refused, and `sendTile` reuses the merges read at load (`mergesById`), never `sheet.model`, which rebuilds the whole sheet. ⚠️ The COLUMN index costs the same way (a row's cells live at `_cells[col - 1]`, so one `XFD` cell makes ExcelJS's `eachRow`/`eachCell`/`hasValues` visit 16,384 slots per row): `worksheetMetadata` builds each sheet's row and cell index from the keys that exist (`Object.keys` of `_rows` and `_cells`, skipping falsy and `Null`-type cells as `eachCell({ includeEmpty: false })` does), reads row heights in that pass and merges from `sheet._merges`, and `sendTile` reads cells from that index (`populatedRowsById`); never call ExcelJS's dense `eachRow`/`eachCell` there. `parseThemePalette` returns the default palette for a theme above 64 KB (64 * 1024 characters), since its patterns are quadratic on unclosed tags. ⚠️ Admission keys every entry on the name ExcelJS will SEE (`excelJsEntryName()`: JSZip's `.`/`..`/empty-segment resolution, then one leading `/` stripped), refuses two entries that land on one name, and treats anything matching ExcelJS's UNANCHORED `xl/worksheets/sheet.xml` as a worksheet, so `/xl/...` or `xl/./...` cannot skip a counter, and ExcelJS parses only a STORE-only archive rebuilt from the entries admission inflated (`buildAdmittedArchive()`), never the fetched bytes; cell text goes through `textContent`, formulas are never evaluated. Bumping either package or editing the worker/core changes `SPREADSHEET_ASSET_VERSION`, which `npm run check:public-assets` pins. xls/ods stay download-only. → [architecture-invariants#file-path-links-terminal--response-viewer](docs/architecture-invariants.md#file-path-links-terminal--response-viewer) +**File-path links (terminal + chat)**: a path an agent prints is clickable on BOTH surfaces and opens the file-preview overlay. ⚠️ ONE pattern (`FILE_PATH_LINK_PATTERN` / `absoluteFilePathPattern()` in constants.js) feeds the xterm link provider AND `_linkifyFilePaths()`, a fresh instance per call (`lastIndex`). The chat linkifier walks TEXT NODES with DOM APIs, never rebuilds sanitized markup as a string. ⚠️ An out-of-workspace path goes through the ATTACHMENT routes (`POST /api/sessions/:id/attachments` with `notify: false`), never by widening `file-content`/`file-raw` or `file-stream-manager`'s `tail -f` allowlist. ⚠️ `TEXT_ATTACHMENT_EXTENSIONS` IS `EDITABLE_EXTENSIONS` (never a second list), and widening READ must never widen RUN: `html`/`htm`/`svg` stay download-only, other text is inert `text/plain`+`nosniff`. Media extensions are single-sourced in `attachment-registry.ts`. ⚠️ **XLSX previews parse in the BROWSER**, never on the server: `spreadsheet-preview.js` fetches the raw route with `?preview=true` (413 above `MAX_XLSX_BROWSER_PREVIEW_BYTES`, 10 MB) and hands the bytes to `spreadsheet-preview-worker.js`, the only place the pinned `exceljs`/`fflate` vendor bundles load (never on page load). `admitXlsx()` caps the ZIP before ExcelJS runs, counting what ExcelJS will EXPAND as well as what it reads (a merge costs its area, a `` past 16384 is refused, validations are never parsed via `ignoreNodes`, and defined names are never expanded: the worker stubs `_definedNames.model`). ⚠️ The INDEX a row or sheet claims is bounded too, since ExcelJS allocates and walks up to it: a `` outside 1-1048576 and an `xl/workbook.xml` `` above `LIMITS.maxSheetId` (65535) are refused, and `sendTile` reuses the merges read at load (`mergesById`), never `sheet.model`, which rebuilds the whole sheet. ⚠️ The COLUMN index costs the same way (a row's cells live at `_cells[col - 1]`, so one `XFD` cell makes ExcelJS's `eachRow`/`eachCell`/`hasValues` visit 16,384 slots per row): `worksheetMetadata` builds each sheet's row and cell index from the keys that exist (`Object.keys` of `_rows` and `_cells`, skipping falsy and `Null`-type cells as `eachCell({ includeEmpty: false })` does), reads row heights in that pass and merges from `sheet._merges`, and `sendTile` reads cells from that index (`populatedRowsById`); never call ExcelJS's dense `eachRow`/`eachCell` there. `parseThemePalette` returns the default palette for a theme above 64 KB (64 * 1024 characters), since its patterns are quadratic on unclosed tags. ⚠️ Merges are capped per sheet (`LIMITS.maxMergesPerSheet`, 2,000) AND workbook-wide (`LIMITS.maxMerges`, 10,000), refused as `merge-limit` before ExcelJS loads, since ExcelJS's `_mergeCellsInternal` checks each merge against every earlier one on its sheet (quadratic). Every `` in `xl/styles.xml` is read with `readTagAttributes()` and refused (`number-format`) when its decoded `formatCode` is over 255 characters or has a `[` after its last `]`: ExcelJS's `isDateFmt` rescans to the end of the code per unclosed `[` once per numeric cell, and the code is echoed into the notice bar. ⚠️ `formatCellValue` caps every cell's display text at `LIMITS.maxCellTextChars` (1,000, ellipsis, never splitting a surrogate pair) for every value shape (rich text, hyperlink text, formula source, errors), since structured clone copies each tile cell's whole string to the page and the load timeout no longer covers tiles. ⚠️ Admission keys every entry on the name ExcelJS will SEE (`excelJsEntryName()`: JSZip's `.`/`..`/empty-segment resolution, then one leading `/` stripped), refuses two entries that land on one name, and treats anything matching ExcelJS's UNANCHORED `xl/worksheets/sheet.xml` as a worksheet, so `/xl/...` or `xl/./...` cannot skip a counter, and ExcelJS parses only a STORE-only archive rebuilt from the entries admission inflated (`buildAdmittedArchive()`), never the fetched bytes; cell text goes through `textContent`, formulas are never evaluated. Bumping either package or editing the worker/core changes `SPREADSHEET_ASSET_VERSION`, which `npm run check:public-assets` pins. xls/ods stay download-only. → [architecture-invariants#file-path-links-terminal--response-viewer](docs/architecture-invariants.md#file-path-links-terminal--response-viewer) **Filesystem path picker** (Link Existing "Browse" + the mobile keyboard's `📁 Path` key): lazy one-directory browsing via `GET /api/filesystem/browse`, with `GET /api/filesystem/preview` for the tapped file. Inserts the path **without** Enter, so the prompt is never submitted; the sibling `⌫ All` key clears only the unsent prompt and must never send the agent's `/clear`. ⚠️ This is a **second file-serving surface and inherits neither the attachment confinement nor its ownership scoping** — it allowlists Home, `CASES_DIR`, `/mnt/d` and `CODEMAN_FILE_PICKER_ROOTS`, blocks sensitive trees, and rejects symlink escapes **after** `realpath`. ⚠️ The optional `sessionId` is an ownership boundary that must be `canAccessOwned`-checked by hand (it does not go through `findSessionOrFail`), and in multi-user mode a non-admin gets only their own `userSpacePath` as a root: per-user spaces live INSIDE `homedir()`, so a `Home` root exposes every other user's workspace. Previews go through the same global conversion limiter, and Markdown/TXT/JSON are served as inert `text/plain`. → [architecture-invariants#filesystem-path-picker](docs/architecture-invariants.md#filesystem-path-picker) diff --git a/docs/architecture-invariants.md b/docs/architecture-invariants.md index 327c9783..6fd2d06f 100644 --- a/docs/architecture-invariants.md +++ b/docs/architecture-invariants.md @@ -371,7 +371,7 @@ A file path an agent prints is a link on both surfaces it can appear on, and cli ⚠️ **The preview overlay must outrank the panel that launched it.** `.file-preview-overlay` sits at `z-index: 5100`, above the response viewer (5000) and its backdrop (4999); at its historical 2000 a path clicked in the chat opened the overlay *behind* the chat, which reads as a dead link. It stays below the toast/picker band (10000+) so a "Saved" toast still lands on top. -**XLSX previews parse in the browser worker, and ExcelJS only ever sees what admission checked.** An `.xlsx` joins the raw routes like any other file (`?preview=true` caps it at 10 MB with a 413); the server never parses it. `spreadsheet-preview.js` hands the bytes to `spreadsheet-preview-worker.js`, the ONLY place the pinned `exceljs`/`fflate` vendor bundles load: fflate and `spreadsheet-xlsx-core.js` at worker start, ExcelJS only after `admitXlsx()` passes, and the page itself never loads either. `admitXlsx()` streams every entry through fflate and enforces the entry count, per-entry and total inflated bytes (64 MB), compression ratio, worksheet, cell, merge and style caps on the actual inflated output, refusing (never truncating) a workbook that trips one. ⚠️ Admission walks LOCAL headers while ExcelJS (JSZip) reads the CENTRAL directory, so overlapping entries (a stored entry hiding a whole `sheet1.xml`, a one-cell decoy later in the stream) once let a file be admitted as 1 cell and parsed as 300k. The worker therefore never hands ExcelJS the fetched bytes: it gets `buildAdmittedArchive()`, a STORE-only `fflate.zipSync(entries, { level: 0 })` of exactly the entries admission inflated, and a name streamed twice is refused. That costs one transient copy of the inflated entries (bounded by the same 64 MB cap) and is pinned by the overlapping-entry fixture in `test/spreadsheet-preview-worker.test.ts`. ⚠️ Admission also has to bound what ExcelJS EXPANDS, not just what it reads: ExcelJS 4.4.0 builds one object per covered cell of a ``, per address of a `` and per index up to ``, so a 6.5 KB file admitted as one cell once cost a gigabyte. The counter therefore charges each merge its full AREA against the cell caps (an unparseable `ref` is refused), refuses a `` whose `min`/`max` is past 16384, and the worker loads with `ignoreNodes: ['dataValidations']` (the preview never shows validations). Defined names expand the same way and live in `xl/workbook.xml`, where admission reads only the `` ids: ExcelJS's `DefinedNames` model setter builds one object per cell of every range (a whole-sheet name exhausted a 4 GB heap), so right after `new ExcelJS.Workbook()` the worker replaces `_definedNames.model` with an `Object.defineProperty` stub whose getter returns `[]` and whose setter drops the value. Print areas and titles are split off in the workbook xform's reconcile, before that setter runs, so they are unaffected, and `defineProperty` throws if a future ExcelJS renames `_definedNames` rather than silently expanding again. ⚠️ Every name admission uses is the one ExcelJS will SEE, not the stored one: JSZip resolves each entry name on load (`utils.resolve`: `.` and empty middle segments dropped, `..` pops a segment) and ExcelJS then strips one leading `/` and matches worksheets with an UNANCHORED `xl/worksheets/sheet.xml`. Checking the stored name once let `/xl/worksheets/sheet1.xml`, `xl/./...`, `xl//...`, `xl/xl/worksheets/sheet1.xml` and `xl/worksheets/sheet1.xml.x` skip every counter. `excelJsEntryName()` mirrors both steps; admission refuses two entries that resolve to one name, keys `inflatedEntries` (so the rebuilt archive) on the resolved name, picks the worksheet and styles counters from it, and counts any name matching ExcelJS's unanchored worksheet pattern as a worksheet. The local-versus-central consistency checks still compare the stored names. The counter reads ``/`` attributes IN ORDER from the tag name with a sticky regex that consumes each quoted value whole (a raw `>` or the other quote character is legal inside a value, so a first-match search could be fed a fake `ref`/`max`), and refuses a tag whose attributes do not parse up to `>` or repeat a name. It counts every `` (ExcelJS keeps one Row object per element, cells or not) against per-sheet and total row caps, and reads each ``'s attributes the same way. ⚠️ The INDEX a row or sheet claims is a cost too, not just how many there are: ExcelJS stores a row at `_rows[r - 1]` and walks `_rows` up to the largest index on every `eachRow` and `sheet.model` (five rows plus one `` took 5 s to load and 1.7 s per tile), and stores a sheet at `_worksheets[sheetId]`, which its `worksheets` getter slices and sorts (`sheetId="30000000"` on a one-cell workbook took 1.6 s and 557 MB). So a `` that is not plain digits in 1 to 1,048,576 is refused (an absent `r` is fine), and a counter for the resolved `xl/workbook.xml` reads every `` tag in order and refuses one that does not parse or whose `sheetId` is not plain digits up to `LIMITS.maxSheetId` (65535; real ids are small, and the `(?=[\s/>])` lookahead keeps the `` container out). The counter also counts every `` in styles.xml with no list-tracking state (a `` inside a comment used to desync it). It carries everything from the last `<` into the next inflate chunk, since `<` can never appear inside an attribute value, and a central-directory `compressedSize` that runs past the file is refused, since the ratio cap divides by it. Fixtures for each are in `test/spreadsheet-preview-worker.test.ts` and `test/spreadsheet-xlsx-core.test.ts`. ⚠️ `sendTile` must never read `sheet.model`, which rebuilds every row and cell model (10-19 ms per tile on a 100k-cell sheet, once per animation frame while scrolling): each sheet's merges are read once in `worksheetMetadata` and kept in `mergesById` next to `populatedRowsById` (replaced on load, cleared on dispose), and a worker test makes the `model` getter throw before requesting a tile. ⚠️ The COLUMN index a cell claims costs the same way: ExcelJS keeps a row's cells at `_cells[col - 1]`, so a row whose only cell sits in `XFD` is a dictionary-mode array, and ExcelJS's `row.eachCell` (`_cells.forEach`) and `row.hasValues` (which `sheet.eachRow` calls) visit every index up to 16,384, about 0.5 ms per row (10,000 such rows took 13 s to load in Chromium, and each tile paid it again; `XFD` is a legal column, so admission cannot refuse it). `worksheetMetadata` therefore never calls ExcelJS's dense `eachRow`/`eachCell`: `populatedRowIndex()` walks `Object.keys(sheet._rows)` and, per row, `Object.keys(row._cells)`, sorted numerically, skipping falsy and `ValueType.Null` cells exactly as `eachCell({ includeEmpty: false })` does and keeping a row only when it holds one such cell (ExcelJS's `hasValues`). Styles, the extent and row heights come from that one pass, merges from `sheet._merges` (the map the `model` getter itself lists), and the per-row cell lists are what `populatedRowsById` keeps, so `sendTile` reads a row's cells from the index instead of `row.eachCell`. Worker tests make `eachRow`, `eachCell`, `hasValues` and the `model` getter throw across load and a tile, and load and tile 3,000 one-`XFD`-cell rows. `parseThemePalette` returns the default palette for a theme above 64 KB (64 * 1024 characters of the decoded XML; real themes are under 10 KB): its `clrScheme` and slot patterns rescan to the end of the text for every unclosed opening tag, and a 1 MB theme of repeated `` took 21.8 s in Chromium. Number formats cap decimals at 30, as Excel does, since `toLocaleString` throws above 100. The worker is a stable URL cache-busted by `SPREADSHEET_ASSET_VERSION` in `spreadsheet-preview.js`, the content hash of the worker, the core and both vendor bundles; `npm run check:public-assets` fails when it drifts, so editing any of those (or bumping either package) means updating the token, or a deploy pairs a new worker with a year-cached old core. +**XLSX previews parse in the browser worker, and ExcelJS only ever sees what admission checked.** An `.xlsx` joins the raw routes like any other file (`?preview=true` caps it at 10 MB with a 413); the server never parses it. `spreadsheet-preview.js` hands the bytes to `spreadsheet-preview-worker.js`, the ONLY place the pinned `exceljs`/`fflate` vendor bundles load: fflate and `spreadsheet-xlsx-core.js` at worker start, ExcelJS only after `admitXlsx()` passes, and the page itself never loads either. `admitXlsx()` streams every entry through fflate and enforces the entry count, per-entry and total inflated bytes (64 MB), compression ratio, worksheet, cell, merge and style caps on the actual inflated output, refusing (never truncating) a workbook that trips one. ⚠️ Admission walks LOCAL headers while ExcelJS (JSZip) reads the CENTRAL directory, so overlapping entries (a stored entry hiding a whole `sheet1.xml`, a one-cell decoy later in the stream) once let a file be admitted as 1 cell and parsed as 300k. The worker therefore never hands ExcelJS the fetched bytes: it gets `buildAdmittedArchive()`, a STORE-only `fflate.zipSync(entries, { level: 0 })` of exactly the entries admission inflated, and a name streamed twice is refused. That costs one transient copy of the inflated entries (bounded by the same 64 MB cap) and is pinned by the overlapping-entry fixture in `test/spreadsheet-preview-worker.test.ts`. ⚠️ Admission also has to bound what ExcelJS EXPANDS, not just what it reads: ExcelJS 4.4.0 builds one object per covered cell of a ``, per address of a `` and per index up to ``, so a 6.5 KB file admitted as one cell once cost a gigabyte. The counter therefore charges each merge its full AREA against the cell caps (an unparseable `ref` is refused), refuses a `` whose `min`/`max` is past 16384, and the worker loads with `ignoreNodes: ['dataValidations']` (the preview never shows validations). Defined names expand the same way and live in `xl/workbook.xml`, where admission reads only the `` ids: ExcelJS's `DefinedNames` model setter builds one object per cell of every range (a whole-sheet name exhausted a 4 GB heap), so right after `new ExcelJS.Workbook()` the worker replaces `_definedNames.model` with an `Object.defineProperty` stub whose getter returns `[]` and whose setter drops the value. Print areas and titles are split off in the workbook xform's reconcile, before that setter runs, so they are unaffected, and `defineProperty` throws if a future ExcelJS renames `_definedNames` rather than silently expanding again. ⚠️ Every name admission uses is the one ExcelJS will SEE, not the stored one: JSZip resolves each entry name on load (`utils.resolve`: `.` and empty middle segments dropped, `..` pops a segment) and ExcelJS then strips one leading `/` and matches worksheets with an UNANCHORED `xl/worksheets/sheet.xml`. Checking the stored name once let `/xl/worksheets/sheet1.xml`, `xl/./...`, `xl//...`, `xl/xl/worksheets/sheet1.xml` and `xl/worksheets/sheet1.xml.x` skip every counter. `excelJsEntryName()` mirrors both steps; admission refuses two entries that resolve to one name, keys `inflatedEntries` (so the rebuilt archive) on the resolved name, picks the worksheet and styles counters from it, and counts any name matching ExcelJS's unanchored worksheet pattern as a worksheet. The local-versus-central consistency checks still compare the stored names. The counter reads ``/`` attributes IN ORDER from the tag name with a sticky regex that consumes each quoted value whole (a raw `>` or the other quote character is legal inside a value, so a first-match search could be fed a fake `ref`/`max`), and refuses a tag whose attributes do not parse up to `>` or repeat a name. It counts every `` (ExcelJS keeps one Row object per element, cells or not) against per-sheet and total row caps, and reads each ``'s attributes the same way. ⚠️ The INDEX a row or sheet claims is a cost too, not just how many there are: ExcelJS stores a row at `_rows[r - 1]` and walks `_rows` up to the largest index on every `eachRow` and `sheet.model` (five rows plus one `` took 5 s to load and 1.7 s per tile), and stores a sheet at `_worksheets[sheetId]`, which its `worksheets` getter slices and sorts (`sheetId="30000000"` on a one-cell workbook took 1.6 s and 557 MB). So a `` that is not plain digits in 1 to 1,048,576 is refused (an absent `r` is fine), and a counter for the resolved `xl/workbook.xml` reads every `` tag in order and refuses one that does not parse or whose `sheetId` is not plain digits up to `LIMITS.maxSheetId` (65535; real ids are small, and the `(?=[\s/>])` lookahead keeps the `` container out). The counter also counts every `` in styles.xml with no list-tracking state (a `` inside a comment used to desync it). It carries everything from the last `<` into the next inflate chunk, since `<` can never appear inside an attribute value, and a central-directory `compressedSize` that runs past the file is refused, since the ratio cap divides by it. Fixtures for each are in `test/spreadsheet-preview-worker.test.ts` and `test/spreadsheet-xlsx-core.test.ts`. ⚠️ `sendTile` must never read `sheet.model`, which rebuilds every row and cell model (10-19 ms per tile on a 100k-cell sheet, once per animation frame while scrolling): each sheet's merges are read once in `worksheetMetadata` and kept in `mergesById` next to `populatedRowsById` (replaced on load, cleared on dispose), and a worker test makes the `model` getter throw before requesting a tile. ⚠️ The COLUMN index a cell claims costs the same way: ExcelJS keeps a row's cells at `_cells[col - 1]`, so a row whose only cell sits in `XFD` is a dictionary-mode array, and ExcelJS's `row.eachCell` (`_cells.forEach`) and `row.hasValues` (which `sheet.eachRow` calls) visit every index up to 16,384, about 0.5 ms per row (10,000 such rows took 13 s to load in Chromium, and each tile paid it again; `XFD` is a legal column, so admission cannot refuse it). `worksheetMetadata` therefore never calls ExcelJS's dense `eachRow`/`eachCell`: `populatedRowIndex()` walks `Object.keys(sheet._rows)` and, per row, `Object.keys(row._cells)`, sorted numerically, skipping falsy and `ValueType.Null` cells exactly as `eachCell({ includeEmpty: false })` does and keeping a row only when it holds one such cell (ExcelJS's `hasValues`). Styles, the extent and row heights come from that one pass, merges from `sheet._merges` (the map the `model` getter itself lists), and the per-row cell lists are what `populatedRowsById` keeps, so `sendTile` reads a row's cells from the index instead of `row.eachCell`. Worker tests make `eachRow`, `eachCell`, `hasValues` and the `model` getter throw across load and a tile, and load and tile 3,000 one-`XFD`-cell rows. `parseThemePalette` returns the default palette for a theme above 64 KB (64 * 1024 characters of the decoded XML; real themes are under 10 KB): its `clrScheme` and slot patterns rescan to the end of the text for every unclosed opening tag, and a 1 MB theme of repeated `` took 21.8 s in Chromium. Number formats cap decimals at 30, as Excel does, since `toLocaleString` throws above 100. ⚠️ Merges are also capped per sheet (`LIMITS.maxMergesPerSheet`, 2,000) and across the workbook (`LIMITS.maxMerges`, 10,000), both refused as `merge-limit` before ExcelJS loads: on top of the area each merge costs, ExcelJS's `Worksheet._mergeCellsInternal` checks every new merge against every earlier one on its sheet (`_.each(this._merges)`, rebuilding `Object.keys` per call), so a sheet's cost is quadratic in its merge count (one sheet at the old 5,000 cap took 1.6 s in the worker harness, and 20 such sheets ran past the 20 s load timeout in Chromium; the new caps keep the worst case near 1.3 s). ⚠️ The styles counter reads every `` (cellXfs' and dxfs' alike; the lookahead skips ``) with `readTagAttributes()` and refuses, as `number-format`, a `formatCode` that is over 255 characters (Excel's own limit) or has a `[` after its last `]`, checked on the value as ExcelJS's XML parser hands it over (entities and character references decoded, so `[` cannot hide a `[`; ExcelJS's own backslash unescape only removes characters, so it cannot reopen a bracket). ExcelJS runs `utils.isDateFmt` once per numeric cell at load, and its `fmt.replace(/\[[^\]]*]/g, '')` rescans to the end of the code for every `[` with no later `]` (a 60,000-character code of `[` cost 2.2 s per numeric cell; even 255 unclosed characters add up at the cell cap), while a closed code is linear. The length cap also bounds the `Unsupported number format: ` notice, and the refusal messages never echo the code. ⚠️ Cell text reaching the page is bounded in the worker: `formatCellValue` caps every display string at `LIMITS.maxCellTextChars` (1,000; the last kept character is an ellipsis and a cut never splits a surrogate pair) for every value shape, strings, rich text (runs are joined only until the cap is passed), hyperlink text, formula source and results, and errors. Without it each tile cell carried its whole string and structured clone copied it once per cell, after the load timeout had already been cleared by the metadata reply: one 1 MB shared string over a 60 x 20 block left the page unresponsive for 150 s at 9.6 GB. A cell is one `nowrap` line with an ellipsis, so nothing past the column width was ever shown. A worker test asserts every tile cell's `text` is within the cap for a long shared string, rich text and a hyperlink. The worker is a stable URL cache-busted by `SPREADSHEET_ASSET_VERSION` in `spreadsheet-preview.js`, the content hash of the worker, the core and both vendor bundles; `npm run check:public-assets` fails when it drifts, so editing any of those (or bumping either package) means updating the token, or a deploy pairs a new worker with a year-cached old core. ### Filesystem path picker diff --git a/src/web/public/spreadsheet-preview.js b/src/web/public/spreadsheet-preview.js index ff630d83..eee04d92 100644 --- a/src/web/public/spreadsheet-preview.js +++ b/src/web/public/spreadsheet-preview.js @@ -20,7 +20,7 @@ (function initSpreadsheetPreview(global) { 'use strict'; - const SPREADSHEET_ASSET_VERSION = '6e843e269018'; + const SPREADSHEET_ASSET_VERSION = 'd83f632d5c69'; const MAX_PREVIEW_BYTES = 10 * 1024 * 1024; const DEFAULT_TIMEOUT_MS = 20000; const MAX_SCROLL_PX = 8000000; diff --git a/src/web/public/spreadsheet-xlsx-core.js b/src/web/public/spreadsheet-xlsx-core.js index fb71f1da..15c2d994 100644 --- a/src/web/public/spreadsheet-xlsx-core.js +++ b/src/web/public/spreadsheet-xlsx-core.js @@ -27,12 +27,24 @@ maxCellsPerSheet: 100000, maxRows: 250000, maxRowsPerSheet: 100000, - maxMergesPerSheet: 5000, + // ExcelJS's `_mergeCellsInternal` checks each new merge against every + // earlier one on its sheet, so a sheet costs the SQUARE of its merge count + // (5,000 on one sheet took 1.6 s). Both caps keep the worst case near 1.3 s. + maxMergesPerSheet: 2000, + maxMerges: 10000, maxStyles: 5000, // ExcelJS stores each sheet at `_worksheets[sheetId]`, so the id is an array // length. Excel numbers sheets from 1 and never reuses an id, so real ids // stay small; 65535 leaves room for heavy editing at a negligible cost. maxSheetId: 65535, + // Excel's own limit on a number format. ExcelJS runs `isDateFmt` on the code + // once per numeric cell, and the code is echoed into the notice bar. + maxNumFmtChars: 255, + // Display text per cell. A cell is one `nowrap` line ending in an ellipsis, + // so nothing past the column width shows; uncapped, every tile cell carried + // its whole string to the page (a 1 MB shared string over a 60 x 20 block + // froze the page's main thread at 9.6 GB). + maxCellTextChars: 1000, }); const MAX_ROW = 1048576; const MAX_COL = 16384; @@ -240,6 +252,35 @@ } } + // Attribute values as the XML parser inside ExcelJS (saxes) hands them over: + // the predefined entities and character references decoded. Checking the raw + // text would let `[` hide a `[` and would overcount `"`. + function decodeXmlAttribute(value) { + return value.replace(/&(?:#x([0-9a-fA-F]+)|#([0-9]+)|(amp|lt|gt|quot|apos));/g, (entity, hex, dec, name) => { + if (name) return { amp: '&', lt: '<', gt: '>', quot: '"', apos: "'" }[name]; + const code = hex ? Number.parseInt(hex, 16) : Number(dec); + return code <= 0x10ffff ? String.fromCodePoint(code) : entity; + }); + } + + // ExcelJS's `isDateFmt` strips `/\[[^\]]*]/g` from the code once per numeric + // cell, and that pattern rescans to the end of the code for every `[` with no + // later `]` (a 60,000-character code of `[` cost 2.2 s a cell; even 255 + // characters added up at the cell cap). Once nothing follows the last `]` + // but text without `[`, each `[` stops at the next `]` and the scan is linear. + // The messages never echo the code. + function checkNumberFormat(attributes, limits) { + const raw = attributes.get('formatCode'); + if (raw === undefined) return; + const code = decodeXmlAttribute(raw); + if (code.length > limits.maxNumFmtChars) { + fail('number-format', `Workbook has a number format longer than ${limits.maxNumFmtChars} characters`); + } + if (code.lastIndexOf('[') > code.lastIndexOf(']')) { + fail('number-format', 'Workbook has a number format with an unclosed bracket'); + } + } + function createXmlCounter(name, counts, limits) { let tail = ''; const decoder = new TextDecoder(); @@ -286,8 +327,8 @@ } sheetMerges += 1; counts.merges += 1; - if (sheetMerges > limits.maxMergesPerSheet) - fail('merge-limit', 'Worksheet exceeds the merged ranges limit'); + if (sheetMerges > limits.maxMergesPerSheet || counts.merges > limits.maxMerges) + fail('merge-limit', 'Workbook exceeds the merged ranges limit'); addCells(mergeArea(attributes)); } } @@ -304,6 +345,12 @@ // sits in can be desynced by a closing tag inside an XML comment. counts.styles += (scan.match(/])/g) || []).length; if (counts.styles > limits.maxStyles) fail('style-limit', 'Workbook exceeds the cell styles limit'); + // Every , cellXfs' and dxfs' alike; the lookahead skips . + for (const match of scan.matchAll(/])/g)) { + const attributes = readTagAttributes(scan, match.index + match[0].length); + if (!attributes) fail('malformed', 'Workbook has a whose attributes do not parse'); + checkNumberFormat(attributes, limits); + } } tail = text.slice(safeEnd); if (tail.length > MAX_CARRIED_TAG) fail('malformed', 'Workbook XML has an oversized tag'); @@ -529,33 +576,68 @@ return 'formula' in value || 'sharedFormula' in value; } + // Cut to LIMITS.maxCellTextChars, ending in an ellipsis, never between the + // halves of a surrogate pair. + function capCellText(text) { + const max = LIMITS.maxCellTextChars; + if (text.length <= max) return text; + let end = max - 1; + const last = text.charCodeAt(end - 1); + if (last >= 0xd800 && last <= 0xdbff) end -= 1; + return `${text.slice(0, end)}…`; + } + + // Joins rich-text runs only until the cap is passed, so a long run is never + // copied whole once per cell. + function richTextPrefix(runs) { + let text = ''; + for (const run of runs) { + if (text.length > LIMITS.maxCellTextChars) break; + if (typeof run?.text === 'string') text += run.text.slice(0, LIMITS.maxCellTextChars + 1); + } + return text; + } + + /** + * A cell's display text and any warning. The text is ALWAYS at most + * `LIMITS.maxCellTextChars` characters, whatever shape the value has. + */ + function formatCellValue(value, format, date1904) { + const formatted = formatCellValueUncapped(value, format, date1904); + return formatted.text.length > LIMITS.maxCellTextChars + ? Object.assign({}, formatted, { text: capCellText(formatted.text) }) + : formatted; + } + // Every non-scalar shape ExcelJS loads a cell value as. Anything not handled // here would otherwise reach String() and render as "[object Object]". - function formatCellValue(value, format, date1904) { + function formatCellValueUncapped(value, format, date1904) { if (value === null || value === undefined) return { text: '' }; if (isDateValue(value)) { const code = String(format || 'General'); const known = DATE_FORMAT.test(code) || TIME_FORMAT.test(code) || DATE_TIME_FORMAT.test(code); const serial = dateToSerial(value, date1904); - if (known) return formatCellValue(serial, code, date1904); + if (known) return formatCellValueUncapped(serial, code, date1904); const fallback = serial % 1 === 0 ? 'yyyy-mm-dd' : 'yyyy-mm-dd hh:mm'; - const formatted = formatCellValue(serial, fallback, date1904); + const formatted = formatCellValueUncapped(serial, fallback, date1904); return /^General$/i.test(code) ? formatted : { text: formatted.text, warning: `Unsupported number format: ${code}` }; } if (typeof value === 'object') { if (isFormulaValue(value)) { - if (value.result !== undefined && value.result !== null) return formatCellValue(value.result, format, date1904); + if (value.result !== undefined && value.result !== null) { + return formatCellValueUncapped(value.result, format, date1904); + } const source = typeof value.formula === 'string' ? `=${value.formula}` : ''; return { text: source, warning: 'Formula has no cached result' }; } if (typeof value.error === 'string') return { text: value.error }; if (Array.isArray(value.richText)) { - return { text: value.richText.map((run) => (typeof run?.text === 'string' ? run.text : '')).join('') }; + return { text: richTextPrefix(value.richText) }; } // Hyperlink: the display text, never the target. The text may be rich. - if ('text' in value) return formatCellValue(value.text, 'General', date1904); + if ('text' in value) return formatCellValueUncapped(value.text, 'General', date1904); return { text: '', warning: 'Unsupported cell value' }; } const code = String(format || 'General'); diff --git a/test/spreadsheet-preview-worker.test.ts b/test/spreadsheet-preview-worker.test.ts index 1c89f1ea..cbf3f9f0 100644 --- a/test/spreadsheet-preview-worker.test.ts +++ b/test/spreadsheet-preview-worker.test.ts @@ -856,3 +856,133 @@ describe('spreadsheet preview worker: rows and cells are indexed by their presen } }, 60_000); }); + +/** Deterministic, poorly compressible text, so the ratio cap does not refuse it first. */ +function noisyText(length: number, seed = 1): string { + let state = seed; + let text = ''; + for (let i = 0; i < length; i += 1) { + state = (state * 1103515245 + 12345) & 0x7fffffff; + text += String.fromCharCode(97 + (state % 26)); + } + return text; +} + +describe('spreadsheet preview worker: cell text reaching the page is bounded', () => { + // Every tile cell carried its whole string and structured clone copied it + // once per cell: one 1 MB shared string over a 60 x 20 block froze the page. + it('caps every tile cell text, shared string, rich text and hyperlink alike', async () => { + const long = noisyText(200_000); + const workbook = new ExcelJS.Workbook(); + const sheet = workbook.addWorksheet('Long'); + for (let row = 1; row <= 20; row += 1) { + for (let col = 1; col <= 10; col += 1) sheet.getCell(row, col).value = long; + } + sheet.getCell(21, 1).value = { richText: [{ text: long }, { font: { bold: true }, text: long }] }; + sheet.getCell(21, 2).value = { text: long, hyperlink: 'https://example.invalid/' }; + sheet.getCell(21, 3).value = 'short'; + const bytes = await writeWorkbook(workbook); + const entries = fflate.unzipSync(new Uint8Array(bytes)); + // One shared string (index 0), referenced by every cell of the block. + expect(fflate.strFromU8(entries['xl/sharedStrings.xml'])).toContain(`${long}`); + expect( + fflate.strFromU8(entries['xl/worksheets/sheet1.xml']).match(/t="s">0<\/v>/g)?.length + ).toBeGreaterThanOrEqual(200); + + const harness = createHarness(); + const metadata = await loadMetadata(harness, bytes); + const tile = await requestTile(harness, metadata.sheets[0].id, { r1: 1, c1: 1, r2: 21, c2: 10 }); + expect(tile.type).toBe('tile'); + const cells = tile.cells as Array<{ row: number; col: number; text: string }>; + expect(cells).toHaveLength(203); + for (const cell of cells) expect(cell.text.length, `${cell.row}:${cell.col}`).toBeLessThanOrEqual(1000); + expect(cells.find((cell) => cell.row === 1 && cell.col === 1)?.text).toBe(`${long.slice(0, 999)}…`); + expect(cells.find((cell) => cell.row === 21 && cell.col === 1)?.text.length).toBe(1000); + expect(cells.find((cell) => cell.row === 21 && cell.col === 2)?.text.length).toBe(1000); + expect(cells.find((cell) => cell.row === 21 && cell.col === 3)?.text).toBe('short'); + }, 60_000); +}); + +/** A workbook of `sheets` one-cell sheets, each carrying `mergesPerSheet` one-row merges. */ +async function mergeHeavyWorkbook(sheets: number, mergesPerSheet: number): Promise { + const workbook = new ExcelJS.Workbook(); + for (let i = 1; i <= sheets; i += 1) workbook.addWorksheet(`S${i}`).getCell('A1').value = 'one'; + const entries = fflate.unzipSync(new Uint8Array(await workbook.xlsx.writeBuffer())); + const merges = Array.from({ length: mergesPerSheet }, (_, i) => ``).join(''); + for (let i = 1; i <= sheets; i += 1) { + const name = `xl/worksheets/sheet${i}.xml`; + const sheet = fflate.strFromU8(entries[name]); + const patched = sheet.replace( + '', + `${merges}` + ); + expect(patched).not.toBe(sheet); + entries[name] = fflate.strToU8(patched); + } + return toArrayBuffer(fflate.zipSync(entries)); +} + +describe('spreadsheet preview worker: merges are bounded per sheet and workbook-wide', () => { + // ExcelJS checks each new merge against every earlier one on its sheet, so + // a sheet's load cost grows with the square of its merge count. + it('refuses one sheet over the per-sheet merge cap before ExcelJS loads', async () => { + const harness = createHarness(); + await harness.send({ type: 'load', bytes: await mergeHeavyWorkbook(1, 2_001) }); + expect(harness.messages.at(-1)).toMatchObject({ type: 'error', code: 'merge-limit' }); + expect(harness.imports.some((url) => url.includes('exceljs'))).toBe(false); + }, 60_000); + + it('refuses sheets each under the per-sheet cap once the workbook total passes 10,000', async () => { + const harness = createHarness(); + await harness.send({ type: 'load', bytes: await mergeHeavyWorkbook(6, 2_000) }); + expect(harness.messages.at(-1)).toMatchObject({ type: 'error', code: 'merge-limit' }); + expect(harness.imports.some((url) => url.includes('exceljs'))).toBe(false); + }, 60_000); + + it('still previews a sheet at the per-sheet merge cap', async () => { + const harness = createHarness(); + const metadata = await loadMetadata(harness, await mergeHeavyWorkbook(1, 2_000)); + expect(metadata.sheets[0].merges).toHaveLength(2_000); + }, 60_000); +}); + +/** A one-cell numeric workbook whose styles.xml declares `formatCode` for the cell. */ +async function numberFormatWorkbook(formatCode: string): Promise { + const workbook = new ExcelJS.Workbook(); + const sheet = workbook.addWorksheet('Data'); + sheet.getCell('A1').value = 1.5; + sheet.getCell('A1').numFmt = '0.000'; + const entries = fflate.unzipSync(new Uint8Array(await workbook.xlsx.writeBuffer())); + const styles = fflate.strFromU8(entries['xl/styles.xml']); + const patched = styles.replace('formatCode="0.000"', `formatCode="${formatCode}"`); + expect(patched).not.toBe(styles); + entries['xl/styles.xml'] = fflate.strToU8(patched); + return toArrayBuffer(fflate.zipSync(entries)); +} + +describe('spreadsheet preview worker: number formats ExcelJS would rescan', () => { + // `isDateFmt` runs `/\[[^\]]*]/g` per numeric cell; a 60,000-character code of + // `[` cost 2.2 s a cell, and the code is echoed into the notice bar. + it.each([ + ['an unclosed [', `0${'['.repeat(200)}`], + ['a code over 255 characters', `${'0'.repeat(300)}.00`], + ])( + 'refuses %s before ExcelJS loads', + async (_label, code) => { + const harness = createHarness(); + await harness.send({ type: 'load', bytes: await numberFormatWorkbook(code) }); + expect(harness.messages.at(-1)).toMatchObject({ type: 'error', code: 'number-format' }); + expect(harness.imports.some((url) => url.includes('exceljs'))).toBe(false); + }, + 60_000 + ); + + it('previews a closed bracketed format and keeps its warning bounded', async () => { + const code = `[Red]${'0'.repeat(200)}`; + const harness = createHarness(); + const metadata = await loadMetadata(harness, await numberFormatWorkbook(code)); + const tile = await requestTile(harness, metadata.sheets[0].id, { r1: 1, c1: 1, r2: 1, c2: 1 }); + expect(tile.cells).toEqual([expect.objectContaining({ row: 1, col: 1, text: '1.5' })]); + expect(tile.warnings).toEqual([`Unsupported number format: ${code}`]); + }, 60_000); +}); diff --git a/test/spreadsheet-xlsx-core.test.ts b/test/spreadsheet-xlsx-core.test.ts index c33af908..427a1f70 100644 --- a/test/spreadsheet-xlsx-core.test.ts +++ b/test/spreadsheet-xlsx-core.test.ts @@ -333,6 +333,101 @@ describe('spreadsheet XLSX core', () => { expect(() => counter.push(fflate.strToU8(xml), true)).toThrowError(/styles limit/i); }); + // ExcelJS's `_mergeCellsInternal` checks every new merge against every earlier + // one on its sheet, so a sheet costs the SQUARE of its merge count: one sheet + // at the old 5,000 cap took 1.6 s, and 20 such sheets ran past the page timeout. + it('caps merges per sheet and across the workbook', () => { + expect(core.LIMITS.maxMergesPerSheet).toBe(2000); + expect(core.LIMITS.maxMerges).toBe(10000); + const merges = (n: number, row = 1) => + Array.from({ length: n }, (_, i) => ``).join(''); + const sheetXml = (n: number) => `${merges(n)}`; + expect(() => core.admitXlsx(workbookZip(sheetXml(4)), fflate, { maxMergesPerSheet: 3 })).toThrowError( + /merged ranges limit/i + ); + expect(() => core.admitXlsx(workbookZip(sheetXml(3)), fflate, { maxMergesPerSheet: 3 })).not.toThrow(); + // Two sheets each under the per-sheet cap still trip the workbook-wide one. + const twoSheets = fflate.zipSync({ + ...fflate.unzipSync(workbookZip(sheetXml(3))), + 'xl/worksheets/sheet2.xml': fflate.strToU8(sheetXml(3)), + }); + expect(() => core.admitXlsx(twoSheets, fflate, { maxMergesPerSheet: 3, maxMerges: 5 })).toThrowError( + /merged ranges limit/i + ); + expect(() => core.admitXlsx(twoSheets, fflate, { maxMergesPerSheet: 3, maxMerges: 6 })).not.toThrow(); + expect(() => core.admitXlsx(twoSheets, fflate, { maxMergesPerSheet: 3, maxMerges: 5 })).toThrowError( + expect.objectContaining({ code: 'merge-limit' }) + ); + }); + + // ExcelJS runs `isDateFmt` once per numeric cell, and its first step, + // `fmt.replace(/\[[^\]]*]/g, '')`, rescans to the end of the code for every + // `[` with no later `]`. The code is also echoed into the notice bar. + it('refuses a formatCode over 255 characters or with a [ after its last ]', () => { + const styles = (numFmts: string) => { + const counts = { cells: 0, merges: 0, styles: 0, rows: 0 }; + const counter = core.createXmlCounter('xl/styles.xml', counts as never, core.LIMITS); + const xml = `${numFmts}`; + return () => counter.push(fflate.strToU8(xml), true); + }; + const fmt = (code: string) => ``; + expect(styles(fmt('0'.repeat(256)))).toThrowError(/longer than 255/i); + expect(styles(fmt('0'.repeat(255)))).not.toThrow(); + for (const code of ['[', '0[', '[Red]0[', '[[[[', '[Red]0.00;[']) { + expect(styles(fmt(code)), code).toThrowError(/unclosed bracket/i); + } + for (const code of ['[Red]0.00', '[$-409]mmm d, yyyy', '[h]:mm:ss', '0.00', '#,##0;[Red]-#,##0', ']', '[[]']) { + expect(styles(fmt(code)), code).not.toThrow(); + } + // The code is checked as ExcelJS decodes it: an entity cannot hide a `[`, + // and an entity-heavy code is measured by its decoded length. + expect(styles(fmt('0['))).toThrowError(/unclosed bracket/i); + expect(styles(fmt('0['))).toThrowError(/unclosed bracket/i); + expect(styles(fmt('"x"'.repeat(60)))).not.toThrow(); + expect(styles(fmt('&'.repeat(256)))).toThrowError(/longer than 255/i); + // Attributes are read in order, so a quoted fake formatCode cannot mask the real one. + expect(styles(``)).toThrowError(/unclosed bracket/i); + expect(styles('')).toThrowError(/do not parse/i); + // `` is the container, not a format. + expect(styles('')).not.toThrow(); + // The refusal never echoes the code itself. + for (const code of ['QZJX[', `${'QZJX'.repeat(64)}0`]) { + expect(styles(fmt(code))).toThrowError( + expect.objectContaining({ code: 'number-format', message: expect.not.stringContaining('QZJX') }) + ); + } + // Every in the file is read, wherever it sits (dxfs carry them too). + const counts = { cells: 0, merges: 0, styles: 0, rows: 0 }; + const counter = core.createXmlCounter('xl/styles.xml', counts as never, core.LIMITS); + const dxf = `${fmt('0[')}`; + expect(() => counter.push(fflate.strToU8(dxf), true)).toThrowError(/unclosed bracket/i); + }); + + it('caps cell display text at 1,000 characters for every value shape', () => { + const long = 'x'.repeat(50_000); + const shapes: Array<[string, unknown]> = [ + ['string', long], + ['rich text', { richText: [{ text: long }, { text: long }] }], + ['hyperlink', { text: long, hyperlink: 'https://example.invalid/' }], + ['rich hyperlink', { text: { richText: [{ text: long }] }, hyperlink: 'https://example.invalid/' }], + ['formula source', { formula: long }], + ['formula result', { formula: 'A1', result: long }], + ['error', { error: long }], + ]; + for (const [label, value] of shapes) { + const { text } = core.formatCellValue(value, 'General'); + expect(text.length, label).toBeLessThanOrEqual(1000); + expect(text.endsWith('…'), label).toBe(true); + } + expect(core.formatCellValue('y'.repeat(1000), 'General').text).toBe('y'.repeat(1000)); + expect(core.formatCellValue({ richText: [{ text: 'a' }, { text: 'b' }] }, 'General').text).toBe('ab'); + // A cut never leaves half of a surrogate pair. + // 'aa' puts a high surrogate at index 998, exactly where a naive cut lands. + const emoji = core.formatCellValue('aa' + '😀'.repeat(600), 'General').text; + expect(emoji.length).toBeLessThanOrEqual(1000); + expect(/[\uD800-\uDBFF](?![\uDC00-\uDFFF])/.test(emoji)).toBe(false); + }); + it('refuses an entry whose declared compressed size runs past the file', () => { const zip = workbookZip(); const view = new DataView(zip.buffer, zip.byteOffset, zip.byteLength);