/** * @fileoverview Pure XLSX admission, formatting, and sparse-grid helpers. * * Loaded only inside spreadsheet-preview-worker.js (never by the page) and by * test/spreadsheet-xlsx-core.test.ts. `admitXlsx()` walks the ZIP central * directory and streams every entry through fflate BEFORE ExcelJS sees the * bytes, enforcing {@link LIMITS}; a workbook that trips any cap is refused * rather than truncated. It returns the entries it inflated, and the worker * hands ExcelJS a STORE-only archive rebuilt from exactly those * (`buildAdmittedArchive()`), never the original bytes: admission follows local * headers while ExcelJS (JSZip) follows the central directory, so overlapping * entries could otherwise show each reader a different file. Entries are * counted and rebuilt under the name ExcelJS will see (`excelJsEntryName()`), * never the stored spelling, so `/xl/...` or `xl/./...` cannot skip a counter. */ (function initSpreadsheetXlsxCore(global) { 'use strict'; const LIMITS = Object.freeze({ maxEntries: 5000, maxInflatedBytes: 64 * 1024 * 1024, maxEntryBytes: 32 * 1024 * 1024, maxCompressionRatio: 100, maxWorksheets: 50, maxCells: 250000, maxCellsPerSheet: 100000, maxRows: 250000, maxRowsPerSheet: 100000, // ExcelJS's `_mergeCellsInternal` checks each new merge against every // earlier one on its sheet, so a sheet costs the SQUARE of its merge count // (5,000 on one sheet took 1.6 s). Both caps keep the worst case near 1.3 s. maxMergesPerSheet: 2000, maxMerges: 10000, maxStyles: 5000, // ExcelJS stores each sheet at `_worksheets[sheetId]`, so the id is an array // length. Excel numbers sheets from 1 and never reuses an id, so real ids // stay small; 65535 leaves room for heavy editing at a negligible cost. maxSheetId: 65535, // Excel's own limit on a number format. ExcelJS runs `isDateFmt` on the code // once per numeric cell, and the code is echoed into the notice bar. maxNumFmtChars: 255, // Display text per cell. A cell is one `nowrap` line ending in an ellipsis, // so nothing past the column width shows; uncapped, every tile cell carried // its whole string to the page (a 1 MB shared string over a 60 x 20 block // froze the page's main thread at 9.6 GB). maxCellTextChars: 1000, // Start tags across every part ExcelJS parses (all but xl/media). ExcelJS // builds an object per element, and outside worksheets nothing else bounded // them: one shared string of 8.3M empty `` runs (a 375 KB file, padded // past the ratio cap with empty stored blocks) cost 570 MB of heap. A // three-sheet 249k-cell workbook with rich text has about 589k. maxElements: 2000000, }); const MAX_ROW = 1048576; const MAX_COL = 16384; class XlsxPreviewError extends Error { constructor(code, message) { super(message); this.name = 'XlsxPreviewError'; this.code = code; } } function fail(code, message) { throw new XlsxPreviewError(code, message); } function mergedLimits(overrides) { return Object.assign({}, LIMITS, overrides || {}); } function u16(bytes, offset) { return new DataView(bytes.buffer, bytes.byteOffset, bytes.byteLength).getUint16(offset, true); } function u32(bytes, offset) { return new DataView(bytes.buffer, bytes.byteOffset, bytes.byteLength).getUint32(offset, true); } /** * The name ExcelJS will give a ZIP entry. JSZip resolves every name on load * (`utils.resolve` in jszip/lib/utils.js, called from lib/load.js): `.` and * empty middle segments are dropped and `..` pops the previous segment. ExcelJS * then strips ONE leading `/` (lib/xlsx/xlsx.js, the `load` loop). Admission * counts, and the admitted archive is rebuilt, under this name, so the * stored spelling of a name cannot steer an entry past the counters. */ function excelJsEntryName(name) { const parts = String(name).split('/'); const resolved = []; for (let index = 0; index < parts.length; index += 1) { const part = parts[index]; // JSZip keeps an empty first or last segment (a leading or trailing `/`). if (part === '.' || (part === '' && index !== 0 && index !== parts.length - 1)) continue; if (part === '..') resolved.pop(); else resolved.push(part); } const joined = resolved.join('/'); return joined[0] === '/' ? joined.slice(1) : joined; } // ExcelJS's own worksheet test, copied verbatim (lib/xlsx/xlsx.js): it is // UNANCHORED, so `xl/xl/worksheets/sheet1.xml` and `.../sheet1.xml.x` are // worksheets too. The older anchored test stays as a conservative superset. const EXCELJS_WORKSHEET = /xl\/worksheets\/sheet(\d+)[.]xml/; function isWorksheetName(name) { return EXCELJS_WORKSHEET.test(name) || /^xl\/worksheets\/[^/]+\.xml$/i.test(name); } function inspectZipDirectory(bytes, overrides) { const limits = mergedLimits(overrides); if (bytes.length >= 4 && bytes[0] === 0xd0 && bytes[1] === 0xcf && bytes[2] === 0x11 && bytes[3] === 0xe0) { fail('encrypted', 'Encrypted or legacy OLE workbooks cannot be previewed'); } let eocd = -1; const floor = Math.max(0, bytes.length - 65557); for (let i = bytes.length - 22; i >= floor; i -= 1) { if (u32(bytes, i) === 0x06054b50) { eocd = i; break; } } if (eocd < 0) fail('malformed', 'Malformed XLSX ZIP directory'); const entryCount = u16(bytes, eocd + 10); const directorySize = u32(bytes, eocd + 12); const directoryOffset = u32(bytes, eocd + 16); if (entryCount === 0xffff || directorySize === 0xffffffff || directoryOffset === 0xffffffff) { fail('zip64', 'ZIP64 workbooks are not supported'); } if (entryCount > limits.maxEntries) fail('entry-limit', `Workbook exceeds ${limits.maxEntries} ZIP entries`); if (directoryOffset + directorySize > eocd) fail('malformed', 'Malformed XLSX central directory bounds'); const entries = []; const excelJsNames = new Set(); let cursor = directoryOffset; for (let i = 0; i < entryCount; i += 1) { if (cursor + 46 > eocd || u32(bytes, cursor) !== 0x02014b50) fail('malformed', 'Malformed XLSX central directory'); const compressedSize = u32(bytes, cursor + 20); const declaredSize = u32(bytes, cursor + 24); const nameLength = u16(bytes, cursor + 28); const extraLength = u16(bytes, cursor + 30); const commentLength = u16(bytes, cursor + 32); const localHeaderOffset = u32(bytes, cursor + 42); if (compressedSize === 0xffffffff || declaredSize === 0xffffffff) fail('zip64', 'ZIP64 entries are not supported'); const end = cursor + 46 + nameLength + extraLength + commentLength; if (end > eocd) fail('malformed', 'Malformed XLSX entry bounds'); const name = new TextDecoder().decode(bytes.subarray(cursor + 46, cursor + 46 + nameLength)); if (localHeaderOffset + 30 > directoryOffset || u32(bytes, localHeaderOffset) !== 0x04034b50) { fail('malformed', 'Malformed XLSX local file header'); } const localNameLength = u16(bytes, localHeaderOffset + 26); const localExtraLength = u16(bytes, localHeaderOffset + 28); const localNameEnd = localHeaderOffset + 30 + localNameLength; if (localNameEnd + localExtraLength > directoryOffset) fail('malformed', 'Malformed XLSX local entry bounds'); // The compression-ratio cap divides by this declared size, so it must // describe bytes that actually exist before the central directory. if (localNameEnd + localExtraLength + compressedSize > directoryOffset) { fail('malformed', 'XLSX entry declares more compressed bytes than the file holds'); } const localName = new TextDecoder().decode(bytes.subarray(localHeaderOffset + 30, localNameEnd)); if (localName !== name) fail('malformed', 'XLSX local and central directory names do not match'); // JSZip keeps only the last of two entries that resolve to one name, and // the rebuilt archive can hold only one, so both are refused. const excelJsName = excelJsEntryName(name); if (excelJsName === '') fail('malformed', 'XLSX entry name resolves to nothing'); if (excelJsNames.has(excelJsName)) fail('malformed', 'Two XLSX entries resolve to the same name'); excelJsNames.add(excelJsName); entries.push({ name, excelJsName, compressedSize, declaredSize, localHeaderOffset }); cursor = end; } return { entries }; } function featureForName(name) { if (name.startsWith('xl/charts/')) return 'charts'; if (name.startsWith('xl/drawings/')) return 'drawings'; if (name.startsWith('xl/pivotCache/')) return 'pivotTables'; if (name.startsWith('xl/externalLinks/')) return 'externalLinks'; if (/vbaProject\.bin$/i.test(name)) return 'macros'; return null; } // Longest single XML tag the streaming counter will carry across chunks. Real // worksheet tags are a few hundred bytes; this only stops a pathological tag // from turning the carried tail into quadratic re-scanning. const MAX_CARRIED_TAG = 256 * 1024; // XML attribute syntax, read in order from just after the tag name. A quoted // value may hold a raw `>` or the other quote character, so a value is always // consumed whole; a tag is only understood if this walk reaches its `>`. const XML_SPACE = '[ \\t\\r\\n]'; const ATTRIBUTE = new RegExp( `${XML_SPACE}+([^ \\t\\r\\n=/>]+)${XML_SPACE}*=${XML_SPACE}*(?:"([^"]*)"|'([^']*)')`, 'y' ); const TAG_END = new RegExp(`${XML_SPACE}*/?>`, 'y'); /** * Parse the attributes of the tag whose name ends at `from` in `text`. * Returns the attributes, or null when they do not parse cleanly up to the * tag's closing `/>` or `>` (or a name repeats, which XML forbids). */ function readTagAttributes(text, from) { const attributes = new Map(); let at = from; for (;;) { ATTRIBUTE.lastIndex = at; const match = ATTRIBUTE.exec(text); if (!match) break; if (attributes.has(match[1])) return null; attributes.set(match[1], match[2] ?? match[3]); at = ATTRIBUTE.lastIndex; } TAG_END.lastIndex = at; return TAG_END.test(text) ? attributes : null; } // ExcelJS expands a merge into one cell object per covered cell at load time, // so a merge costs its AREA, not one tag. function mergeArea(attributes) { const range = parseRange(attributes.get('ref')); if (!range) fail('malformed', 'Worksheet has a merged range that does not parse'); return (Math.abs(range.r2 - range.r1) + 1) * (Math.abs(range.c2 - range.c1) + 1); } // ExcelJS builds one column object for every index up to ``, unclamped. function checkColumnSpan(attributes) { for (const name of ['min', 'max']) { const value = attributes.get(name); if (value === undefined) continue; const index = Number(value); if (!Number.isInteger(index) || index < 1 || index > MAX_COL) { fail('malformed', `Worksheet column ${name} is outside 1-${MAX_COL}`); } } } // ExcelJS stores each row at `_rows[r - 1]`, and eachRow and `sheet.model` // walk every index up to the largest, so the index a row CLAIMS is a cost. function checkRowIndex(attributes) { const value = attributes.get('r'); if (value === undefined) return; if (!/^[0-9]+$/.test(value) || Number(value) < 1 || Number(value) > MAX_ROW) { fail('malformed', `Worksheet row index is outside 1-${MAX_ROW}`); } } // ExcelJS stores each sheet at `_worksheets[sheetId]` (an absent id parses to // NaN, a plain property, so it is harmless). function checkSheetId(attributes, limits) { const value = attributes.get('sheetId'); if (value === undefined) return; if (!/^[0-9]+$/.test(value) || Number(value) > limits.maxSheetId) { fail('malformed', `Workbook sheetId is not a number up to ${limits.maxSheetId}`); } } // Attribute values as the XML parser inside ExcelJS (saxes) hands them over: // the predefined entities and character references decoded. Checking the raw // text would let `[` hide a `[` and would overcount `"`. function decodeXmlAttribute(value) { return value.replace(/&(?:#x([0-9a-fA-F]+)|#([0-9]+)|(amp|lt|gt|quot|apos));/g, (entity, hex, dec, name) => { if (name) return { amp: '&', lt: '<', gt: '>', quot: '"', apos: "'" }[name]; const code = hex ? Number.parseInt(hex, 16) : Number(dec); return code <= 0x10ffff ? String.fromCodePoint(code) : entity; }); } // ExcelJS's `isDateFmt` strips `/\[[^\]]*]/g` from the code once per numeric // cell, and that pattern rescans to the end of the code for every `[` with no // later `]` (a 60,000-character code of `[` cost 2.2 s a cell; even 255 // characters added up at the cell cap). Once nothing follows the last `]` // but text without `[`, each `[` stops at the next `]` and the scan is linear. // The messages never echo the code. function checkNumberFormat(attributes, limits) { const raw = attributes.get('formatCode'); if (raw === undefined) return; const code = decodeXmlAttribute(raw); if (code.length > limits.maxNumFmtChars) { fail('number-format', `Workbook has a number format longer than ${limits.maxNumFmtChars} characters`); } if (code.lastIndexOf('[') > code.lastIndexOf(']')) { fail('number-format', 'Workbook has a number format with an unclosed bracket'); } } // Exactly the names ExcelJS hands to `_processMediaEntry` and to no parser: // its other patterns are unanchored, so `xl/media/xl/drawings/a.xml` is still // parsed as a drawing, and only a single `.` segment is pure bytes. const EXCELJS_MEDIA = /^xl\/media\/[a-zA-Z0-9]+[.][a-zA-Z0-9]{3,4}$/; const START_TAG = /<[A-Za-z_]/g; function countStartTags(text) { let count = 0; START_TAG.lastIndex = 0; while (START_TAG.exec(text)) count += 1; return count; } function createXmlCounter(name, counts, limits) { let tail = ''; const decoder = new TextDecoder(); let sheetCells = 0; let sheetMerges = 0; let sheetRows = 0; const excelJsName = excelJsEntryName(name); const worksheet = isWorksheetName(excelJsName); const styles = excelJsName === 'xl/styles.xml'; const workbook = excelJsName === 'xl/workbook.xml'; const media = EXCELJS_MEDIA.test(excelJsName); // Only these parts read tag attributes, so only they carry a whole tag. const readsTags = worksheet || styles || workbook; const addCells = (cells) => { sheetCells += cells; counts.cells += cells; if (sheetCells > limits.maxCellsPerSheet || counts.cells > limits.maxCells) fail('cell-limit', 'Workbook exceeds the cells limit'); }; return { push(chunk, final) { if (media) return; const text = tail + decoder.decode(chunk, { stream: !final }); // `<` can never appear inside an attribute value, so every tag before the // last `<` is complete. Carry everything from that `<` into the next scan // so a tag cut by a chunk edge is always read in one piece. Any other part // only needs its start tags counted, so it carries at most a trailing `<` // (a long text node or a binary part is never mistaken for a huge tag). let safeEnd = text.length; if (!final && readsTags) safeEnd = Math.max(0, text.lastIndexOf('<')); else if (!final && text.endsWith('<')) safeEnd = text.length - 1; const scan = text.slice(0, safeEnd); counts.elements += countStartTags(scan); if (counts.elements > limits.maxElements) fail('element-limit', 'Workbook exceeds the XML elements limit'); if (worksheet) { addCells((scan.match(/)/g) || []).length); // ExcelJS keeps a Row object for every , with or without cells. const rows = (scan.match(/])/g) || []).length; sheetRows += rows; counts.rows += rows; if (sheetRows > limits.maxRowsPerSheet || counts.rows > limits.maxRows) fail('row-limit', 'Workbook exceeds the rows limit'); for (const match of scan.matchAll(/<(mergeCell|col|row)(?=[\s/>])/g)) { const attributes = readTagAttributes(scan, match.index + match[0].length); if (!attributes) fail('malformed', `Worksheet has a <${match[1]}> whose attributes do not parse`); if (match[1] === 'row') { checkRowIndex(attributes); continue; } if (match[1] === 'col') { checkColumnSpan(attributes); continue; } sheetMerges += 1; counts.merges += 1; if (sheetMerges > limits.maxMergesPerSheet || counts.merges > limits.maxMerges) fail('merge-limit', 'Workbook exceeds the merged ranges limit'); addCells(mergeArea(attributes)); } } if (workbook) { // `` is the container; the lookahead keeps it from matching. for (const match of scan.matchAll(/])/g)) { const attributes = readTagAttributes(scan, match.index + match[0].length); if (!attributes) fail('malformed', 'Workbook has a whose attributes do not parse'); checkSheetId(attributes, limits); } } if (styles) { // Every counts, cellStyleXfs included: tracking which list a tag // sits in can be desynced by a closing tag inside an XML comment. counts.styles += (scan.match(/])/g) || []).length; if (counts.styles > limits.maxStyles) fail('style-limit', 'Workbook exceeds the cell styles limit'); // Every , cellXfs' and dxfs' alike; the lookahead skips . for (const match of scan.matchAll(/])/g)) { const attributes = readTagAttributes(scan, match.index + match[0].length); if (!attributes) fail('malformed', 'Workbook has a whose attributes do not parse'); checkNumberFormat(attributes, limits); } } tail = text.slice(safeEnd); if (tail.length > MAX_CARRIED_TAG) fail('malformed', 'Workbook XML has an oversized tag'); }, }; } function admitXlsx(bytes, zipApi, overrides) { const limits = mergedLimits(overrides); const directory = inspectZipDirectory(bytes, limits); const directoryByName = new Map(directory.entries.map((entry) => [entry.name, entry])); const expectedEntries = new Map(); for (const entry of directory.entries) expectedEntries.set(entry.name, (expectedEntries.get(entry.name) || 0) + 1); const streamedEntries = new Map(); const inflatedEntries = Object.create(null); const counts = { worksheets: 0, cells: 0, rows: 0, merges: 0, styles: 0, elements: 0 }; const features = new Set(); let totalInflated = 0; let seenEntries = 0; let thrown; const unzip = new zipApi.Unzip((file) => { if (!directoryByName.has(file.name)) fail('malformed', 'Local XLSX entry is absent from the central directory'); // The admitted archive is rebuilt from these entries, which cannot hold two // files under one name, so a duplicate name is refused rather than dropped. if (streamedEntries.has(file.name)) fail('malformed', 'Duplicate XLSX entry name'); streamedEntries.set(file.name, (streamedEntries.get(file.name) || 0) + 1); seenEntries += 1; if (seenEntries > limits.maxEntries) fail('entry-limit', 'Workbook exceeds the ZIP entries limit'); // Everything below keys on the name ExcelJS will see, never the stored one. const name = directoryByName.get(file.name).excelJsName; if (isWorksheetName(name)) { counts.worksheets += 1; if (counts.worksheets > limits.maxWorksheets) fail('worksheet-limit', 'Workbook exceeds the worksheet limit'); } const feature = featureForName(name); if (feature) features.add(feature); const counter = createXmlCounter(name, counts, limits); let entryInflated = 0; const chunks = []; file.ondata = (error, chunk, final) => { if (error) throw error; chunks.push(chunk.slice()); entryInflated += chunk.length; totalInflated += chunk.length; if (entryInflated > limits.maxEntryBytes) fail('entry-size', 'Inflated ZIP entry exceeds the entry limit'); if (totalInflated > limits.maxInflatedBytes) fail('inflated-size', 'Workbook exceeds the inflated bytes limit'); const compressed = directoryByName.get(file.name)?.compressedSize || 1; if (entryInflated / Math.max(1, compressed) > limits.maxCompressionRatio) { fail('compression-ratio', 'ZIP entry exceeds the compression ratio limit'); } counter.push(chunk, final); if (final) { const data = new Uint8Array(entryInflated); let offset = 0; for (const part of chunks) { data.set(part, offset); offset += part.length; } chunks.length = 0; inflatedEntries[name] = data; } }; file.start(); }); unzip.register(zipApi.UnzipInflate); try { const inputChunkBytes = 64 * 1024; for (let offset = 0; offset < bytes.length && !thrown; offset += inputChunkBytes) { const end = Math.min(bytes.length, offset + inputChunkBytes); unzip.push(bytes.subarray(offset, end), end === bytes.length); } } catch (error) { thrown = error; } if (thrown) throw thrown; for (const [name, count] of expectedEntries) { if (streamedEntries.get(name) !== count) fail('malformed', 'Central XLSX entry was not streamed for admission'); } for (const name of streamedEntries.keys()) { if (!(directoryByName.get(name).excelJsName in inflatedEntries)) { fail('malformed', 'XLSX entry did not finish streaming for admission'); } } return { counts, features: Array.from(features), inflatedBytes: totalInflated, entries: inflatedEntries }; } // The ONLY bytes ExcelJS may parse: the entries admission itself inflated and // counted, re-packed uncompressed. Its size is ~ the inflated total (already // capped at LIMITS.maxInflatedBytes) plus per-entry headers. function buildAdmittedArchive(admission, zipApi) { return zipApi.zipSync(admission.entries, { level: 0 }); } function parseCellRef(ref) { const match = /^\$?([A-Z]{1,3})\$?([1-9]\d*)$/i.exec(String(ref || '')); if (!match) return null; let col = 0; for (const char of match[1].toUpperCase()) col = col * 26 + char.charCodeAt(0) - 64; const row = Number(match[2]); return row <= MAX_ROW && col <= MAX_COL ? { row, col } : null; } function parseRange(range) { const parts = String(range).split(':'); const start = parseCellRef(parts[0]); const end = parseCellRef(parts[1] || parts[0]); return start && end ? { r1: start.row, c1: start.col, r2: end.row, c2: end.col } : null; } function deriveExtent(cells, merges) { let rows = 0; let cols = 0; for (const ref of cells) { const cell = parseCellRef(ref); if (cell) { rows = Math.max(rows, cell.row); cols = Math.max(cols, cell.col); } } for (const merge of merges) { const range = parseRange(merge); if (range) { rows = Math.max(rows, range.r2); cols = Math.max(cols, range.c2); } } return { rows, cols }; } function createSparseAxis(count, defaultSize, overrides) { const sorted = Array.from(overrides || []) .filter(([index, size]) => index >= 1 && index <= count && Number.isFinite(size)) .map(([index, size]) => [index, Math.max(0, size)]) .sort((a, b) => a[0] - b[0]); const base = Math.max(0, defaultSize); // deltas[i] is the summed size change of overrides[0..i], so an offset is // one binary search instead of a walk over every override. const deltas = []; let delta = 0; for (const [, size] of sorted) { delta += size - base; deltas.push(delta); } return { count: Math.max(0, count), defaultSize: base, overrides: sorted, deltas }; } // Number of overrides whose index is below `bounded`. function overridesBefore(overrides, bounded) { let low = 0; let high = overrides.length; while (low < high) { const mid = (low + high) >> 1; if (overrides[mid][0] < bounded) low = mid + 1; else high = mid; } return low; } function axisOffset(axis, index) { const bounded = Math.max(1, Math.min(axis.count + 1, index)); const before = overridesBefore(axis.overrides, bounded); return (bounded - 1) * axis.defaultSize + (before ? axis.deltas[before - 1] : 0); } function axisIndexAt(axis, offset) { let low = 1; let high = Math.max(1, axis.count); const target = Math.max(0, offset); while (low < high) { const mid = Math.floor((low + high + 1) / 2); if (axisOffset(axis, mid) <= target) low = mid; else high = mid - 1; } return low; } function computeViewport(axis, offset, viewportSize, overscan) { const pad = Math.max(0, overscan || 0); const start = Math.max(1, axisIndexAt(axis, offset) - pad); const end = Math.min(axis.count, axisIndexAt(axis, offset + Math.max(0, viewportSize)) + pad); return [start, end]; } function intersectingMerges(merges, viewport) { return merges.filter((merge) => { const range = parseRange(merge); return ( range && range.r1 <= viewport.r2 && range.r2 >= viewport.r1 && range.c1 <= viewport.c2 && range.c2 >= viewport.c1 ); }); } // Rounded to whole milliseconds: `new Date(fraction)` truncates, which turned // midnight minus a float error into the previous day. function excelDate(serial, date1904) { if (date1904) return new Date(Math.round(Date.UTC(1904, 0, 1) + Number(serial) * 86400000)); const numeric = Number(serial); const adjusted = numeric >= 60 ? numeric - 1 : numeric; return new Date(Math.round(Date.UTC(1899, 11, 31) + adjusted * 86400000)); } function isDateValue(value) { return Object.prototype.toString.call(value) === '[object Date]' && Number.isFinite(value.getTime()); } // Inverse of ExcelJS's own `excelToDate()` (utils.js), which builds the Date // from the serial in UTC. Recovering the serial keeps the result independent of // the viewer's timezone; `String(date)` rendered it in local time, a day early // at any negative UTC offset. function dateToSerial(date, date1904) { return 25569 + date.getTime() / 86400000 - (date1904 ? 1462 : 0); } const DATE_FORMAT = /^[ymd\-/ ]+$/i; // An optional trailing AM/PM (built-in format 18 is `h:mm AM/PM`). const TIME_FORMAT = /^[hms: ]+(?:AM\/PM)?$/i; const DATE_TIME_FORMAT = /^[ymdhis\-/: ]+$/i; function isFormulaValue(value) { return 'formula' in value || 'sharedFormula' in value; } // Cut to LIMITS.maxCellTextChars, ending in an ellipsis, never between the // halves of a surrogate pair. function capCellText(text) { const max = LIMITS.maxCellTextChars; if (text.length <= max) return text; let end = max - 1; const last = text.charCodeAt(end - 1); if (last >= 0xd800 && last <= 0xdbff) end -= 1; return `${text.slice(0, end)}…`; } // Joins rich-text runs only until the cap is passed, so a long run is never // copied whole once per cell. It also visits at most maxCellTextChars + 1 // runs: an empty run (``, 4 bytes) adds no text, so a length check alone // walked every run, for every cell sharing the string, on every tile. Any // maxCellTextChars + 1 non-empty runs already pass the cap. function richTextPrefix(runs) { let text = ''; const visit = Math.min(runs.length, LIMITS.maxCellTextChars + 1); for (let index = 0; index < visit; index += 1) { if (text.length > LIMITS.maxCellTextChars) break; const run = runs[index]; if (typeof run?.text === 'string') text += run.text.slice(0, LIMITS.maxCellTextChars + 1); } return text; } // Excel displays at most 15 significant digits, so `=0.1+0.2` shows 0.3, // never the binary float's 0.30000000000000004. function generalNumber(value) { return Number.isFinite(value) ? String(Number(value.toPrecision(15))) : String(value); } const UNSUPPORTED_FORMAT_PREFIX = 'Unsupported number format: '; /** * Folds every unsupported number format warning into one counted entry when * there is more than one, so a sheet with a code per cell cannot grow the * notice bar without bound. A lone warning is kept as is; its code is at * most 255 characters (admission refuses longer ones). Order is preserved. */ function foldWarnings(warnings) { const formats = warnings.filter((warning) => String(warning).startsWith(UNSUPPORTED_FORMAT_PREFIX)); if (formats.length < 2) return warnings.slice(); const folded = []; let placed = false; for (const warning of warnings) { if (!String(warning).startsWith(UNSUPPORTED_FORMAT_PREFIX)) folded.push(warning); else if (!placed) { folded.push(`${formats.length} unsupported number formats`); placed = true; } } return folded; } /** * A cell's display text and any warning. The text is ALWAYS at most * `LIMITS.maxCellTextChars` characters, whatever shape the value has. */ function formatCellValue(value, format, date1904) { const formatted = formatCellValueUncapped(value, format, date1904); return formatted.text.length > LIMITS.maxCellTextChars ? Object.assign({}, formatted, { text: capCellText(formatted.text) }) : formatted; } // Every non-scalar shape ExcelJS loads a cell value as. Anything not handled // here would otherwise reach String() and render as "[object Object]". function formatCellValueUncapped(value, format, date1904) { if (value === null || value === undefined) return { text: '' }; if (isDateValue(value)) { const code = String(format || 'General'); const known = DATE_FORMAT.test(code) || TIME_FORMAT.test(code) || DATE_TIME_FORMAT.test(code); const serial = dateToSerial(value, date1904); if (known) return formatCellValueUncapped(serial, code, date1904); const fallback = serial % 1 === 0 ? 'yyyy-mm-dd' : 'yyyy-mm-dd hh:mm'; const formatted = formatCellValueUncapped(serial, fallback, date1904); return /^General$/i.test(code) ? formatted : { text: formatted.text, warning: `${UNSUPPORTED_FORMAT_PREFIX}${code}` }; } if (typeof value === 'object') { if (isFormulaValue(value)) { if (value.result !== undefined && value.result !== null) { return formatCellValueUncapped(value.result, format, date1904); } const source = typeof value.formula === 'string' ? `=${value.formula}` : ''; return { text: source, warning: 'Formula has no cached result' }; } if (typeof value.error === 'string') return { text: value.error }; if (Array.isArray(value.richText)) { return { text: richTextPrefix(value.richText) }; } // Hyperlink: the display text, never the target. The text may be rich. if ('text' in value) return formatCellValueUncapped(value.text, 'General', date1904); return { text: '', warning: 'Unsupported cell value' }; } const code = String(format || 'General'); if (typeof value !== 'number') return { text: String(value) }; if (/^General$/i.test(code)) return { text: generalNumber(value) }; if (DATE_FORMAT.test(code)) { const date = excelDate(value, Boolean(date1904)); const yyyy = date.getUTCFullYear(); const mm = String(date.getUTCMonth() + 1).padStart(2, '0'); const dd = String(date.getUTCDate()).padStart(2, '0'); return { text: `${yyyy}-${mm}-${dd}` }; } if (TIME_FORMAT.test(code)) { const seconds = Math.round((value - Math.floor(value)) * 86400) % 86400; const hours = Math.floor(seconds / 3600); const mm = String(Math.floor((seconds % 3600) / 60)).padStart(2, '0'); const ss = String(seconds % 60).padStart(2, '0'); if (/AM\/PM$/i.test(code)) { const hour12 = String(hours % 12 || 12); const hh = /hh/i.test(code) ? hour12.padStart(2, '0') : hour12; const clock = /s/i.test(code) ? `${hh}:${mm}:${ss}` : `${hh}:${mm}`; return { text: `${clock} ${hours < 12 ? 'AM' : 'PM'}` }; } const hh = String(hours).padStart(2, '0'); return { text: `${hh}:${mm}:${ss}` }; } if (DATE_TIME_FORMAT.test(code)) { const date = excelDate(value, Boolean(date1904)); const yyyy = date.getUTCFullYear(); const mm = String(date.getUTCMonth() + 1).padStart(2, '0'); const dd = String(date.getUTCDate()).padStart(2, '0'); const hh = String(date.getUTCHours()).padStart(2, '0'); const minutes = String(date.getUTCMinutes()).padStart(2, '0'); return { text: `${yyyy}-${mm}-${dd} ${hh}:${minutes}` }; } const percent = code.includes('%'); // toLocaleString throws above 100 fraction digits; Excel itself caps at 30. const decimals = Math.min(code.match(/\.([0#]+)/)?.[1].length || 0, 30); const numericPattern = /^[€£¥$]?[#,0]+(?:\.[0#]+)?%?$/; if (numericPattern.test(code)) { const currency = /^[€£¥$]/.exec(code)?.[0] || ''; const numeric = percent ? value * 100 : value; const useGrouping = code.includes(','); return { text: currency + numeric.toLocaleString('en-US', { useGrouping, minimumFractionDigits: decimals, maximumFractionDigits: decimals, }) + (percent ? '%' : ''), }; } return { text: generalNumber(value), warning: `${UNSUPPORTED_FORMAT_PREFIX}${code}` }; } // Colour resolution -------------------------------------------------------- // // ExcelJS surfaces theme and indexed palette colours WITHOUT an `argb` key // (`{theme,tint}` / `{indexed}`), and theme colours are what Excel emits by // default. Dropping them left cells rendering the workbook's font colour on // the skin's own background, which is how black-on-dark (invisible) text got // shipped. Everything below is pure so `test/spreadsheet-xlsx-core.test.ts` // can pin it without a DOM. // styles.xml `theme="N"` order. NOTE: theme1.xml's lists // dk1, lt1, dk2, lt2 — indices 0/1 and 2/3 are SWAPPED between the two. const THEME_SLOT_ORDER = Object.freeze([ 'lt1', 'dk1', 'lt2', 'dk2', 'accent1', 'accent2', 'accent3', 'accent4', 'accent5', 'accent6', 'hlink', 'folHlink', ]); // Default Office theme, used when theme1.xml is missing or unparseable. const DEFAULT_THEME_PALETTE = Object.freeze([ '#ffffff', '#000000', '#eeece1', '#1f497d', '#4f81bd', '#c0504d', '#9bbb59', '#8064a2', '#4bacc6', '#f79646', '#0000ff', '#800080', ]); // Legacy 64-entry indexed palette. Indices 64/65 are the "auto" foreground and // background sentinels and deliberately have no entry here. // prettier-ignore const INDEXED_PALETTE = Object.freeze([ '#000000', '#ffffff', '#ff0000', '#00ff00', '#0000ff', '#ffff00', '#ff00ff', '#00ffff', '#000000', '#ffffff', '#ff0000', '#00ff00', '#0000ff', '#ffff00', '#ff00ff', '#00ffff', '#800000', '#008000', '#000080', '#808000', '#800080', '#008080', '#c0c0c0', '#808080', '#9999ff', '#993366', '#ffffcc', '#ccffff', '#660066', '#ff8080', '#0066cc', '#ccccff', '#000080', '#ff00ff', '#ffff00', '#00ffff', '#800080', '#800000', '#008080', '#0000ff', '#00ccff', '#ccffff', '#ccffcc', '#ffff99', '#99ccff', '#ff99cc', '#cc99ff', '#ffcc99', '#3366ff', '#33cccc', '#99cc00', '#ffcc00', '#ff9900', '#ff6600', '#666699', '#969696', '#003366', '#339966', '#003300', '#333300', '#993300', '#993366', '#333399', '#333333', ]); const MIN_CONTRAST_RATIO = 4.5; // A workbook with no fill renders on Excel's implicit white sheet background, // never on the viewer skin's `var(--bg-primary)`. const IMPLICIT_SHEET_BACKGROUND = '#ffffff'; const IMPLICIT_SHEET_FOREGROUND = '#000000'; function hexToRgb(hex) { const match = /^#([0-9a-f]{2})([0-9a-f]{2})([0-9a-f]{2})$/i.exec(String(hex || '')); if (!match) return null; return [Number.parseInt(match[1], 16), Number.parseInt(match[2], 16), Number.parseInt(match[3], 16)]; } function rgbToHex(rgb) { let hex = '#'; for (const channel of rgb) { const bounded = Math.max(0, Math.min(255, Math.round(channel))); hex += (bounded < 16 ? '0' : '') + bounded.toString(16); } return hex; } function rgbToHsl(rgb) { const r = rgb[0] / 255; const g = rgb[1] / 255; const b = rgb[2] / 255; const max = Math.max(r, g, b); const min = Math.min(r, g, b); const l = (max + min) / 2; if (max === min) return { h: 0, s: 0, l }; const delta = max - min; const s = l > 0.5 ? delta / (2 - max - min) : delta / (max + min); let h; if (max === r) h = (g - b) / delta + (g < b ? 6 : 0); else if (max === g) h = (b - r) / delta + 2; else h = (r - g) / delta + 4; return { h: h / 6, s, l }; } function hueToChannel(p, q, hue) { let t = hue; if (t < 0) t += 1; if (t > 1) t -= 1; if (t < 1 / 6) return p + (q - p) * 6 * t; if (t < 1 / 2) return q; if (t < 2 / 3) return p + (q - p) * (2 / 3 - t) * 6; return p; } function hslToRgb(hsl) { if (hsl.s === 0) { const gray = hsl.l * 255; return [gray, gray, gray]; } const q = hsl.l < 0.5 ? hsl.l * (1 + hsl.s) : hsl.l + hsl.s - hsl.l * hsl.s; const p = 2 * hsl.l - q; return [ hueToChannel(p, q, hsl.h + 1 / 3) * 255, hueToChannel(p, q, hsl.h) * 255, hueToChannel(p, q, hsl.h - 1 / 3) * 255, ]; } // Excel tint acts on HSL luminance: negative darkens, positive lightens. function applyTint(hex, tint) { const amount = Number(tint); if (!Number.isFinite(amount) || amount === 0) return hex; const rgb = hexToRgb(hex); if (!rgb) return hex; const bounded = Math.max(-1, Math.min(1, amount)); const hsl = rgbToHsl(rgb); const luminance = bounded < 0 ? hsl.l * (1 + bounded) : hsl.l * (1 - bounded) + bounded; return rgbToHex(hslToRgb({ h: hsl.h, s: hsl.s, l: Math.max(0, Math.min(1, luminance)) })); } function parseSchemeColor(fragment) { const srgb = /<(?:[A-Za-z0-9_]+:)?srgbClr\b[^>]*\bval="([0-9A-Fa-f]{6})"/.exec(fragment); if (srgb) return `#${srgb[1].toLowerCase()}`; const sys = /<(?:[A-Za-z0-9_]+:)?sysClr\b[^>]*\blastClr="([0-9A-Fa-f]{6})"/.exec(fragment); if (sys) return `#${sys[1].toLowerCase()}`; return undefined; } // Real theme1.xml files are under 10 KB. The scheme and slot patterns below // rescan to the end of the text for every opening tag that has no close, so a // padded theme is quadratic (1 MB of `` took 21.8 s); above this // many characters (UTF-16 code units of the decoded XML) the default palette // is used instead. const MAX_THEME_XML_CHARS = 64 * 1024; function parseThemePalette(xml) { const palette = DEFAULT_THEME_PALETTE.slice(); const text = typeof xml === 'string' ? xml : ''; if (text.length > MAX_THEME_XML_CHARS) return palette; const scheme = /<(?:[A-Za-z0-9_]+:)?clrScheme\b[^>]*>([\s\S]*?)<\/(?:[A-Za-z0-9_]+:)?clrScheme\s*>/.exec(text); if (!scheme) return palette; // Fresh pattern per call: a shared /g regex would carry lastIndex across calls. const slots = /<(?:[A-Za-z0-9_]+:)?(lt1|dk1|lt2|dk2|accent[1-6]|hlink|folHlink)\b[^>]*>([\s\S]*?)<\/(?:[A-Za-z0-9_]+:)?\1\s*>/g; let match = slots.exec(scheme[1]); while (match) { const index = THEME_SLOT_ORDER.indexOf(match[1]); const resolved = index >= 0 ? parseSchemeColor(match[2]) : undefined; if (resolved) palette[index] = resolved; match = slots.exec(scheme[1]); } return palette; } function themePaletteOrDefault(palette) { return Array.isArray(palette) && palette.length === THEME_SLOT_ORDER.length ? palette : DEFAULT_THEME_PALETTE; } // Accepts every colour shape ExcelJS emits: {argb}, {theme,tint}, {indexed}. function resolveColor(color, palette) { if (!color || typeof color !== 'object') return undefined; const argb = color.argb; if (typeof argb === 'string' && /^[A-Fa-f0-9]{8}$/.test(argb)) return `#${argb.slice(2).toLowerCase()}`; const theme = color.theme; if (Number.isInteger(theme) && theme >= 0 && theme < THEME_SLOT_ORDER.length) { const base = themePaletteOrDefault(palette)[theme]; return typeof base === 'string' ? applyTint(base, color.tint) : undefined; } const indexed = color.indexed; if (Number.isInteger(indexed) && indexed >= 0 && indexed < INDEXED_PALETTE.length) return INDEXED_PALETTE[indexed]; return undefined; } function channelLuminance(channel) { const value = channel / 255; return value <= 0.03928 ? value / 12.92 : Math.pow((value + 0.055) / 1.055, 2.4); } function relativeLuminance(hex) { const rgb = hexToRgb(hex); if (!rgb) return 0; return 0.2126 * channelLuminance(rgb[0]) + 0.7152 * channelLuminance(rgb[1]) + 0.0722 * channelLuminance(rgb[2]); } function contrastRatio(a, b) { const first = relativeLuminance(a); const second = relativeLuminance(b); return (Math.max(first, second) + 0.05) / (Math.min(first, second) + 0.05); } function ensureContrast(foreground, background, minRatio) { const minimum = Number.isFinite(minRatio) ? minRatio : MIN_CONTRAST_RATIO; if (contrastRatio(foreground, background) >= minimum) return foreground; return contrastRatio('#000000', background) >= contrastRatio('#ffffff', background) ? '#000000' : '#ffffff'; } // Returns a SELF-CONSISTENT pair, or nothing at all. Emitting only one half is // what let workbook text land on the skin's background and vanish. function resolveCellColors(fillColor, fontColor, palette) { const fill = resolveColor(fillColor, palette); const font = resolveColor(fontColor, palette); if (!fill && !font) return {}; const background = fill || IMPLICIT_SHEET_BACKGROUND; return { background, foreground: ensureContrast(font || IMPLICIT_SHEET_FOREGROUND, background) }; } global.CodemanSpreadsheetXlsxCore = Object.freeze({ LIMITS, MAX_ROW, MAX_COL, XlsxPreviewError, inspectZipDirectory, createXmlCounter, admitXlsx, buildAdmittedArchive, excelJsEntryName, parseCellRef, parseRange, deriveExtent, createSparseAxis, axisOffset, axisIndexAt, computeViewport, intersectingMerges, formatCellValue, foldWarnings, DEFAULT_THEME_PALETTE, INDEXED_PALETTE, MIN_CONTRAST_RATIO, parseThemePalette, resolveColor, contrastRatio, ensureContrast, resolveCellColors, }); })(typeof self !== 'undefined' ? self : globalThis);