/**
* @fileoverview Pure XLSX admission, formatting, and sparse-grid helpers.
*
* Loaded only inside spreadsheet-preview-worker.js (never by the page) and by
* test/spreadsheet-xlsx-core.test.ts. `admitXlsx()` walks the ZIP central
* directory and streams every entry through fflate BEFORE ExcelJS sees the
* bytes, enforcing {@link LIMITS}; a workbook that trips any cap is refused
* rather than truncated. It returns the entries it inflated, and the worker
* hands ExcelJS a STORE-only archive rebuilt from exactly those
* (`buildAdmittedArchive()`), never the original bytes: admission follows local
* headers while ExcelJS (JSZip) follows the central directory, so overlapping
* entries could otherwise show each reader a different file. Entries are
* counted and rebuilt under the name ExcelJS will see (`excelJsEntryName()`),
* never the stored spelling, so `/xl/...` or `xl/./...` cannot skip a counter.
*/
(function initSpreadsheetXlsxCore(global) {
'use strict';
const LIMITS = Object.freeze({
maxEntries: 5000,
maxInflatedBytes: 64 * 1024 * 1024,
maxEntryBytes: 32 * 1024 * 1024,
maxCompressionRatio: 100,
maxWorksheets: 50,
maxCells: 250000,
maxCellsPerSheet: 100000,
maxRows: 250000,
maxRowsPerSheet: 100000,
// ExcelJS's `_mergeCellsInternal` checks each new merge against every
// earlier one on its sheet, so a sheet costs the SQUARE of its merge count
// (5,000 on one sheet took 1.6 s). Both caps keep the worst case near 1.3 s.
maxMergesPerSheet: 2000,
maxMerges: 10000,
maxStyles: 5000,
// ExcelJS stores each sheet at `_worksheets[sheetId]`, so the id is an array
// length. Excel numbers sheets from 1 and never reuses an id, so real ids
// stay small; 65535 leaves room for heavy editing at a negligible cost.
maxSheetId: 65535,
// Excel's own limit on a number format. ExcelJS runs `isDateFmt` on the code
// once per numeric cell, and the code is echoed into the notice bar.
maxNumFmtChars: 255,
// Display text per cell. A cell is one `nowrap` line ending in an ellipsis,
// so nothing past the column width shows; uncapped, every tile cell carried
// its whole string to the page (a 1 MB shared string over a 60 x 20 block
// froze the page's main thread at 9.6 GB).
maxCellTextChars: 1000,
// Start tags across every part ExcelJS parses (all but xl/media). ExcelJS
// builds an object per element, and outside worksheets nothing else bounded
// them: one shared string of 8.3M empty `` runs (a 375 KB file, padded
// past the ratio cap with empty stored blocks) cost 570 MB of heap. A
// three-sheet 249k-cell workbook with rich text has about 589k.
maxElements: 2000000,
});
const MAX_ROW = 1048576;
const MAX_COL = 16384;
class XlsxPreviewError extends Error {
constructor(code, message) {
super(message);
this.name = 'XlsxPreviewError';
this.code = code;
}
}
function fail(code, message) {
throw new XlsxPreviewError(code, message);
}
function mergedLimits(overrides) {
return Object.assign({}, LIMITS, overrides || {});
}
function u16(bytes, offset) {
return new DataView(bytes.buffer, bytes.byteOffset, bytes.byteLength).getUint16(offset, true);
}
function u32(bytes, offset) {
return new DataView(bytes.buffer, bytes.byteOffset, bytes.byteLength).getUint32(offset, true);
}
/**
* The name ExcelJS will give a ZIP entry. JSZip resolves every name on load
* (`utils.resolve` in jszip/lib/utils.js, called from lib/load.js): `.` and
* empty middle segments are dropped and `..` pops the previous segment. ExcelJS
* then strips ONE leading `/` (lib/xlsx/xlsx.js, the `load` loop). Admission
* counts, and the admitted archive is rebuilt, under this name, so the
* stored spelling of a name cannot steer an entry past the counters.
*/
function excelJsEntryName(name) {
const parts = String(name).split('/');
const resolved = [];
for (let index = 0; index < parts.length; index += 1) {
const part = parts[index];
// JSZip keeps an empty first or last segment (a leading or trailing `/`).
if (part === '.' || (part === '' && index !== 0 && index !== parts.length - 1)) continue;
if (part === '..') resolved.pop();
else resolved.push(part);
}
const joined = resolved.join('/');
return joined[0] === '/' ? joined.slice(1) : joined;
}
// ExcelJS's own worksheet test, copied verbatim (lib/xlsx/xlsx.js): it is
// UNANCHORED, so `xl/xl/worksheets/sheet1.xml` and `.../sheet1.xml.x` are
// worksheets too. The older anchored test stays as a conservative superset.
const EXCELJS_WORKSHEET = /xl\/worksheets\/sheet(\d+)[.]xml/;
function isWorksheetName(name) {
return EXCELJS_WORKSHEET.test(name) || /^xl\/worksheets\/[^/]+\.xml$/i.test(name);
}
function inspectZipDirectory(bytes, overrides) {
const limits = mergedLimits(overrides);
if (bytes.length >= 4 && bytes[0] === 0xd0 && bytes[1] === 0xcf && bytes[2] === 0x11 && bytes[3] === 0xe0) {
fail('encrypted', 'Encrypted or legacy OLE workbooks cannot be previewed');
}
let eocd = -1;
const floor = Math.max(0, bytes.length - 65557);
for (let i = bytes.length - 22; i >= floor; i -= 1) {
if (u32(bytes, i) === 0x06054b50) {
eocd = i;
break;
}
}
if (eocd < 0) fail('malformed', 'Malformed XLSX ZIP directory');
const entryCount = u16(bytes, eocd + 10);
const directorySize = u32(bytes, eocd + 12);
const directoryOffset = u32(bytes, eocd + 16);
if (entryCount === 0xffff || directorySize === 0xffffffff || directoryOffset === 0xffffffff) {
fail('zip64', 'ZIP64 workbooks are not supported');
}
if (entryCount > limits.maxEntries) fail('entry-limit', `Workbook exceeds ${limits.maxEntries} ZIP entries`);
if (directoryOffset + directorySize > eocd) fail('malformed', 'Malformed XLSX central directory bounds');
const entries = [];
const excelJsNames = new Set();
let cursor = directoryOffset;
for (let i = 0; i < entryCount; i += 1) {
if (cursor + 46 > eocd || u32(bytes, cursor) !== 0x02014b50)
fail('malformed', 'Malformed XLSX central directory');
const compressedSize = u32(bytes, cursor + 20);
const declaredSize = u32(bytes, cursor + 24);
const nameLength = u16(bytes, cursor + 28);
const extraLength = u16(bytes, cursor + 30);
const commentLength = u16(bytes, cursor + 32);
const localHeaderOffset = u32(bytes, cursor + 42);
if (compressedSize === 0xffffffff || declaredSize === 0xffffffff)
fail('zip64', 'ZIP64 entries are not supported');
const end = cursor + 46 + nameLength + extraLength + commentLength;
if (end > eocd) fail('malformed', 'Malformed XLSX entry bounds');
const name = new TextDecoder().decode(bytes.subarray(cursor + 46, cursor + 46 + nameLength));
if (localHeaderOffset + 30 > directoryOffset || u32(bytes, localHeaderOffset) !== 0x04034b50) {
fail('malformed', 'Malformed XLSX local file header');
}
const localNameLength = u16(bytes, localHeaderOffset + 26);
const localExtraLength = u16(bytes, localHeaderOffset + 28);
const localNameEnd = localHeaderOffset + 30 + localNameLength;
if (localNameEnd + localExtraLength > directoryOffset) fail('malformed', 'Malformed XLSX local entry bounds');
// The compression-ratio cap divides by this declared size, so it must
// describe bytes that actually exist before the central directory.
if (localNameEnd + localExtraLength + compressedSize > directoryOffset) {
fail('malformed', 'XLSX entry declares more compressed bytes than the file holds');
}
const localName = new TextDecoder().decode(bytes.subarray(localHeaderOffset + 30, localNameEnd));
if (localName !== name) fail('malformed', 'XLSX local and central directory names do not match');
// JSZip keeps only the last of two entries that resolve to one name, and
// the rebuilt archive can hold only one, so both are refused.
const excelJsName = excelJsEntryName(name);
if (excelJsName === '') fail('malformed', 'XLSX entry name resolves to nothing');
if (excelJsNames.has(excelJsName)) fail('malformed', 'Two XLSX entries resolve to the same name');
excelJsNames.add(excelJsName);
entries.push({ name, excelJsName, compressedSize, declaredSize, localHeaderOffset });
cursor = end;
}
return { entries };
}
function featureForName(name) {
if (name.startsWith('xl/charts/')) return 'charts';
if (name.startsWith('xl/drawings/')) return 'drawings';
if (name.startsWith('xl/pivotCache/')) return 'pivotTables';
if (name.startsWith('xl/externalLinks/')) return 'externalLinks';
if (/vbaProject\.bin$/i.test(name)) return 'macros';
return null;
}
// Longest single XML tag the streaming counter will carry across chunks. Real
// worksheet tags are a few hundred bytes; this only stops a pathological tag
// from turning the carried tail into quadratic re-scanning.
const MAX_CARRIED_TAG = 256 * 1024;
// XML attribute syntax, read in order from just after the tag name. A quoted
// value may hold a raw `>` or the other quote character, so a value is always
// consumed whole; a tag is only understood if this walk reaches its `>`.
const XML_SPACE = '[ \\t\\r\\n]';
const ATTRIBUTE = new RegExp(
`${XML_SPACE}+([^ \\t\\r\\n=/>]+)${XML_SPACE}*=${XML_SPACE}*(?:"([^"]*)"|'([^']*)')`,
'y'
);
const TAG_END = new RegExp(`${XML_SPACE}*/?>`, 'y');
/**
* Parse the attributes of the tag whose name ends at `from` in `text`.
* Returns the attributes, or null when they do not parse cleanly up to the
* tag's closing `/>` or `>` (or a name repeats, which XML forbids).
*/
function readTagAttributes(text, from) {
const attributes = new Map();
let at = from;
for (;;) {
ATTRIBUTE.lastIndex = at;
const match = ATTRIBUTE.exec(text);
if (!match) break;
if (attributes.has(match[1])) return null;
attributes.set(match[1], match[2] ?? match[3]);
at = ATTRIBUTE.lastIndex;
}
TAG_END.lastIndex = at;
return TAG_END.test(text) ? attributes : null;
}
// ExcelJS expands a merge into one cell object per covered cell at load time,
// so a merge costs its AREA, not one tag.
function mergeArea(attributes) {
const range = parseRange(attributes.get('ref'));
if (!range) fail('malformed', 'Worksheet has a merged range that does not parse');
return (Math.abs(range.r2 - range.r1) + 1) * (Math.abs(range.c2 - range.c1) + 1);
}
// ExcelJS builds one column object for every index up to `
`, unclamped.
function checkColumnSpan(attributes) {
for (const name of ['min', 'max']) {
const value = attributes.get(name);
if (value === undefined) continue;
const index = Number(value);
if (!Number.isInteger(index) || index < 1 || index > MAX_COL) {
fail('malformed', `Worksheet column ${name} is outside 1-${MAX_COL}`);
}
}
}
// ExcelJS stores each row at `_rows[r - 1]`, and eachRow and `sheet.model`
// walk every index up to the largest, so the index a row CLAIMS is a cost.
function checkRowIndex(attributes) {
const value = attributes.get('r');
if (value === undefined) return;
if (!/^[0-9]+$/.test(value) || Number(value) < 1 || Number(value) > MAX_ROW) {
fail('malformed', `Worksheet row index is outside 1-${MAX_ROW}`);
}
}
// ExcelJS stores each sheet at `_worksheets[sheetId]` (an absent id parses to
// NaN, a plain property, so it is harmless).
function checkSheetId(attributes, limits) {
const value = attributes.get('sheetId');
if (value === undefined) return;
if (!/^[0-9]+$/.test(value) || Number(value) > limits.maxSheetId) {
fail('malformed', `Workbook sheetId is not a number up to ${limits.maxSheetId}`);
}
}
// Attribute values as the XML parser inside ExcelJS (saxes) hands them over:
// the predefined entities and character references decoded. Checking the raw
// text would let `[` hide a `[` and would overcount `"`.
function decodeXmlAttribute(value) {
return value.replace(/&(?:#x([0-9a-fA-F]+)|#([0-9]+)|(amp|lt|gt|quot|apos));/g, (entity, hex, dec, name) => {
if (name) return { amp: '&', lt: '<', gt: '>', quot: '"', apos: "'" }[name];
const code = hex ? Number.parseInt(hex, 16) : Number(dec);
return code <= 0x10ffff ? String.fromCodePoint(code) : entity;
});
}
// ExcelJS's `isDateFmt` strips `/\[[^\]]*]/g` from the code once per numeric
// cell, and that pattern rescans to the end of the code for every `[` with no
// later `]` (a 60,000-character code of `[` cost 2.2 s a cell; even 255
// characters added up at the cell cap). Once nothing follows the last `]`
// but text without `[`, each `[` stops at the next `]` and the scan is linear.
// The messages never echo the code.
function checkNumberFormat(attributes, limits) {
const raw = attributes.get('formatCode');
if (raw === undefined) return;
const code = decodeXmlAttribute(raw);
if (code.length > limits.maxNumFmtChars) {
fail('number-format', `Workbook has a number format longer than ${limits.maxNumFmtChars} characters`);
}
if (code.lastIndexOf('[') > code.lastIndexOf(']')) {
fail('number-format', 'Workbook has a number format with an unclosed bracket');
}
}
// Exactly the names ExcelJS hands to `_processMediaEntry` and to no parser:
// its other patterns are unanchored, so `xl/media/xl/drawings/a.xml` is still
// parsed as a drawing, and only a single `.` segment is pure bytes.
const EXCELJS_MEDIA = /^xl\/media\/[a-zA-Z0-9]+[.][a-zA-Z0-9]{3,4}$/;
const START_TAG = /<[A-Za-z_]/g;
function countStartTags(text) {
let count = 0;
START_TAG.lastIndex = 0;
while (START_TAG.exec(text)) count += 1;
return count;
}
function createXmlCounter(name, counts, limits) {
let tail = '';
const decoder = new TextDecoder();
let sheetCells = 0;
let sheetMerges = 0;
let sheetRows = 0;
const excelJsName = excelJsEntryName(name);
const worksheet = isWorksheetName(excelJsName);
const styles = excelJsName === 'xl/styles.xml';
const workbook = excelJsName === 'xl/workbook.xml';
const media = EXCELJS_MEDIA.test(excelJsName);
// Only these parts read tag attributes, so only they carry a whole tag.
const readsTags = worksheet || styles || workbook;
const addCells = (cells) => {
sheetCells += cells;
counts.cells += cells;
if (sheetCells > limits.maxCellsPerSheet || counts.cells > limits.maxCells)
fail('cell-limit', 'Workbook exceeds the cells limit');
};
return {
push(chunk, final) {
if (media) return;
const text = tail + decoder.decode(chunk, { stream: !final });
// `<` can never appear inside an attribute value, so every tag before the
// last `<` is complete. Carry everything from that `<` into the next scan
// so a tag cut by a chunk edge is always read in one piece. Any other part
// only needs its start tags counted, so it carries at most a trailing `<`
// (a long text node or a binary part is never mistaken for a huge tag).
let safeEnd = text.length;
if (!final && readsTags) safeEnd = Math.max(0, text.lastIndexOf('<'));
else if (!final && text.endsWith('<')) safeEnd = text.length - 1;
const scan = text.slice(0, safeEnd);
counts.elements += countStartTags(scan);
if (counts.elements > limits.maxElements) fail('element-limit', 'Workbook exceeds the XML elements limit');
if (worksheet) {
addCells((scan.match(/)/g) || []).length);
// ExcelJS keeps a Row object for every , with or without cells.
const rows = (scan.match(/
])/g) || []).length;
sheetRows += rows;
counts.rows += rows;
if (sheetRows > limits.maxRowsPerSheet || counts.rows > limits.maxRows)
fail('row-limit', 'Workbook exceeds the rows limit');
for (const match of scan.matchAll(/<(mergeCell|col|row)(?=[\s/>])/g)) {
const attributes = readTagAttributes(scan, match.index + match[0].length);
if (!attributes) fail('malformed', `Worksheet has a <${match[1]}> whose attributes do not parse`);
if (match[1] === 'row') {
checkRowIndex(attributes);
continue;
}
if (match[1] === 'col') {
checkColumnSpan(attributes);
continue;
}
sheetMerges += 1;
counts.merges += 1;
if (sheetMerges > limits.maxMergesPerSheet || counts.merges > limits.maxMerges)
fail('merge-limit', 'Workbook exceeds the merged ranges limit');
addCells(mergeArea(attributes));
}
}
if (workbook) {
// `` is the container; the lookahead keeps it from matching.
for (const match of scan.matchAll(/])/g)) {
const attributes = readTagAttributes(scan, match.index + match[0].length);
if (!attributes) fail('malformed', 'Workbook has a whose attributes do not parse');
checkSheetId(attributes, limits);
}
}
if (styles) {
// Every counts, cellStyleXfs included: tracking which list a tag
// sits in can be desynced by a closing tag inside an XML comment.
counts.styles += (scan.match(/])/g) || []).length;
if (counts.styles > limits.maxStyles) fail('style-limit', 'Workbook exceeds the cell styles limit');
// Every , cellXfs' and dxfs' alike; the lookahead skips .
for (const match of scan.matchAll(/])/g)) {
const attributes = readTagAttributes(scan, match.index + match[0].length);
if (!attributes) fail('malformed', 'Workbook has a whose attributes do not parse');
checkNumberFormat(attributes, limits);
}
}
tail = text.slice(safeEnd);
if (tail.length > MAX_CARRIED_TAG) fail('malformed', 'Workbook XML has an oversized tag');
},
};
}
function admitXlsx(bytes, zipApi, overrides) {
const limits = mergedLimits(overrides);
const directory = inspectZipDirectory(bytes, limits);
const directoryByName = new Map(directory.entries.map((entry) => [entry.name, entry]));
const expectedEntries = new Map();
for (const entry of directory.entries) expectedEntries.set(entry.name, (expectedEntries.get(entry.name) || 0) + 1);
const streamedEntries = new Map();
const inflatedEntries = Object.create(null);
const counts = { worksheets: 0, cells: 0, rows: 0, merges: 0, styles: 0, elements: 0 };
const features = new Set();
let totalInflated = 0;
let seenEntries = 0;
let thrown;
const unzip = new zipApi.Unzip((file) => {
if (!directoryByName.has(file.name)) fail('malformed', 'Local XLSX entry is absent from the central directory');
// The admitted archive is rebuilt from these entries, which cannot hold two
// files under one name, so a duplicate name is refused rather than dropped.
if (streamedEntries.has(file.name)) fail('malformed', 'Duplicate XLSX entry name');
streamedEntries.set(file.name, (streamedEntries.get(file.name) || 0) + 1);
seenEntries += 1;
if (seenEntries > limits.maxEntries) fail('entry-limit', 'Workbook exceeds the ZIP entries limit');
// Everything below keys on the name ExcelJS will see, never the stored one.
const name = directoryByName.get(file.name).excelJsName;
if (isWorksheetName(name)) {
counts.worksheets += 1;
if (counts.worksheets > limits.maxWorksheets) fail('worksheet-limit', 'Workbook exceeds the worksheet limit');
}
const feature = featureForName(name);
if (feature) features.add(feature);
const counter = createXmlCounter(name, counts, limits);
let entryInflated = 0;
const chunks = [];
file.ondata = (error, chunk, final) => {
if (error) throw error;
chunks.push(chunk.slice());
entryInflated += chunk.length;
totalInflated += chunk.length;
if (entryInflated > limits.maxEntryBytes) fail('entry-size', 'Inflated ZIP entry exceeds the entry limit');
if (totalInflated > limits.maxInflatedBytes) fail('inflated-size', 'Workbook exceeds the inflated bytes limit');
const compressed = directoryByName.get(file.name)?.compressedSize || 1;
if (entryInflated / Math.max(1, compressed) > limits.maxCompressionRatio) {
fail('compression-ratio', 'ZIP entry exceeds the compression ratio limit');
}
counter.push(chunk, final);
if (final) {
const data = new Uint8Array(entryInflated);
let offset = 0;
for (const part of chunks) {
data.set(part, offset);
offset += part.length;
}
chunks.length = 0;
inflatedEntries[name] = data;
}
};
file.start();
});
unzip.register(zipApi.UnzipInflate);
try {
const inputChunkBytes = 64 * 1024;
for (let offset = 0; offset < bytes.length && !thrown; offset += inputChunkBytes) {
const end = Math.min(bytes.length, offset + inputChunkBytes);
unzip.push(bytes.subarray(offset, end), end === bytes.length);
}
} catch (error) {
thrown = error;
}
if (thrown) throw thrown;
for (const [name, count] of expectedEntries) {
if (streamedEntries.get(name) !== count) fail('malformed', 'Central XLSX entry was not streamed for admission');
}
for (const name of streamedEntries.keys()) {
if (!(directoryByName.get(name).excelJsName in inflatedEntries)) {
fail('malformed', 'XLSX entry did not finish streaming for admission');
}
}
return { counts, features: Array.from(features), inflatedBytes: totalInflated, entries: inflatedEntries };
}
// The ONLY bytes ExcelJS may parse: the entries admission itself inflated and
// counted, re-packed uncompressed. Its size is ~ the inflated total (already
// capped at LIMITS.maxInflatedBytes) plus per-entry headers.
function buildAdmittedArchive(admission, zipApi) {
return zipApi.zipSync(admission.entries, { level: 0 });
}
function parseCellRef(ref) {
const match = /^\$?([A-Z]{1,3})\$?([1-9]\d*)$/i.exec(String(ref || ''));
if (!match) return null;
let col = 0;
for (const char of match[1].toUpperCase()) col = col * 26 + char.charCodeAt(0) - 64;
const row = Number(match[2]);
return row <= MAX_ROW && col <= MAX_COL ? { row, col } : null;
}
function parseRange(range) {
const parts = String(range).split(':');
const start = parseCellRef(parts[0]);
const end = parseCellRef(parts[1] || parts[0]);
return start && end ? { r1: start.row, c1: start.col, r2: end.row, c2: end.col } : null;
}
function deriveExtent(cells, merges) {
let rows = 0;
let cols = 0;
for (const ref of cells) {
const cell = parseCellRef(ref);
if (cell) {
rows = Math.max(rows, cell.row);
cols = Math.max(cols, cell.col);
}
}
for (const merge of merges) {
const range = parseRange(merge);
if (range) {
rows = Math.max(rows, range.r2);
cols = Math.max(cols, range.c2);
}
}
return { rows, cols };
}
function createSparseAxis(count, defaultSize, overrides) {
const sorted = Array.from(overrides || [])
.filter(([index, size]) => index >= 1 && index <= count && Number.isFinite(size))
.map(([index, size]) => [index, Math.max(0, size)])
.sort((a, b) => a[0] - b[0]);
const base = Math.max(0, defaultSize);
// deltas[i] is the summed size change of overrides[0..i], so an offset is
// one binary search instead of a walk over every override.
const deltas = [];
let delta = 0;
for (const [, size] of sorted) {
delta += size - base;
deltas.push(delta);
}
return { count: Math.max(0, count), defaultSize: base, overrides: sorted, deltas };
}
// Number of overrides whose index is below `bounded`.
function overridesBefore(overrides, bounded) {
let low = 0;
let high = overrides.length;
while (low < high) {
const mid = (low + high) >> 1;
if (overrides[mid][0] < bounded) low = mid + 1;
else high = mid;
}
return low;
}
function axisOffset(axis, index) {
const bounded = Math.max(1, Math.min(axis.count + 1, index));
const before = overridesBefore(axis.overrides, bounded);
return (bounded - 1) * axis.defaultSize + (before ? axis.deltas[before - 1] : 0);
}
function axisIndexAt(axis, offset) {
let low = 1;
let high = Math.max(1, axis.count);
const target = Math.max(0, offset);
while (low < high) {
const mid = Math.floor((low + high + 1) / 2);
if (axisOffset(axis, mid) <= target) low = mid;
else high = mid - 1;
}
return low;
}
function computeViewport(axis, offset, viewportSize, overscan) {
const pad = Math.max(0, overscan || 0);
const start = Math.max(1, axisIndexAt(axis, offset) - pad);
const end = Math.min(axis.count, axisIndexAt(axis, offset + Math.max(0, viewportSize)) + pad);
return [start, end];
}
function intersectingMerges(merges, viewport) {
return merges.filter((merge) => {
const range = parseRange(merge);
return (
range &&
range.r1 <= viewport.r2 &&
range.r2 >= viewport.r1 &&
range.c1 <= viewport.c2 &&
range.c2 >= viewport.c1
);
});
}
// Rounded to whole milliseconds: `new Date(fraction)` truncates, which turned
// midnight minus a float error into the previous day.
function excelDate(serial, date1904) {
if (date1904) return new Date(Math.round(Date.UTC(1904, 0, 1) + Number(serial) * 86400000));
const numeric = Number(serial);
const adjusted = numeric >= 60 ? numeric - 1 : numeric;
return new Date(Math.round(Date.UTC(1899, 11, 31) + adjusted * 86400000));
}
function isDateValue(value) {
return Object.prototype.toString.call(value) === '[object Date]' && Number.isFinite(value.getTime());
}
// Inverse of ExcelJS's own `excelToDate()` (utils.js), which builds the Date
// from the serial in UTC. Recovering the serial keeps the result independent of
// the viewer's timezone; `String(date)` rendered it in local time, a day early
// at any negative UTC offset.
function dateToSerial(date, date1904) {
return 25569 + date.getTime() / 86400000 - (date1904 ? 1462 : 0);
}
const DATE_FORMAT = /^[ymd\-/ ]+$/i;
// An optional trailing AM/PM (built-in format 18 is `h:mm AM/PM`).
const TIME_FORMAT = /^[hms: ]+(?:AM\/PM)?$/i;
const DATE_TIME_FORMAT = /^[ymdhis\-/: ]+$/i;
function isFormulaValue(value) {
return 'formula' in value || 'sharedFormula' in value;
}
// Cut to LIMITS.maxCellTextChars, ending in an ellipsis, never between the
// halves of a surrogate pair.
function capCellText(text) {
const max = LIMITS.maxCellTextChars;
if (text.length <= max) return text;
let end = max - 1;
const last = text.charCodeAt(end - 1);
if (last >= 0xd800 && last <= 0xdbff) end -= 1;
return `${text.slice(0, end)}…`;
}
// Joins rich-text runs only until the cap is passed, so a long run is never
// copied whole once per cell. It also visits at most maxCellTextChars + 1
// runs: an empty run (``, 4 bytes) adds no text, so a length check alone
// walked every run, for every cell sharing the string, on every tile. Any
// maxCellTextChars + 1 non-empty runs already pass the cap.
function richTextPrefix(runs) {
let text = '';
const visit = Math.min(runs.length, LIMITS.maxCellTextChars + 1);
for (let index = 0; index < visit; index += 1) {
if (text.length > LIMITS.maxCellTextChars) break;
const run = runs[index];
if (typeof run?.text === 'string') text += run.text.slice(0, LIMITS.maxCellTextChars + 1);
}
return text;
}
// Excel displays at most 15 significant digits, so `=0.1+0.2` shows 0.3,
// never the binary float's 0.30000000000000004.
function generalNumber(value) {
return Number.isFinite(value) ? String(Number(value.toPrecision(15))) : String(value);
}
const UNSUPPORTED_FORMAT_PREFIX = 'Unsupported number format: ';
/**
* Folds every unsupported number format warning into one counted entry when
* there is more than one, so a sheet with a code per cell cannot grow the
* notice bar without bound. A lone warning is kept as is; its code is at
* most 255 characters (admission refuses longer ones). Order is preserved.
*/
function foldWarnings(warnings) {
const formats = warnings.filter((warning) => String(warning).startsWith(UNSUPPORTED_FORMAT_PREFIX));
if (formats.length < 2) return warnings.slice();
const folded = [];
let placed = false;
for (const warning of warnings) {
if (!String(warning).startsWith(UNSUPPORTED_FORMAT_PREFIX)) folded.push(warning);
else if (!placed) {
folded.push(`${formats.length} unsupported number formats`);
placed = true;
}
}
return folded;
}
/**
* A cell's display text and any warning. The text is ALWAYS at most
* `LIMITS.maxCellTextChars` characters, whatever shape the value has.
*/
function formatCellValue(value, format, date1904) {
const formatted = formatCellValueUncapped(value, format, date1904);
return formatted.text.length > LIMITS.maxCellTextChars
? Object.assign({}, formatted, { text: capCellText(formatted.text) })
: formatted;
}
// Every non-scalar shape ExcelJS loads a cell value as. Anything not handled
// here would otherwise reach String() and render as "[object Object]".
function formatCellValueUncapped(value, format, date1904) {
if (value === null || value === undefined) return { text: '' };
if (isDateValue(value)) {
const code = String(format || 'General');
const known = DATE_FORMAT.test(code) || TIME_FORMAT.test(code) || DATE_TIME_FORMAT.test(code);
const serial = dateToSerial(value, date1904);
if (known) return formatCellValueUncapped(serial, code, date1904);
const fallback = serial % 1 === 0 ? 'yyyy-mm-dd' : 'yyyy-mm-dd hh:mm';
const formatted = formatCellValueUncapped(serial, fallback, date1904);
return /^General$/i.test(code)
? formatted
: { text: formatted.text, warning: `${UNSUPPORTED_FORMAT_PREFIX}${code}` };
}
if (typeof value === 'object') {
if (isFormulaValue(value)) {
if (value.result !== undefined && value.result !== null) {
return formatCellValueUncapped(value.result, format, date1904);
}
const source = typeof value.formula === 'string' ? `=${value.formula}` : '';
return { text: source, warning: 'Formula has no cached result' };
}
if (typeof value.error === 'string') return { text: value.error };
if (Array.isArray(value.richText)) {
return { text: richTextPrefix(value.richText) };
}
// Hyperlink: the display text, never the target. The text may be rich.
if ('text' in value) return formatCellValueUncapped(value.text, 'General', date1904);
return { text: '', warning: 'Unsupported cell value' };
}
const code = String(format || 'General');
if (typeof value !== 'number') return { text: String(value) };
if (/^General$/i.test(code)) return { text: generalNumber(value) };
if (DATE_FORMAT.test(code)) {
const date = excelDate(value, Boolean(date1904));
const yyyy = date.getUTCFullYear();
const mm = String(date.getUTCMonth() + 1).padStart(2, '0');
const dd = String(date.getUTCDate()).padStart(2, '0');
return { text: `${yyyy}-${mm}-${dd}` };
}
if (TIME_FORMAT.test(code)) {
const seconds = Math.round((value - Math.floor(value)) * 86400) % 86400;
const hours = Math.floor(seconds / 3600);
const mm = String(Math.floor((seconds % 3600) / 60)).padStart(2, '0');
const ss = String(seconds % 60).padStart(2, '0');
if (/AM\/PM$/i.test(code)) {
const hour12 = String(hours % 12 || 12);
const hh = /hh/i.test(code) ? hour12.padStart(2, '0') : hour12;
const clock = /s/i.test(code) ? `${hh}:${mm}:${ss}` : `${hh}:${mm}`;
return { text: `${clock} ${hours < 12 ? 'AM' : 'PM'}` };
}
const hh = String(hours).padStart(2, '0');
return { text: `${hh}:${mm}:${ss}` };
}
if (DATE_TIME_FORMAT.test(code)) {
const date = excelDate(value, Boolean(date1904));
const yyyy = date.getUTCFullYear();
const mm = String(date.getUTCMonth() + 1).padStart(2, '0');
const dd = String(date.getUTCDate()).padStart(2, '0');
const hh = String(date.getUTCHours()).padStart(2, '0');
const minutes = String(date.getUTCMinutes()).padStart(2, '0');
return { text: `${yyyy}-${mm}-${dd} ${hh}:${minutes}` };
}
const percent = code.includes('%');
// toLocaleString throws above 100 fraction digits; Excel itself caps at 30.
const decimals = Math.min(code.match(/\.([0#]+)/)?.[1].length || 0, 30);
const numericPattern = /^[€£¥$]?[#,0]+(?:\.[0#]+)?%?$/;
if (numericPattern.test(code)) {
const currency = /^[€£¥$]/.exec(code)?.[0] || '';
const numeric = percent ? value * 100 : value;
const useGrouping = code.includes(',');
return {
text:
currency +
numeric.toLocaleString('en-US', {
useGrouping,
minimumFractionDigits: decimals,
maximumFractionDigits: decimals,
}) +
(percent ? '%' : ''),
};
}
return { text: generalNumber(value), warning: `${UNSUPPORTED_FORMAT_PREFIX}${code}` };
}
// Colour resolution --------------------------------------------------------
//
// ExcelJS surfaces theme and indexed palette colours WITHOUT an `argb` key
// (`{theme,tint}` / `{indexed}`), and theme colours are what Excel emits by
// default. Dropping them left cells rendering the workbook's font colour on
// the skin's own background, which is how black-on-dark (invisible) text got
// shipped. Everything below is pure so `test/spreadsheet-xlsx-core.test.ts`
// can pin it without a DOM.
// styles.xml `theme="N"` order. NOTE: theme1.xml's lists
// dk1, lt1, dk2, lt2 — indices 0/1 and 2/3 are SWAPPED between the two.
const THEME_SLOT_ORDER = Object.freeze([
'lt1',
'dk1',
'lt2',
'dk2',
'accent1',
'accent2',
'accent3',
'accent4',
'accent5',
'accent6',
'hlink',
'folHlink',
]);
// Default Office theme, used when theme1.xml is missing or unparseable.
const DEFAULT_THEME_PALETTE = Object.freeze([
'#ffffff',
'#000000',
'#eeece1',
'#1f497d',
'#4f81bd',
'#c0504d',
'#9bbb59',
'#8064a2',
'#4bacc6',
'#f79646',
'#0000ff',
'#800080',
]);
// Legacy 64-entry indexed palette. Indices 64/65 are the "auto" foreground and
// background sentinels and deliberately have no entry here.
// prettier-ignore
const INDEXED_PALETTE = Object.freeze([
'#000000', '#ffffff', '#ff0000', '#00ff00', '#0000ff', '#ffff00', '#ff00ff', '#00ffff',
'#000000', '#ffffff', '#ff0000', '#00ff00', '#0000ff', '#ffff00', '#ff00ff', '#00ffff',
'#800000', '#008000', '#000080', '#808000', '#800080', '#008080', '#c0c0c0', '#808080',
'#9999ff', '#993366', '#ffffcc', '#ccffff', '#660066', '#ff8080', '#0066cc', '#ccccff',
'#000080', '#ff00ff', '#ffff00', '#00ffff', '#800080', '#800000', '#008080', '#0000ff',
'#00ccff', '#ccffff', '#ccffcc', '#ffff99', '#99ccff', '#ff99cc', '#cc99ff', '#ffcc99',
'#3366ff', '#33cccc', '#99cc00', '#ffcc00', '#ff9900', '#ff6600', '#666699', '#969696',
'#003366', '#339966', '#003300', '#333300', '#993300', '#993366', '#333399', '#333333',
]);
const MIN_CONTRAST_RATIO = 4.5;
// A workbook with no fill renders on Excel's implicit white sheet background,
// never on the viewer skin's `var(--bg-primary)`.
const IMPLICIT_SHEET_BACKGROUND = '#ffffff';
const IMPLICIT_SHEET_FOREGROUND = '#000000';
function hexToRgb(hex) {
const match = /^#([0-9a-f]{2})([0-9a-f]{2})([0-9a-f]{2})$/i.exec(String(hex || ''));
if (!match) return null;
return [Number.parseInt(match[1], 16), Number.parseInt(match[2], 16), Number.parseInt(match[3], 16)];
}
function rgbToHex(rgb) {
let hex = '#';
for (const channel of rgb) {
const bounded = Math.max(0, Math.min(255, Math.round(channel)));
hex += (bounded < 16 ? '0' : '') + bounded.toString(16);
}
return hex;
}
function rgbToHsl(rgb) {
const r = rgb[0] / 255;
const g = rgb[1] / 255;
const b = rgb[2] / 255;
const max = Math.max(r, g, b);
const min = Math.min(r, g, b);
const l = (max + min) / 2;
if (max === min) return { h: 0, s: 0, l };
const delta = max - min;
const s = l > 0.5 ? delta / (2 - max - min) : delta / (max + min);
let h;
if (max === r) h = (g - b) / delta + (g < b ? 6 : 0);
else if (max === g) h = (b - r) / delta + 2;
else h = (r - g) / delta + 4;
return { h: h / 6, s, l };
}
function hueToChannel(p, q, hue) {
let t = hue;
if (t < 0) t += 1;
if (t > 1) t -= 1;
if (t < 1 / 6) return p + (q - p) * 6 * t;
if (t < 1 / 2) return q;
if (t < 2 / 3) return p + (q - p) * (2 / 3 - t) * 6;
return p;
}
function hslToRgb(hsl) {
if (hsl.s === 0) {
const gray = hsl.l * 255;
return [gray, gray, gray];
}
const q = hsl.l < 0.5 ? hsl.l * (1 + hsl.s) : hsl.l + hsl.s - hsl.l * hsl.s;
const p = 2 * hsl.l - q;
return [
hueToChannel(p, q, hsl.h + 1 / 3) * 255,
hueToChannel(p, q, hsl.h) * 255,
hueToChannel(p, q, hsl.h - 1 / 3) * 255,
];
}
// Excel tint acts on HSL luminance: negative darkens, positive lightens.
function applyTint(hex, tint) {
const amount = Number(tint);
if (!Number.isFinite(amount) || amount === 0) return hex;
const rgb = hexToRgb(hex);
if (!rgb) return hex;
const bounded = Math.max(-1, Math.min(1, amount));
const hsl = rgbToHsl(rgb);
const luminance = bounded < 0 ? hsl.l * (1 + bounded) : hsl.l * (1 - bounded) + bounded;
return rgbToHex(hslToRgb({ h: hsl.h, s: hsl.s, l: Math.max(0, Math.min(1, luminance)) }));
}
function parseSchemeColor(fragment) {
const srgb = /<(?:[A-Za-z0-9_]+:)?srgbClr\b[^>]*\bval="([0-9A-Fa-f]{6})"/.exec(fragment);
if (srgb) return `#${srgb[1].toLowerCase()}`;
const sys = /<(?:[A-Za-z0-9_]+:)?sysClr\b[^>]*\blastClr="([0-9A-Fa-f]{6})"/.exec(fragment);
if (sys) return `#${sys[1].toLowerCase()}`;
return undefined;
}
// Real theme1.xml files are under 10 KB. The scheme and slot patterns below
// rescan to the end of the text for every opening tag that has no close, so a
// padded theme is quadratic (1 MB of `` took 21.8 s); above this
// many characters (UTF-16 code units of the decoded XML) the default palette
// is used instead.
const MAX_THEME_XML_CHARS = 64 * 1024;
function parseThemePalette(xml) {
const palette = DEFAULT_THEME_PALETTE.slice();
const text = typeof xml === 'string' ? xml : '';
if (text.length > MAX_THEME_XML_CHARS) return palette;
const scheme = /<(?:[A-Za-z0-9_]+:)?clrScheme\b[^>]*>([\s\S]*?)<\/(?:[A-Za-z0-9_]+:)?clrScheme\s*>/.exec(text);
if (!scheme) return palette;
// Fresh pattern per call: a shared /g regex would carry lastIndex across calls.
const slots =
/<(?:[A-Za-z0-9_]+:)?(lt1|dk1|lt2|dk2|accent[1-6]|hlink|folHlink)\b[^>]*>([\s\S]*?)<\/(?:[A-Za-z0-9_]+:)?\1\s*>/g;
let match = slots.exec(scheme[1]);
while (match) {
const index = THEME_SLOT_ORDER.indexOf(match[1]);
const resolved = index >= 0 ? parseSchemeColor(match[2]) : undefined;
if (resolved) palette[index] = resolved;
match = slots.exec(scheme[1]);
}
return palette;
}
function themePaletteOrDefault(palette) {
return Array.isArray(palette) && palette.length === THEME_SLOT_ORDER.length ? palette : DEFAULT_THEME_PALETTE;
}
// Accepts every colour shape ExcelJS emits: {argb}, {theme,tint}, {indexed}.
function resolveColor(color, palette) {
if (!color || typeof color !== 'object') return undefined;
const argb = color.argb;
if (typeof argb === 'string' && /^[A-Fa-f0-9]{8}$/.test(argb)) return `#${argb.slice(2).toLowerCase()}`;
const theme = color.theme;
if (Number.isInteger(theme) && theme >= 0 && theme < THEME_SLOT_ORDER.length) {
const base = themePaletteOrDefault(palette)[theme];
return typeof base === 'string' ? applyTint(base, color.tint) : undefined;
}
const indexed = color.indexed;
if (Number.isInteger(indexed) && indexed >= 0 && indexed < INDEXED_PALETTE.length) return INDEXED_PALETTE[indexed];
return undefined;
}
function channelLuminance(channel) {
const value = channel / 255;
return value <= 0.03928 ? value / 12.92 : Math.pow((value + 0.055) / 1.055, 2.4);
}
function relativeLuminance(hex) {
const rgb = hexToRgb(hex);
if (!rgb) return 0;
return 0.2126 * channelLuminance(rgb[0]) + 0.7152 * channelLuminance(rgb[1]) + 0.0722 * channelLuminance(rgb[2]);
}
function contrastRatio(a, b) {
const first = relativeLuminance(a);
const second = relativeLuminance(b);
return (Math.max(first, second) + 0.05) / (Math.min(first, second) + 0.05);
}
function ensureContrast(foreground, background, minRatio) {
const minimum = Number.isFinite(minRatio) ? minRatio : MIN_CONTRAST_RATIO;
if (contrastRatio(foreground, background) >= minimum) return foreground;
return contrastRatio('#000000', background) >= contrastRatio('#ffffff', background) ? '#000000' : '#ffffff';
}
// Returns a SELF-CONSISTENT pair, or nothing at all. Emitting only one half is
// what let workbook text land on the skin's background and vanish.
function resolveCellColors(fillColor, fontColor, palette) {
const fill = resolveColor(fillColor, palette);
const font = resolveColor(fontColor, palette);
if (!fill && !font) return {};
const background = fill || IMPLICIT_SHEET_BACKGROUND;
return { background, foreground: ensureContrast(font || IMPLICIT_SHEET_FOREGROUND, background) };
}
global.CodemanSpreadsheetXlsxCore = Object.freeze({
LIMITS,
MAX_ROW,
MAX_COL,
XlsxPreviewError,
inspectZipDirectory,
createXmlCounter,
admitXlsx,
buildAdmittedArchive,
excelJsEntryName,
parseCellRef,
parseRange,
deriveExtent,
createSparseAxis,
axisOffset,
axisIndexAt,
computeViewport,
intersectingMerges,
formatCellValue,
foldWarnings,
DEFAULT_THEME_PALETTE,
INDEXED_PALETTE,
MIN_CONTRAST_RATIO,
parseThemePalette,
resolveColor,
contrastRatio,
ensureContrast,
resolveCellColors,
});
})(typeof self !== 'undefined' ? self : globalThis);