diff --git a/package.json b/package.json index fb0fdb7..9094c84 100644 --- a/package.json +++ b/package.json @@ -33,12 +33,13 @@ "typecheck": "vue-tsc --noEmit", "lint": "eslint src demo", "test": "vitest run", + "test:engine": "tsx --test \"tests/engine/test_*.ts\"", "test:watch": "vitest", "test:fixtures": "tsx tests/engine/run-all-fixtures.ts", "test:calculator": "tsx tests/engine/run-calculator-tests.ts", "test:evaluator": "tsx tests/engine/run-evaluator-tests.ts", "test:functions": "tsx tests/engine/test-functions.ts", - "test:all": "vitest run && tsx tests/engine/run-all-fixtures.ts" + "test:all": "vitest run && yarn run test:engine && tsx tests/engine/run-all-fixtures.ts" }, "peerDependencies": { "gui-chat-protocol": "^2.0.0", diff --git a/src/engine/calculator.ts b/src/engine/calculator.ts index 1721bb3..0464594 100644 --- a/src/engine/calculator.ts +++ b/src/engine/calculator.ts @@ -4,18 +4,35 @@ * Core calculation engine with circular reference detection and cross-sheet support */ -import { formatNumber } from "./formatter"; -import { columnToIndex } from "./parser"; +import { formatCellForDisplay } from "./cellFormatting"; import { evaluateFormula as evaluateFormulaFn } from "./evaluator"; +import { expandRangeOrCell, parseSingleCellRef } from "./formulaRefs"; import { parseDate, getDefaultDateFormat } from "./date-parser"; -import type { - SheetData, - CellValue, - CalculatedSheet, - CalculationError, - FormulaInfo, - SpreadsheetCell, -} from "./types"; +import type { SheetData, CellValue, CalculatedSheet, CalculationError, FormulaInfo, SpreadsheetCell, CalculateOptions } from "./types"; +import { isObj } from "./guards"; +import { isEmptyCell } from "./cellEmpty"; +import { errorMessage } from "./guards"; +import { classifyThrownError, invalidRefError } from "./formulaError"; +import { isSpreadsheetErrorValue, spreadsheetError } from "./spreadsheet-errors"; + +// The grid a reference should read from, plus the sheet-name-stripped ref and +// whether it points at the sheet currently being calculated (which decides +// whether recursive formula evaluation is allowed for the cell it lands on). +interface ResolvedSheetRef { + sheetData: (SpreadsheetCell | CellValue)[][]; + ref: string; + isCurrentSheet: boolean; +} + +// Where a cell sits in the grid CURRENTLY being calculated, carrying the row it +// belongs to so a formula's result is written back without re-indexing. Only a +// same-sheet cell has one: a cross-sheet read is resolved by that sheet's own +// calculateSheet pass, so it has no position here to recurse from. +interface CellPosition { + cells: any[]; + row: number; + col: number; +} /** * Normalize malformed data structures @@ -24,7 +41,7 @@ import type { * @param data - Potentially malformed sheet data * @returns Normalized 2D array */ -function normalizeData(data: any): SpreadsheetCell[][] { +export function normalizeData(data: any): SpreadsheetCell[][] { // Handle null/undefined if (!data) { return []; @@ -48,7 +65,7 @@ function normalizeData(data: any): SpreadsheetCell[][] { // If data is a flat array of cell objects, convert to 2D by pairing cells // Pattern: [cell1, cell2, cell3, cell4] -> [[cell1, cell2], [cell3, cell4]] // This handles the case where models output flat arrays instead of rows - if (typeof data[0] === "object" && data[0] !== null) { + if (isObj(data[0])) { const rows: SpreadsheetCell[][] = []; for (let i = 0; i < data.length; i += 2) { const row = [data[i]]; @@ -71,11 +88,11 @@ function normalizeData(data: any): SpreadsheetCell[][] { * @param data - Raw sheet data * @returns Processed data with dates converted to serial numbers */ -function preprocessDates(data: SpreadsheetCell[][]): SpreadsheetCell[][] { +function preprocessDates(data: SpreadsheetCell[][], preferDDMMYYYY: boolean): SpreadsheetCell[][] { return data.map((row) => row.map((cell) => { // Skip if not a cell object or if it has a formula - if (!cell || typeof cell !== "object" || !("v" in cell)) { + if (!isObj(cell) || !("v" in cell)) { return cell; } @@ -83,13 +100,13 @@ function preprocessDates(data: SpreadsheetCell[][]): SpreadsheetCell[][] { // Only parse strings that aren't formulas if (typeof value === "string" && !value.startsWith("=")) { - const dateSerial = parseDate(value); + const dateSerial = parseDate(value, preferDDMMYYYY); if (dateSerial !== null) { // It's a date! Convert to serial number return { v: dateSerial, - f: cell.f || getDefaultDateFormat(value), // Use existing format or detect from input + f: cell.f || getDefaultDateFormat(value, preferDDMMYYYY), // Use existing format or detect from input }; } } @@ -100,6 +117,12 @@ function preprocessDates(data: SpreadsheetCell[][]): SpreadsheetCell[][] { ); } +// `skipFormatting` is internal, not a public knob: a cross-sheet reference +// computes its target sheet only to READ values, so the display-formatting pass +// is skipped there — a date must stay a serial, not become "03/04/2025" that a +// downstream parseFloat reads as 3 (issue #2332). +type SheetCalculateOptions = CalculateOptions & { skipFormatting?: boolean }; + /** * Calculate formulas in a single sheet * @@ -107,20 +130,19 @@ function preprocessDates(data: SpreadsheetCell[][]): SpreadsheetCell[][] { * @param allSheets - All sheets for cross-sheet references * @returns Calculated sheet with formulas evaluated */ -export function calculateSheet( - sheet: SheetData, - allSheets?: SheetData[], -): CalculatedSheet { +export function calculateSheet(sheet: SheetData, allSheets?: SheetData[], options: SheetCalculateOptions = {}): CalculatedSheet { + const preferDDMMYYYY = options.preferDDMMYYYY ?? false; + const skipFormatting = options.skipFormatting ?? false; // Normalize malformed data structures first const normalizedData = normalizeData(sheet.data); // Pre-process dates before calculation - const processedData = preprocessDates(normalizedData); + const processedData = preprocessDates(normalizedData, preferDDMMYYYY); // Also preprocess all sheets if provided const processedAllSheets = allSheets?.map((s) => ({ ...s, - data: preprocessDates(normalizeData(s.data)), + data: preprocessDates(normalizeData(s.data), preferDDMMYYYY), })); const data = processedData; @@ -138,12 +160,50 @@ export function calculateSheet( // Track cells being calculated to detect circular references const calculating = new Set(); + // Cells whose result is already stored in `calculated`, of ANY type. A number + // check alone missed string/error results, so a cell referenced before the + // top loop reached it was re-evaluated (and, once formulas can throw, would + // re-emit its error). Membership here means "read the cached value, do not + // re-run". + const evaluated = new Set(); + + // Evaluate one formula cell, guarding circular references, caching the result, + // and turning a thrown failure into a typed errors[] entry plus the Excel + // error value in the cell — never a swallowed bare string/number (#2359). + const resolveFormulaCell = (formulaText: string, { cells, row, col }: CellPosition): CellValue => { + const cellKey = `${row},${col}`; + if (calculating.has(cellKey)) { + errors.push({ cell: { row, col }, formula: formulaText, error: "Circular reference detected", type: "circular" }); + return 0; + } + if (evaluated.has(cellKey)) return cells[col]; + calculating.add(cellKey); + try { + const result = evaluateFormula(formulaText.substring(1)); // drop leading "=" + cells[col] = result; + return result; + } catch (error) { + const { type, display } = classifyThrownError(error); + errors.push({ cell: { row, col }, formula: formulaText, error: errorMessage(error), type }); + // The error VALUE, not its text: a cell that reads this one must see a + // real error, and the display pass renders it back to `#DIV/0!`. + const errorValue = spreadsheetError(display); + cells[col] = errorValue; + return errorValue; + } finally { + calculating.delete(cellKey); + evaluated.add(cellKey); + } + }; // Helper to extract raw value from cell with recursive formula evaluation - const getRawValue = (cell: any, row?: number, col?: number): CellValue => { + const getRawValue = (cell: any, position?: CellPosition): CellValue => { // Handle null/undefined cells - treat as 0 if (cell === null || cell === undefined) return 0; + // An already-calculated cell can hold a formula error; it stays an error. + if (isSpreadsheetErrorValue(cell)) return cell; + if (typeof cell === "number") return cell; // Handle string values (for legacy or calculated cells) @@ -175,65 +235,22 @@ export function calculateSheet( } // Handle new cell format {v, f} - if (typeof cell === "object" && cell !== null && "v" in cell) { + if (isObj(cell) && "v" in cell) { const value = cell.v; // If value is a string starting with "=", it's a formula if (typeof value === "string" && value.startsWith("=")) { - // Check if we have row/col info to evaluate recursively - if (row !== undefined && col !== undefined) { - const cellKey = `${row},${col}`; - - // Check for circular reference - if (calculating.has(cellKey)) { - console.warn( - `Circular reference detected at row ${row}, col ${col}`, - ); - errors.push({ - cell: { row, col }, - formula: value, - error: "Circular reference detected", - type: "circular", - }); - return 0; - } - - // Check if already calculated (result is cached as a number) - const calculatedCell = calculated[row][col]; - if (typeof calculatedCell === "number") { - return calculatedCell; - } - - // Recursively evaluate the formula - calculating.add(cellKey); - try { - const formula = value.substring(1); // Remove "=" prefix - const result = evaluateFormula(formula); - calculating.delete(cellKey); - - // Cache the calculated result (preserve strings and numbers) - calculated[row][col] = result; - - return result; - } catch (error) { - calculating.delete(cellKey); - console.error( - `Error evaluating formula at row ${row}, col ${col}:`, - error, - ); - errors.push({ - cell: { row, col }, - formula: value, - error: error instanceof Error ? error.message : String(error), - type: "unknown", - }); - return 0; - } - } - return 0; // No position info, can't evaluate + // Only evaluatable when we know the cell's position (for recursion + + // circular tracking); otherwise treat as 0. + return position ? resolveFormulaCell(value, position) : 0; } - // Try to parse as number, but preserve strings - const num = parseFloat(value); - return isNaN(num) ? value : num; + // Try to parse as number, but preserve original type on failure + if (typeof value === "number") return value; + if (typeof value === "boolean") return value; + if (typeof value === "string") { + const num = parseFloat(value); + return isNaN(num) ? value : num; + } + return String(value); } // Try to parse cell as number, but preserve strings @@ -241,137 +258,87 @@ export function calculateSheet( return isNaN(num) ? cell : num; }; + // Resolve a possibly cross-sheet reference to the grid it reads from, plus the + // sheet-name-stripped ref. A same-sheet ref returns the current `calculated` + // grid; a cross-sheet ref computes and caches its target sheet. null means the + // named sheet does not exist — the caller picks the terminal action (#REF! for + // a single cell, [] for a range). The two-stage cache seed (raw copy published + // BEFORE recursing, real result after) is the cross-sheet infinite-loop guard. + const resolveSheetData = (fullRef: string): ResolvedSheetRef | null => { + const sheetMatch = fullRef.match(/^(?:'([^']+)'|([^!]+))!(.+)$/); + if (!sheetMatch) return { sheetData: calculated, ref: fullRef, isCurrentSheet: true }; + + // Exactly one of the two name branches participates; the reference part + // always does. A shortfall would mean the pattern and this read disagree, so + // it lands on the caller's "sheet not found" path rather than a guess. + const [, quotedName, plainName, innerRef] = sheetMatch; + const targetSheetName = quotedName ?? plainName; + if (targetSheetName === undefined || innerRef === undefined) return null; + + // Check cache first to prevent infinite loops + const cached = sheetsCache.get(targetSheetName); + if (cached) return { sheetData: cached, ref: innerRef, isCurrentSheet: false }; + + const targetSheet = processedAllSheets?.find((s) => s.name === targetSheetName); + if (!targetSheet || !targetSheet.data) return null; + + // Seed the cache with a raw copy BEFORE recursing so a cyclic back-reference + // finds this sheet mid-flight, then overwrite it with the calculated result. + // Resolve cross-sheet values RAW (skip display formatting) so a date cell + // reads as its serial, not the presentation string "03/04/2025". + const targetCalculated = targetSheet.data.map((row) => [...row]); + sheetsCache.set(targetSheetName, targetCalculated); + const targetResult = calculateSheet(targetSheet, processedAllSheets, { preferDDMMYYYY, skipFormatting: true }); + sheetsCache.set(targetSheetName, targetResult.data); + return { sheetData: targetResult.data, ref: innerRef, isCurrentSheet: false }; + }; + // Helper to get cell value by reference (e.g., "B2", "$B$2", or "'Sheet1'!B2") const getCellValue = (ref: string): CellValue => { - let sheetData: any[][] = calculated; - let cellRef = ref; - let isCurrentSheet = true; - - // Check for cross-sheet reference (e.g., 'Sheet Name'!B2 or Sheet1!B2) - const sheetMatch = ref.match(/^(?:'([^']+)'|([^!]+))!(.+)$/); - if (sheetMatch) { - const targetSheetName = sheetMatch[1] || sheetMatch[2]; // Quoted or unquoted sheet name - cellRef = sheetMatch[3]; // Cell reference part - isCurrentSheet = false; - - // Check cache first to prevent infinite loops - if (sheetsCache.has(targetSheetName)) { - sheetData = sheetsCache.get(targetSheetName)!; - } else { - // Find the sheet in all sheets - const targetSheet = processedAllSheets?.find( - (s) => s.name === targetSheetName, - ); - if (targetSheet && targetSheet.data) { - // Calculate formulas for the target sheet with cache - const targetCalculated = targetSheet.data.map((row) => [...row]); - sheetsCache.set(targetSheetName, targetCalculated); - - // Recursively calculate the target sheet - const targetResult = calculateSheet(targetSheet, processedAllSheets); - sheetsCache.set(targetSheetName, targetResult.data); - sheetData = targetResult.data as any[][]; - } else { - return 0; // Sheet not found - } - } - } - - // Remove $ symbols for absolute references - const cleanRef = cellRef.replace(/\$/g, ""); - const match = cleanRef.match(/^([A-Z]+)(\d+)$/); - if (!match) return 0; + const resolved = resolveSheetData(ref); + if (!resolved) throw invalidRefError(ref); // Sheet not found → #REF! + const { sheetData, ref: cellRef, isCurrentSheet } = resolved; - const col = columnToIndex(match[1]); // A=0, B=1, ..., Z=25, AA=26, etc. - const row = parseInt(match[2]) - 1; // 1-indexed to 0-indexed + // `$` symbols and the A1 shape are parsed by the shared single-cell reader. + const coord = parseSingleCellRef(cellRef); + if (!coord) return 0; - if ( - row < 0 || - row >= sheetData.length || - col < 0 || - col >= sheetData[row].length - ) { - return 0; - } + const { row, col } = coord; + const gridRow = row >= 0 ? sheetData[row] : undefined; + if (!gridRow || col < 0 || col >= gridRow.length) return 0; - const cell = sheetData[row][col]; - // Pass row/col only if this is the current sheet (for recursive evaluation) - return getRawValue( - cell, - isCurrentSheet ? row : undefined, - isCurrentSheet ? col : undefined, - ); + // Pass the position only if this is the current sheet (for recursive evaluation) + return getRawValue(gridRow[col], isCurrentSheet ? { cells: gridRow, row, col } : undefined); }; - const collectRangeValues = ( - range: string, - options: { numericOnly: boolean }, - ): CellValue[] => { - let sheetData: any[][] = calculated; - let rangeRef = range; - let isCurrentSheet = true; - - // Check for cross-sheet reference - const sheetMatch = range.match(/^(?:'([^']+)'|([^!]+))!(.+)$/); - if (sheetMatch) { - const targetSheetName = sheetMatch[1] || sheetMatch[2]; - rangeRef = sheetMatch[3]; - isCurrentSheet = false; - - // Check cache first - if (sheetsCache.has(targetSheetName)) { - sheetData = sheetsCache.get(targetSheetName)!; - } else { - // Find and calculate the target sheet - const targetSheet = processedAllSheets?.find( - (s) => s.name === targetSheetName, - ); - if (targetSheet && targetSheet.data) { - const targetCalculated = targetSheet.data.map((row) => [...row]); - sheetsCache.set(targetSheetName, targetCalculated); - - // Recursively calculate the target sheet - const targetResult = calculateSheet(targetSheet, processedAllSheets); - sheetsCache.set(targetSheetName, targetResult.data); - sheetData = targetResult.data as any[][]; - } else { - return []; - } - } - } - - const match = rangeRef.match(/^([A-Z]+)(\d+):([A-Z]+)(\d+)$/); - if (!match) return []; + const collectRangeValues = (range: string, options: { numericOnly: boolean }): CellValue[] => { + const resolved = resolveSheetData(range); + if (!resolved) return []; // Sheet not found → empty range + const { sheetData, ref: rangeRef, isCurrentSheet } = resolved; - const startCol = columnToIndex(match[1]); - const startRow = parseInt(match[2]) - 1; - const endCol = columnToIndex(match[3]); - const endRow = parseInt(match[4]) - 1; + const coords = expandRangeOrCell(rangeRef); + if (!coords) return []; const values: CellValue[] = []; - for (let row = startRow; row <= endRow; row++) { - for (let col = startCol; col <= endCol; col++) { - if ( - row >= 0 && - row < sheetData.length && - col >= 0 && - col < sheetData[row].length - ) { - const cell = sheetData[row][col]; - // Pass row/col only if current sheet (for recursive evaluation) - const rawValue = getRawValue( - cell, - isCurrentSheet ? row : undefined, - isCurrentSheet ? col : undefined, - ); - - if (options.numericOnly) { - if (!isNaN(rawValue as number)) { - values.push(rawValue); - } - } else { + for (const { row, col } of coords) { + const gridRow = row >= 0 ? sheetData[row] : undefined; + if (gridRow && col >= 0 && col < gridRow.length) { + const cell = gridRow[col]; + // Pass the position only if current sheet (for recursive evaluation) + const rawValue = getRawValue(cell, isCurrentSheet ? { cells: gridRow, row, col } : undefined); + + if (options.numericOnly) { + // A blank cell is not a value. Dropping it from the NUMERIC list keeps + // SUM unchanged (a blank read as 0) while stopping it from inflating + // AVERAGE's denominator and COUNT's tally (#2358). The raw list keeps + // every cell so SUMIF/AVERAGEIF's criteria and value ranges stay + // row-aligned; dropping there would shift indexes and aggregate the + // wrong rows (Codex review). + if (!isEmptyCell(cell) && !isNaN(rawValue as number)) { values.push(rawValue); } + } else { + values.push(rawValue); } } } @@ -379,12 +346,10 @@ export function calculateSheet( }; // Helper to get numeric-only range values (legacy behavior) - const getRangeValues = (range: string): CellValue[] => - collectRangeValues(range, { numericOnly: true }); + const getRangeValues = (range: string): CellValue[] => collectRangeValues(range, { numericOnly: true }); // Helper to get raw range values including text - const getRangeValuesRaw = (range: string): CellValue[] => - collectRangeValues(range, { numericOnly: false }); + const getRangeValuesRaw = (range: string): CellValue[] => collectRangeValues(range, { numericOnly: false }); // Evaluate a formula with context const evaluateFormula = (formula: string): CellValue => { @@ -393,111 +358,62 @@ export function calculateSheet( getRangeValues, getRangeValuesRaw, evaluateFormula, + preferDDMMYYYY, }); }; - // Process all cells and calculate formulas - for (let rowIdx = 0; rowIdx < data.length; rowIdx++) { - for (let colIdx = 0; colIdx < data[rowIdx].length; colIdx++) { - const originalCell = data[rowIdx][colIdx]; - const calculatedCell = calculated[rowIdx][colIdx]; - - // Skip if cell was already calculated recursively - if ( - typeof calculatedCell === "number" && - originalCell && - typeof originalCell === "object" && - "f" in originalCell - ) { - // Cell was already evaluated - keep it as number for now - // Formatting will be applied at the end - continue; - } + // Compute one cell into its calculated row. A cell not in {v, f} format is + // left as-is (it is already a plain value). + const calculateCell = (originalCell: SpreadsheetCell, position: CellPosition): void => { + if (!isObj(originalCell) || !("v" in originalCell)) return; + const value = originalCell.v; - // Handle cell format {v, f} - if ( - originalCell && - typeof originalCell === "object" && - "v" in originalCell - ) { - const value = originalCell.v; - - // Check if value is a formula (string starting with "=") - if (typeof value === "string" && value.startsWith("=")) { - // Remove the "=" prefix and evaluate the formula - const formula = value.substring(1); - - // Track formula info - formulas.push({ - cell: { row: rowIdx, col: colIdx }, - formula: value, - dependencies: [], // TODO: Extract dependencies from formula - result: 0, // Will be updated below - }); - - const result = evaluateFormula(formula); - - // Update formula result - formulas[formulas.length - 1].result = result; - - // Store result as-is (formatting will be applied at the end) - calculated[rowIdx][colIdx] = result; - } else { - // Regular value cell (not a formula) - // Convert to plain value (important for range evaluation) - calculated[rowIdx][colIdx] = value; - } - } - // If cell is not in {v, f} format, leave it as-is (already a plain value) + // A plain value is copied through, so range evaluation reads it. + if (typeof value !== "string" || !value.startsWith("=")) { + position.cells[position.col] = value; + return; } - } - // Final formatting pass: apply formatting to all cells with format codes - for (let rowIdx = 0; rowIdx < data.length; rowIdx++) { - for (let colIdx = 0; colIdx < data[rowIdx].length; colIdx++) { - const originalCell = data[rowIdx][colIdx]; - const calculatedValue = calculated[rowIdx][colIdx]; - - if ( - originalCell && - typeof originalCell === "object" && - "v" in originalCell - ) { - const isFormula = - typeof originalCell.v === "string" && originalCell.v.startsWith("="); - - // Apply formatting if cell has a format code and calculated value is a number - if ( - "f" in originalCell && - originalCell.f && - typeof calculatedValue === "number" - ) { - calculated[rowIdx][colIdx] = formatNumber( - calculatedValue, - originalCell.f, - ); - } - // Auto-format date serial numbers from formulas without explicit format - else if ( - isFormula && - typeof calculatedValue === "number" && - calculatedValue >= 36000 && - calculatedValue <= 63499 && - Number.isInteger(calculatedValue) && - (!("f" in originalCell) || !originalCell.f) - ) { - // Check if this looks like a date serial number - // 36000 = Jul 1998, 63499 = Dec 2073 - // Must be integer (dates without time component) - // Avoids formatting calculated averages/sums as dates - // Apply default date format - calculated[rowIdx][colIdx] = formatNumber( - calculatedValue, - "MM/DD/YYYY", - ); - } - } - } + const info: FormulaInfo = { + cell: { row: position.row, col: position.col }, + formula: value, + dependencies: [], // TODO: Extract dependencies from formula + result: 0, // Will be updated below + }; + formulas.push(info); + // Route through the protected path so a thrown failure is classified into + // errors[] instead of escaping this walk, and a cell already resolved via + // another formula's recursion is read from cache (#2359). + info.result = resolveFormulaCell(value, position); + // Store result as-is (formatting will be applied at the end) + position.cells[position.col] = info.result; + }; + + // Walk every cell of the grid. `calculated` is built as a row-for-row copy of + // `data` and never resized, so a missing row cannot happen; skipping one keeps + // the walk total rather than asserting the invariant at each cell. + const forEachCell = (visit: (originalCell: SpreadsheetCell, position: CellPosition) => void): void => { + data.forEach((dataRow, row) => { + const cells = calculated[row]; + if (!cells) return; + dataRow.forEach((originalCell, col) => visit(originalCell, { cells, row, col })); + }); + }; + + // Process all cells and calculate formulas. A cell already evaluated through + // another formula's recursion keeps its number; formatting comes at the end. + forEachCell((originalCell, position) => { + const alreadyCalculated = typeof position.cells[position.col] === "number" && isObj(originalCell) && "f" in originalCell; + if (!alreadyCalculated) calculateCell(originalCell, position); + }); + + // Final display-formatting pass: turn raw serials into presentation strings. + // Skipped when this sheet is computed only to resolve a cross-sheet reference, + // so the referencing cell reads the underlying value, not a display string. + if (!skipFormatting) { + forEachCell((originalCell, { cells, col }) => { + cells[col] = formatCellForDisplay(originalCell, cells[col], preferDDMMYYYY); + }); } return { @@ -514,6 +430,6 @@ export function calculateSheet( * @param sheets - Array of sheets to calculate * @returns Array of calculated sheets */ -export function calculateWorkbook(sheets: SheetData[]): CalculatedSheet[] { - return sheets.map((sheet) => calculateSheet(sheet, sheets)); +export function calculateWorkbook(sheets: SheetData[], options: CalculateOptions = {}): CalculatedSheet[] { + return sheets.map((sheet) => calculateSheet(sheet, sheets, options)); } diff --git a/src/engine/cellBuilder.ts b/src/engine/cellBuilder.ts new file mode 100644 index 0000000..5eaf209 --- /dev/null +++ b/src/engine/cellBuilder.ts @@ -0,0 +1,81 @@ +/** + * Build a SpreadsheetCell from the raw input captured by the mini + * editor (type + value / formula / format). Extracted from + * `saveMiniEditor` in `src/plugins/spreadsheet/View.vue` where it + * was inlined as ~30 lines of nested if/else that pushed the + * surrounding function over the cognitive-complexity threshold. + * + * Pure — no refs, no DOM, no side effects. Given the same inputs + * it always returns the same SpreadsheetCell. Tested in + * `test/plugins/spreadsheet/engine/test_cellBuilder.ts`. + */ + +import type { SpreadsheetCell } from "./types.js"; + +/** Inputs to the cell builder. Mirrors the mini editor refs in the + * View but as plain values so unit tests don't need a Vue runtime. */ +export interface MiniEditorInput { + /** "string" → value is stored as-is as a string. + * Anything else → the `formula` field is parsed (formula / number / raw string). */ + type: string; + /** Used when type === "string". Coerced to string. */ + value: unknown; + /** Used when type !== "string". Trimmed before classification. */ + formula?: string; + /** Optional format code (e.g. "$#,##0.00"). */ + format?: string; +} + +// Anchored at the start of the input (after optional unary +/-) so we +// only treat expressions that clearly begin with a function call as +// formulas. Unanchored would match "abc FOO(" inside ordinary text. +const FORMULA_FUNCTION_CALL = /^[-+]?\s*[A-Z]+\s*\(/i; + +// `A1 + B2` style — cell reference next to an arithmetic operator. +const FORMULA_CELL_OP = /[A-Z]+\d+\s*[+\-*/^]/; + +// `6/100`, `5 * 2` — arithmetic between two literal numbers. +const FORMULA_NUMERIC_OP = /\d+\s*[+\-*/^]\s*\d+/; + +// Strict numeric literal. `parseFloat` accepts trailing junk +// ("42abc" → 42) which silently corrupts user input; this anchor +// ensures the ENTIRE trimmed string is a number. +const STRICT_NUMBER = /^[-+]?(?:\d+\.?\d*|\.\d+)(?:[eE][-+]?\d+)?$/; + +/** + * Best-effort formula detection. The rules are conservative enough + * that plain text like "hello world" stays as text, but any input + * with arithmetic operators or function calls is treated as a + * formula and gets the "=" prefix the engine expects. + */ +export function looksLikeFormula(input: string): boolean { + return FORMULA_FUNCTION_CALL.test(input) || FORMULA_CELL_OP.test(input) || FORMULA_NUMERIC_OP.test(input); +} + +/** + * Parse the raw (non-string-type) editor input into a cell value. + * Priority: formula > number > raw string > empty string. + */ +export function parseNonStringInput(raw: string): number | string { + const input = raw.trim(); + if (input === "") return ""; + if (looksLikeFormula(input)) return `=${input}`; + return STRICT_NUMBER.test(input) ? Number(input) : input; +} + +/** + * Build the full SpreadsheetCell from a mini editor input record. + * String type short-circuits to `{ v: String(value) }`. + * Everything else goes through parseNonStringInput for formula / + * number / text classification, then optionally attaches `f`. + */ +export function buildCellFromInput(input: MiniEditorInput): SpreadsheetCell { + if (input.type === "string") { + return { v: String(input.value) }; + } + const cell: SpreadsheetCell = { v: parseNonStringInput(input.formula ?? "") }; + if (input.format && input.format.length > 0) { + cell.f = input.format; + } + return cell; +} diff --git a/src/engine/cellEmpty.ts b/src/engine/cellEmpty.ts new file mode 100644 index 0000000..55c2398 --- /dev/null +++ b/src/engine/cellEmpty.ts @@ -0,0 +1,26 @@ +/** + * Telling a genuinely empty cell apart from one that holds the number 0. + * + * The calculator reads a blank cell as 0 for arithmetic (`=A1+1` on a blank A1 + * is 1, as in Excel). But an aggregate must not: `AVERAGE` divides by the count + * of real values, and `COUNT` counts numbers — a blank that reads as 0 inflates + * the denominator and the count. So range collection needs to skip the blanks, + * which means distinguishing them from a stored 0, which this does. + */ + +import { isObj } from "./guards"; + +/** True when a cell holds no value at all — absent, null, or an empty/whitespace + * string, in either the bare or the `{ v }` form. A cell containing the number + * 0, `false`, or any non-empty text is NOT empty. */ +export function isEmptyCell(cell: unknown): boolean { + if (cell === null || cell === undefined) return true; + if (typeof cell === "string") return cell.trim() === ""; + if (isObj(cell)) { + if (!("v" in cell)) return true; + const value = (cell as { v: unknown }).v; + if (value === null || value === undefined) return true; + return typeof value === "string" && value.trim() === ""; + } + return false; +} diff --git a/src/engine/cellFormatting.ts b/src/engine/cellFormatting.ts new file mode 100644 index 0000000..aea68ad --- /dev/null +++ b/src/engine/cellFormatting.ts @@ -0,0 +1,59 @@ +/** + * Cell display formatting + * + * Turns a cell's raw calculated value into its display value (currency, + * percentage, date, ...). Pure — no engine state — so cross-sheet reference + * resolution can deliberately SKIP it and keep raw serial numbers, while the + * final output pass applies it for presentation. + */ + +import { formatNumber } from "./formatter"; +import { isRecord } from "./guards"; +import { isSpreadsheetErrorValue } from "./spreadsheet-errors"; +import type { CellValue, SpreadsheetCell, StoredCellValue } from "./types"; + +// Integer serials the engine is willing to auto-format as dates without an +// explicit format code: ~Jul 1998 (36000) through ~Dec 2073 (63499). Narrow on +// purpose so ordinary sums/averages are not mistaken for dates. +const DATE_SERIAL_MIN = 36000; +const DATE_SERIAL_MAX = 63499; + +const isSpreadsheetCell = (value: unknown): value is SpreadsheetCell => isRecord(value) && "v" in value; + +/** An integer within the date-serial window — a calculated number the engine + * should display as a date when the cell carries no explicit format. */ +export const isLikelyDateSerial = (value: CellValue): boolean => + typeof value === "number" && Number.isInteger(value) && value >= DATE_SERIAL_MIN && value <= DATE_SERIAL_MAX; + +/** + * Resolve the display value of one cell from its original definition and its + * calculated value. + * + * - A formula error renders as its code, so the cell still reads `#NUM!`. + * - Explicit format code wins (currency, percentage, date, ...). + * - A formula that produced a date serial auto-formats as a date. + * - Everything else (text, plain numbers, empty) passes through unchanged. + * + * The result is always a STORED value: this is the boundary where a computed + * error becomes the text a cell shows and a workbook serializes. + */ +export const formatCellForDisplay = (originalCell: unknown, calculatedValue: CellValue, preferDDMMYYYY: boolean): StoredCellValue => { + if (isSpreadsheetErrorValue(calculatedValue)) { + return calculatedValue.code; + } + if (!isSpreadsheetCell(originalCell) || typeof calculatedValue !== "number") { + return calculatedValue; + } + + const explicitFormat = typeof originalCell.f === "string" ? originalCell.f : ""; + if (explicitFormat) { + return formatNumber(calculatedValue, explicitFormat); + } + + const isFormula = typeof originalCell.v === "string" && originalCell.v.startsWith("="); + if (isFormula && isLikelyDateSerial(calculatedValue)) { + return formatNumber(calculatedValue, preferDDMMYYYY ? "DD/MM/YYYY" : "MM/DD/YYYY"); + } + + return calculatedValue; +}; diff --git a/src/engine/coerce-boolean.ts b/src/engine/coerce-boolean.ts new file mode 100644 index 0000000..5b56c78 --- /dev/null +++ b/src/engine/coerce-boolean.ts @@ -0,0 +1,26 @@ +import type { CellValue } from "./types"; +import { isSpreadsheetErrorValue } from "./spreadsheet-errors"; + +/** Excel-style truthiness, shared by IF and AND/OR/NOT so the same value cannot + * read as true in one function and false in another. A number is false only + * when 0; blank and empty text are false; the words `true`/`false` are their + * logical values (case-insensitively); a numeric string follows its number + * (`"0"` → false); any other non-empty text is true. */ +export function coerceToBoolean(value: CellValue | null | undefined): boolean { + if (typeof value === "boolean") return value; + if (value === null || value === undefined) return false; + if (typeof value === "number") return value !== 0; + // Pinned: an error reads as non-empty text, i.e. true — the same answer the + // error strings gave before they became values. + if (isSpreadsheetErrorValue(value)) return true; + + const text = value.trim(); + if (text === "") return false; + + const lowered = text.toLowerCase(); + if (lowered === "true") return true; + if (lowered === "false") return false; + + const asNumber = Number(text); + return Number.isNaN(asNumber) ? true : asNumber !== 0; +} diff --git a/src/engine/condition.ts b/src/engine/condition.ts new file mode 100644 index 0000000..5f571e3 --- /dev/null +++ b/src/engine/condition.ts @@ -0,0 +1,182 @@ +/** + * Evaluating a spreadsheet condition without running it as code. + * + * A condition is one comparison, or a bare value tested for truthiness. That is + * the whole grammar — small enough to read directly, which is the point: the + * previous implementation handed the substituted text to `eval`, so a cell + * containing `globalThis.x = 1` executed when any IFS referenced it. + */ + +import type { CellValue } from "./types"; +import { isSpreadsheetErrorValue } from "./spreadsheet-errors"; + +export type ComparisonOperator = ">=" | "<=" | "<>" | "!=" | "==" | "=" | ">" | "<"; + +// Longest first: `>=` must win over `>`, and `<>` / `<=` over `<`. +const OPERATORS: readonly ComparisonOperator[] = [">=", "<=", "<>", "!=", "==", "=", ">", "<"]; + +export interface Comparison { + left: string; + operator: ComparisonOperator; + right: string; +} + +/** Remove parentheses that wrap the WHOLE expression, repeatedly. `(1>0)` is + * the same condition as `1>0`, but `(A)=(B)` is not `A)=(B` — the leading `(` + * closes before the end, so it wraps only its own operand and must stay. + * Unbalanced input is left untouched rather than guessed at. */ +export function stripOuterParens(condition: string): string { + let text = condition.trim(); + while (text.startsWith("(") && text.endsWith(")")) { + let depth = 0; + let quote: string | null = null; + let wrapsAll = true; + for (let index = 0; index < text.length; index++) { + const char = text[index]; + if (quote !== null) { + if (char === "\\") + index++; // skip the escaped character + else if (char === quote) quote = null; + continue; + } + if (char === '"' || char === "'") { + quote = char; + continue; + } + if (char === "(") depth++; + else if (char === ")") { + depth--; + // Back to zero before the end means this `(` closed early. + if (depth === 0 && index < text.length - 1) { + wrapsAll = false; + break; + } + if (depth < 0) return text; // unbalanced + } + } + if (!wrapsAll || depth !== 0) return text; + text = text.slice(1, -1).trim(); + } + return text; +} + +/** Split a condition into its two sides, or null when it holds no comparison. + * Only the FIRST top-level operator counts — `a>b>c` is not a chain here, and + * treating it as one is what let `1=1=1` reach a JS parser before. + * + * "Top-level" means outside quotes: a cell holding `a>b` substitutes into the + * condition as `"a>b"`, and splitting on that `>` would compare two fragments + * of one string literal. */ +export function splitComparison(condition: string): Comparison | null { + const text = stripOuterParens(condition); + let quote: string | null = null; + for (let index = 0; index < text.length; index++) { + const char = text[index]; + if (quote !== null) { + if (char === "\\") + index++; // skip the escaped character + else if (char === quote) quote = null; + continue; + } + if (char === '"' || char === "'") { + quote = char; + continue; + } + for (const operator of OPERATORS) { + if (!text.startsWith(operator, index)) continue; + return { left: text.slice(0, index).trim(), operator, right: text.slice(index + operator.length).trim() }; + } + } + return null; +} + +/** Strip one matching pair of surrounding quotes, and undo the `\"` / `\\` + * escaping that `renderConditionOperand` applies when it quotes a cell value. */ +function unquote(text: string): { value: string; quoted: boolean } { + const isQuoted = text.length >= 2 && ((text.startsWith('"') && text.endsWith('"')) || (text.startsWith("'") && text.endsWith("'"))); + if (!isQuoted) return { value: text, quoted: false }; + const inner = text.slice(1, -1).replace(/\\(["'\\])/g, "$1"); + return { value: inner, quoted: true }; +} + +/** Read an operand as the value it denotes: a quoted string stays text, a + * numeric literal becomes a number, `TRUE`/`FALSE` become booleans, and + * anything else stays the text it already is. */ +export function readOperand(raw: string): CellValue { + const { value, quoted } = unquote(raw.trim()); + if (quoted) return value; + if (value === "") return ""; + const upper = value.toUpperCase(); + if (upper === "TRUE") return true; + if (upper === "FALSE") return false; + // `Number` rather than `parseFloat`: it rejects trailing garbage, so "12abc" + // stays text instead of becoming 12. + const numeric = Number(value); + return Number.isNaN(numeric) ? value : numeric; +} + +function compareValues(left: CellValue, right: CellValue): number | null { + if (typeof left === "number" && typeof right === "number") return left - right; + if (typeof left === "boolean" || typeof right === "boolean") return null; + return String(left).localeCompare(String(right)); +} + +function applyOperator(operator: ComparisonOperator, left: CellValue, right: CellValue): boolean { + // Equality does not need an ordering, so it works for every type pair — + // including the boolean combinations `compareValues` refuses to order. + if (operator === "=" || operator === "==") return left === right; + if (operator === "<>" || operator === "!=") return left !== right; + const ordering = compareValues(left, right); + if (ordering === null) return false; + if (operator === ">") return ordering > 0; + if (operator === ">=") return ordering >= 0; + if (operator === "<") return ordering < 0; + return ordering <= 0; +} + +/** Whether a resolved condition value counts as satisfied. Mirrors the + * spreadsheet convention rather than JavaScript's: 0 and an empty string are + * false, every other value is true. */ +function valueIsTruthy(value: CellValue): boolean { + if (typeof value === "boolean") return value; + if (typeof value === "number") return value !== 0; + return value !== ""; +} + +/** True when a bare (non-comparison) condition counts as satisfied. */ +export function isTruthyCondition(raw: string): boolean { + return valueIsTruthy(readOperand(stripOuterParens(raw))); +} + +/** Evaluate a condition — one comparison, or a value tested for truthiness. + * Never executes its input. */ +export function evaluateCondition(condition: string): boolean { + const comparison = splitComparison(condition); + if (!comparison) return isTruthyCondition(condition); + return applyOperator(comparison.operator, readOperand(comparison.left), readOperand(comparison.right)); +} + +/** Like `evaluateCondition`, but each operand is resolved by `evaluate` — so a + * caller holding the engine can compute arithmetic and sub-expressions + * (`5+1>10`) instead of reading each side as a bare string. It still never runs + * the condition as code: it only splits on the top-level comparison and + * compares the two resolved values. */ +export function evaluateConditionValues(condition: string, evaluate: (operand: string) => CellValue): boolean { + const comparison = splitComparison(condition); + if (!comparison) return valueIsTruthy(evaluate(stripOuterParens(condition))); + return applyOperator(comparison.operator, evaluate(comparison.left), evaluate(comparison.right)); +} + +/** Render a cell's value as an operand for a condition string. A string is + * quoted, with its own quotes and backslashes escaped, so its contents cannot + * be re-parsed as operators — `evaluateCondition` then unquotes it back to the + * original text. A missing value becomes an empty string; numbers and booleans + * render as themselves. */ +export function renderConditionOperand(value: CellValue | null | undefined): string { + if (value === null || value === undefined) return '""'; + // A formula error renders as its quoted code, so a condition compares it as + // the text a cell shows rather than as a bare `#NUM!` token. + const text = isSpreadsheetErrorValue(value) ? value.code : value; + if (typeof text === "string") return `"${text.replace(/\\/g, "\\\\").replace(/"/g, '\\"')}"`; + return text.toString(); +} diff --git a/src/engine/date-locale.ts b/src/engine/date-locale.ts new file mode 100644 index 0000000..850bfa1 --- /dev/null +++ b/src/engine/date-locale.ts @@ -0,0 +1,47 @@ +/** + * Which order an ambiguous slash date is written in, per locale. + * + * Only dates whose two leading numbers are BOTH 12 or under are ambiguous — + * `13/04/2025` can only be day-first, and `04/13/2025` can only be month-first, + * so those decide themselves. This is purely about `03/04/2025`. + */ + +// Day-first locales among the ones the app ships. The rest are month-first +// here: `en` because the app cannot see the region (`en-GB` is folded to `en` +// before it reaches any plugin), and ja / zh / ko because their conventional +// order is year-month-day, which puts the month before the day in a two-part +// date just as US order does. +const DAY_FIRST_LANGUAGES = new Set(["es", "pt", "fr", "de", "it", "nl", "ru", "pl", "tr", "id", "vi", "th"]); + +// English regions that write month-first. Everywhere else that speaks English +// writes day-first. +const MONTH_FIRST_EN_REGIONS = new Set(["us", "ca", "ph"]); + +/** The region subtag, or "" when the tag carries none. Scanning rather than + * taking position 1: a script subtag (`en-Latn-US`) sits between the language + * and the region, and reading it as the region flips month-first English + * locales to day-first. A region is two letters or three digits; a + * single-character subtag starts an extension, so nothing past it is one. */ +function regionOf(subtags: readonly string[]): string { + for (const subtag of subtags) { + if (subtag.length === 1) break; + if (/^[a-z]{2}$/.test(subtag) || /^[0-9]{3}$/.test(subtag)) return subtag; + } + return ""; +} + +/** True when an ambiguous `A/B/YYYY` should read as day-first. Accepts a full + * BCP 47 tag or a bare language subtag; the region is used when present so a + * caller that can supply `en-GB` gets the right answer even though the app's + * own locale resolution discards it. */ +export function prefersDayFirst(locale: string | undefined | null): boolean { + if (!locale) return false; + const [language = "", ...rest] = locale.toLowerCase().split(/[-_]/); + // English splits on region rather than language. A bare `en` carries no + // region to split on, so it keeps the US default. + if (language === "en") { + const region = regionOf(rest); + return region !== "" && !MONTH_FIRST_EN_REGIONS.has(region); + } + return DAY_FIRST_LANGUAGES.has(language); +} diff --git a/src/engine/date-parser.ts b/src/engine/date-parser.ts index 2f6a2f9..1cd1c58 100644 --- a/src/engine/date-parser.ts +++ b/src/engine/date-parser.ts @@ -4,11 +4,7 @@ * Parse various date string formats into Excel serial numbers. */ -import { - dateToSerial, - MONTH_NAMES_SHORT, - MONTH_NAMES_FULL, -} from "./date-utils"; +import { dateToSerial, MONTH_NAMES_SHORT, MONTH_NAMES_FULL } from "./date-utils"; /** * Check if a string looks like a date @@ -45,6 +41,37 @@ export function isDateLike(str: string): boolean { return datePatterns.some((pattern) => pattern.test(str.trim())); } +/** + * Build the Excel serial number for a year/month/day triple, or null when the + * triple is not a valid calendar date. Every dated branch of `parseDate` ends in + * this same validate → Date.UTC → dateToSerial step. + */ +export function serialFromParts(year: number, month: number, day: number): number | null { + if (!isValidDate(year, month, day)) return null; + return dateToSerial(new Date(Date.UTC(year, month - 1, day))); +} + +/** + * The three capture groups of a date pattern, as a tuple the caller can + * destructure. `null` when the text does not match — or when a group did not + * participate, which every pattern here makes impossible, so it lands on the + * same "not a date" answer rather than reading an absent group as text. + */ +function matchDateParts(text: string, pattern: RegExp): [string, string, string] | null { + const [, first, second, third] = text.match(pattern) ?? []; + if (first === undefined || second === undefined || third === undefined) return null; + return [first, second, third]; +} + +const SLASH_DATE_PATTERN = /^(\d{1,2})\/(\d{1,2})\/(\d{2,4})$/; +const MAX_MONTH = 12; + +/** Whether `A/B/YYYY` reads day-first. A number no month can hold decides on its + * own; an ambiguous pair follows the reading preference. Shared with + * `getDefaultDateFormat` so a slash date renders in the order it was READ. */ +const readsDayFirst = (first: number, second: number, preferDDMMYYYY: boolean): boolean => + first > MAX_MONTH || (second <= MAX_MONTH && first <= MAX_MONTH && preferDDMMYYYY); + /** * Parse a month name to month number (1-12) */ @@ -52,15 +79,11 @@ function parseMonthName(monthStr: string): number | null { const month = monthStr.toLowerCase(); // Try short names - const shortIndex = MONTH_NAMES_SHORT.findIndex( - (m) => m.toLowerCase() === month, - ); + const shortIndex = MONTH_NAMES_SHORT.findIndex((m) => m.toLowerCase() === month); if (shortIndex !== -1) return shortIndex + 1; // Try full names - const fullIndex = MONTH_NAMES_FULL.findIndex( - (m) => m.toLowerCase() === month, - ); + const fullIndex = MONTH_NAMES_FULL.findIndex((m) => m.toLowerCase() === month); if (fullIndex !== -1) return fullIndex + 1; return null; @@ -80,108 +103,69 @@ function parseMonthName(monthStr: string): number | null { * @param preferDDMMYYYY - Prefer DD/MM/YYYY over MM/DD/YYYY for ambiguous dates (default: false) * @returns Serial number or null if not a valid date */ -export function parseDate( - dateStr: string, - preferDDMMYYYY: boolean = false, -): number | null { +export function parseDate(dateStr: string, preferDDMMYYYY: boolean = false): number | null { if (!isDateLike(dateStr)) return null; const trimmed = dateStr.trim(); // Try YYYY-MM-DD or YYYY/MM/DD (ISO format) - const isoMatch = trimmed.match(/^(\d{4})[-/](\d{1,2})[-/](\d{1,2})$/); - if (isoMatch) { - const year = parseInt(isoMatch[1]); - const month = parseInt(isoMatch[2]); - const day = parseInt(isoMatch[3]); - - if (isValidDate(year, month, day)) { - const date = new Date(Date.UTC(year, month - 1, day)); - return dateToSerial(date); - } - return null; + const isoParts = matchDateParts(trimmed, /^(\d{4})[-/](\d{1,2})[-/](\d{1,2})$/); + if (isoParts) { + const [year, month, day] = isoParts; + return serialFromParts(parseInt(year), parseInt(month), parseInt(day)); } // Try DD-MMM-YYYY or D-MMM-YYYY - const dmmyMatch = trimmed.match(/^(\d{1,2})-([A-Za-z]{3})-(\d{2,4})$/); - if (dmmyMatch) { - const day = parseInt(dmmyMatch[1]); - const monthName = dmmyMatch[2]; - let year = parseInt(dmmyMatch[3]); - - // Handle 2-digit years - if (year < 100) { - year = year < 30 ? 2000 + year : 1900 + year; - } - - const month = parseMonthName(monthName); - if (month && isValidDate(year, month, day)) { - const date = new Date(Date.UTC(year, month - 1, day)); - return dateToSerial(date); - } - return null; + const dmmyParts = matchDateParts(trimmed, /^(\d{1,2})-([A-Za-z]{3})-(\d{2,4})$/); + if (dmmyParts) { + const [day, monthName, year] = dmmyParts; + return serialFromNamedMonth(expandTwoDigitYear(parseInt(year)), monthName, parseInt(day)); } // Try MMM D, YYYY or MMMM D, YYYY - const mmmMatch = trimmed.match(/^([A-Za-z]{3,9})\s+(\d{1,2}),?\s+(\d{4})$/); - if (mmmMatch) { - const monthName = mmmMatch[1]; - const day = parseInt(mmmMatch[2]); - const year = parseInt(mmmMatch[3]); - - const month = parseMonthName(monthName); - if (month && isValidDate(year, month, day)) { - const date = new Date(Date.UTC(year, month - 1, day)); - return dateToSerial(date); - } - return null; + const mmmParts = matchDateParts(trimmed, /^([A-Za-z]{3,9})\s+(\d{1,2}),?\s+(\d{4})$/); + if (mmmParts) { + const [monthName, day, year] = mmmParts; + return serialFromNamedMonth(parseInt(year), monthName, parseInt(day)); } // Try D MMM YYYY - const dMmmMatch = trimmed.match(/^(\d{1,2})\s+([A-Za-z]{3,9})\s+(\d{4})$/); - if (dMmmMatch) { - const day = parseInt(dMmmMatch[1]); - const monthName = dMmmMatch[2]; - const year = parseInt(dMmmMatch[3]); - - const month = parseMonthName(monthName); - if (month && isValidDate(year, month, day)) { - const date = new Date(Date.UTC(year, month - 1, day)); - return dateToSerial(date); - } - return null; + const dMmmParts = matchDateParts(trimmed, /^(\d{1,2})\s+([A-Za-z]{3,9})\s+(\d{4})$/); + if (dMmmParts) { + const [day, monthName, year] = dMmmParts; + return serialFromNamedMonth(parseInt(year), monthName, parseInt(day)); } // Try MM/DD/YYYY or DD/MM/YYYY - const slashMatch = trimmed.match(/^(\d{1,2})\/(\d{1,2})\/(\d{2,4})$/); - if (slashMatch) { - const first = parseInt(slashMatch[1]); - const second = parseInt(slashMatch[2]); - let year = parseInt(slashMatch[3]); - - // Handle 2-digit years - if (year < 100) { - year = year < 30 ? 2000 + year : 1900 + year; - } - - // Determine if it's MM/DD/YYYY or DD/MM/YYYY - // If first > 12, it must be DD/MM; if second > 12, it must be MM/DD - // Otherwise use preference (default to MM/DD for US format) - const isDayFirst = - first > 12 || (second <= 12 && first <= 12 && preferDDMMYYYY); - const month = isDayFirst ? second : first; - const day = isDayFirst ? first : second; - - if (isValidDate(year, month, day)) { - const date = new Date(Date.UTC(year, month - 1, day)); - return dateToSerial(date); - } - return null; + const slashParts = matchDateParts(trimmed, SLASH_DATE_PATTERN); + if (slashParts) { + const [first, second, year] = slashParts; + // If first > 12, it must be DD/MM; if second > 12, it must be MM/DD. + // Otherwise use preference (default to MM/DD for US format). + const dayFirst = readsDayFirst(parseInt(first), parseInt(second), preferDDMMYYYY); + const month = dayFirst ? parseInt(second) : parseInt(first); + const day = dayFirst ? parseInt(first) : parseInt(second); + return serialFromParts(expandTwoDigitYear(parseInt(year)), month, day); } return null; } +/** A two-digit year reads as this century up to 29, the previous one after. */ +const TWO_DIGIT_YEAR_LIMIT = 100; +const CENTURY_PIVOT = 30; + +function expandTwoDigitYear(year: number): number { + if (year >= TWO_DIGIT_YEAR_LIMIT) return year; + return year < CENTURY_PIVOT ? 2000 + year : 1900 + year; +} + +function serialFromNamedMonth(year: number, monthName: string, day: number): number | null { + const month = parseMonthName(monthName); + if (month === null) return null; + return serialFromParts(year, month, day); +} + /** * Validate that a date is valid */ @@ -195,11 +179,7 @@ function isValidDate(year: number, month: number, day: number): boolean { const date = new Date(Date.UTC(year, month - 1, day)); // If the date rolls over to the next month, it's invalid - return ( - date.getUTCFullYear() === year && - date.getUTCMonth() === month - 1 && - date.getUTCDate() === day - ); + return date.getUTCFullYear() === year && date.getUTCMonth() === month - 1 && date.getUTCDate() === day; } /** @@ -208,7 +188,7 @@ function isValidDate(year: number, month: number, day: number): boolean { * @param originalStr - Original date string * @returns Appropriate format code */ -export function getDefaultDateFormat(originalStr: string): string { +export function getDefaultDateFormat(originalStr: string, preferDDMMYYYY: boolean = false): string { const trimmed = originalStr.trim(); // YYYY-MM-DD → use same format @@ -216,6 +196,14 @@ export function getDefaultDateFormat(originalStr: string): string { return "YYYY-MM-DD"; } + // YYYY/MM/DD parses as ISO, so it must keep a year-first label. Without this + // branch it fell through to the slash default and re-rendered as MM/DD or + // DD/MM — the same digits in a different order, which reads as a different + // date (Codex review). + if (/^\d{4}\/\d{1,2}\/\d{1,2}$/.test(trimmed)) { + return "YYYY/MM/DD"; + } + // DD-MMM-YYYY → use same format if (/^\d{1,2}-[A-Za-z]{3}-\d{2,4}$/.test(trimmed)) { return "DD-MMM-YYYY"; @@ -231,6 +219,16 @@ export function getDefaultDateFormat(originalStr: string): string { return "MMMM D, YYYY"; } - // Default to MM/DD/YYYY for slash-separated dates - return "MM/DD/YYYY"; + // A slash date must render in the order it was READ, or the cell shows the + // user's own input with its two halves swapped. `parseDate` takes the first + // number as the day whenever it cannot be a month, whatever the preference + // says, so that case is decided here the same way. + const slashParts = matchDateParts(trimmed, SLASH_DATE_PATTERN); + if (slashParts) { + const [first, second] = slashParts; + return readsDayFirst(parseInt(first), parseInt(second), preferDDMMYYYY) ? "DD/MM/YYYY" : "MM/DD/YYYY"; + } + + // Anything unrecognised keeps the reading order's default. + return preferDDMMYYYY ? "DD/MM/YYYY" : "MM/DD/YYYY"; } diff --git a/src/engine/date-utils.ts b/src/engine/date-utils.ts index e75ff7d..e5f37f1 100644 --- a/src/engine/date-utils.ts +++ b/src/engine/date-utils.ts @@ -41,25 +41,16 @@ export const serialToDate = (serial: number): Date => { return date; }; +/** A name list that always has a first entry, so a formatter can fall back to it + * when a date is unreadable and its month/weekday index comes out NaN. */ +export type NameList = readonly [string, ...string[]]; + /** * Month names for formatting */ -export const MONTH_NAMES_SHORT = [ - "Jan", - "Feb", - "Mar", - "Apr", - "May", - "Jun", - "Jul", - "Aug", - "Sep", - "Oct", - "Nov", - "Dec", -]; +export const MONTH_NAMES_SHORT: NameList = ["Jan", "Feb", "Mar", "Apr", "May", "Jun", "Jul", "Aug", "Sep", "Oct", "Nov", "Dec"]; -export const MONTH_NAMES_FULL = [ +export const MONTH_NAMES_FULL: NameList = [ "January", "February", "March", @@ -77,22 +68,6 @@ export const MONTH_NAMES_FULL = [ /** * Day names for formatting */ -export const DAY_NAMES_SHORT = [ - "Sun", - "Mon", - "Tue", - "Wed", - "Thu", - "Fri", - "Sat", -]; +export const DAY_NAMES_SHORT: NameList = ["Sun", "Mon", "Tue", "Wed", "Thu", "Fri", "Sat"]; -export const DAY_NAMES_FULL = [ - "Sunday", - "Monday", - "Tuesday", - "Wednesday", - "Thursday", - "Friday", - "Saturday", -]; +export const DAY_NAMES_FULL: NameList = ["Sunday", "Monday", "Tuesday", "Wednesday", "Thursday", "Friday", "Saturday"]; diff --git a/src/engine/datedif.ts b/src/engine/datedif.ts new file mode 100644 index 0000000..8e5ca53 --- /dev/null +++ b/src/engine/datedif.ts @@ -0,0 +1,93 @@ +/** + * DATEDIF — complete elapsed time between two dates, in a chosen unit. + * + * Pure: two Excel serials and a unit in, a number (or a formula error value) + * out. The unit branches each have their own boundary handling (month-end + * borrowing, year wraparound), which is exactly what makes them worth testing + * apart from the handler that reads the arguments. + */ + +import { serialToDate } from "./date-utils"; +import { NUM_ERROR, type SpreadsheetError } from "./spreadsheet-errors"; + +const MS_PER_DAY = 24 * 60 * 60 * 1000; +const MONTHS_PER_YEAR = 12; + +/** `date` advanced by `months`, clamping the day to the target month's length so + * adding a month to Jan 30 lands on the last day of a shorter month instead of + * overflowing into the next one. */ +function addMonthsClamped(date: Date, months: number): Date { + const monthIndex = date.getUTCMonth() + months; + const year = date.getUTCFullYear() + Math.floor(monthIndex / MONTHS_PER_YEAR); + const month = ((monthIndex % MONTHS_PER_YEAR) + MONTHS_PER_YEAR) % MONTHS_PER_YEAR; + const lastDayOfMonth = new Date(Date.UTC(year, month + 1, 0)).getUTCDate(); + const day = Math.min(date.getUTCDate(), lastDayOfMonth); + return new Date(Date.UTC(year, month, day)); +} + +/** Complete `unit`s between two dates, or `#NUM!` when start is after end or + * the unit is not one of Y / M / D / MD / YM / YD. `unit` is matched + * case-insensitively. */ +export function computeDatedif(startSerial: number, endSerial: number, unit: string): number | SpreadsheetError { + if (startSerial > endSerial) return NUM_ERROR; + + const startDate = serialToDate(startSerial); + const endDate = serialToDate(endSerial); + + const yearDiff = endDate.getUTCFullYear() - startDate.getUTCFullYear(); + const monthDiff = endDate.getUTCMonth() - startDate.getUTCMonth(); + const dayDiff = endDate.getUTCDate() - startDate.getUTCDate(); + + switch (unit.toUpperCase()) { + case "Y": { + // Complete years: back off one if the end has not yet reached the + // start's month-and-day within its year. + const years = yearDiff; + return monthDiff < 0 || (monthDiff === 0 && dayDiff < 0) ? years - 1 : years; + } + + case "M": { + // Complete months, backing off one when the day-of-month has not been reached. + const months = yearDiff * MONTHS_PER_YEAR + monthDiff; + return dayDiff < 0 ? months - 1 : months; + } + + case "D": + return Math.floor(endSerial - startSerial); + + case "MD": { + // Days left after the complete months "M" counts. Anchoring on start plus + // those months keeps the result non-negative and self-consistent (start + + // M months + MD days == end). Subtracting the calendar month before `end` + // instead goes negative when the start day outruns that month's length — + // Jan 30 → Mar 1 borrowed Feb's 28 days and returned -1. + const completeMonths = yearDiff * MONTHS_PER_YEAR + monthDiff - (dayDiff < 0 ? 1 : 0); + const anchor = addMonthsClamped(startDate, completeMonths); + // Compare whole days only. `end` may carry a time-of-day (a datetime + // serial); anchor is already UTC midnight, so strip end's time too or the + // remainder would swing with the clock (Codex review). + const endMidnight = Date.UTC(endDate.getUTCFullYear(), endDate.getUTCMonth(), endDate.getUTCDate()); + return Math.round((endMidnight - anchor.getTime()) / MS_PER_DAY); + } + + case "YM": { + // Month difference, ignoring years — wraps into 0..11. + const ym = dayDiff < 0 ? monthDiff - 1 : monthDiff; + return ym < 0 ? ym + MONTHS_PER_YEAR : ym; + } + + case "YD": { + // Day difference, ignoring years: move the start into the end's year, + // stepping back a year if that would put it after the end. + const startInEndYear = new Date(startDate); + startInEndYear.setUTCFullYear(endDate.getUTCFullYear()); + if (startInEndYear.getTime() - endDate.getTime() > 0) { + startInEndYear.setUTCFullYear(endDate.getUTCFullYear() - 1); + } + return Math.floor((endDate.getTime() - startInEndYear.getTime()) / MS_PER_DAY); + } + + default: + return NUM_ERROR; + } +} diff --git a/src/engine/engine.ts b/src/engine/engine.ts index 65fd0a9..516fffb 100644 --- a/src/engine/engine.ts +++ b/src/engine/engine.ts @@ -5,12 +5,8 @@ */ import { calculateSheet, calculateWorkbook } from "./calculator"; -import type { - SheetData, - CalculatedSheet, - EngineOptions, - SpreadsheetCell, -} from "./types"; +import type { SheetData, CalculatedSheet, EngineOptions, SpreadsheetCell } from "./types"; +import { isObj } from "./guards"; /** * SpreadsheetEngine - Main calculation engine class @@ -49,6 +45,7 @@ export class SpreadsheetEngine { maxIterations: options.maxIterations ?? 100, enableCrossSheetRefs: options.enableCrossSheetRefs ?? true, strictMode: options.strictMode ?? false, + preferDDMMYYYY: options.preferDDMMYYYY ?? false, }; } @@ -77,7 +74,7 @@ export class SpreadsheetEngine { * ``` */ calculate(sheet: SheetData, allSheets?: SheetData[]): CalculatedSheet { - return calculateSheet(sheet, allSheets); + return calculateSheet(sheet, allSheets, { preferDDMMYYYY: this.options.preferDDMMYYYY }); } /** @@ -99,7 +96,7 @@ export class SpreadsheetEngine { * ``` */ calculateWorkbook(sheets: SheetData[]): CalculatedSheet[] { - return calculateWorkbook(sheets); + return calculateWorkbook(sheets, { preferDDMMYYYY: this.options.preferDDMMYYYY }); } /** @@ -146,15 +143,12 @@ export class SpreadsheetEngine { * ]); * ``` */ - createSheet( - name: string, - data: Array>, - ): SheetData { + createSheet(name: string, data: Array>): SheetData { return { name, data: data.map((row) => row.map((cell) => { - if (typeof cell === "object" && cell !== null && "v" in cell) { + if (isObj(cell) && "v" in cell) { return cell as SpreadsheetCell; } return { v: cell }; diff --git a/src/engine/evaluator.ts b/src/engine/evaluator.ts index a555727..922db3b 100644 --- a/src/engine/evaluator.ts +++ b/src/engine/evaluator.ts @@ -4,9 +4,12 @@ * Evaluates spreadsheet formulas including functions, cell references, and arithmetic */ -import { functionRegistry } from "./registry"; +import { functionRegistry, tooFewArgumentsError } from "./registry"; import type { CellValue } from "./types"; import { parseDate } from "./date-parser"; +import { caretToPow, replaceConcatOperator, rewriteComparisonEq, isSafeArithmetic, isSafeComparison } from "./translateFormula"; +import { divZeroError, unknownError, nameError, propagatedError } from "./formulaError"; +import { errorCodeOf, isSpreadsheetErrorValue } from "./spreadsheet-errors"; /** * Evaluation context for formulas @@ -14,8 +17,155 @@ import { parseDate } from "./date-parser"; export interface EvaluatorContext { getCellValue: (ref: string) => CellValue; getRangeValues: (range: string) => CellValue[]; - getRangeValuesRaw?: (range: string) => CellValue[]; + getRangeValuesRaw?: ((range: string) => CellValue[]) | undefined; evaluateFormula: (formula: string) => CellValue; + /** Reading order for an ambiguous slash date; see engine/date-locale.ts. */ + preferDDMMYYYY?: boolean | undefined; +} + +/** Render a cell's value as the text that stands in for it inside an + * expression. Strings are quoted so they cannot be read as identifiers or + * operators, and their own quotes and backslashes are escaped so the literal + * cannot be closed early. A missing value becomes 0, matching how blanks are + * treated everywhere else in the engine. */ +export function renderOperand(value: CellValue | null | undefined): string { + if (value === null || value === undefined) return "0"; + const text = isSpreadsheetErrorValue(value) ? value.code : value; + if (typeof text === "string") return `"${text.replace(/\\/g, "\\\\").replace(/"/g, '\\"')}"`; + return text.toString(); +} + +/** Replace every quoted string literal with an empty pair of quotes, honouring + * `\` escapes so an escaped quote does not end the literal early. Used to + * validate the STRUCTURE of a concat/arithmetic expression without letting the + * arbitrary CONTENT of a string decide whether the whole thing looks safe. */ +export function maskStringLiterals(expr: string): string { + let out = ""; + let quote: string | null = null; + for (let index = 0; index < expr.length; index++) { + const char = expr[index]; + if (quote !== null) { + if (char === "\\") { + index++; // skip the escaped character — it is part of the literal + continue; + } + if (char === quote) { + quote = null; + out += char; + } + continue; + } + if (char === '"' || char === "'") { + quote = char; + out += char; + continue; + } + out += char; + } + return out; +} + +/** Whether a `&`-to-`+` concatenation expression is safe to evaluate. Checks the + * structure with string CONTENT masked out — a string literal may hold any + * character (a `!`, a `\`), and validating those against a character allowlist + * rejected valid formulas like `=A1&"!"` and every escaped operand. Once the + * literals are masked, only the joining structure remains to validate. */ +export function isSafeConcatExpression(expr: string): boolean { + // A boolean cell renders as a bare `true` / `false` here (renderOperand) — + // the only non-literal identifier the substitution can emit. Drop those words + // before validating so a boolean operand (`=A1&"!"` with A1 = true) still + // evaluates, while any other identifier keeps the structure from matching and + // never reaches `new Function`. + const structure = maskStringLiterals(expr).replace(/\b(?:true|false)\b/g, ""); + return /^[\d+\-*/(). "']*$/.test(structure); +} + +/** Index just past the string literal that opens at `start` (a quote char), + * honouring `\` escapes so an escaped quote does not close it early. Returns + * `expr.length` when the literal is never closed. */ +export function endOfStringLiteral(expr: string, start: number): number { + const quote = expr[start]; + let index = start + 1; + while (index < expr.length) { + if (expr[index] === "\\") { + index += 2; // skip the escaped character — it is part of the literal + continue; + } + if (expr[index] === quote) return index + 1; + index++; + } + return expr.length; +} + +/** A `'Sheet Name'!A1` reference beginning at `start` (a `'`), or null when the + * quotes open a plain string literal rather than a sheet-qualified reference. */ +function matchQuotedSheetRef(expr: string, start: number): string | null { + const endQuote = expr.indexOf("'", start + 1); + if (endQuote === -1 || expr[endQuote + 1] !== "!") return null; + const cellPart = expr.substring(endQuote + 2).match(/^(\$?[A-Z]+\$?\d+)/); + if (!cellPart) return null; + return expr.substring(start, endQuote + 2 + cellPart[0].length); +} + +/** A `Sheet!A1` or bare `A1` / `$A$1` reference beginning at `start`, or null. */ +function matchUnquotedRef(expr: string, start: number): string | null { + const rest = expr.substring(start); + const sheetMatch = rest.match(/^([A-Z][A-Z0-9]*)!/i); + if (sheetMatch) { + const cellPart = rest.substring(sheetMatch[0].length).match(/^(\$?[A-Z]+\$?\d+)/); + if (cellPart) return sheetMatch[0] + cellPart[0]; + } + const cellMatch = rest.match(/^(\$?[A-Z]+\$?\d+)/); + return cellMatch ? cellMatch[0] : null; +} + +/** Every cell reference in an expression, as `{ref, start}` spans in source + * order. Quoted string literals are skipped whole: a `"A1"` in the text is a + * constant, not a reference, and substituting it would turn `="A1"&"!"` into + * A1's value. A `'` opens either a `'Sheet'!A1` reference or a string literal — + * only the former is a reference; the latter is skipped like a `"` literal. */ +export interface CellRefSpan { + ref: string; + start: number; +} + +export function findCellRefs(expr: string): CellRefSpan[] { + const cellRefs: CellRefSpan[] = []; + let i = 0; + while (i < expr.length) { + const char = expr[i]; + if (char === '"') { + i = endOfStringLiteral(expr, i); + continue; + } + if (char === "'") { + const sheetRef = matchQuotedSheetRef(expr, i); + if (sheetRef) { + cellRefs.push({ ref: sheetRef, start: i }); + i += sheetRef.length; + } else { + i = endOfStringLiteral(expr, i); + } + continue; + } + const ref = matchUnquotedRef(expr, i); + if (ref) { + cellRefs.push({ ref, start: i }); + i += ref.length; + continue; + } + i++; + } + return cellRefs; +} + +/** Replace every reference span in `expr` with the text `render` gives for it, + * walking BACK TO FRONT so the spans ahead of each edit stay valid. A global + * string replace rewrote every occurrence of the shorter reference first, so + * `=A1+A10` had its `A10` broken into `0` and produced a plausible + * wrong number (#2357). */ +export function substituteCellRefs(expr: string, cellRefs: CellRefSpan[], render: (ref: string) => string): string { + return [...cellRefs].reverse().reduce((text, { ref, start }) => text.slice(0, start) + render(ref) + text.slice(start + ref.length), expr); } /** @@ -71,6 +221,27 @@ export function parseFunctionArgs(argsStr: string): string[] { return args; } +/** Run `new Function` on an expression the callers have already gated with a + * character allowlist. A JS parse/eval failure becomes #ERROR! — a genuinely + * broken formula that used to be swallowed into a bare string (#2359). A + * non-finite NUMBER result is only reachable by dividing by zero (comparisons + * yield booleans, concatenation yields strings), so it becomes #DIV/0!. */ +function evalValidatedExpression(jsExpr: string): CellValue { + let result: unknown; + try { + // eslint-disable -- sonarjs/code-eval + result = new Function(`return (${jsExpr})`)(); + } catch { + throw unknownError(); + } + if (typeof result === "number") { + if (!Number.isFinite(result)) throw divZeroError(); + return result; + } + if (typeof result === "string" || typeof result === "boolean") return result; + throw unknownError(); +} + /** * Evaluate a formula string * @@ -80,333 +251,239 @@ export function parseFunctionArgs(argsStr: string): string[] { * - Arithmetic: 2+3, A1*B1, (A1+B1)/2 * - Nested expressions: ROUND(SUM(A1:A10)/COUNT(A1:A10), 2) * + * A genuine failure THROWS (a typed FormulaError or any handler error) rather + * than returning the raw formula text; the calculator classifies it into + * errors[] and shows the Excel error value in the cell (#2359). + * * @param formula - Formula string (without leading =) * @param context - Evaluation context with cell/range accessors * @returns Evaluated result (number or string) */ -export function evaluateFormula( - formula: string, - context: EvaluatorContext, -): CellValue { - try { - // Handle string literals - remove surrounding quotes - // But NOT string concatenations (which contain & operators) - const trimmed = formula.trim(); - if ( - ((trimmed.startsWith('"') && trimmed.endsWith('"')) || - (trimmed.startsWith("'") && trimmed.endsWith("'"))) && - !trimmed.includes("&") // Exclude string concatenations - ) { - const stringValue = trimmed.slice(1, -1); // Remove first and last character (quotes) - - // Auto-parse date strings to serial numbers for compatibility with date arithmetic - // This allows formulas like =HLOOKUP("6/1/2024", ...) to work with parsed date cells - const dateSerial = parseDate(stringValue); - if (dateSerial !== null) { - return dateSerial; - } - - return stringValue; +export function evaluateFormula(formula: string, context: EvaluatorContext): CellValue { + // Handle string literals - remove surrounding quotes + // But NOT string concatenations (which contain & operators) + const trimmed = formula.trim(); + if ( + ((trimmed.startsWith('"') && trimmed.endsWith('"')) || (trimmed.startsWith("'") && trimmed.endsWith("'"))) && + !trimmed.includes("&") // Exclude string concatenations + ) { + const stringValue = trimmed.slice(1, -1); // Remove first and last character (quotes) + + // Auto-parse date strings to serial numbers for compatibility with date arithmetic + // This allows formulas like =HLOOKUP("6/1/2024", ...) to work with parsed date cells + const dateSerial = parseDate(stringValue, context.preferDDMMYYYY); + if (dateSerial !== null) { + return dateSerial; } - // Check if it's a SIMPLE function call (not a complex expression) - // We need to ensure the formula is JUST a function, not "FUNC(...) + something" - const funcMatch = formula.match(/^([A-Z]+)\((.*)\)$/i); - if (funcMatch) { - const [, funcName, argsStr] = funcMatch; - - // Check that the closing paren is actually the end of the function - // by counting parentheses in argsStr - let parenDepth = 0; - let isValidFunction = true; - for (const char of argsStr) { - if (char === "(") parenDepth++; - else if (char === ")") { - parenDepth--; - if (parenDepth < 0) { - // More closing parens than opening - this means we matched too much - isValidFunction = false; - break; - } - } - } - - // Normalize function name to uppercase for registry lookup - const normalizedFuncName = funcName.toUpperCase(); - const func = functionRegistry.get(normalizedFuncName); - - if (func && isValidFunction) { - const args = parseFunctionArgs(argsStr); + return stringValue; + } - // Validate argument count - if (func.minArgs !== undefined && args.length < func.minArgs) { - throw new Error( - `${normalizedFuncName} requires at least ${func.minArgs} argument${func.minArgs !== 1 ? "s" : ""}`, - ); - } - if (func.maxArgs !== undefined && args.length > func.maxArgs) { - throw new Error( - `${normalizedFuncName} accepts at most ${func.maxArgs} argument${func.maxArgs !== 1 ? "s" : ""}`, - ); + // Check if it's a SIMPLE function call (not a complex expression) + // We need to ensure the formula is JUST a function, not "FUNC(...) + something" + const [, funcName, argsStr] = formula.match(/^([A-Z]+)\((.*)\)$/i) ?? []; + if (funcName !== undefined && argsStr !== undefined) { + // Check that the closing paren is actually the end of the function + // by counting parentheses in argsStr + let parenDepth = 0; + let isValidFunction = true; + for (const char of argsStr) { + if (char === "(") parenDepth++; + else if (char === ")") { + parenDepth--; + if (parenDepth < 0) { + // More closing parens than opening - this means we matched too much + isValidFunction = false; + break; } - - // Execute function with context - return func.handler(args, { - getCellValue: context.getCellValue, - getRangeValues: context.getRangeValues, - getRangeValuesRaw: context.getRangeValuesRaw, - evaluateFormula: context.evaluateFormula, - }); } } - // Handle simple arithmetic expressions with cell references - // First, replace any function calls within the expression - let expr = formula; - - // Find and evaluate function calls (e.g., TODAY(), SUM(A1:A10), LOWER(A1), etc.) - // Use a simpler approach: find function names followed by parentheses - // and manually parse the matching closing parenthesis - let searchIndex = 0; - const maxIterations = 100; // Prevent infinite loops - let iterations = 0; - - while (searchIndex < expr.length && iterations < maxIterations) { - iterations++; - const funcNameMatch = expr.substring(searchIndex).match(/^([A-Z]+)\(/i); - if (!funcNameMatch) { - // No more functions found, move to next character - searchIndex++; - if (searchIndex >= expr.length) break; - continue; - } + // Normalize function name to uppercase for registry lookup + const normalizedFuncName = funcName.toUpperCase(); + const func = functionRegistry.get(normalizedFuncName); - const funcStartIndex = searchIndex; - const funcName = funcNameMatch[1]; - const argsStartIndex = searchIndex + funcName.length + 1; - - // Find matching closing parenthesis - let depth = 1; - let argsEndIndex = argsStartIndex; - let inString = false; - let stringChar = ""; - - while (argsEndIndex < expr.length && depth > 0) { - const char = expr[argsEndIndex]; - const prevChar = argsEndIndex > 0 ? expr[argsEndIndex - 1] : ""; - - // Track string boundaries - if ((char === '"' || char === "'") && prevChar !== "\\") { - if (!inString) { - inString = true; - stringChar = char; - } else if (char === stringChar) { - inString = false; - stringChar = ""; - } - } + if (func && isValidFunction) { + const args = parseFunctionArgs(argsStr); - // Only count parens outside of strings - if (!inString) { - if (char === "(") depth++; - else if (char === ")") depth--; - } - argsEndIndex++; + // Validate argument count + if (func.minArgs !== undefined && args.length < func.minArgs) { + throw tooFewArgumentsError(normalizedFuncName, func.minArgs); } - - if (depth === 0) { - const fullMatch = expr.substring(funcStartIndex, argsEndIndex); - const result = context.evaluateFormula(fullMatch); - // For string results, wrap in quotes; for numbers, wrap in parentheses - const replacement = - typeof result === "string" ? `"${result}"` : `(${result})`; - expr = - expr.substring(0, funcStartIndex) + - replacement + - expr.substring(argsEndIndex); - // Continue from after the replacement - searchIndex = funcStartIndex + replacement.length; - } else { - searchIndex++; + if (func.maxArgs !== undefined && args.length > func.maxArgs) { + throw new Error(`${normalizedFuncName} accepts at most ${func.maxArgs} argument${func.maxArgs !== 1 ? "s" : ""}`); } + + // Execute function with context + return func.handler(args, { + functionName: normalizedFuncName, + getCellValue: context.getCellValue, + getRangeValues: context.getRangeValues, + getRangeValuesRaw: context.getRangeValuesRaw, + evaluateFormula: context.evaluateFormula, + }); } - // Then replace cell references with their values - // Match cell references manually to avoid complex regex - const cellRefs: string[] = []; - let i = 0; - while (i < expr.length) { - // Check for cross-sheet reference (quoted or unquoted) - let ref = ""; - if (expr[i] === "'") { - // Quoted sheet name - const endQuote = expr.indexOf("'", i + 1); - if (endQuote !== -1 && expr[endQuote + 1] === "!") { - const cellPart = expr - .substring(endQuote + 2) - .match(/^(\$?[A-Z]+\$?\d+)/); - if (cellPart) { - ref = expr.substring(i, endQuote + 2 + cellPart[0].length); - cellRefs.push(ref); - i += ref.length; - continue; - } - } - } else { - // Unquoted sheet name or simple cell ref - const sheetMatch = expr.substring(i).match(/^([A-Z][A-Z0-9]*)!/i); - if (sheetMatch) { - const cellPart = expr - .substring(i + sheetMatch[0].length) - .match(/^(\$?[A-Z]+\$?\d+)/); - if (cellPart) { - ref = sheetMatch[0] + cellPart[0]; - cellRefs.push(ref); - i += ref.length; - continue; - } - } - // Simple cell reference - const cellMatch = expr.substring(i).match(/^(\$?[A-Z]+\$?\d+)/); - if (cellMatch) { - ref = cellMatch[0]; - cellRefs.push(ref); - i += ref.length; - continue; - } - } - i++; + // A balanced `NAME(...)` spanning the whole formula whose NAME is not a + // registered function is an unrecognized name — Excel's #NAME?. Surfacing + // it here also stops the arithmetic path below from recursing on it forever + // and leaving the literal text in the cell (#2359). + if (isValidFunction && !func) { + throw nameError(normalizedFuncName); } + } - if (cellRefs.length > 0) { - for (const ref of cellRefs) { - const value = context.getCellValue(ref); - // Escape special regex characters - const escapedRef = ref.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"); - // Wrap string values in quotes for proper evaluation - // Handle null/undefined values by treating them as 0 - let replacement: string; - if (value === null || value === undefined) { - replacement = "0"; - } else if (typeof value === "string") { - replacement = `"${value}"`; - } else { - replacement = value.toString(); - } - expr = expr.replace(new RegExp(escapedRef, "g"), replacement); - } + // Handle simple arithmetic expressions with cell references + // First, replace any function calls within the expression + let expr = formula; + + // Find and evaluate function calls (e.g., TODAY(), SUM(A1:A10), LOWER(A1), etc.) + // Use a simpler approach: find function names followed by parentheses + // and manually parse the matching closing parenthesis + let searchIndex = 0; + const maxIterations = 100; // Prevent infinite loops + let iterations = 0; + + while (searchIndex < expr.length && iterations < maxIterations) { + iterations++; + const [, funcName] = expr.substring(searchIndex).match(/^([A-Z]+)\(/i) ?? []; + if (funcName === undefined) { + // No more functions found, move to next character + searchIndex++; + if (searchIndex >= expr.length) break; + continue; } - // Parse date strings in arithmetic expressions (e.g., "06/01/2025" → serial number) - // This allows formulas like =B3-"06/01/2025" to work correctly - expr = expr.replace(/"([^"]+)"/g, (match, dateStr) => { - const dateSerial = parseDate(dateStr); - if (dateSerial !== null) { - return dateSerial.toString(); - } - return match; // Keep original if not a date - }); - - // Replace ^ with ** for exponentiation - expr = expr.replace(/\^/g, "**"); - - // Check if this is a string concatenation expression (contains & and quoted strings) - const hasStringConcat = expr.includes("&"); - const hasQuotedStrings = /["']/.test(expr); - - // If it contains string concatenation, handle it specially - if (hasStringConcat && hasQuotedStrings) { - try { - // Convert & to + for JavaScript string concatenation - // We need to be careful to only replace & that are not inside strings - let inString = false; - let stringChar = ""; - let result = ""; - - for (let index = 0; index < expr.length; index++) { - const char = expr[index]; - const prevChar = index > 0 ? expr[index - 1] : ""; - - // Handle string boundaries - if ((char === '"' || char === "'") && prevChar !== "\\") { - if (!inString) { - inString = true; - stringChar = char; - } else if (char === stringChar) { - inString = false; - stringChar = ""; - } - } - - // Replace & with + when not in a string - if (char === "&" && !inString) { - result += "+"; - } else { - result += char; - } - } + const funcStartIndex = searchIndex; + const argsStartIndex = searchIndex + funcName.length + 1; + + // Find matching closing parenthesis + let depth = 1; + let argsEndIndex = argsStartIndex; + let inString = false; + let stringChar = ""; - // Validate the expression contains only safe characters - // Allow: numbers, letters, strings (with quotes), operators, parentheses, whitespace, @, ., comma - if (/^[a-zA-Z0-9+\-*/(). "'@,]+$/.test(result)) { - // eslint-disable -- sonarjs/code-eval - const evalResult = new Function(`return (${result})`)(); - return evalResult; + while (argsEndIndex < expr.length && depth > 0) { + const char = expr[argsEndIndex]; + const prevChar = argsEndIndex > 0 ? expr[argsEndIndex - 1] : ""; + + // Track string boundaries + if ((char === '"' || char === "'") && prevChar !== "\\") { + if (!inString) { + inString = true; + stringChar = char; + } else if (char === stringChar) { + inString = false; + stringChar = ""; } - } catch (error) { - console.error( - `Failed to evaluate string concatenation: ${expr}`, - error, - ); - return formula; } - } - // Safely evaluate comparison expressions (e.g., 5=6, (5)>(6)) - // Allow numbers, comparison operators (=, !=, <, >, <=, >=), parentheses, whitespace - if (/^[\d+\-*/(). <>!=]+$/.test(expr)) { - try { - // Replace = with == for JavaScript comparison (but not <= or >=) - const jsExpr = expr.replace(/([^<>!])=([^=])/g, "$1==$2"); - - // Use Function constructor which is safer than eval - // eslint-disable -- sonarjs/code-eval - const result = new Function(`return (${jsExpr})`)(); - return result; - } catch { - return formula; + // Only count parens outside of strings + if (!inString) { + if (char === "(") depth++; + else if (char === ")") depth--; } + argsEndIndex++; } - // Safely evaluate arithmetic expressions using Function constructor instead of eval - // Allow numbers, operators, parentheses, whitespace, and decimal points - if (/^[\d+\-*/(). ]+$/.test(expr)) { - try { - // Use Function constructor which is safer than eval because: - // 1. The expression is strictly validated (only numbers and math operators) - // 2. No access to local scope variables - // 3. No this binding issues - // This is safe because we validate the expression first - // eslint-disable -- sonarjs/code-eval - const result = new Function(`return (${expr})`)(); - return result; - } catch { - return formula; - } + if (depth === 0) { + const fullMatch = expr.substring(funcStartIndex, argsEndIndex); + const result = context.evaluateFormula(fullMatch); + // A nested call that errored poisons the expression the same way an + // errored reference does — substituting it produced `"#NUM!"+1` text. + // Only the error VALUE counts: text that merely spells an error (CONCAT) + // is an ordinary operand. + if (isSpreadsheetErrorValue(result)) throw propagatedError(result.code); + // For string results, wrap in quotes; for numbers, wrap in parentheses + const replacement = typeof result === "string" ? `"${result}"` : `(${result})`; + expr = expr.substring(0, funcStartIndex) + replacement + expr.substring(argsEndIndex); + // Continue from after the replacement + searchIndex = funcStartIndex + replacement.length; + } else { + searchIndex++; } + } + + // Then replace cell references with their values. Detection skips quoted + // string literals so a `"A1"` constant is not read as a reference. + const cellRefs = findCellRefs(expr); + + // A formula that is nothing but one reference returns that cell's value + // directly. Rendering it into an expression first would mean escaping the + // text, and the escapes would survive into the result — `=A1` on a cell + // holding `say "hi"` would come back `say \"hi\"`. Surrounding whitespace + // (`= A1`, `=A1 `) is part of "nothing but one reference", so compare the + // span against the trimmed expression rather than the raw one. + const soleRef = cellRefs.length === 1 ? cellRefs[0] : undefined; + if (soleRef) { + const before = expr.slice(0, soleRef.start); + const after = expr.slice(soleRef.start + soleRef.ref.length); + if (before.trim() === "" && after.trim() === "") { + return context.getCellValue(soleRef.ref); + } + } - // If the final expression is a quoted string literal, unwrap it - const trimmedExpr = expr.trim(); - if ( - (trimmedExpr.startsWith('"') && trimmedExpr.endsWith('"')) || - (trimmedExpr.startsWith("'") && trimmedExpr.endsWith("'")) - ) { - return trimmedExpr.slice(1, -1); // Remove quotes + expr = substituteCellRefs(expr, cellRefs, (ref) => { + const value = context.getCellValue(ref); + // A referenced cell holding an error poisons the whole expression: + // rendering it would produce `"#DIV/0!"+1` garbage, so propagate the error + // instead of substituting it (#2359, Excel behaviour). A cell whose stored + // TEXT spells an error counts here too — Excel stores a typed error literal + // as an error, and arithmetic over it is never meaningful. + const referencedErrorCode = errorCodeOf(value); + if (referencedErrorCode !== null) throw propagatedError(referencedErrorCode); + return renderOperand(value); + }); + + // Parse date strings in arithmetic expressions (e.g., "06/01/2025" → serial number) + // This allows formulas like =B3-"06/01/2025" to work correctly + expr = expr.replace(/"([^"]+)"/g, (match, dateStr) => { + const dateSerial = parseDate(dateStr, context.preferDDMMYYYY); + if (dateSerial !== null) { + return dateSerial.toString(); + } + return match; // Keep original if not a date + }); + + // Replace ^ with ** for exponentiation + expr = caretToPow(expr); + + // Check if this is a string concatenation expression (contains & and quoted strings) + const hasStringConcat = expr.includes("&"); + const hasQuotedStrings = /["']/.test(expr); + + // If it contains string concatenation, handle it specially. Convert & to + + // for JS concatenation (leaving any & inside a literal untouched), then + // validate the STRUCTURE with literals masked — a literal may hold any + // character, so masking is what lets `=A1&"!"` and escaped operands through + // while still rejecting a genuinely unsafe expression. A non-safe structure + // falls through to the paths below rather than erroring. + if (hasStringConcat && hasQuotedStrings) { + const result = replaceConcatOperator(expr); + if (isSafeConcatExpression(result)) { + return evalValidatedExpression(result); } + } - return expr; // Return processed expression (with cell refs replaced, etc.) - } catch (error) { - console.error(`Failed to evaluate formula: ${formula}`, error); - return formula; + // Safely evaluate comparison expressions (e.g., 5=6, (5)>(6)). The allowlist + // (numbers, `= != < > <= >=`, parens, whitespace) gates the evaluation. It is + // a superset of the arithmetic allowlist, so a plain `1/0` is handled here — + // evalValidatedExpression turns its Infinity into #DIV/0! (#2359). + if (isSafeComparison(expr)) { + return evalValidatedExpression(rewriteComparisonEq(expr)); } + + // Safely evaluate arithmetic expressions (numbers, operators, parens, + // whitespace, decimal points). + if (isSafeArithmetic(expr)) { + return evalValidatedExpression(expr); + } + + // If the final expression is a quoted string literal, unwrap it + const trimmedExpr = expr.trim(); + if ((trimmedExpr.startsWith('"') && trimmedExpr.endsWith('"')) || (trimmedExpr.startsWith("'") && trimmedExpr.endsWith("'"))) { + return trimmedExpr.slice(1, -1); // Remove quotes + } + + return expr; // Return processed expression (with cell refs replaced, etc.) } diff --git a/src/engine/financial-math.ts b/src/engine/financial-math.ts new file mode 100644 index 0000000..d1361c0 --- /dev/null +++ b/src/engine/financial-math.ts @@ -0,0 +1,114 @@ +/** + * Annuity math for the financial functions, as pure numeric helpers. + * + * Excel's cash-flow sign convention: money received is positive, money paid out + * is negative — so a loan's `pv` is positive and its payment and interest come + * back negative. Keeping the per-period split as plain number-in / number-out + * functions lets it be tested against known Excel values without routing string + * arguments back through the formula evaluator. + */ + +import { NUM_ERROR, type SpreadsheetError } from "./spreadsheet-errors"; + +/** Future value after `nper` periods. `type` is 0 (end of period) or 1 (begin). */ +export function computeFv(rate: number, nper: number, pmt: number, pv: number, type: number): number { + if (rate === 0) return -(pv + pmt * nper); + const growth = Math.pow(1 + rate, nper); + return -pv * growth - (pmt * (growth - 1) * (1 + rate * type)) / rate; +} + +/** Constant per-period payment amortising a present value `pv` to `fv`. */ +export function computePmt(rate: number, nper: number, pv: number, fv: number, type: number): number { + if (rate === 0) return -(fv + pv) / nper; + const growth = Math.pow(1 + rate, nper); + return (-rate * (fv + pv * growth)) / ((growth - 1) * (1 + rate * type)); +} + +/** Interest portion of the `per`-th payment. */ +export function computeIpmt(rate: number, per: number, nper: number, pv: number, fv: number, type: number): number { + const pmt = computePmt(rate, nper, pv, fv, type); + if (per === 1 && type === 1) return 0; + const priorPeriods = type === 1 ? per - 2 : per - 1; + // Interest is the rate applied to the balance outstanding at the start of the + // period. computeFv already carries the payment-negative sign, so this product + // is the interest as a (negative) cash outflow — negating it again inverted the + // sign (IPMT came back +1250 for a -1250 payment) and PPMT amplified the error. + const interest = computeFv(rate, priorPeriods, pmt, pv, type) * rate; + return type === 1 ? interest / (1 + rate) : interest; +} + +/** Principal portion of the `per`-th payment (total payment minus interest). */ +export function computePpmt(rate: number, per: number, nper: number, pv: number, fv: number, type: number): number { + return computePmt(rate, nper, pv, fv, type) - computeIpmt(rate, per, nper, pv, fv, type); +} + +/** Present value of an annuity that grows to `fv` after `nper` payments of `pmt`. */ +export function computePv(rate: number, nper: number, pmt: number, fv: number, type: number): number { + if (rate === 0) return -(fv + pmt * nper); + const growth = Math.pow(1 + rate, nper); + return (-fv - (pmt * (growth - 1) * (1 + rate * type)) / rate) / growth; +} + +/** Number of periods needed to amortise `pv` to `fv` at constant `pmt`. */ +export function computeNper(rate: number, pmt: number, pv: number, fv: number, type: number): number { + if (rate === 0) return -(fv + pv) / pmt; + const pmtWithType = pmt * (1 + rate * type); + return Math.log((pmtWithType - fv * rate) / (pmtWithType + pv * rate)) / Math.log(1 + rate); +} + +// Newton-Raphson is iterative; a valid annuity / cash-flow series converges well +// inside these bounds. A series with no real solution never does, so exhausting +// the loop means "no answer" — Excel reports #NUM! there, and returning the last +// iterate instead would hand back a divergent value or NaN as if it were a rate. +const NEWTON_MAX_ITERATIONS = 100; +const NEWTON_TOLERANCE = 1e-7; + +/** Interest rate per period solving the annuity equation, via Newton-Raphson. */ +export function computeRate(nper: number, pmt: number, pv: number, fv: number, type: number, guess: number): number | SpreadsheetError { + let rate = guess; + for (let i = 0; i < NEWTON_MAX_ITERATIONS; i++) { + if (Math.abs(rate) < NEWTON_TOLERANCE) rate = NEWTON_TOLERANCE; // avoid division by zero + const growth = Math.pow(1 + rate, nper); + const f = pv * growth + pmt * ((growth - 1) / rate) * (1 + rate * type) + fv; + const df = + nper * pv * Math.pow(1 + rate, nper - 1) + + (pmt * (1 + rate * type) * (nper * Math.pow(1 + rate, nper - 1) * rate - (growth - 1))) / (rate * rate) + + pmt * type * ((growth - 1) / rate); + const newRate = rate - f / df; + if (Math.abs(newRate - rate) < NEWTON_TOLERANCE) return Number.isFinite(newRate) ? newRate : NUM_ERROR; + rate = newRate; + } + return NUM_ERROR; +} + +/** Net present value of `cashflows`, each discounted one period further into the future. */ +export function computeNpv(rate: number, cashflows: number[]): number { + return cashflows.reduce((npv, value, index) => npv + value / Math.pow(1 + rate, index + 1), 0); +} + +/** The present value of `values` (element 0 at period 0) discounted at `rate`, + * together with its derivative — the pair one Newton-Raphson step needs. */ +function discountedSeries(values: number[], rate: number): { npv: number; dnpv: number } { + return values.reduce( + (totals, value, period) => { + const factor = Math.pow(1 + rate, period); + return { npv: totals.npv + value / factor, dnpv: totals.dnpv - (period * value) / (factor * (1 + rate)) }; + }, + { npv: 0, dnpv: 0 }, + ); +} + +/** Internal rate of return of `values` (element 0 at period 0), via Newton-Raphson. */ +export function computeIrr(values: number[], guess: number): number | SpreadsheetError { + let rate = guess; + for (let i = 0; i < NEWTON_MAX_ITERATIONS; i++) { + const { npv, dnpv } = discountedSeries(values, rate); + if (Math.abs(npv) < NEWTON_TOLERANCE) return rate; + // A flat derivative leaves nowhere to step: the series has no root here. + if (Math.abs(dnpv) < NEWTON_TOLERANCE) return NUM_ERROR; + const newRate = rate - npv / dnpv; + if (Math.abs(newRate - rate) < NEWTON_TOLERANCE) return Number.isFinite(newRate) ? newRate : NUM_ERROR; + rate = newRate; + } + return NUM_ERROR; +} diff --git a/src/engine/formatter.ts b/src/engine/formatter.ts index bd7728f..d7775a5 100644 --- a/src/engine/formatter.ts +++ b/src/engine/formatter.ts @@ -4,13 +4,7 @@ * Handles Excel-style format codes for currency, percentages, decimals, dates, etc. */ -import { - serialToDate, - MONTH_NAMES_SHORT, - MONTH_NAMES_FULL, - DAY_NAMES_SHORT, - DAY_NAMES_FULL, -} from "./date-utils"; +import { serialToDate, MONTH_NAMES_SHORT, MONTH_NAMES_FULL, DAY_NAMES_SHORT, DAY_NAMES_FULL } from "./date-utils"; /** * Check if a format code is for dates @@ -37,6 +31,12 @@ function isDateFormat(format: string): boolean { * @param format - Date format code * @returns Formatted date string */ +// Every token, longest-first within each family so `MMMM` wins over `MMM` and +// `MM`. Matched in ONE pass: a sequence of `replace` calls re-scans its own +// output, so an inserted "March" had its `M` rewritten by the later month-number +// step and "AM/PM" was destroyed before the meridiem branch could see it. +const DATE_TOKEN_RE = /YYYY|YY|MMMM|MMM|MM|M|dddd|ddd|DD|D|AM\/PM|am\/pm|HH|H|hh|h|mm|ss/g; + function formatDate(serial: number, format: string): string { const date = serialToDate(serial); @@ -48,59 +48,56 @@ function formatDate(serial: number, format: string): string { const seconds = date.getUTCSeconds(); const dayOfWeek = date.getUTCDay(); // 0-6 - let result = format; - - // Replace year tokens - result = result.replace(/YYYY/g, year.toString()); - result = result.replace(/YY/g, (year % 100).toString().padStart(2, "0")); - - // Replace month tokens (order matters - do longer patterns first) - result = result.replace( - /MMMM/g, - MONTH_NAMES_FULL[month] || MONTH_NAMES_FULL[0], - ); - result = result.replace( - /MMM/g, - MONTH_NAMES_SHORT[month] || MONTH_NAMES_SHORT[0], - ); - result = result.replace(/MM/g, (month + 1).toString().padStart(2, "0")); - result = result.replace(/M/g, (month + 1).toString()); - - // Replace day tokens - result = result.replace(/DD/g, day.toString().padStart(2, "0")); - result = result.replace(/D/g, day.toString()); - - // Replace day of week tokens - result = result.replace( - /dddd/g, - DAY_NAMES_FULL[dayOfWeek] || DAY_NAMES_FULL[0], - ); - result = result.replace( - /ddd/g, - DAY_NAMES_SHORT[dayOfWeek] || DAY_NAMES_SHORT[0], - ); - - // Replace time tokens - // Handle 12-hour format with AM/PM - if (result.includes("AM/PM") || result.includes("am/pm")) { - const isPM = hours >= 12; - const hours12 = hours % 12 || 12; // 0 becomes 12 - - result = result.replace(/h/g, hours12.toString()); - result = result.replace(/AM\/PM/g, isPM ? "PM" : "AM"); - result = result.replace(/am\/pm/g, isPM ? "pm" : "am"); - } else { - // 24-hour format - result = result.replace(/HH/g, hours.toString().padStart(2, "0")); - result = result.replace(/H/g, hours.toString()); - result = result.replace(/h/g, hours.toString()); - } + // `h` means the 12-hour clock only when the format also asks for a meridiem; + // decided from the ORIGINAL format, before any substitution. + const uses12Hour = /AM\/PM|am\/pm/.test(format); + const hours12 = hours % 12 || 12; // 0 becomes 12 + const isPM = hours >= 12; + + const replacements: Record = { + YYYY: year.toString(), + YY: (year % 100).toString().padStart(2, "0"), + MMMM: MONTH_NAMES_FULL[month] ?? MONTH_NAMES_FULL[0], + MMM: MONTH_NAMES_SHORT[month] ?? MONTH_NAMES_SHORT[0], + MM: (month + 1).toString().padStart(2, "0"), + M: (month + 1).toString(), + dddd: DAY_NAMES_FULL[dayOfWeek] ?? DAY_NAMES_FULL[0], + ddd: DAY_NAMES_SHORT[dayOfWeek] ?? DAY_NAMES_SHORT[0], + DD: day.toString().padStart(2, "0"), + D: day.toString(), + "AM/PM": isPM ? "PM" : "AM", + "am/pm": isPM ? "pm" : "am", + HH: hours.toString().padStart(2, "0"), + H: hours.toString(), + hh: (uses12Hour ? hours12 : hours).toString().padStart(2, "0"), + h: (uses12Hour ? hours12 : hours).toString(), + mm: minutes.toString().padStart(2, "0"), + ss: seconds.toString().padStart(2, "0"), + }; + + return format.replace(DATE_TOKEN_RE, (token) => replacements[token] ?? token); +} - result = result.replace(/mm/g, minutes.toString().padStart(2, "0")); - result = result.replace(/ss/g, seconds.toString().padStart(2, "0")); +const THOUSANDS_GROUP_SIZE = 3; - return result; -} +/** + * Insert thousands separators into a run of digits ("1234567" → "1,234,567"). + * Regex free on purpose: the classic `\B(?=(\d{3})+(?!\d))` lookahead + * backtracks badly on a long run of digits. + */ +export const groupThousands = (digits: string): string => + Array.from(digits).reduce((grouped, digit, index) => { + const needsSeparator = index > 0 && (digits.length - index) % THOUSANDS_GROUP_SIZE === 0; + return needsSeparator ? `${grouped},${digit}` : `${grouped}${digit}`; + }, ""); + +/** Group the integer part of a formatted number ("1234567.89" → "1,234,567.89"), + * leaving any fractional part after the decimal point untouched. */ +export const addThousandSeparators = (formatted: string): string => { + const [whole, ...fraction] = formatted.split("."); + if (whole === undefined) return formatted; + return [groupThousands(whole), ...fraction].join("."); +}; /** * Format a number according to Excel format code @@ -128,26 +125,10 @@ export function formatNumber(value: number, format: string): string { // Handle currency formats if (format.includes("$")) { const decimals = (format.match(/\.0+/) || [""])[0].length - 1; - const hasComma = format.includes(","); - - let formatted = Math.abs(value).toFixed(decimals >= 0 ? decimals : 0); - if (hasComma) { - // Add thousand separators without regex to avoid performance issues - const parts = formatted.split("."); - const integerPart = parts[0]; - let result = ""; - for (let i = integerPart.length - 1, count = 0; i >= 0; i--, count++) { - if (count > 0 && count % 3 === 0) { - result = "," + result; - } - result = integerPart[i] + result; - } - parts[0] = result; - formatted = parts.join("."); - } - formatted = "$" + formatted; - if (value < 0) formatted = "-" + formatted; - return formatted; + const magnitude = Math.abs(value).toFixed(decimals >= 0 ? decimals : 0); + const grouped = format.includes(",") ? addThousandSeparators(magnitude) : magnitude; + const withSymbol = "$" + grouped; + return value < 0 ? "-" + withSymbol : withSymbol; } // Handle percentage @@ -159,21 +140,8 @@ export function formatNumber(value: number, format: string): string { // Handle comma separator if (format.includes(",")) { const decimals = (format.match(/\.0+/) || [""])[0].length - 1; - let formatted = Math.abs(value).toFixed(decimals >= 0 ? decimals : 0); - // Add thousand separators without regex to avoid performance issues - const parts = formatted.split("."); - const integerPart = parts[0]; - let result = ""; - for (let i = integerPart.length - 1, count = 0; i >= 0; i--, count++) { - if (count > 0 && count % 3 === 0) { - result = "," + result; - } - result = integerPart[i] + result; - } - parts[0] = result; - formatted = parts.join("."); - if (value < 0) formatted = "-" + formatted; - return formatted; + const grouped = addThousandSeparators(Math.abs(value).toFixed(decimals >= 0 ? decimals : 0)); + return value < 0 ? "-" + grouped : grouped; } // Handle decimal places diff --git a/src/engine/formulaError.ts b/src/engine/formulaError.ts new file mode 100644 index 0000000..7e34282 --- /dev/null +++ b/src/engine/formulaError.ts @@ -0,0 +1,67 @@ +/** + * Typed formula-evaluation errors + * + * A failed formula must surface as a typed error with an Excel-style error value + * in the cell, never as a swallowed bare string (issue #2359). This module holds + * the error taxonomy, the throwable carrier, and the pure helpers that classify a + * thrown error and propagate an error value across cell references. No engine + * state is captured — every export is input → output only. + */ + +import type { CalculationError } from "./types"; +import type { SpreadsheetErrorCode } from "./spreadsheet-errors"; + +/** The evaluator-facing error kinds: every `CalculationError["type"]` except + * `circular`, which the calculator raises directly (not via a thrown error). */ +export type FormulaErrorType = Exclude; + +/** The error code shown for each kind, matching Excel's error literals. */ +export const FORMULA_ERROR_VALUES: Record = { + div_zero: "#DIV/0!", + invalid_ref: "#REF!", + syntax: "#NAME?", + unknown: "#ERROR!", +}; + +/** Reverse map for propagation: an error code back to the kind it represents. + * Codes without a dedicated kind (`#N/A`, `#NUM!`, …) propagate as `unknown`. */ +const ERROR_VALUE_TO_TYPE: Record = { + [FORMULA_ERROR_VALUES.div_zero]: "div_zero", + [FORMULA_ERROR_VALUES.invalid_ref]: "invalid_ref", + [FORMULA_ERROR_VALUES.syntax]: "syntax", +}; + +/** A recoverable formula failure carrying the kind and the code to show. */ +export class FormulaError extends Error { + constructor( + readonly errorType: FormulaErrorType, + readonly display: SpreadsheetErrorCode, + message?: string, + ) { + super(message ?? display); + this.name = "FormulaError"; + } +} + +export const isFormulaError = (error: unknown): error is FormulaError => error instanceof FormulaError; + +export const divZeroError = (): FormulaError => new FormulaError("div_zero", FORMULA_ERROR_VALUES.div_zero, "Division by zero"); + +export const invalidRefError = (ref: string): FormulaError => new FormulaError("invalid_ref", FORMULA_ERROR_VALUES.invalid_ref, `Invalid reference: ${ref}`); + +export const nameError = (funcName: string): FormulaError => new FormulaError("syntax", FORMULA_ERROR_VALUES.syntax, `Unknown function: ${funcName}`); + +export const unknownError = (message?: string): FormulaError => new FormulaError("unknown", FORMULA_ERROR_VALUES.unknown, message); + +/** The FormulaError that re-raises an error read from a referenced cell or a + * nested call, so `=A1+1` inherits A1's error instead of corrupting into + * `"#DIV/0!"+1`. */ +export const propagatedError = (value: SpreadsheetErrorCode): FormulaError => + new FormulaError(ERROR_VALUE_TO_TYPE[value] ?? "unknown", value, `Propagated error: ${value}`); + +/** Map any thrown value onto the typed entry the calculator records. A + * `FormulaError` keeps its own kind and display; anything else is `unknown`. */ +export const classifyThrownError = (error: unknown): { type: FormulaErrorType; display: SpreadsheetErrorCode } => { + if (isFormulaError(error)) return { type: error.errorType, display: error.display }; + return { type: "unknown", display: FORMULA_ERROR_VALUES.unknown }; +}; diff --git a/src/engine/formulaRefs.ts b/src/engine/formulaRefs.ts new file mode 100644 index 0000000..c6cb9a6 --- /dev/null +++ b/src/engine/formulaRefs.ts @@ -0,0 +1,202 @@ +/** + * Extract the set of cells that a formula references. + * + * Extracted from `src/plugins/spreadsheet/View.vue` (was the body of + * `extractCellReferences`, cognitive complexity 32). The original + * function combined regex scanning, range expansion, single-cell + * parsing, and deduplication all in one body; splitting each concern + * into a named helper brings the top-level function well under the + * sonarjs/cognitive-complexity threshold of 15 and makes the pure + * logic unit-testable in isolation (see + * `test/plugins/spreadsheet/engine/test_formulaRefs.ts`). + * + * Tracks #175. No behavioural change — the wrapper in View.vue + * still returns exactly the same `{ row, col }` list as before. + */ + +import { columnToIndex } from "./parser.js"; + +export interface CellCoord { + row: number; + col: number; +} + +// `A1:B10`, `$A$1:$B$10`, `Sheet` refs are out of scope here — the +// caller only passes the formula body, and cross-sheet ranges never +// reached the original regex anyway. Keeping the patterns identical +// to the pre-refactor code preserves behaviour exactly. +const RANGE_REGEX = /\$?[A-Z]+\$?\d+:\$?[A-Z]+\$?\d+/g; +const CELL_REGEX = /\$?[A-Z]+\$?\d+/g; + +// Excel formulas start with `=`. Strip it for uniform handling. +// Keeps any inner `=` intact (Excel does not allow them but the +// caller may pass partial text during live editing). +export function stripFormulaPrefix(formula: string): string { + return formula.startsWith("=") ? formula.slice(1) : formula; +} + +const A1_REGEX = /^([A-Z]+)(\d+)$/; + +// One `A1` token as its column index (0-based) and its row NUMBER as written +// (1-based). The single place a column letter + row digit pair is read, so the +// range and single-cell parsers below cannot drift apart. +function parseA1Token(token: string): { col: number; rowNumber: number } | null { + const [, column, row] = token.match(A1_REGEX) ?? []; + if (column === undefined || row === undefined) return null; + return { col: columnToIndex(column), rowNumber: parseInt(row, 10) }; +} + +// The two endpoints of an `A1:B3` range, or null for anything that is not +// exactly two `A1` tokens joined by one colon. +function parseRangeEndpoints(body: string): { start: { col: number; rowNumber: number }; end: { col: number; rowNumber: number } } | null { + const [startToken, endToken, ...extra] = body.split(":"); + if (startToken === undefined || endToken === undefined || extra.length > 0) return null; + const start = parseA1Token(startToken); + const end = parseA1Token(endToken); + return start && end ? { start, end } : null; +} + +// Expand a single range token (`A1:B3`, `$A$1:$C$5`) into every +// coordinate the range covers. Returns an empty array for malformed +// input so callers never have to handle exceptions; the worst case +// is "we silently ignored a weird-looking substring," which matches +// the original inline behaviour. +export function expandRange(rangeStr: string): CellCoord[] { + const endpoints = parseRangeEndpoints(rangeStr.replace(/\$/g, "")); + if (!endpoints) return []; + const { start, end } = endpoints; + const cells: CellCoord[] = []; + for (let row = start.rowNumber - 1; row <= end.rowNumber - 1; row++) { + for (let col = start.col; col <= end.col; col++) { + cells.push({ row, col }); + } + } + return cells; +} + +// Expand a range OR a single cell into coordinates, upcasing first so +// lowercase references (`a1:b2`, which spreadsheets accept) are not dropped, +// and falling back to a single cell when there is no colon. `collectRangeValues` +// in the calculator used a range-only, case-sensitive regex, so `A1`, +// `$A$1:$A$10` and `a1:a10` all silently produced no values (#2356). Ordering +// matches `expandRange`: top-to-bottom, left-to-right. +export function expandRangeOrCell(ref: string): CellCoord[] | null { + const upper = ref.trim().toUpperCase(); + if (upper.includes(":")) { + const cells = expandRange(upper); + return cells.length > 0 ? cells : null; + } + const single = parseSingleCellRef(upper); + return single ? [single] : null; +} + +// Parse a single cell ref (`A1`, `$A$1`, `AA100`) into a coord. +// Returns null for malformed input rather than throwing — keeps the +// caller's loop flat (the engine-layer `parseCellRef` throws, which +// is fine for the evaluator but wrong for a best-effort scanner). +export function parseSingleCellRef(refStr: string): CellCoord | null { + const parsed = parseA1Token(refStr.replace(/\$/g, "")); + return parsed ? { col: parsed.col, row: parsed.rowNumber - 1 } : null; +} + +// Numeric bounds of a `A2:C10` range, with any `Sheet1!` / `'My Sheet'!` +// prefix kept verbatim so callers can rebuild sheet-qualified refs. Columns +// are 0-based (via `columnToIndex`); rows stay 1-based, matching A1 notation. +export interface RangeBounds { + sheetPrefix: string; + startCol: number; + startRow: number; + endCol: number; + endRow: number; +} + +// Split an optional sheet prefix from a range, then parse the `A2:C10` body. +// The prefix is everything up to and including the last `!`, so a quoted sheet +// name that itself contains no `!` (the common case) is preserved intact. The +// lookup functions each carried their own copy of this parse; one copy ran a +// sheet-unaware regex against the whole string and threw on `Sheet1!A2:C10` +// before the sheet-aware copy could run (#2390). Returns null for anything that +// is not a two-endpoint range so callers surface one "Invalid range" message. +export function parseRangeBounds(range: string): RangeBounds | null { + const bang = range.lastIndexOf("!"); + const sheetPrefix = bang >= 0 ? range.slice(0, bang + 1) : ""; + const body = bang >= 0 ? range.slice(bang + 1) : range; + const endpoints = parseRangeEndpoints(body); + if (!endpoints) return null; + const { start, end } = endpoints; + return { sheetPrefix, startCol: start.col, startRow: start.rowNumber, endCol: end.col, endRow: end.rowNumber }; +} + +// Excel's `0` row/column index selects the entire row/column. This scalar engine +// returns a single cell, so `0` is representable only when that dimension is one +// line long; otherwise it is out of range. +const WHOLE_LINE_INDEX = 0; +const FIRST_INDEX = 1; + +// Map a 1-based INDEX position onto a 0-based offset within a `size`-long line, +// truncating toward zero as Excel does. Returns null when the position falls +// outside the line — including a `0` (whole-line) request the scalar engine +// cannot collapse to one cell. Callers turn null into #REF!. +function lineOffset(position: number, size: number): number | null { + const index = Math.trunc(position); + if (!Number.isFinite(index)) return null; + if (index === WHOLE_LINE_INDEX) return size === 1 ? 0 : null; + if (index < FIRST_INDEX || index > size) return null; + return index - FIRST_INDEX; +} + +// Resolve INDEX(range, rowNum, colNum)'s 1-based position to an absolute cell +// within the range, or null (→ #REF!) when it falls outside. The former handler +// computed the target with no bounds check, so an out-of-range index silently +// read a cell outside the range (#2390: `INDEX(A1:A3,5)` read A5, `INDEX(A2:B5, +// 0,1)` read A1). Returned column is 0-based; row is 1-based (A1 notation). +export function resolveIndexTarget(bounds: RangeBounds, rowNum: number, colNum: number): { colIndex: number; row: number } | null { + const rowOffset = lineOffset(rowNum, bounds.endRow - bounds.startRow + 1); + const colOffset = lineOffset(colNum, bounds.endCol - bounds.startCol + 1); + if (rowOffset === null || colOffset === null) return null; + return { colIndex: bounds.startCol + colOffset, row: bounds.startRow + rowOffset }; +} + +// VLOOKUP's col_index_num / HLOOKUP's row_index_num as a 0-based offset inside +// the table, or null (→ #REF!) when it points outside. Excel rejects an index +// past the table's width; without the check the handler addressed a cell beyond +// the range and returned whatever lived there — usually a silent 0 (#2360). +export function resolveTableOffset(position: number, size: number): number | null { + // Not `lineOffset`: INDEX reads a `0` position as "the whole line", which it + // can collapse to one cell when the line is one long. A lookup index has no + // such meaning — VLOOKUP's columns are numbered from 1 — so `0` is out of + // range even for a single-column table (Codex review). + const index = Math.trunc(position); + if (!Number.isFinite(index)) return null; + if (index < FIRST_INDEX || index > size) return null; + return index - FIRST_INDEX; +} + +// Top-level: scan the formula, expand any ranges, then pick up +// remaining single-cell refs, deduplicating as we go. Kept short +// (~15 lines) so the cognitive-complexity signal lands on the +// helpers if anything grows here. +export function extractCellReferences(formula: string): CellCoord[] { + const clean = stripFormulaPrefix(formula); + const refs: CellCoord[] = []; + const seen = new Set(); + const addUnique = (coord: CellCoord): void => { + const key = `${coord.row},${coord.col}`; + if (seen.has(key)) return; + seen.add(key); + refs.push(coord); + }; + + for (const range of clean.match(RANGE_REGEX) ?? []) { + for (const coord of expandRange(range)) addUnique(coord); + } + // Strip matched ranges so the cell-regex doesn't re-emit their + // endpoints as standalone refs (mirrors the original's second + // `.replace(rangeRegex, "")` pass). + const withoutRanges = clean.replace(RANGE_REGEX, ""); + for (const cellStr of withoutRanges.match(CELL_REGEX) ?? []) { + const coord = parseSingleCellRef(cellStr); + if (coord) addUnique(coord); + } + return refs; +} diff --git a/src/engine/functions/date.ts b/src/engine/functions/date.ts index 0e6e33a..311ab70 100644 --- a/src/engine/functions/date.ts +++ b/src/engine/functions/date.ts @@ -4,51 +4,30 @@ * January 1, 1900 is serial number 1. */ -import { - functionRegistry, - toNumber, - toString, - type FunctionHandler, -} from "../registry"; +import { functionRegistry, requiredArg, toNumber, toString, type FunctionHandler } from "../registry"; +import { computeDatedif } from "../datedif"; import { dateToSerial, serialToDate } from "../date-utils"; -const MS_PER_DAY = 24 * 60 * 60 * 1000; - -const nowHandler: FunctionHandler = (args) => { - if (args.length !== 0) throw new Error("NOW requires 0 arguments"); +const nowHandler: FunctionHandler = () => { // Return current date and time as serial number // We need to adjust for timezone offset because Excel dates are "local" usually // But for simplicity we'll use local time converted to serial const now = new Date(); // Create a UTC date that matches the local time components - const localAsUtc = new Date( - Date.UTC( - now.getFullYear(), - now.getMonth(), - now.getDate(), - now.getHours(), - now.getMinutes(), - now.getSeconds(), - ), - ); + const localAsUtc = new Date(Date.UTC(now.getFullYear(), now.getMonth(), now.getDate(), now.getHours(), now.getMinutes(), now.getSeconds())); return dateToSerial(localAsUtc); }; -const todayHandler: FunctionHandler = (args) => { - if (args.length !== 0) throw new Error("TODAY requires 0 arguments"); +const todayHandler: FunctionHandler = () => { const now = new Date(); - const today = new Date( - Date.UTC(now.getFullYear(), now.getMonth(), now.getDate()), - ); + const today = new Date(Date.UTC(now.getFullYear(), now.getMonth(), now.getDate())); return dateToSerial(today); }; const dateHandler: FunctionHandler = (args, context) => { - if (args.length !== 3) throw new Error("DATE requires 3 arguments"); - - const year = toNumber(context.evaluateFormula(args[0])); - const month = toNumber(context.evaluateFormula(args[1])); - const day = toNumber(context.evaluateFormula(args[2])); + const year = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + const month = toNumber(context.evaluateFormula(requiredArg(context, args, 1))); + const day = toNumber(context.evaluateFormula(requiredArg(context, args, 2))); // JS Date constructor handles overflow (e.g. month 13 becomes Jan of next year) // Month is 0-indexed in JS, 1-indexed in Excel @@ -57,11 +36,9 @@ const dateHandler: FunctionHandler = (args, context) => { }; const timeHandler: FunctionHandler = (args, context) => { - if (args.length !== 3) throw new Error("TIME requires 3 arguments"); - - const hour = toNumber(context.evaluateFormula(args[0])); - const minute = toNumber(context.evaluateFormula(args[1])); - const second = toNumber(context.evaluateFormula(args[2])); + const hour = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + const minute = toNumber(context.evaluateFormula(requiredArg(context, args, 1))); + const second = toNumber(context.evaluateFormula(requiredArg(context, args, 2))); // Time is a fraction of a day // 1 hour = 1/24 @@ -76,29 +53,25 @@ const timeHandler: FunctionHandler = (args, context) => { }; const yearHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("YEAR requires 1 argument"); - const serial = toNumber(context.evaluateFormula(args[0])); + const serial = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); const date = serialToDate(serial); return date.getUTCFullYear(); }; const monthHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("MONTH requires 1 argument"); - const serial = toNumber(context.evaluateFormula(args[0])); + const serial = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); const date = serialToDate(serial); return date.getUTCMonth() + 1; // 1-indexed }; const dayHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("DAY requires 1 argument"); - const serial = toNumber(context.evaluateFormula(args[0])); + const serial = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); const date = serialToDate(serial); return date.getUTCDate(); }; const hourHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("HOUR requires 1 argument"); - const serial = toNumber(context.evaluateFormula(args[0])); + const serial = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); // Get fractional part const timePart = serial - Math.floor(serial); const totalSeconds = Math.round(timePart * 86400); @@ -106,103 +79,25 @@ const hourHandler: FunctionHandler = (args, context) => { }; const minuteHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("MINUTE requires 1 argument"); - const serial = toNumber(context.evaluateFormula(args[0])); + const serial = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); const timePart = serial - Math.floor(serial); const totalSeconds = Math.round(timePart * 86400); return Math.floor((totalSeconds % 3600) / 60); }; const secondHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("SECOND requires 1 argument"); - const serial = toNumber(context.evaluateFormula(args[0])); + const serial = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); const timePart = serial - Math.floor(serial); const totalSeconds = Math.round(timePart * 86400); return totalSeconds % 60; }; const datedifHandler: FunctionHandler = (args, context) => { - if (args.length !== 3) throw new Error("DATEDIF requires 3 arguments"); - - const startSerial = toNumber(context.evaluateFormula(args[0])); - const endSerial = toNumber(context.evaluateFormula(args[1])); - const unit = toString(context.evaluateFormula(args[2])).toUpperCase(); - - if (startSerial > endSerial) return "#NUM!"; - - const startDate = serialToDate(startSerial); - const endDate = serialToDate(endSerial); - - const yearDiff = endDate.getUTCFullYear() - startDate.getUTCFullYear(); - const monthDiff = endDate.getUTCMonth() - startDate.getUTCMonth(); - const dayDiff = endDate.getUTCDate() - startDate.getUTCDate(); - - switch (unit) { - case "Y": { - // Complete years - let years = yearDiff; - if (monthDiff < 0 || (monthDiff === 0 && dayDiff < 0)) { - years--; - } - return years; - } - - case "M": { - // Complete months - let months = yearDiff * 12 + monthDiff; - if (dayDiff < 0) { - months--; - } - return months; - } - - case "D": - // Complete days - return Math.floor(endSerial - startSerial); - - case "MD": { - // Difference in days, ignoring months and years - // This is tricky. It's basically day of month difference, but handling wrap around - // E.g. Jan 30 to Mar 1. - // Standard implementation: - const startD = startDate.getUTCDate(); - const endD = endDate.getUTCDate(); - - if (endD >= startD) return endD - startD; - - // Need to borrow days from previous month - const prevMonthDate = new Date( - Date.UTC(endDate.getUTCFullYear(), endDate.getUTCMonth(), 0), - ); - return prevMonthDate.getUTCDate() - startD + endD; - } - - case "YM": { - // Difference in months, ignoring years - let ym = monthDiff; - if (dayDiff < 0) ym--; - if (ym < 0) ym += 12; - return ym; - } - - case "YD": { - // Difference in days, ignoring years - // Treat start date as being in the same year as end date - // If start > end (after adjusting year), move start to previous year - const startCopy = new Date(startDate); - startCopy.setUTCFullYear(endDate.getUTCFullYear()); - - const diff = (startCopy.getTime() - endDate.getTime()) / MS_PER_DAY; - if (diff > 0) { - startCopy.setUTCFullYear(endDate.getUTCFullYear() - 1); - } - - return Math.floor((endDate.getTime() - startCopy.getTime()) / MS_PER_DAY); - } + const startSerial = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + const endSerial = toNumber(context.evaluateFormula(requiredArg(context, args, 1))); + const unit = toString(context.evaluateFormula(requiredArg(context, args, 2))); - default: - return "#NUM!"; - } + return computeDatedif(startSerial, endSerial, unit); }; // Register functions @@ -311,8 +206,7 @@ functionRegistry.register({ handler: datedifHandler, minArgs: 3, maxArgs: 3, - description: - "Calculates the number of days, months, or years between two dates", + description: "Calculates the number of days, months, or years between two dates", examples: ['DATEDIF(A1, B1, "Y")', 'DATEDIF(DATE(2020,1,1), TODAY(), "D")'], category: "Date & Time", }); diff --git a/src/engine/functions/financial.ts b/src/engine/functions/financial.ts index b0ed85a..4cbc19a 100644 --- a/src/engine/functions/financial.ts +++ b/src/engine/functions/financial.ts @@ -2,7 +2,8 @@ * Financial Functions */ -import { functionRegistry, toNumber, type FunctionHandler } from "../registry"; +import { functionRegistry, requiredArg, toNumber, type FunctionContext, type FunctionHandler } from "../registry"; +import { computeFv, computePmt, computeIpmt, computePpmt, computePv, computeNper, computeRate, computeNpv, computeIrr } from "../financial-math"; /** * FV - Future Value @@ -15,25 +16,13 @@ import { functionRegistry, toNumber, type FunctionHandler } from "../registry"; * - type: 0 = end of period, 1 = beginning of period (optional, default 0) */ const fvHandler: FunctionHandler = (args, context) => { - if (args.length < 3 || args.length > 5) { - throw new Error("FV requires 3 to 5 arguments"); - } - - const rate = toNumber(context.evaluateFormula(args[0])); - const nper = toNumber(context.evaluateFormula(args[1])); - const pmt = toNumber(context.evaluateFormula(args[2])); - const pv = args.length >= 4 ? toNumber(context.evaluateFormula(args[3])) : 0; - const type = - args.length >= 5 ? toNumber(context.evaluateFormula(args[4])) : 0; - - if (rate === 0) { - return -(pv + pmt * nper); - } + const rate = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + const nper = toNumber(context.evaluateFormula(requiredArg(context, args, 1))); + const pmt = toNumber(context.evaluateFormula(requiredArg(context, args, 2))); + const pv = args.length >= 4 ? toNumber(context.evaluateFormula(requiredArg(context, args, 3))) : 0; + const type = args.length >= 5 ? toNumber(context.evaluateFormula(requiredArg(context, args, 4))) : 0; - const pvFactor = Math.pow(1 + rate, nper); - const fv = -pv * pvFactor - (pmt * (pvFactor - 1) * (1 + rate * type)) / rate; - - return fv; + return computeFv(rate, nper, pmt, pv, type); }; /** @@ -42,26 +31,13 @@ const fvHandler: FunctionHandler = (args, context) => { * PV(rate, nper, pmt, [fv], [type]) */ const pvHandler: FunctionHandler = (args, context) => { - if (args.length < 3 || args.length > 5) { - throw new Error("PV requires 3 to 5 arguments"); - } - - const rate = toNumber(context.evaluateFormula(args[0])); - const nper = toNumber(context.evaluateFormula(args[1])); - const pmt = toNumber(context.evaluateFormula(args[2])); - const fv = args.length >= 4 ? toNumber(context.evaluateFormula(args[3])) : 0; - const type = - args.length >= 5 ? toNumber(context.evaluateFormula(args[4])) : 0; - - if (rate === 0) { - return -(fv + pmt * nper); - } - - const pvFactor = Math.pow(1 + rate, nper); - const pv = - (-fv - (pmt * (pvFactor - 1) * (1 + rate * type)) / rate) / pvFactor; + const rate = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + const nper = toNumber(context.evaluateFormula(requiredArg(context, args, 1))); + const pmt = toNumber(context.evaluateFormula(requiredArg(context, args, 2))); + const fv = args.length >= 4 ? toNumber(context.evaluateFormula(requiredArg(context, args, 3))) : 0; + const type = args.length >= 5 ? toNumber(context.evaluateFormula(requiredArg(context, args, 4))) : 0; - return pv; + return computePv(rate, nper, pmt, fv, type); }; /** @@ -70,26 +46,13 @@ const pvHandler: FunctionHandler = (args, context) => { * PMT(rate, nper, pv, [fv], [type]) */ const pmtHandler: FunctionHandler = (args, context) => { - if (args.length < 3 || args.length > 5) { - throw new Error("PMT requires 3 to 5 arguments"); - } - - const rate = toNumber(context.evaluateFormula(args[0])); - const nper = toNumber(context.evaluateFormula(args[1])); - const pv = toNumber(context.evaluateFormula(args[2])); - const fv = args.length >= 4 ? toNumber(context.evaluateFormula(args[3])) : 0; - const type = - args.length >= 5 ? toNumber(context.evaluateFormula(args[4])) : 0; + const rate = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + const nper = toNumber(context.evaluateFormula(requiredArg(context, args, 1))); + const pv = toNumber(context.evaluateFormula(requiredArg(context, args, 2))); + const fv = args.length >= 4 ? toNumber(context.evaluateFormula(requiredArg(context, args, 3))) : 0; + const type = args.length >= 5 ? toNumber(context.evaluateFormula(requiredArg(context, args, 4))) : 0; - if (rate === 0) { - return -(fv + pv) / nper; - } - - const pvFactor = Math.pow(1 + rate, nper); - const pmt = - (-rate * (fv + pv * pvFactor)) / ((pvFactor - 1) * (1 + rate * type)); - - return pmt; + return computePmt(rate, nper, pv, fv, type); }; /** @@ -98,27 +61,13 @@ const pmtHandler: FunctionHandler = (args, context) => { * NPER(rate, pmt, pv, [fv], [type]) */ const nperHandler: FunctionHandler = (args, context) => { - if (args.length < 3 || args.length > 5) { - throw new Error("NPER requires 3 to 5 arguments"); - } + const rate = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + const pmt = toNumber(context.evaluateFormula(requiredArg(context, args, 1))); + const pv = toNumber(context.evaluateFormula(requiredArg(context, args, 2))); + const fv = args.length >= 4 ? toNumber(context.evaluateFormula(requiredArg(context, args, 3))) : 0; + const type = args.length >= 5 ? toNumber(context.evaluateFormula(requiredArg(context, args, 4))) : 0; - const rate = toNumber(context.evaluateFormula(args[0])); - const pmt = toNumber(context.evaluateFormula(args[1])); - const pv = toNumber(context.evaluateFormula(args[2])); - const fv = args.length >= 4 ? toNumber(context.evaluateFormula(args[3])) : 0; - const type = - args.length >= 5 ? toNumber(context.evaluateFormula(args[4])) : 0; - - if (rate === 0) { - return -(fv + pv) / pmt; - } - - const pmtWithType = pmt * (1 + rate * type); - const nper = - Math.log((pmtWithType - fv * rate) / (pmtWithType + pv * rate)) / - Math.log(1 + rate); - - return nper; + return computeNper(rate, pmt, pv, fv, type); }; /** @@ -128,161 +77,63 @@ const nperHandler: FunctionHandler = (args, context) => { * Uses Newton-Raphson method for iteration */ const rateHandler: FunctionHandler = (args, context) => { - if (args.length < 3 || args.length > 6) { - throw new Error("RATE requires 3 to 6 arguments"); - } - - const nper = toNumber(context.evaluateFormula(args[0])); - const pmt = toNumber(context.evaluateFormula(args[1])); - const pv = toNumber(context.evaluateFormula(args[2])); - const fv = args.length >= 4 ? toNumber(context.evaluateFormula(args[3])) : 0; - const type = - args.length >= 5 ? toNumber(context.evaluateFormula(args[4])) : 0; - const guess = - args.length >= 6 ? toNumber(context.evaluateFormula(args[5])) : 0.1; - - // Use Newton-Raphson method to find rate - let rate = guess; - const maxIterations = 100; - const tolerance = 1e-7; - - for (let i = 0; i < maxIterations; i++) { - if (Math.abs(rate) < tolerance) { - rate = tolerance; // Avoid division by zero - } - - const y = Math.pow(1 + rate, nper); - const f = pv * y + pmt * ((y - 1) / rate) * (1 + rate * type) + fv; - - const df = - nper * pv * Math.pow(1 + rate, nper - 1) + - (pmt * - (1 + rate * type) * - (nper * Math.pow(1 + rate, nper - 1) * rate - - (Math.pow(1 + rate, nper) - 1))) / - (rate * rate) + - pmt * type * ((Math.pow(1 + rate, nper) - 1) / rate); + const nper = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + const pmt = toNumber(context.evaluateFormula(requiredArg(context, args, 1))); + const pv = toNumber(context.evaluateFormula(requiredArg(context, args, 2))); + const fv = args.length >= 4 ? toNumber(context.evaluateFormula(requiredArg(context, args, 3))) : 0; + const type = args.length >= 5 ? toNumber(context.evaluateFormula(requiredArg(context, args, 4))) : 0; + const guess = args.length >= 6 ? toNumber(context.evaluateFormula(requiredArg(context, args, 5))) : 0.1; + + return computeRate(nper, pmt, pv, fv, type, guess); +}; - const newRate = rate - f / df; +// IPMT and PPMT take the SAME six operands — (rate, per, nper, pv, [fv], [type]) +// — and differ only in which annuity component they compute. One factory parses +// the shared shape so the two handlers can't drift in their optional-arg defaults. +type PeriodicComponent = (rate: number, per: number, nper: number, pv: number, fv: number, type: number) => number; - if (Math.abs(newRate - rate) < tolerance) { - return newRate; - } +const makePeriodicComponentHandler = + (compute: PeriodicComponent): FunctionHandler => + (args, context) => { + const rate = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + const per = toNumber(context.evaluateFormula(requiredArg(context, args, 1))); + const nper = toNumber(context.evaluateFormula(requiredArg(context, args, 2))); + const pv = toNumber(context.evaluateFormula(requiredArg(context, args, 3))); + const fv = args.length >= 5 ? toNumber(context.evaluateFormula(requiredArg(context, args, 4))) : 0; + const type = args.length >= 6 ? toNumber(context.evaluateFormula(requiredArg(context, args, 5))) : 0; - rate = newRate; - } - - return rate; -}; + return compute(rate, per, nper, pv, fv, type); + }; /** * IPMT - Interest Payment * Calculates the interest payment for a given period for an investment based on periodic, constant payments and a constant interest rate. * IPMT(rate, per, nper, pv, [fv], [type]) */ -const ipmtHandler: FunctionHandler = (args, context) => { - if (args.length < 4 || args.length > 6) { - throw new Error("IPMT requires 4 to 6 arguments"); - } - - const rate = toNumber(context.evaluateFormula(args[0])); - const per = toNumber(context.evaluateFormula(args[1])); - const type = - args.length >= 6 ? toNumber(context.evaluateFormula(args[5])) : 0; - - // Calculate payment first - const pmt = pmtHandler( - [ - args[0], - args[2], - args[3], - ...(args.length >= 5 ? [args[4]] : []), - ...(args.length >= 6 ? [args[5]] : []), - ], - context, - ); - - if (per === 1 && type === 1) { - return 0; // No interest in first period when payment is at beginning - } - - // Calculate remaining balance at previous period - const fvPrevious = fvHandler( - [ - args[0], - String(type === 1 ? per - 2 : per - 1), - String(pmt), - args[3], - ...(args.length >= 6 ? [args[5]] : []), - ], - context, - ); - - const ipmt = -fvPrevious * rate; - - return type === 1 ? ipmt / (1 + rate) : ipmt; -}; +const ipmtHandler: FunctionHandler = makePeriodicComponentHandler(computeIpmt); /** * PPMT - Principal Payment * Calculates the payment on the principal for a given period for an investment based on periodic, constant payments and a constant interest rate. * PPMT(rate, per, nper, pv, [fv], [type]) */ -const ppmtHandler: FunctionHandler = (args, context) => { - if (args.length < 4 || args.length > 6) { - throw new Error("PPMT requires 4 to 6 arguments"); - } - - // Calculate total payment - const pmt = pmtHandler( - [ - args[0], - args[2], - args[3], - ...(args.length >= 5 ? [args[4]] : []), - ...(args.length >= 6 ? [args[5]] : []), - ], - context, - ); - - // Calculate interest payment - const ipmt = ipmtHandler(args, context); - - // Principal payment = Total payment - Interest payment - return toNumber(pmt) - toNumber(ipmt); -}; +const ppmtHandler: FunctionHandler = makePeriodicComponentHandler(computePpmt); /** * NPV - Net Present Value * Calculates the net present value of an investment based on a discount rate and a series of future cash flows. * NPV(rate, value1, [value2], ...) */ -const npvHandler: FunctionHandler = (args, context) => { - if (args.length < 2) { - throw new Error("NPV requires at least 2 arguments"); - } +// Flatten NPV's value arguments into one ordered list of numeric cash flows: +// ranges expand in place (numeric cells only), scalars contribute one value. +// The order defines each flow's discount period, so ranges and the scalars +// after them must stay in argument order. +const collectCashFlows = (valueArgs: string[], context: FunctionContext): number[] => + valueArgs.flatMap((arg) => (arg.includes(":") ? context.getRangeValues(arg) : [context.evaluateFormula(arg)]).map(toNumber)); - const rate = toNumber(context.evaluateFormula(args[0])); - let npv = 0; - - // Process each cash flow - for (let i = 1; i < args.length; i++) { - // Check if argument is a range - if (args[i].includes(":")) { - const values = context.getRangeValues(args[i]); - for (let j = 0; j < values.length; j++) { - const value = toNumber(values[j]); - // Period starts from i for first range element, then continues - const period = i + j; - npv += value / Math.pow(1 + rate, period); - } - } else { - const value = toNumber(context.evaluateFormula(args[i])); - npv += value / Math.pow(1 + rate, i); - } - } - - return npv; +const npvHandler: FunctionHandler = (args, context) => { + const rate = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + return computeNpv(rate, collectCashFlows(args.slice(1), context)); }; /** @@ -292,51 +143,14 @@ const npvHandler: FunctionHandler = (args, context) => { * Uses Newton-Raphson method for iteration */ const irrHandler: FunctionHandler = (args, context) => { - if (args.length < 1 || args.length > 2) { - throw new Error("IRR requires 1 or 2 arguments"); - } - - const values = context.getRangeValues(args[0]).map(toNumber); - const guess = - args.length === 2 ? toNumber(context.evaluateFormula(args[1])) : 0.1; + const values = context.getRangeValues(requiredArg(context, args, 0)).map(toNumber); + const guess = args.length === 2 ? toNumber(context.evaluateFormula(requiredArg(context, args, 1))) : 0.1; if (values.length === 0) { throw new Error("IRR requires at least one value"); } - // Use Newton-Raphson method - let rate = guess; - const maxIterations = 100; - const tolerance = 1e-7; - - for (let i = 0; i < maxIterations; i++) { - let npv = 0; - let dnpv = 0; - - for (let j = 0; j < values.length; j++) { - const factor = Math.pow(1 + rate, j); - npv += values[j] / factor; - dnpv -= (j * values[j]) / (factor * (1 + rate)); - } - - if (Math.abs(npv) < tolerance) { - return rate; - } - - if (Math.abs(dnpv) < tolerance) { - throw new Error("IRR cannot converge"); - } - - const newRate = rate - npv / dnpv; - - if (Math.abs(newRate - rate) < tolerance) { - return newRate; - } - - rate = newRate; - } - - return rate; + return computeIrr(values, guess); }; // Register all financial functions @@ -345,8 +159,7 @@ functionRegistry.register({ handler: fvHandler, minArgs: 3, maxArgs: 5, - description: - "Calculates the future value of an investment based on periodic, constant payments and a constant interest rate", + description: "Calculates the future value of an investment based on periodic, constant payments and a constant interest rate", examples: ["FV(0.06/12, 12, -100, -1000, 1)", "FV(A1, A2, A3)"], category: "Financial", }); @@ -356,8 +169,7 @@ functionRegistry.register({ handler: pvHandler, minArgs: 3, maxArgs: 5, - description: - "Calculates the present value of an investment based on periodic, constant payments and a constant interest rate", + description: "Calculates the present value of an investment based on periodic, constant payments and a constant interest rate", examples: ["PV(0.08/12, 12*20, 500, 0, 0)", "PV(A1, A2, A3)"], category: "Financial", }); @@ -367,8 +179,7 @@ functionRegistry.register({ handler: pmtHandler, minArgs: 3, maxArgs: 5, - description: - "Calculates the payment for a loan based on constant payments and a constant interest rate", + description: "Calculates the payment for a loan based on constant payments and a constant interest rate", examples: ["PMT(0.06/12, 30*12, 250000)", "PMT(A1, A2, A3)"], category: "Financial", }); @@ -378,8 +189,7 @@ functionRegistry.register({ handler: nperHandler, minArgs: 3, maxArgs: 5, - description: - "Calculates the number of periods for an investment based on periodic, constant payments and a constant interest rate", + description: "Calculates the number of periods for an investment based on periodic, constant payments and a constant interest rate", examples: ["NPER(0.06/12, -1000, 50000, 0, 0)", "NPER(A1, A2, A3)"], category: "Financial", }); @@ -389,8 +199,7 @@ functionRegistry.register({ handler: rateHandler, minArgs: 3, maxArgs: 6, - description: - "Calculates the interest rate per period of an annuity using iteration", + description: "Calculates the interest rate per period of an annuity using iteration", examples: ["RATE(12, -100, 1000, 0, 0, 0.1)", "RATE(A1, A2, A3)"], category: "Financial", }); @@ -400,8 +209,7 @@ functionRegistry.register({ handler: ipmtHandler, minArgs: 4, maxArgs: 6, - description: - "Calculates the interest payment for a given period for an investment", + description: "Calculates the interest payment for a given period for an investment", examples: ["IPMT(0.06/12, 1, 30*12, 250000)", "IPMT(A1, A2, A3, A4)"], category: "Financial", }); @@ -411,8 +219,7 @@ functionRegistry.register({ handler: ppmtHandler, minArgs: 4, maxArgs: 6, - description: - "Calculates the payment on the principal for a given period for an investment", + description: "Calculates the payment on the principal for a given period for an investment", examples: ["PPMT(0.06/12, 1, 30*12, 250000)", "PPMT(A1, A2, A3, A4)"], category: "Financial", }); @@ -421,8 +228,7 @@ functionRegistry.register({ name: "NPV", handler: npvHandler, minArgs: 2, - description: - "Calculates the net present value of an investment based on a discount rate and a series of future cash flows", + description: "Calculates the net present value of an investment based on a discount rate and a series of future cash flows", examples: ["NPV(0.1, -10000, 3000, 4200, 6800)", "NPV(A1, B1:B10)"], category: "Financial", }); @@ -432,8 +238,7 @@ functionRegistry.register({ handler: irrHandler, minArgs: 1, maxArgs: 2, - description: - "Calculates the internal rate of return for a series of cash flows", + description: "Calculates the internal rate of return for a series of cash flows", examples: ["IRR(A1:A5, 0.1)", "IRR(B1:B10)"], category: "Financial", }); diff --git a/src/engine/functions/logical.ts b/src/engine/functions/logical.ts index aa65377..515df31 100644 --- a/src/engine/functions/logical.ts +++ b/src/engine/functions/logical.ts @@ -2,32 +2,21 @@ * Logical Functions */ - - -import { functionRegistry, type FunctionHandler } from "../registry"; +import { evaluateConditionValues, readOperand, renderConditionOperand } from "../condition"; +import { findCellRefs, substituteCellRefs } from "../evaluator"; +import { functionRegistry, requiredArg, type FunctionHandler } from "../registry"; +import { isErrorResult, isSpreadsheetErrorValue, NA_ERROR } from "../spreadsheet-errors"; +import { coerceToBoolean } from "../coerce-boolean"; +import type { CellValue } from "../types"; const ifHandler: FunctionHandler = (args, context) => { - if (args.length !== 3) throw new Error("IF requires 3 arguments"); - - const condition = args[0]; - const trueValue = args[1]; - const falseValue = args[2]; + const condition = requiredArg(context, args, 0); + const trueValue = requiredArg(context, args, 1); + const falseValue = requiredArg(context, args, 2); // Evaluate condition - use evaluateFormula to handle nested functions like MONTH() const conditionValue = context.evaluateFormula(condition); - - // Convert to boolean - let conditionResult = false; - if (typeof conditionValue === "boolean") { - conditionResult = conditionValue; - } else if (typeof conditionValue === "number") { - conditionResult = conditionValue !== 0; - } else if (typeof conditionValue === "string") { - conditionResult = - conditionValue.toLowerCase() === "true" || conditionValue !== ""; - } else { - conditionResult = !!conditionValue; - } + const conditionResult = coerceToBoolean(conditionValue); // Return the appropriate value based on condition const resultValue = conditionResult ? trueValue : falseValue; @@ -37,40 +26,17 @@ const ifHandler: FunctionHandler = (args, context) => { return resultValue.slice(1, -1); } - // If result is a nested formula, evaluate it recursively - if (/^(SUM|AVERAGE|MAX|MIN|COUNT|IF|AND|OR|NOT)\(/i.test(resultValue)) { - return context.evaluateFormula(resultValue); - } - - // Otherwise evaluate as expression - let expr = resultValue; - - const refs = resultValue.match( - /(?:'[^']+'|[^'!\s]+)![A-Z]+\d+|\$?[A-Z]+\$?\d+/g, - ); - if (refs) { - for (const ref of refs) { - const value = context.getCellValue(ref); - const escapedRef = ref.replace(/\$/g, "\\$").replace(/'/g, "\\'"); - expr = expr.replace( - new RegExp(escapedRef.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"), "g"), - String(value), - ); - } - } - - const numResult = parseFloat(expr); - return isNaN(numResult) ? expr : numResult; + // Everything else — a nested call, an arithmetic expression, a reference — is + // evaluated by the engine. Hand-rolling it here silently returned a plausible + // wrong value twice over: a hard-coded list of nine function names sent + // `ROUND(A1,1)` back as its own text, and the fallback read `A1+1` through + // `parseFloat("3+1")`, yielding 3. + return context.evaluateFormula(resultValue); }; const andHandler: FunctionHandler = (args, context) => { - if (args.length === 0) throw new Error("AND requires at least 1 argument"); - for (const arg of args) { - const value = context.evaluateFormula(arg.trim()); - // Check if value is falsy (0, false, empty string, etc.) - // Note: !value already covers false, so we check for 0 and "0" explicitly - if (!value || value === 0 || value === "0") { + if (!coerceToBoolean(context.evaluateFormula(arg.trim()))) { return false; } } @@ -78,12 +44,8 @@ const andHandler: FunctionHandler = (args, context) => { }; const orHandler: FunctionHandler = (args, context) => { - if (args.length === 0) throw new Error("OR requires at least 1 argument"); - for (const arg of args) { - const value = context.evaluateFormula(arg.trim()); - // Check if value is truthy (non-zero, non-empty) - if (value && value !== 0 && value !== "0") { + if (coerceToBoolean(context.evaluateFormula(arg.trim()))) { return true; } } @@ -91,83 +53,67 @@ const orHandler: FunctionHandler = (args, context) => { }; const notHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("NOT requires 1 argument"); - - const value = context.evaluateFormula(args[0]); - // Note: !value already covers false - return !value || value === 0 || value === "0"; + return !coerceToBoolean(context.evaluateFormula(requiredArg(context, args, 0))); }; const iferrorHandler: FunctionHandler = (args, context) => { - if (args.length !== 2) throw new Error("IFERROR requires 2 arguments"); - try { - const result = context.evaluateFormula(args[0]); - // Check if result is an error (NaN, Infinity, etc.) - if ( - result === null || - result === undefined || - (typeof result === "number" && (isNaN(result) || !isFinite(result))) - ) { - return context.evaluateFormula(args[1]); + const result = context.evaluateFormula(requiredArg(context, args, 0)); + // Catches NaN/∞ and the formula error VALUES functions return (a math domain + // miss like SQRT(-1) → #NUM!), so IFERROR(SQRT(-1), 0) is 0. Text that + // merely spells an error is not an error value, so it passes through + // whether it was written as a literal (IFERROR("#NUM!", 42)) or computed + // (IFERROR(CONCAT("#N","UM!"), 42)) — the computed case is why errors carry + // provenance at all (#2451). + if (isErrorResult(result)) { + return context.evaluateFormula(requiredArg(context, args, 1)); } return result; } catch { // If evaluation throws an error, return the fallback value - return context.evaluateFormula(args[1]); + return context.evaluateFormula(requiredArg(context, args, 1)); } }; const ifnaHandler: FunctionHandler = (args, context) => { - if (args.length !== 2) throw new Error("IFNA requires 2 arguments"); - - const result = context.evaluateFormula(args[0]); - // Check if result is N/A (could be represented as specific error value) - if (result === null || result === undefined || result === "#N/A") { - return context.evaluateFormula(args[1]); + const result = context.evaluateFormula(requiredArg(context, args, 0)); + const isNotAvailable = isSpreadsheetErrorValue(result) && result.code === NA_ERROR.code; + if (result === null || result === undefined || isNotAvailable) { + return context.evaluateFormula(requiredArg(context, args, 1)); } return result; }; const ifsHandler: FunctionHandler = (args, context) => { if (args.length < 2 || args.length % 2 !== 0) { - throw new Error( - "IFS requires an even number of arguments (condition-value pairs)", - ); + throw new Error("IFS requires an even number of arguments (condition-value pairs)"); } // Iterate through condition-value pairs for (let i = 0; i < args.length; i += 2) { - const condition = args[i]; - const value = args[i + 1]; - - // Evaluate condition - let condExpr = condition; - - const cellRefs = condition.match( - /(?:'[^']+'|[^'!\s]+)![A-Z]+\d+|\$?[A-Z]+\$?\d+/g, - ); - if (cellRefs) { - for (const ref of cellRefs) { - const cellValue = context.getCellValue(ref); - const escapedRef = ref.replace(/\$/g, "\\$").replace(/'/g, "\\'"); - condExpr = condExpr.replace( - new RegExp(escapedRef.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"), "g"), - String(cellValue), - ); - } - } - - // Evaluate the condition - let conditionResult = false; - - if (/>=|<=|>|<|==|!=/.test(condExpr)) { - conditionResult = eval(condExpr); - } else { - conditionResult = !!eval(condExpr); - } - - if (conditionResult) { + const condition = requiredArg(context, args, i); + const value = requiredArg(context, args, i + 1); + + // Substitute references by POSITION (back to front), skipping any that sit + // inside a quoted string literal: `IFS(A1="B2", …)` must compare A1 to the + // TEXT "B2", not to cell B2's value (Codex review). findCellRefs already + // skips literals and matches absolute / sheet-qualified refs, so this also + // avoids the earlier regex double-escaping. renderConditionOperand quotes a + // text cell so its own operators are not re-parsed as comparisons. + const condExpr = substituteCellRefs(condition, findCellRefs(condition), (ref) => renderConditionOperand(context.getCellValue(ref))); + + // Parsed, not executed. This used to call `eval` on `condExpr`, which is + // the substituted text — so a cell containing `globalThis.x = 1` ran as + // code whenever an IFS referenced it, and so did anything written into the + // formula itself. `readOperand` resolves the simple operands (TRUE/FALSE -> + // boolean, quoted text, numbers); an arithmetic expression it leaves as raw + // text is handed to the engine's safe evaluator so `A1+1>10` is computed. + // Only the top-level comparison is applied — the condition is never run. + const evaluateOperand = (operand: string): CellValue => { + const parsed = readOperand(operand); + return typeof parsed === "string" && parsed === operand.trim() ? context.evaluateFormula(operand) : parsed; + }; + if (evaluateConditionValues(condExpr, evaluateOperand)) { // If result is a quoted string, return without quotes if (/^["'](.*)["']$/.test(value)) { @@ -179,18 +125,12 @@ const ifsHandler: FunctionHandler = (args, context) => { } // If no conditions match, return error - return "#N/A"; + return NA_ERROR; }; -const trueHandler: FunctionHandler = (args) => { - if (args.length !== 0) throw new Error("TRUE requires 0 arguments"); - return true; -}; +const trueHandler: FunctionHandler = () => true; -const falseHandler: FunctionHandler = (args) => { - if (args.length !== 0) throw new Error("FALSE requires 0 arguments"); - return false; -}; +const falseHandler: FunctionHandler = () => false; // Register all logical functions functionRegistry.register({ @@ -236,12 +176,8 @@ functionRegistry.register({ handler: iferrorHandler, minArgs: 2, maxArgs: 2, - description: - "Returns a value if expression is an error, otherwise returns the expression", - examples: [ - "IFERROR(A1/B1, 0)", - 'IFERROR(VLOOKUP(A1, B1:C10, 2), "Not found")', - ], + description: "Returns a value if expression is an error, otherwise returns the expression", + examples: ["IFERROR(A1/B1, 0)", 'IFERROR(VLOOKUP(A1, B1:C10, 2), "Not found")'], category: "Logical", }); @@ -250,8 +186,7 @@ functionRegistry.register({ handler: ifnaHandler, minArgs: 2, maxArgs: 2, - description: - "Returns a value if expression is #N/A, otherwise returns the expression", + description: "Returns a value if expression is #N/A, otherwise returns the expression", examples: ['IFNA(A1, "N/A")', "IFNA(MATCH(A1, B1:B10), 0)"], category: "Logical", }); @@ -261,10 +196,7 @@ functionRegistry.register({ handler: ifsHandler, minArgs: 2, description: "Checks multiple conditions and returns the first true result", - examples: [ - 'IFS(A1>90, "A", A1>80, "B", A1>70, "C")', - 'IFS(B1="Yes", 1, B1="No", 0)', - ], + examples: ['IFS(A1>90, "A", A1>80, "B", A1>70, "C")', 'IFS(B1="Yes", 1, B1="No", 0)'], category: "Logical", }); diff --git a/src/engine/functions/lookup-math.ts b/src/engine/functions/lookup-math.ts new file mode 100644 index 0000000..aa12348 --- /dev/null +++ b/src/engine/functions/lookup-math.ts @@ -0,0 +1,28 @@ +/** + * Pure lookup rules, separated from the range-reading handlers so they can be + * unit-tested directly. + */ + +import type { CellValue } from "../types"; +import { isSpreadsheetErrorValue } from "../spreadsheet-errors"; + +/** + * Interpret VLOOKUP/HLOOKUP's 4th argument (range_lookup): TRUE (or omitted) = + * approximate match, FALSE = exact. + * + * The literal `TRUE` reaches the handler as the STRING `"TRUE"` — the evaluator + * leaves bare words unquoted — so an accept-only-`true | 1 | "1"` check silently + * fell back to exact match and returned `#N/A` for a valid approximate lookup + * (#2360). Read it the way Excel coerces a logical instead: a boolean as-is, a + * non-zero number as TRUE, and the words `TRUE` / `FALSE` (case-insensitive) + * and `"1"` / `"0"` as their logical value. Anything else (blank, stray text) + * is treated as FALSE, matching Excel's coercion of a blank range_lookup — and + * so is a formula error, which is neither of the two logicals. + */ +export function isApproximateMatch(rangeLookup: CellValue): boolean { + if (typeof rangeLookup === "boolean") return rangeLookup; + if (typeof rangeLookup === "number") return rangeLookup !== 0; + if (isSpreadsheetErrorValue(rangeLookup)) return false; + const normalized = rangeLookup.trim().toUpperCase(); + return normalized === "TRUE" || normalized === "1"; +} diff --git a/src/engine/functions/lookup.ts b/src/engine/functions/lookup.ts index fd3f4ec..de96924 100644 --- a/src/engine/functions/lookup.ts +++ b/src/engine/functions/lookup.ts @@ -2,33 +2,35 @@ * Lookup and Reference Functions */ -import { - functionRegistry, - toNumber, - parseCriteria, - type FunctionHandler, -} from "../registry"; +import { functionRegistry, rawRangeReader, requiredArg, toNumber, parseCriteria, type FunctionHandler, type FunctionContext } from "../registry"; +import { indexToColumn } from "../parser"; +import { parseRangeBounds, resolveIndexTarget, resolveTableOffset } from "../formulaRefs"; +import { isApproximateMatch } from "./lookup-math"; import type { CellValue } from "../types"; - -// Helper to convert Excel column letters to 0-based index (A=0, Z=25, AA=26, etc.) -const colToIndex = (col: string): number => { - let result = 0; - for (let i = 0; i < col.length; i++) { - result = result * 26 + (col.charCodeAt(i) - 64); - } - return result - 1; -}; - -// Helper to convert 0-based index to Excel column letters (0=A, 25=Z, 26=AA, etc.) -const indexToCol = (index: number): string => { - let col = ""; - let num = index + 1; - while (num > 0) { - const remainder = (num - 1) % 26; - col = String.fromCharCode(65 + remainder) + col; - num = Math.floor((num - 1) / 26); - } - return col; +import { NA_ERROR, REF_ERROR } from "../spreadsheet-errors"; + +const inclusiveRange = (start: number, end: number): number[] => Array.from({ length: Math.max(0, end - start + 1) }, (_, i) => start + i); + +// Read a vertical slice (one column, `startRow`..`endRow`) as VLOOKUP's lookup column. +const columnValues = (context: FunctionContext, sheetPrefix: string, colStr: string, startRow: number, endRow: number): CellValue[] => + inclusiveRange(startRow, endRow).map((r) => context.getCellValue(`${sheetPrefix}${colStr}${r}`)); + +// Read a horizontal slice (one row, `startCol`..`endCol`) as HLOOKUP's lookup row. +const rowValues = (context: FunctionContext, sheetPrefix: string, row: number, startCol: number, endCol: number): CellValue[] => + inclusiveRange(startCol, endCol).map((c) => context.getCellValue(`${sheetPrefix}${indexToColumn(c)}${row}`)); + +/** The index of the LAST value that satisfies `matches`, or -1. Array.findIndex + * only walks forward, and XLOOKUP's `searchMode: -1` searches from the end. */ +const findLastMatchIndex = (values: CellValue[], matches: (value: CellValue) => boolean): number => + values.reduce((found, value, index) => (matches(value) ? index : found), -1); + +/** The index of the last value in a SORTED list before `keepGoing` stops + * holding, or -1 when the very first value already fails. The list is assumed + * sorted (as Excel's approximate match requires), so the first failure ends the + * run. */ +const lastIndexWhile = (values: CellValue[], keepGoing: (value: CellValue) => boolean): number => { + const stop = values.findIndex((value) => !keepGoing(value)); + return (stop === -1 ? values.length : stop) - 1; }; // Helper to find match index @@ -45,271 +47,110 @@ const findMatchIndex = ( // Exact match if (matchType === 0) { - // Handle wildcards for strings if it's an exact match request - if ( - typeof lookupValue === "string" && - (lookupValue.includes("*") || lookupValue.includes("?")) - ) { - const criteriaFn = parseCriteria(lookupValue); - - if (searchMode === 1) { - return lookupArray.findIndex((item) => criteriaFn(item)); - } else { - for (let i = lookupArray.length - 1; i >= 0; i--) { - if (criteriaFn(lookupArray[i])) return i; - } - return -1; - } - } - - if (searchMode === 1) { - return lookupArray.findIndex((item) => item == lookupValue); // Loose equality for "10" == 10 - } else { - for (let i = lookupArray.length - 1; i >= 0; i--) { - if (lookupArray[i] == lookupValue) return i; - } - return -1; - } + // Handle wildcards for strings if it's an exact match request. + // Loose equality otherwise, for "10" == 10. + const usesWildcards = typeof lookupValue === "string" && (lookupValue.includes("*") || lookupValue.includes("?")); + const isMatch = usesWildcards ? parseCriteria(lookupValue) : (item: CellValue) => item == lookupValue; + return searchMode === 1 ? lookupArray.findIndex(isMatch) : findLastMatchIndex(lookupArray, isMatch); } // Approximate match (requires sorted array) // We'll assume the user knows what they are doing regarding sorting, as per Excel behavior - if (matchType === 1) { - // Less than or equal to - // Array must be sorted ascending - let bestIdx = -1; - for (let i = 0; i < lookupArray.length; i++) { - const item = lookupArray[i]; - if (compare(item, lookupValue) <= 0) { - bestIdx = i; - } else { - // Since it's sorted ascending, once we exceed, we can stop - break; - } - } - return bestIdx; - } + // Less than or equal to, over an array sorted ascending: once we exceed, stop. + if (matchType === 1) return lastIndexWhile(lookupArray, (item) => compare(item, lookupValue) <= 0); - if (matchType === -1) { - // Greater than or equal to - // Array must be sorted descending - let bestIdx = -1; - for (let i = 0; i < lookupArray.length; i++) { - const item = lookupArray[i]; - if (compare(item, lookupValue) >= 0) { - bestIdx = i; - } else { - break; - } - } - return bestIdx; - } + // Greater than or equal to, over an array sorted descending. + if (matchType === -1) return lastIndexWhile(lookupArray, (item) => compare(item, lookupValue) >= 0); return -1; }; const vlookupHandler: FunctionHandler = (args, context) => { - if (args.length < 3 || args.length > 4) { - throw new Error("VLOOKUP requires 3 or 4 arguments"); - } - - const lookupValue = context.evaluateFormula(args[0]); - const tableArrayRange = args[1]; - const colIndexNum = toNumber(context.evaluateFormula(args[2])); - const rangeLookup = - args.length === 4 ? context.evaluateFormula(args[3]) : true; - - // Convert rangeLookup to boolean/number logic - // TRUE/1/omitted = approximate match (default) - // FALSE/0 = exact match - const isApprox = - rangeLookup === true || rangeLookup === 1 || rangeLookup === "1"; - const matchType = isApprox ? 1 : 0; - - // Get the full table data - // We need to parse the range string to get dimensions - const match = tableArrayRange.match(/^([A-Z]+)(\d+):([A-Z]+)(\d+)$/); - if (!match) throw new Error("Invalid table array range"); - - // We need to get the first column for looking up - // And the specific column for the result - // This is a bit tricky with the current getRangeValues which flattens everything - // We need to manually reconstruct the table structure or request specific cells - - // Let's parse the range to get start/end col/row - // Note: This relies on the context.getCellValue implementation details or we need to implement - // a smarter way to get 2D data. - // For now, we will iterate row by row. - - // Parse range manually to get boundaries - // We can't easily use context.getRangeValues because it flattens 2D arrays to 1D - // So we will iterate through the rows of the first column - - // Extract sheet name if present - let sheetName = ""; - let rangePart = tableArrayRange; - if (tableArrayRange.includes("!")) { - const parts = tableArrayRange.split("!"); - sheetName = parts[0] + "!"; - rangePart = parts[1]; - } - - const rangeMatch = rangePart.match(/^([A-Z]+)(\d+):([A-Z]+)(\d+)$/); - if (!rangeMatch) throw new Error("Invalid range format"); - - const startColStr = rangeMatch[1]; - const startRow = parseInt(rangeMatch[2]); - const endRow = parseInt(rangeMatch[4]); - - const startColIdx = colToIndex(startColStr); - const resultColIdx = startColIdx + colIndexNum - 1; - const resultColStr = indexToCol(resultColIdx); - - // Build lookup array (first column) - const lookupArray: CellValue[] = []; - for (let r = startRow; r <= endRow; r++) { - const cellRef = `${sheetName}${startColStr}${r}`; - lookupArray.push(context.getCellValue(cellRef)); - } - + const lookupValue = context.evaluateFormula(requiredArg(context, args, 0)); + const bounds = parseRangeBounds(requiredArg(context, args, 1)); + if (!bounds) throw new Error("Invalid table array range"); + const colIndexNum = toNumber(context.evaluateFormula(requiredArg(context, args, 2))); + const rangeLookup = args.length === 4 ? context.evaluateFormula(requiredArg(context, args, 3)) : true; + const matchType = isApproximateMatch(rangeLookup) ? 1 : 0; + + // Excel rejects a col_index_num past the table's width; reading on regardless + // addressed a cell outside the range and usually returned a silent 0 (#2360). + // Checked before the search: an out-of-range index is an argument error, so it + // wins over the #N/A a missing key would otherwise mask it with (Codex review). + const colOffset = resolveTableOffset(colIndexNum, bounds.endCol - bounds.startCol + 1); + if (colOffset === null) return REF_ERROR; + + const startColStr = indexToColumn(bounds.startCol); + const lookupArray = columnValues(context, bounds.sheetPrefix, startColStr, bounds.startRow, bounds.endRow); const matchIdx = findMatchIndex(lookupValue, lookupArray, matchType); + if (matchIdx === -1) return NA_ERROR; - if (matchIdx === -1) return "#N/A"; - - // Get result - const resultRow = startRow + matchIdx; - const resultRef = `${sheetName}${resultColStr}${resultRow}`; - - return context.getCellValue(resultRef); + const resultColStr = indexToColumn(bounds.startCol + colOffset); + const resultRow = bounds.startRow + matchIdx; + return context.getCellValue(`${bounds.sheetPrefix}${resultColStr}${resultRow}`); }; const hlookupHandler: FunctionHandler = (args, context) => { - if (args.length < 3 || args.length > 4) { - throw new Error("HLOOKUP requires 3 or 4 arguments"); - } - - const lookupValue = context.evaluateFormula(args[0]); - const tableArrayRange = args[1]; - const rowIndexNum = toNumber(context.evaluateFormula(args[2])); - const rangeLookup = - args.length === 4 ? context.evaluateFormula(args[3]) : true; - - const isApprox = - rangeLookup === true || rangeLookup === 1 || rangeLookup === "1"; - const matchType = isApprox ? 1 : 0; - - // Extract sheet name if present - let sheetName = ""; - let rangePart = tableArrayRange; - if (tableArrayRange.includes("!")) { - const parts = tableArrayRange.split("!"); - sheetName = parts[0] + "!"; - rangePart = parts[1]; - } + const lookupValue = context.evaluateFormula(requiredArg(context, args, 0)); + const bounds = parseRangeBounds(requiredArg(context, args, 1)); + if (!bounds) throw new Error("Invalid range format"); + const rowIndexNum = toNumber(context.evaluateFormula(requiredArg(context, args, 2))); + const rangeLookup = args.length === 4 ? context.evaluateFormula(requiredArg(context, args, 3)) : true; + const matchType = isApproximateMatch(rangeLookup) ? 1 : 0; - const rangeMatch = rangePart.match(/^([A-Z]+)(\d+):([A-Z]+)(\d+)$/); - if (!rangeMatch) throw new Error("Invalid range format"); - - const startColStr = rangeMatch[1]; - const endColStr = rangeMatch[3]; - const startRow = parseInt(rangeMatch[2]); - - const startColIdx = colToIndex(startColStr); - const endColIdx = colToIndex(endColStr); - - // Build lookup array (first row) - const lookupArray: CellValue[] = []; - for (let c = startColIdx; c <= endColIdx; c++) { - const colStr = indexToCol(c); - const cellRef = `${sheetName}${colStr}${startRow}`; - lookupArray.push(context.getCellValue(cellRef)); - } + const rowOffset = resolveTableOffset(rowIndexNum, bounds.endRow - bounds.startRow + 1); + if (rowOffset === null) return REF_ERROR; + const lookupArray = rowValues(context, bounds.sheetPrefix, bounds.startRow, bounds.startCol, bounds.endCol); const matchIdx = findMatchIndex(lookupValue, lookupArray, matchType); + if (matchIdx === -1) return NA_ERROR; - if (matchIdx === -1) return "#N/A"; - - // Get result - const resultColIdx = startColIdx + matchIdx; - const resultColStr = indexToCol(resultColIdx); - const resultRow = startRow + rowIndexNum - 1; - const resultRef = `${sheetName}${resultColStr}${resultRow}`; - - return context.getCellValue(resultRef); + const resultColStr = indexToColumn(bounds.startCol + matchIdx); + const resultRow = bounds.startRow + rowOffset; + return context.getCellValue(`${bounds.sheetPrefix}${resultColStr}${resultRow}`); }; const matchHandler: FunctionHandler = (args, context) => { - if (args.length < 2 || args.length > 3) { - throw new Error("MATCH requires 2 or 3 arguments"); - } + const lookupValue = context.evaluateFormula(requiredArg(context, args, 0)); + const lookupArrayRange = requiredArg(context, args, 1); + const matchType = args.length === 3 ? toNumber(context.evaluateFormula(requiredArg(context, args, 2))) : 1; - const lookupValue = context.evaluateFormula(args[0]); - const lookupArrayRange = args[1]; - const matchType = - args.length === 3 ? toNumber(context.evaluateFormula(args[2])) : 1; - - const lookupArray = context.getRangeValues(lookupArrayRange); + // Raw, not numeric-only: MATCH answers with a POSITION, so a dropped text + // cell renumbers everything after it and returns a different row (#2765). + const lookupArray = rawRangeReader(context)(lookupArrayRange); const index = findMatchIndex(lookupValue, lookupArray, matchType); - return index === -1 ? "#N/A" : index + 1; // 1-based index + return index === -1 ? NA_ERROR : index + 1; // 1-based index }; const indexHandler: FunctionHandler = (args, context) => { - if (args.length < 2 || args.length > 4) { - throw new Error("INDEX requires 2 to 4 arguments"); - } - - const arrayRange = args[0]; - const rowNum = toNumber(context.evaluateFormula(args[1])); - const colNum = - args.length >= 3 ? toNumber(context.evaluateFormula(args[2])) : 1; // Default to 1 if omitted (for 1D arrays) - - // Parse range to find the specific cell - let sheetName = ""; - let rangePart = arrayRange; - if (arrayRange.includes("!")) { - const parts = arrayRange.split("!"); - sheetName = parts[0] + "!"; - rangePart = parts[1]; - } - - const rangeMatch = rangePart.match(/^([A-Z]+)(\d+):([A-Z]+)(\d+)$/); - if (!rangeMatch) throw new Error("Invalid range format"); - - const startColStr = rangeMatch[1]; - const startRow = parseInt(rangeMatch[2]); - - const startColIdx = colToIndex(startColStr); - - // Calculate target cell - // rowNum and colNum are 1-based relative to the range - const targetRow = startRow + rowNum - 1; - const targetColIdx = startColIdx + colNum - 1; - const targetColStr = indexToCol(targetColIdx); - - const ref = `${sheetName}${targetColStr}${targetRow}`; - return context.getCellValue(ref); + const bounds = parseRangeBounds(requiredArg(context, args, 0)); + if (!bounds) throw new Error("Invalid range format"); + const rowNum = toNumber(context.evaluateFormula(requiredArg(context, args, 1))); + const colNum = args.length >= 3 ? toNumber(context.evaluateFormula(requiredArg(context, args, 2))) : 1; // Default to 1 if omitted (for 1D arrays) + + const target = resolveIndexTarget(bounds, rowNum, colNum); + if (!target) return REF_ERROR; + return context.getCellValue(`${bounds.sheetPrefix}${indexToColumn(target.colIndex)}${target.row}`); }; const xlookupHandler: FunctionHandler = (args, context) => { - if (args.length < 3 || args.length > 6) { - throw new Error("XLOOKUP requires 3 to 6 arguments"); - } - - const lookupValue = context.evaluateFormula(args[0]); - const lookupArrayRange = args[1]; - const returnArrayRange = args[2]; - const ifNotFound = - args.length >= 4 ? context.evaluateFormula(args[3]) : "#N/A"; - const matchMode = - args.length >= 5 ? toNumber(context.evaluateFormula(args[4])) : 0; - const searchMode = - args.length >= 6 ? toNumber(context.evaluateFormula(args[5])) : 1; - - const lookupArray = context.getRangeValues(lookupArrayRange); - const returnArray = context.getRangeValues(returnArrayRange); + const lookupValue = context.evaluateFormula(requiredArg(context, args, 0)); + const lookupArrayRange = requiredArg(context, args, 1); + const returnArrayRange = requiredArg(context, args, 2); + const ifNotFound = args.length >= 4 ? context.evaluateFormula(requiredArg(context, args, 3)) : NA_ERROR; + const matchMode = args.length >= 5 ? toNumber(context.evaluateFormula(requiredArg(context, args, 4))) : 0; + const searchMode = args.length >= 6 ? toNumber(context.evaluateFormula(requiredArg(context, args, 5))) : 1; + + // Both raw, and for a second reason beyond MATCH's: the numeric-only reader + // filters these two INDEPENDENTLY, so a text cell in one range shifts it + // against the other and the match index reads a different row's value — a + // wrong number with no error, which is the worst outcome a formula can have. + const readRange = rawRangeReader(context); + const lookupArray = readRange(lookupArrayRange); + const returnArray = readRange(returnArrayRange); // XLOOKUP match modes: // 0 = Exact match (default) @@ -358,11 +199,10 @@ const xlookupHandler: FunctionHandler = (args, context) => { if (matchIdx === -1) return ifNotFound; - if (matchIdx >= 0 && matchIdx < returnArray.length) { - return returnArray[matchIdx]; - } - - return "#N/A"; + // A match past the end of the return range has nothing to return — the two + // ranges are independent arguments and need not be the same length. + const found = returnArray[matchIdx]; + return found === undefined ? NA_ERROR : found; }; // Register functions @@ -371,8 +211,7 @@ functionRegistry.register({ handler: vlookupHandler, minArgs: 3, maxArgs: 4, - description: - "Looks for a value in the leftmost column of a table, and then returns a value in the same row from a column you specify", + description: "Looks for a value in the leftmost column of a table, and then returns a value in the same row from a column you specify", examples: ["VLOOKUP(105, A2:C10, 2)", 'VLOOKUP("Smith", A2:E10, 5, FALSE)'], category: "Lookup & Reference", }); @@ -382,8 +221,7 @@ functionRegistry.register({ handler: hlookupHandler, minArgs: 3, maxArgs: 4, - description: - "Looks for a value in the top row of a table, and then returns a value in the same column from a row you specify", + description: "Looks for a value in the top row of a table, and then returns a value in the same column from a row you specify", examples: ['HLOOKUP("Axles", A1:C10, 2, TRUE)'], category: "Lookup & Reference", }); @@ -393,8 +231,7 @@ functionRegistry.register({ handler: matchHandler, minArgs: 2, maxArgs: 3, - description: - "Returns the relative position of an item in an array that matches a specified value", + description: "Returns the relative position of an item in an array that matches a specified value", examples: ["MATCH(25, A1:A10, 0)", 'MATCH("b", A1:A5, 0)'], category: "Lookup & Reference", }); @@ -404,8 +241,7 @@ functionRegistry.register({ handler: indexHandler, minArgs: 2, maxArgs: 4, - description: - "Returns the value of an element in a table or an array, selected by the row and column number indexes", + description: "Returns the value of an element in a table or an array, selected by the row and column number indexes", examples: ["INDEX(A1:B5, 2, 2)", "INDEX(A1:A10, 5)"], category: "Lookup & Reference", }); @@ -415,11 +251,7 @@ functionRegistry.register({ handler: xlookupHandler, minArgs: 3, maxArgs: 6, - description: - "Searches a range or an array, and returns an item corresponding to the first match it finds", - examples: [ - "XLOOKUP(A1, B1:B10, C1:C10)", - 'XLOOKUP("USA", Countries, Populations)', - ], + description: "Searches a range or an array, and returns an item corresponding to the first match it finds", + examples: ["XLOOKUP(A1, B1:B10, C1:C10)", 'XLOOKUP("USA", Countries, Populations)'], category: "Lookup & Reference", }); diff --git a/src/engine/functions/mathematical.ts b/src/engine/functions/mathematical.ts index 278b439..f444f3a 100644 --- a/src/engine/functions/mathematical.ts +++ b/src/engine/functions/mathematical.ts @@ -2,129 +2,113 @@ * Mathematical Functions */ -import { functionRegistry, toNumber, type FunctionHandler } from "../registry"; +import { functionRegistry, requiredArg, toNumber, type FunctionHandler } from "../registry"; +import { + roundTo, + roundUpTo, + roundDownTo, + floorToSignificance, + ceilingToSignificance, + modulo, + power, + safeLog, + safeLog10, + safeSqrt, + logWithBase, +} from "../math-ops"; +import { toScalarNumber } from "../numericCoercion"; +import { isSpreadsheetErrorValue } from "../spreadsheet-errors"; const roundHandler: FunctionHandler = (args, context) => { - if (args.length !== 2) throw new Error("ROUND requires 2 arguments"); - const number = toNumber(context.evaluateFormula(args[0])); - const digits = toNumber(context.evaluateFormula(args[1])); - const multiplier = Math.pow(10, digits); - return Math.round(number * multiplier) / multiplier; + const number = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + const digits = toNumber(context.evaluateFormula(requiredArg(context, args, 1))); + return roundTo(number, digits); }; const roundupHandler: FunctionHandler = (args, context) => { - if (args.length !== 2) throw new Error("ROUNDUP requires 2 arguments"); - const number = toNumber(context.evaluateFormula(args[0])); - const digits = toNumber(context.evaluateFormula(args[1])); - const multiplier = Math.pow(10, digits); - return Math.ceil(number * multiplier) / multiplier; + const number = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + const digits = toNumber(context.evaluateFormula(requiredArg(context, args, 1))); + return roundUpTo(number, digits); }; const rounddownHandler: FunctionHandler = (args, context) => { - if (args.length !== 2) throw new Error("ROUNDDOWN requires 2 arguments"); - const number = toNumber(context.evaluateFormula(args[0])); - const digits = toNumber(context.evaluateFormula(args[1])); - const multiplier = Math.pow(10, digits); - return Math.floor(number * multiplier) / multiplier; + const number = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + const digits = toNumber(context.evaluateFormula(requiredArg(context, args, 1))); + return roundDownTo(number, digits); }; const floorHandler: FunctionHandler = (args, context) => { - if (args.length !== 2) throw new Error("FLOOR requires 2 arguments"); - const number = toNumber(context.evaluateFormula(args[0])); - const significance = toNumber(context.evaluateFormula(args[1])); - if (significance === 0) return 0; - return Math.floor(number / significance) * significance; + const number = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + const significance = toNumber(context.evaluateFormula(requiredArg(context, args, 1))); + return floorToSignificance(number, significance); }; const ceilingHandler: FunctionHandler = (args, context) => { - if (args.length !== 2) throw new Error("CEILING requires 2 arguments"); - const number = toNumber(context.evaluateFormula(args[0])); - const significance = toNumber(context.evaluateFormula(args[1])); - if (significance === 0) return 0; - return Math.ceil(number / significance) * significance; + const number = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + const significance = toNumber(context.evaluateFormula(requiredArg(context, args, 1))); + return ceilingToSignificance(number, significance); }; const absHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("ABS requires 1 argument"); - const number = toNumber(context.evaluateFormula(args[0])); - return Math.abs(number); + const number = toScalarNumber(context.evaluateFormula(requiredArg(context, args, 0))); + return isSpreadsheetErrorValue(number) ? number : Math.abs(number); }; const powerHandler: FunctionHandler = (args, context) => { - if (args.length !== 2) throw new Error("POWER requires 2 arguments"); - const base = toNumber(context.evaluateFormula(args[0])); - const exponent = toNumber(context.evaluateFormula(args[1])); - return Math.pow(base, exponent); + const base = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + const exponent = toNumber(context.evaluateFormula(requiredArg(context, args, 1))); + return power(base, exponent); }; const sqrtHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("SQRT requires 1 argument"); - const number = toNumber(context.evaluateFormula(args[0])); - return Math.sqrt(number); + const number = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + return safeSqrt(number); }; const modHandler: FunctionHandler = (args, context) => { - if (args.length !== 2) throw new Error("MOD requires 2 arguments"); - const number = toNumber(context.evaluateFormula(args[0])); - const divisor = toNumber(context.evaluateFormula(args[1])); - if (divisor === 0) return 0; - return number % divisor; + const number = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + const divisor = toNumber(context.evaluateFormula(requiredArg(context, args, 1))); + return modulo(number, divisor); }; const intHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("INT requires 1 argument"); - const number = toNumber(context.evaluateFormula(args[0])); + const number = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); return Math.floor(number); }; const truncHandler: FunctionHandler = (args, context) => { - if (args.length < 1 || args.length > 2) { - throw new Error("TRUNC requires 1 or 2 arguments"); - } - const number = toNumber(context.evaluateFormula(args[0])); - const digits = - args.length === 2 ? toNumber(context.evaluateFormula(args[1])) : 0; + const number = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + const digits = args.length === 2 ? toNumber(context.evaluateFormula(requiredArg(context, args, 1))) : 0; const multiplier = Math.pow(10, digits); return Math.trunc(number * multiplier) / multiplier; }; const signHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("SIGN requires 1 argument"); - const number = toNumber(context.evaluateFormula(args[0])); - return Math.sign(number); + const number = toScalarNumber(context.evaluateFormula(requiredArg(context, args, 0))); + return isSpreadsheetErrorValue(number) ? number : Math.sign(number); }; -const piHandler: FunctionHandler = (args) => { - if (args.length !== 0) throw new Error("PI requires 0 arguments"); - return Math.PI; -}; +const piHandler: FunctionHandler = () => Math.PI; const expHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("EXP requires 1 argument"); - const number = toNumber(context.evaluateFormula(args[0])); + const number = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); return Math.exp(number); }; const lnHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("LN requires 1 argument"); - const number = toNumber(context.evaluateFormula(args[0])); - return Math.log(number); + const number = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + return safeLog(number); }; const logHandler: FunctionHandler = (args, context) => { - if (args.length < 1 || args.length > 2) { - throw new Error("LOG requires 1 or 2 arguments"); - } - const number = toNumber(context.evaluateFormula(args[0])); - const base = - args.length === 2 ? toNumber(context.evaluateFormula(args[1])) : 10; - return Math.log(number) / Math.log(base); + const number = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + const base = args.length === 2 ? toNumber(context.evaluateFormula(requiredArg(context, args, 1))) : 10; + return logWithBase(number, base); }; const log10Handler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("LOG10 requires 1 argument"); - const number = toNumber(context.evaluateFormula(args[0])); - return Math.log10(number); + const number = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + return safeLog10(number); }; // Register all mathematical functions diff --git a/src/engine/functions/statistical-math.ts b/src/engine/functions/statistical-math.ts new file mode 100644 index 0000000..cd0ad94 --- /dev/null +++ b/src/engine/functions/statistical-math.ts @@ -0,0 +1,91 @@ +/** + * Pure statistical rules, separated from the range-reading handlers so they can + * be unit-tested directly. + */ + +import { DIV_ZERO_ERROR, NA_ERROR, NUM_ERROR, type SpreadsheetError } from "../spreadsheet-errors"; + +const arithmeticMean = (values: number[]): number => values.reduce((sum, value) => sum + value, 0) / values.length; + +/** + * The arithmetic mean, or `#DIV/0!` when there is nothing to average. + * + * Excel divides by the count of NUMBERS, so a range of blanks or text leaves a + * zero denominator and reports the division rather than a 0 that reads like a + * genuine average of zeros. + */ +export function computeAverage(values: number[]): number | SpreadsheetError { + if (values.length === 0) return DIV_ZERO_ERROR; + return arithmeticMean(values); +} + +/** + * The middle value (the mean of the middle two when the count is even), or + * `#NUM!` when there is no value to sit in the middle. + * + * Excel's code here is `#NUM!`, not AVERAGE's `#DIV/0!` — nothing is divided by + * zero, the median of an empty set simply does not exist. Sorts a copy so the + * caller's array keeps its order. + */ +export function computeMedian(values: number[]): number | SpreadsheetError { + if (values.length === 0) return NUM_ERROR; + return arithmeticMean(middleValues([...values].sort((a, b) => a - b))); +} + +/** The one or two values sitting in the middle of a sorted list — two when the + * count is even, one when it is odd. Sliced rather than indexed, so the middle + * of an empty list is simply nothing. */ +function middleValues(sorted: number[]): number[] { + const half = Math.floor(sorted.length / 2); + return sorted.slice(sorted.length % 2 === 0 ? half - 1 : half, half + 1); +} + +/** + * The most frequently occurring value, or `#N/A` when no value repeats. + * + * Excel's MODE is undefined for an all-distinct set, so returning the first + * element (as a naive "highest frequency wins" loop does when every count is 1) + * is a silent wrong answer. Ties resolve to the value that appears first, which + * Map insertion order preserves. + */ +export function computeMode(values: number[]): number | SpreadsheetError { + const frequency = new Map(); + for (const value of values) { + frequency.set(value, (frequency.get(value) ?? 0) + 1); + } + + let topFrequency = 0; + let mode: number | SpreadsheetError = NA_ERROR; + for (const [value, count] of frequency.entries()) { + if (count > topFrequency) { + topFrequency = count; + mode = value; + } + } + + const REPEAT_THRESHOLD = 2; + return topFrequency >= REPEAT_THRESHOLD ? mode : NA_ERROR; +} + +/** Sample size below which a sample variance/stdev is undefined. */ +const MIN_SAMPLE_SIZE = 2; + +/** + * Sample variance (Excel VAR): the mean squared deviation divided by `n - 1`, + * not `n`. Dividing by `n` is the POPULATION variance (Excel's VARP); using it + * for VAR understates the spread. Fewer than two values leave no `n - 1` to + * divide by, so Excel reports `#DIV/0!` rather than a silent 0. + */ +export function sampleVariance(values: number[]): number | SpreadsheetError { + if (values.length < MIN_SAMPLE_SIZE) return DIV_ZERO_ERROR; + const mean = arithmeticMean(values); + const sumSquaredDiffs = values.reduce((sum, value) => sum + (value - mean) ** 2, 0); + return sumSquaredDiffs / (values.length - 1); +} + +/** Sample standard deviation (Excel STDEV): the square root of the sample + * variance, and `#DIV/0!` on the same fewer-than-two-values boundary. */ +export function sampleStdev(values: number[]): number | SpreadsheetError { + const variance = sampleVariance(values); + return typeof variance === "number" ? Math.sqrt(variance) : variance; +} diff --git a/src/engine/functions/statistical.ts b/src/engine/functions/statistical.ts index d4a8f01..a6995a2 100644 --- a/src/engine/functions/statistical.ts +++ b/src/engine/functions/statistical.ts @@ -4,230 +4,163 @@ import { functionRegistry, + rawRangeReader, + requiredArg, toNumber, parseCriteria, type FunctionContext, type FunctionHandler, + type RangeGetter, } from "../registry"; +import { computeAverage, computeMedian, computeMode, sampleStdev, sampleVariance } from "./statistical-math"; +import { DIV_ZERO_ERROR } from "../spreadsheet-errors"; +import { holdsNumber } from "../numericCoercion"; +import type { CellValue } from "../types"; -const isLetter = (char: string): boolean => /[A-Z]/i.test(char); +// Excel accepts up to 255 arguments for its aggregate functions. +const MAX_AGGREGATE_ARGS = 255; -const isCellReference = (segment: string): boolean => { - if (!segment) return false; - let index = 0; - if (segment[index] === "$") index++; - const colStart = index; - while (index < segment.length && isLetter(segment[index])) { - index++; - } - if (index === colStart) return false; // Require at least one column letter - if (segment[index] === "$") index++; - if (index >= segment.length) return false; // Require row digits - for (; index < segment.length; index++) { - const char = segment[index]; - if (char < "0" || char > "9") { - return false; - } - } - return true; -}; +// `A1`, `$A$1`, `AA100`: column letters then row digits, each half optionally +// prefixed by `$`. +const CELL_REFERENCE_PATTERN = /^\$?[A-Z]+\$?\d+$/i; + +const isCellReference = (segment: string): boolean => CELL_REFERENCE_PATTERN.test(segment); + +// Everything after the last `!` — the reference without its sheet name, or the +// whole string when it carries none. +const withoutSheetPrefix = (value: string): string => value.slice(value.lastIndexOf("!") + 1); const isRangeReference = (value: string): boolean => { if (!value) return false; - const rangePart = value.includes("!") ? value.split("!").slice(-1)[0] : value; - const [start, end] = rangePart.split(":"); + const [start, end] = withoutSheetPrefix(value).split(":"); if (!start || !end) return false; return isCellReference(start) && isCellReference(end); }; -const collectNumericValues = ( - args: string[], - context: FunctionContext, -): number[] => { - const values: number[] = []; +// A bare cell reference (`A1`, `Sheet1!B2`) is read through the RANGE path, not +// evaluated as a scalar: the scalar path coerces a blank or text cell to 0, so +// COUNT(A999) counted an empty cell as a value. The range path yields nothing +// for a cell that holds nothing, which is what the count functions need. +const isReference = (arg: string): boolean => isRangeReference(arg) || isCellReference(withoutSheetPrefix(arg)); + +/** One value an argument contributed, tagged by where it came from. A range cell + * was already filtered by the range getter; a scalar is whatever the argument + * evaluated to and may hold no number at all. */ +interface ArgumentValue { + value: CellValue; + isScalar: boolean; +} + +const collectArgumentValues = (args: string[], context: FunctionContext, readRange: RangeGetter): ArgumentValue[] => { + const collected: ArgumentValue[] = []; for (const rawArg of args) { const arg = rawArg?.trim(); if (!arg) continue; - if (isRangeReference(arg)) { - const rangeValues = context.getRangeValues(arg).map(toNumber); - values.push(...rangeValues); + if (isReference(arg)) { + readRange(arg).forEach((value) => collected.push({ value, isScalar: false })); } else { - const evaluated = context.evaluateFormula(arg); - values.push(toNumber(evaluated)); + collected.push({ value: context.evaluateFormula(arg), isScalar: true }); } } - return values; + return collected; }; +const collectNumericValues = (args: string[], context: FunctionContext): number[] => + collectArgumentValues(args, context, context.getRangeValues).map(({ value }) => toNumber(value)); + +// Same walk as `collectNumericValues`, but keeping each cell as it is: COUNTA +// counts non-empty cells, so text must survive the trip. +const collectRawValues = (args: string[], context: FunctionContext): CellValue[] => + collectArgumentValues(args, context, rawRangeReader(context)).map(({ value }) => value); + const sumHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("SUM requires 1 argument"); - const values = context.getRangeValues(args[0]); - return values.reduce((sum: number, val) => sum + toNumber(val), 0); + const values = collectNumericValues(args, context); + return values.reduce((sum: number, value) => sum + value, 0); }; -const averageHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("AVERAGE requires 1 argument"); - const values = context.getRangeValues(args[0]); - if (values.length === 0) return 0; - const sum = values.reduce((acc: number, val) => acc + toNumber(val), 0); - return sum / values.length; -}; +// Multi-argument collection (#2360) feeding the empty-range error rule (#2501). +const averageHandler: FunctionHandler = (args, context) => computeAverage(collectNumericValues(args, context)); const maxHandler: FunctionHandler = (args, context) => { - if (args.length === 0) { - throw new Error("MAX requires at least 1 argument"); - } const values = collectNumericValues(args, context); return values.length > 0 ? Math.max(...values) : 0; }; const minHandler: FunctionHandler = (args, context) => { - if (args.length === 0) { - throw new Error("MIN requires at least 1 argument"); - } const values = collectNumericValues(args, context); return values.length > 0 ? Math.min(...values) : 0; }; -const countHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("COUNT requires 1 argument"); - const values = context.getRangeValues(args[0]); - return values.length; -}; - -const medianHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("MEDIAN requires 1 argument"); - const values = context - .getRangeValues(args[0]) - .map(toNumber) - .sort((a, b) => a - b); - - if (values.length === 0) return 0; - const mid = Math.floor(values.length / 2); - return values.length % 2 === 0 - ? (values[mid - 1] + values[mid]) / 2 - : values[mid]; -}; +// COUNT counts NUMBERS, so it cannot go through the lenient numeric collection +// the other aggregates share: `toNumber("text")` is 0, which made COUNT("text") +// answer 1 where Excel answers 0 (Codex review). A range cell reached the list +// only by being numeric already; a scalar has to be asked. +const countsAsNumber = ({ value, isScalar }: ArgumentValue): boolean => !isScalar || holdsNumber(value); -const modeHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("MODE requires 1 argument"); - const values = context.getRangeValues(args[0]).map(toNumber); +const countHandler: FunctionHandler = (args, context) => collectArgumentValues(args, context, context.getRangeValues).filter(countsAsNumber).length; - if (values.length === 0) return 0; +const medianHandler: FunctionHandler = (args, context) => computeMedian(collectNumericValues(args, context)); - // Count frequency of each value - const frequency = new Map(); - for (const val of values) { - frequency.set(val, (frequency.get(val) || 0) + 1); - } - - // Find the value with highest frequency - let maxFreq = 0; - let mode = values[0]; - for (const [val, freq] of frequency.entries()) { - if (freq > maxFreq) { - maxFreq = freq; - mode = val; - } - } - - return mode; +const modeHandler: FunctionHandler = (args, context) => { + return computeMode(collectNumericValues(args, context)); }; const stdevHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("STDEV requires 1 argument"); - const values = context.getRangeValues(args[0]).map(toNumber); - - if (values.length === 0) return 0; - - const mean = values.reduce((sum, val) => sum + val, 0) / values.length; - const squaredDiffs = values.map((val) => Math.pow(val - mean, 2)); - const variance = - squaredDiffs.reduce((sum, val) => sum + val, 0) / values.length; - return Math.sqrt(variance); + return sampleStdev(collectNumericValues(args, context)); }; const varHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("VAR requires 1 argument"); - const values = context.getRangeValues(args[0]).map(toNumber); - - if (values.length === 0) return 0; - - const mean = values.reduce((sum, val) => sum + val, 0) / values.length; - const squaredDiffs = values.map((val) => Math.pow(val - mean, 2)); - return squaredDiffs.reduce((sum, val) => sum + val, 0) / values.length; + return sampleVariance(collectNumericValues(args, context)); }; const countaHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("COUNTA requires 1 argument"); - const values = - context.getRangeValuesRaw?.(args[0]) ?? context.getRangeValues(args[0]); + const values = collectRawValues(args, context); // Count non-empty cells - return values.filter((v) => v !== null && v !== undefined && v !== "").length; + return values.filter((value) => value !== null && value !== undefined && value !== "").length; }; const countifHandler: FunctionHandler = (args, context) => { - if (args.length !== 2) throw new Error("COUNTIF requires 2 arguments"); - const values = - context.getRangeValuesRaw?.(args[0]) ?? context.getRangeValues(args[0]); - const criteria = args[1].trim(); - const compareFn = parseCriteria(criteria); - return values.filter(compareFn).length; + const values = rawRangeReader(context)(requiredArg(context, args, 0)); + return values.filter(parseCriteria(requiredArg(context, args, 1).trim())).length; }; -const sumifHandler: FunctionHandler = (args, context) => { - if (args.length < 2 || args.length > 3) { - throw new Error("SUMIF requires 2 or 3 arguments"); - } - - const criteriaRange = - context.getRangeValuesRaw?.(args[0]) ?? context.getRangeValues(args[0]); - const criteria = args[1].trim(); - const sumRange = - args.length === 3 - ? context.getRangeValues(args[2]) - : context.getRangeValues(args[0]); +/** The criteria range, the matcher and the value range SUMIF and AVERAGEIF both + * read. Both value ranges are RAW, not numeric-only: dropping blanks would + * shift the value range out of alignment with the (raw) criteria range, so a + * blank would pull a later row's number into an earlier match (#2358). */ +const readConditionalRanges = (args: string[], context: FunctionContext) => { + const criteriaRef = requiredArg(context, args, 0); + const readRaw = rawRangeReader(context); + const valueRef = args.length === 3 ? requiredArg(context, args, 2) : criteriaRef; + return { + criteriaRange: readRaw(criteriaRef), + valueRange: readRaw(valueRef), + matches: parseCriteria(requiredArg(context, args, 1).trim()), + }; +}; - const compareFn = parseCriteria(criteria); +/** Sum and count the values whose row in `criteriaRange` matches. The two + * ranges stay row-aligned, so a matched row with no value contributes 0. */ +const aggregateMatchedRows = (criteriaRange: CellValue[], valueRange: CellValue[], matches: (value: CellValue) => boolean) => + criteriaRange.reduce( + (totals, criteriaValue, index) => (matches(criteriaValue) ? { sum: totals.sum + toNumber(valueRange[index] ?? 0), count: totals.count + 1 } : totals), + { sum: 0, count: 0 }, + ); - let sum = 0; - for (let i = 0; i < criteriaRange.length; i++) { - if (compareFn(criteriaRange[i])) { - sum += toNumber(sumRange[i] ?? 0); - } - } - - return sum; +const sumifHandler: FunctionHandler = (args, context) => { + const { criteriaRange, valueRange, matches } = readConditionalRanges(args, context); + return aggregateMatchedRows(criteriaRange, valueRange, matches).sum; }; const averageifHandler: FunctionHandler = (args, context) => { - if (args.length < 2 || args.length > 3) { - throw new Error("AVERAGEIF requires 2 or 3 arguments"); - } - - const criteriaRange = - context.getRangeValuesRaw?.(args[0]) ?? context.getRangeValues(args[0]); - const criteria = args[1].trim(); - const avgRange = - args.length === 3 - ? context.getRangeValues(args[2]) - : context.getRangeValues(args[0]); - - const compareFn = parseCriteria(criteria); - - let sum = 0; - let count = 0; - for (let i = 0; i < criteriaRange.length; i++) { - if (compareFn(criteriaRange[i])) { - sum += toNumber(avgRange[i] ?? 0); - count++; - } - } - - return count > 0 ? sum / count : 0; + const { criteriaRange, valueRange, matches } = readConditionalRanges(args, context); + const { sum, count } = aggregateMatchedRows(criteriaRange, valueRange, matches); + // Excel returns #DIV/0! when no cell matches (the average of nothing is + // undefined), rather than a silent 0. + return count > 0 ? sum / count : DIV_ZERO_ERROR; }; // Register all statistical functions @@ -235,7 +168,7 @@ functionRegistry.register({ name: "SUM", handler: sumHandler, minArgs: 1, - maxArgs: 1, + maxArgs: MAX_AGGREGATE_ARGS, description: "Returns the sum of all numbers in a range", examples: ["SUM(A1:A10)", "SUM(B2:B20)"], category: "Statistical", @@ -245,7 +178,7 @@ functionRegistry.register({ name: "AVERAGE", handler: averageHandler, minArgs: 1, - maxArgs: 1, + maxArgs: MAX_AGGREGATE_ARGS, description: "Returns the average (arithmetic mean) of numbers in a range", examples: ["AVERAGE(A1:A10)", "AVERAGE(B2:B20)"], category: "Statistical", @@ -273,7 +206,7 @@ functionRegistry.register({ name: "COUNT", handler: countHandler, minArgs: 1, - maxArgs: 1, + maxArgs: MAX_AGGREGATE_ARGS, description: "Counts the number of cells in a range", examples: ["COUNT(A1:A10)", "COUNT(B2:B20)"], category: "Statistical", @@ -283,7 +216,7 @@ functionRegistry.register({ name: "MEDIAN", handler: medianHandler, minArgs: 1, - maxArgs: 1, + maxArgs: MAX_AGGREGATE_ARGS, description: "Returns the median (middle) value in a range", examples: ["MEDIAN(A1:A10)", "MEDIAN(B2:B20)"], category: "Statistical", @@ -293,7 +226,7 @@ functionRegistry.register({ name: "MODE", handler: modeHandler, minArgs: 1, - maxArgs: 1, + maxArgs: MAX_AGGREGATE_ARGS, description: "Returns the most frequently occurring value in a range", examples: ["MODE(A1:A10)", "MODE(B2:B20)"], category: "Statistical", @@ -303,7 +236,7 @@ functionRegistry.register({ name: "STDEV", handler: stdevHandler, minArgs: 1, - maxArgs: 1, + maxArgs: MAX_AGGREGATE_ARGS, description: "Returns the standard deviation of numbers in a range", examples: ["STDEV(A1:A10)", "STDEV(B2:B20)"], category: "Statistical", @@ -313,7 +246,7 @@ functionRegistry.register({ name: "VAR", handler: varHandler, minArgs: 1, - maxArgs: 1, + maxArgs: MAX_AGGREGATE_ARGS, description: "Returns the variance of numbers in a range", examples: ["VAR(A1:A10)", "VAR(B2:B20)"], category: "Statistical", @@ -323,7 +256,7 @@ functionRegistry.register({ name: "COUNTA", handler: countaHandler, minArgs: 1, - maxArgs: 1, + maxArgs: MAX_AGGREGATE_ARGS, description: "Counts the number of non-empty cells in a range", examples: ["COUNTA(A1:A10)", "COUNTA(B2:B20)"], category: "Statistical", diff --git a/src/engine/functions/text.ts b/src/engine/functions/text.ts index 9a67cce..85b85d4 100644 --- a/src/engine/functions/text.ts +++ b/src/engine/functions/text.ts @@ -2,12 +2,67 @@ * Text Functions */ -import { functionRegistry, toString, type FunctionHandler } from "../registry"; +import { functionRegistry, requiredArg, toString, type FunctionHandler } from "../registry"; +import { VALUE_ERROR, isSpreadsheetErrorValue, type SpreadsheetError } from "../spreadsheet-errors"; +import { formatWithPattern } from "../textFormat"; + +// A letter or a combining mark. A decomposed accented letter (e + U+0301) is two +// code points; counting the mark as part of the word stops PROPER from treating +// the base letter that follows it as a new word (éclair → Éclair, not ÉClair). +const isWordCharacter = (char: string): boolean => /[\p{L}\p{M}]/u.test(char); + +// Excel PROPER capitalises a letter at the start of the text or after any +// non-letter (space, punctuation, digit) and lowercases the rest — so word +// boundaries include "'" and "-", which a space-only split misses. +export const toProperCase = (text: string): string => { + const chars = Array.from(text); + const cased = chars.map((char, index) => { + const previous = chars[index - 1]; + return previous === undefined || !isWordCharacter(previous) ? char.toUpperCase() : char.toLowerCase(); + }); + return cased.join(""); +}; -const concatenateHandler: FunctionHandler = (args, context) => { - if (args.length === 0) - throw new Error("CONCATENATE requires at least 1 argument"); +// Excel LEFT/RIGHT reject a negative count with #VALUE!; 0 and over-length +// counts keep substring's clamping. +// Excel truncates a fractional count toward zero and rejects a non-finite or +// negative one with #VALUE! (LEFT/RIGHT with "x" or -1). Normalising once keeps +// LEFT and RIGHT consistent instead of each feeding a raw Number() to substring. +const normalizeCharCount = (count: number): number | SpreadsheetError => { + // Test the sign before truncating: Math.trunc(-0.5) is -0, which is not < 0. + if (!Number.isFinite(count) || count < 0) return VALUE_ERROR; + return Math.trunc(count); +}; + +export const takeLeft = (text: string, count: number): string | SpreadsheetError => { + const chars = normalizeCharCount(count); + return isSpreadsheetErrorValue(chars) ? chars : text.substring(0, chars); +}; + +export const takeRight = (text: string, count: number): string | SpreadsheetError => { + const chars = normalizeCharCount(count); + return isSpreadsheetErrorValue(chars) ? chars : text.substring(text.length - chars); +}; + +// Replace the nth (1-based) occurrence; split/join keeps matches non-overlapping, +// matching the replace-all path. +const replaceNthOccurrence = (text: string, oldText: string, newText: string, nth: number): string => { + const parts = text.split(oldText); + if (nth > parts.length - 1) return text; + return parts.slice(0, nth).join(oldText) + newText + parts.slice(nth).join(oldText); +}; + +// Excel SUBSTITUTE: empty old_text returns the text unchanged (never inserts +// between characters); a supplied instance ≤ 0 or non-finite is a #VALUE! error. +export const substituteText = (text: string, oldText: string, newText: string, instance?: number): string | SpreadsheetError => { + if (oldText === "") return text; + if (instance === undefined) return text.split(oldText).join(newText); + const nth = Math.trunc(instance); + if (!Number.isFinite(nth) || nth <= 0) return VALUE_ERROR; + return replaceNthOccurrence(text, oldText, newText, nth); +}; +const concatenateHandler: FunctionHandler = (args, context) => { return args .map((arg) => { const value = context.evaluateFormula(arg.trim()); @@ -19,217 +74,156 @@ const concatenateHandler: FunctionHandler = (args, context) => { const concatHandler: FunctionHandler = concatenateHandler; // Alias const leftHandler: FunctionHandler = (args, context) => { - if (args.length < 1 || args.length > 2) { - throw new Error("LEFT requires 1 or 2 arguments"); - } - - const text = toString(context.evaluateFormula(args[0])); - const numChars = - args.length === 2 ? Number(context.evaluateFormula(args[1])) : 1; + const text = toString(context.evaluateFormula(requiredArg(context, args, 0))); + const numChars = args.length === 2 ? Number(context.evaluateFormula(requiredArg(context, args, 1))) : 1; - return text.substring(0, numChars); + return takeLeft(text, numChars); }; const rightHandler: FunctionHandler = (args, context) => { - if (args.length < 1 || args.length > 2) { - throw new Error("RIGHT requires 1 or 2 arguments"); - } + const text = toString(context.evaluateFormula(requiredArg(context, args, 0))); + const numChars = args.length === 2 ? Number(context.evaluateFormula(requiredArg(context, args, 1))) : 1; - const text = toString(context.evaluateFormula(args[0])); - const numChars = - args.length === 2 ? Number(context.evaluateFormula(args[1])) : 1; + return takeRight(text, numChars); +}; - return text.substring(text.length - numChars); +/** Excel MID: 1-based start, count of characters. `substring` SWAPS its bounds + * when they are reversed, so a negative count read backwards from `start` and + * returned earlier characters — `MID("Hello",3,-1)` gave "e" instead of an + * error. Both arguments are validated here instead. */ +export const takeMid = (text: string, start: number, count: number): string | SpreadsheetError => { + const chars = normalizeCharCount(count); + if (isSpreadsheetErrorValue(chars)) return chars; + if (!Number.isFinite(start) || start < 1) return VALUE_ERROR; + const from = Math.trunc(start) - 1; + return text.substring(from, from + chars); }; const midHandler: FunctionHandler = (args, context) => { - if (args.length !== 3) throw new Error("MID requires 3 arguments"); + const text = toString(context.evaluateFormula(requiredArg(context, args, 0))); + const start = Number(context.evaluateFormula(requiredArg(context, args, 1))); + const numChars = Number(context.evaluateFormula(requiredArg(context, args, 2))); - const text = toString(context.evaluateFormula(args[0])); - const start = Number(context.evaluateFormula(args[1])) - 1; // 1-indexed to 0-indexed - const numChars = Number(context.evaluateFormula(args[2])); - - return text.substring(start, start + numChars); + return takeMid(text, start, numChars); }; const lenHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("LEN requires 1 argument"); - - const text = toString(context.evaluateFormula(args[0])); + const text = toString(context.evaluateFormula(requiredArg(context, args, 0))); return text.length; }; const upperHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("UPPER requires 1 argument"); - - const text = toString(context.evaluateFormula(args[0])); + const text = toString(context.evaluateFormula(requiredArg(context, args, 0))); return text.toUpperCase(); }; const lowerHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("LOWER requires 1 argument"); - - const text = toString(context.evaluateFormula(args[0])); + const text = toString(context.evaluateFormula(requiredArg(context, args, 0))); return text.toLowerCase(); }; const properHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("PROPER requires 1 argument"); - - const text = toString(context.evaluateFormula(args[0])); - return text - .toLowerCase() - .split(" ") - .map((word) => - word.length > 0 ? word[0].toUpperCase() + word.slice(1) : "", - ) - .join(" "); + const text = toString(context.evaluateFormula(requiredArg(context, args, 0))); + return toProperCase(text); }; const trimHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("TRIM requires 1 argument"); - - const text = toString(context.evaluateFormula(args[0])); + const text = toString(context.evaluateFormula(requiredArg(context, args, 0))); // Trim leading/trailing spaces and replace multiple spaces with single space return text.trim().replace(/\s+/g, " "); }; const substituteHandler: FunctionHandler = (args, context) => { - if (args.length < 3 || args.length > 4) { - throw new Error("SUBSTITUTE requires 3 or 4 arguments"); - } + const text = toString(context.evaluateFormula(requiredArg(context, args, 0))); + const oldText = toString(context.evaluateFormula(requiredArg(context, args, 1))); + const newText = toString(context.evaluateFormula(requiredArg(context, args, 2))); + const instance = args.length === 4 ? Number(context.evaluateFormula(requiredArg(context, args, 3))) : undefined; - const text = toString(context.evaluateFormula(args[0])); - const oldText = toString(context.evaluateFormula(args[1])); - const newText = toString(context.evaluateFormula(args[2])); - - if (args.length === 4) { - // Replace specific instance - const instance = Number(context.evaluateFormula(args[3])); - let count = 0; - let index = 0; - - while (index < text.length) { - const pos = text.indexOf(oldText, index); - if (pos === -1) break; - - count++; - if (count === instance) { - return ( - text.substring(0, pos) + - newText + - text.substring(pos + oldText.length) - ); - } - index = pos + 1; - } - return text; // Instance not found - } else { - // Replace all instances - return text.split(oldText).join(newText); - } + return substituteText(text, oldText, newText, instance); }; const replaceHandler: FunctionHandler = (args, context) => { - if (args.length !== 4) throw new Error("REPLACE requires 4 arguments"); - - const oldText = toString(context.evaluateFormula(args[0])); - const startPos = Number(context.evaluateFormula(args[1])) - 1; // 1-indexed to 0-indexed - const numChars = Number(context.evaluateFormula(args[2])); - const newText = toString(context.evaluateFormula(args[3])); - - return ( - oldText.substring(0, startPos) + - newText + - oldText.substring(startPos + numChars) - ); + const oldText = toString(context.evaluateFormula(requiredArg(context, args, 0))); + const startPos = Number(context.evaluateFormula(requiredArg(context, args, 1))) - 1; // 1-indexed to 0-indexed + const numChars = Number(context.evaluateFormula(requiredArg(context, args, 2))); + const newText = toString(context.evaluateFormula(requiredArg(context, args, 3))); + + return oldText.substring(0, startPos) + newText + oldText.substring(startPos + numChars); }; -const findHandler: FunctionHandler = (args, context) => { - if (args.length < 2 || args.length > 3) { - throw new Error("FIND requires 2 or 3 arguments"); - } +// FIND and SEARCH share everything but case sensitivity: same 0-based start, +// same #VALUE! on no match, same 1-based result. SEARCH folds case first. +export const locateSubstring = (find: string, within: string, start: number, options: { caseInsensitive: boolean }): number | SpreadsheetError => { + const needle = options.caseInsensitive ? find.toLowerCase() : find; + const haystack = options.caseInsensitive ? within.toLowerCase() : within; + const index = haystack.indexOf(needle, start); + return index === -1 ? VALUE_ERROR : index + 1; // 1-indexed position +}; - const findText = toString(context.evaluateFormula(args[0])); - const withinText = toString(context.evaluateFormula(args[1])); - const startPos = - args.length === 3 ? Number(context.evaluateFormula(args[2])) - 1 : 0; +const findHandler: FunctionHandler = (args, context) => { + const findText = toString(context.evaluateFormula(requiredArg(context, args, 0))); + const withinText = toString(context.evaluateFormula(requiredArg(context, args, 1))); + const startPos = args.length === 3 ? Number(context.evaluateFormula(requiredArg(context, args, 2))) - 1 : 0; - const index = withinText.indexOf(findText, startPos); - return index === -1 ? "#VALUE!" : index + 1; // Return 1-indexed position + return locateSubstring(findText, withinText, startPos, { caseInsensitive: false }); }; const searchHandler: FunctionHandler = (args, context) => { - if (args.length < 2 || args.length > 3) { - throw new Error("SEARCH requires 2 or 3 arguments"); - } - - const findText = toString(context.evaluateFormula(args[0])); - const withinText = toString(context.evaluateFormula(args[1])); - const startPos = - args.length === 3 ? Number(context.evaluateFormula(args[2])) - 1 : 0; - - // SEARCH is case-insensitive - const lowerFind = findText.toLowerCase(); - const lowerWithin = withinText.toLowerCase(); + const findText = toString(context.evaluateFormula(requiredArg(context, args, 0))); + const withinText = toString(context.evaluateFormula(requiredArg(context, args, 1))); + const startPos = args.length === 3 ? Number(context.evaluateFormula(requiredArg(context, args, 2))) - 1 : 0; - const index = lowerWithin.indexOf(lowerFind, startPos); - return index === -1 ? "#VALUE!" : index + 1; // Return 1-indexed position + return locateSubstring(findText, withinText, startPos, { caseInsensitive: true }); }; -const textHandler: FunctionHandler = (args, context) => { - if (args.length !== 2) throw new Error("TEXT requires 2 arguments"); - - const value = context.evaluateFormula(args[0]); - const format = toString(context.evaluateFormula(args[1])).replace( - // eslint-disable -- sonarjs/anchor-precedence - /^["']|["']$/g, - "", - ); - - // Simple format code handling - if (typeof value === "number") { - // Handle common format codes - if (format.includes("$")) { - const decimals = (format.match(/\.0+/) || [""])[0].length - 1; - return "$" + value.toFixed(decimals >= 0 ? decimals : 2); - } - if (format.includes("%")) { - const decimals = (format.match(/\.0+/) || [""])[0].length - 1; - return (value * 100).toFixed(decimals >= 0 ? decimals : 2) + "%"; - } - if (format.includes("0")) { - const decimals = (format.match(/\.0+/) || [""])[0].length - 1; - return value.toFixed(decimals >= 0 ? decimals : 0); - } - } +// A format code written as a literal still carries its quotes when it reaches +// the handler. +const stripSurroundingQuotes = (text: string): string => text.replace(/^["']/, "").replace(/["']$/, ""); - return toString(value); +const textHandler: FunctionHandler = (args, context) => { + const value = context.evaluateFormula(requiredArg(context, args, 0)); + const format = stripSurroundingQuotes(toString(context.evaluateFormula(requiredArg(context, args, 1)))); + if (typeof value !== "number") return toString(value); + return formatWithPattern(value, format) ?? toString(value); }; -const valueHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("VALUE requires 1 argument"); - - const text = toString(context.evaluateFormula(args[0])); - - // Remove currency symbols and commas - const cleaned = text.replace(/[$,]/g, "").trim(); +/** Read a numeric string in full, or null. `Number` rather than `parseFloat`: + * `parseFloat` stops at the first character it cannot read, so `VALUE("12abc")` + * came back 12 where Excel reports #VALUE!. An empty string is not a number + * here even though `Number("")` is 0. */ +// Decimal or scientific notation only. `Number` alone also accepts JS-only +// spellings a spreadsheet never should — `0x10` → 16, `0b10` → 2, `Infinity` — +// so the shape is checked before converting. +const DECIMAL_NUMBER = /^[+-]?(?:\d+(?:\.\d*)?|\.\d+)(?:[eE][+-]?\d+)?$/; + +const wholeNumberOrNull = (text: string): number | null => { + if (!DECIMAL_NUMBER.test(text)) return null; + const parsed = Number(text); + // The pattern admits an exponent that overflows to Infinity (`1e999`), which + // is no more a spreadsheet number than the literal spelling is. + return Number.isFinite(parsed) ? parsed : null; +}; - // Handle percentages - if (cleaned.includes("%")) { - const num = parseFloat(cleaned.replace("%", "")); - return isNaN(num) ? "#VALUE!" : num / 100; +/** Excel VALUE: the WHOLE string must be a number once its currency symbols and + * thousands separators are stripped; trailing text is an error, not a prefix to + * salvage. */ +export const parseValueText = (raw: string): number | SpreadsheetError => { + const cleaned = raw.replace(/[$,]/g, "").trim(); + if (cleaned.endsWith("%")) { + const percent = wholeNumberOrNull(cleaned.slice(0, -1).trim()); + return percent === null ? VALUE_ERROR : percent / 100; } + const parsed = wholeNumberOrNull(cleaned); + return parsed === null ? VALUE_ERROR : parsed; +}; - const num = parseFloat(cleaned); - return isNaN(num) ? "#VALUE!" : num; +const valueHandler: FunctionHandler = (args, context) => { + return parseValueText(toString(context.evaluateFormula(requiredArg(context, args, 0)))); }; const exactHandler: FunctionHandler = (args, context) => { - if (args.length !== 2) throw new Error("EXACT requires 2 arguments"); - - const text1 = toString(context.evaluateFormula(args[0])); - const text2 = toString(context.evaluateFormula(args[1])); + const text1 = toString(context.evaluateFormula(requiredArg(context, args, 0))); + const text2 = toString(context.evaluateFormula(requiredArg(context, args, 1))); return text1 === text2; }; @@ -248,8 +242,7 @@ functionRegistry.register({ name: "CONCAT", handler: concatHandler, minArgs: 1, - description: - "Joins several text strings into one string (same as CONCATENATE)", + description: "Joins several text strings into one string (same as CONCATENATE)", examples: ['CONCAT("Hello", " ", "World")', "CONCAT(A1, B1)"], category: "Text", }); @@ -340,10 +333,7 @@ functionRegistry.register({ minArgs: 3, maxArgs: 4, description: "Replaces old text with new text in a string", - examples: [ - 'SUBSTITUTE("Hello World", "World", "Earth")', - 'SUBSTITUTE(A1, "old", "new", 1)', - ], + examples: ['SUBSTITUTE("Hello World", "World", "Earth")', 'SUBSTITUTE(A1, "old", "new", 1)'], category: "Text", }); @@ -353,10 +343,7 @@ functionRegistry.register({ minArgs: 4, maxArgs: 4, description: "Replaces part of a text string with a different text string", - examples: [ - 'REPLACE("Hello World", 7, 5, "Earth")', - 'REPLACE(A1, 1, 3, "New")', - ], + examples: ['REPLACE("Hello World", 7, 5, "Earth")', 'REPLACE(A1, 1, 3, "New")'], category: "Text", }); diff --git a/src/engine/guards.ts b/src/engine/guards.ts new file mode 100644 index 0000000..6610250 --- /dev/null +++ b/src/engine/guards.ts @@ -0,0 +1,32 @@ +// The three runtime helpers the engine needs, kept here so the engine stays what +// its own header claims: framework-agnostic, with nothing above it to import. +// +// They arrive with the engine that was developed in MulmoClaude, where they came +// from that repo's own leaf package. Copied rather than depended on: a +// gui-chat plugin must not take a dependency on a host's packages. + +/** Narrow `unknown` to a plain object (not null, not array). */ +export function isRecord(value: unknown): value is Record { + return typeof value === "object" && value !== null && !Array.isArray(value); +} + +/** Narrow `unknown` to any object (not null, arrays allowed). + * Use `isRecord` when you need to access string keys. */ +export function isObj(value: unknown): value is object { + return typeof value === "object" && value !== null; +} + +const hasStringProp = (value: unknown, key: K): value is Record & Record => + isRecord(value) && typeof value[key] === "string"; + +/** The message to show for a thrown value. A non-Error object with a non-empty + * string `details` (the gRPC convention) or `message` surfaces that field — + * `details` wins — instead of the `[object Object]` a bare `String(err)` + * would print. */ +export function errorMessage(err: unknown, fallback?: string): string { + if (err instanceof Error) return err.message; + if (hasStringProp(err, "details") && err.details) return err.details; + if (hasStringProp(err, "message") && err.message) return err.message; + if (fallback !== undefined) return fallback; + return String(err); +} diff --git a/src/engine/index.ts b/src/engine/index.ts index 50e157b..84f55da 100644 --- a/src/engine/index.ts +++ b/src/engine/index.ts @@ -9,9 +9,20 @@ export * from "./types"; // Export utilities export * from "./parser"; +export * from "./condition"; +export * from "./date-locale"; +export * from "./cellEmpty"; +export * from "./datedif"; export * from "./formatter"; +export * from "./translateFormula"; +export * from "./formulaError"; +export * from "./spreadsheet-errors"; export * from "./evaluator"; export * from "./calculator"; +export * from "./formulaRefs"; +export * from "./cellBuilder"; +export * from "./responseDecoder"; +export * from "./jsonCellLocator"; // Export function registry export * from "./registry"; diff --git a/src/engine/jsonCellLocator.ts b/src/engine/jsonCellLocator.ts new file mode 100644 index 0000000..b996fb3 --- /dev/null +++ b/src/engine/jsonCellLocator.ts @@ -0,0 +1,111 @@ +/** + * Locate the start offset of a specific cell value inside a + * pretty-printed JSON spreadsheet document. Extracted from + * `handleTableClick` in `src/plugins/spreadsheet/View.vue` where the + * inline scanner pushed cognitive complexity to 163. + * + * The scanner walks the raw editor text character-by-character, + * tracking string boundaries, bracket depth, and object depth to + * find the n-th cell of the m-th row within a named sheet's `data` + * array. We deliberately do NOT parse the JSON — we need the + * character offset inside the user's text buffer (which may not be + * valid JSON mid-edit), so a positional scan is required. + * + * Pure — no refs, no DOM. Returns -1 if the cell can't be located. + * Tested in `test/plugins/spreadsheet/engine/test_jsonCellLocator.ts`. + */ + +// Advance `pos` through `text` until we reach the `rowIndex`-th +// opening `[` after `startPos` (counting from -1 so the first `[` +// encountered is index 0). Returns the position just after that +// opening bracket, or -1 if we ran off the end. +function findRowOpenBracket(text: string, startPos: number, rowIndex: number): number { + let currentRow = -1; + let inString = false; + for (let i = startPos; i < text.length; i++) { + const c = text[i]; + const prevChar = i > 0 ? text[i - 1] : ""; + // Track string literal boundaries so that a `[` inside a cell + // value like `"has [bracket]"` doesn't get mistaken for a row + // opener and throw off the row offset. + if (c === '"' && prevChar !== "\\") { + inString = !inString; + continue; + } + if (!inString && c === "[") { + currentRow++; + if (currentRow === rowIndex) return i + 1; + } + } + return -1; +} + +// Starting just inside the row's `[`, scan for the start offset of +// the `colIndex`-th cell. Tracks string/object/bracket state so +// commas inside cell objects don't miscounted as cell separators. +// Returns -1 if the row ends before we reach colIndex. +function findCellStartWithinRow(text: string, rowStart: number, colIndex: number): number { + let currentCol = 0; + let inString = false; + let inObject = 0; + let bracketDepth = 1; // we already stepped past one `[` + + for (let i = rowStart; i < text.length; i++) { + const c = text[i]; + const prevChar = i > 0 ? text[i - 1] : ""; + + if (c === '"' && prevChar !== "\\") { + inString = !inString; + } + + if (!inString) { + if (c === "[") bracketDepth++; + if (c === "]") { + bracketDepth--; + if (bracketDepth === 0) return -1; // row ended before colIndex + } + if (c === "{") inObject++; + if (c === "}") inObject--; + if (c === "," && inObject === 0 && bracketDepth === 1) { + currentCol++; + } + } + + // Once currentCol matches, skip any structural whitespace / + // opening bracket / comma and return the first content char. + if (currentCol === colIndex) { + if (c !== " " && c !== "\n" && c !== "\t" && c !== "[" && c !== ",") { + return i; + } + } + } + return -1; +} + +/** + * Given the pretty-printed JSON editor text, a sheet name, and + * (rowIndex, colIndex), return the character offset of that cell's + * JSON token. The returned offset can be fed into + * `textarea.setSelectionRange(offset, offset + cellJsonLength)` to + * highlight the cell. + * + * Returns -1 if the sheet isn't found or the (row, col) is out of + * range. Never throws. + */ +export function findCellJsonPosition(editorText: string, sheetName: string, rowIndex: number, colIndex: number): number { + // JSON.stringify escapes embedded quotes/backslashes so the marker + // matches the way the sheet name actually appears in editorText. + const sheetStartMarker = `"name": ${JSON.stringify(sheetName)}`; + const dataStartMarker = `"data": [`; + + const sheetPos = editorText.indexOf(sheetStartMarker); + if (sheetPos === -1) return -1; + + const dataPos = editorText.indexOf(dataStartMarker, sheetPos); + if (dataPos === -1) return -1; + + const rowStart = findRowOpenBracket(editorText, dataPos + dataStartMarker.length, rowIndex); + if (rowStart === -1) return -1; + + return findCellStartWithinRow(editorText, rowStart, colIndex); +} diff --git a/src/engine/math-ops.ts b/src/engine/math-ops.ts new file mode 100644 index 0000000..44e4667 --- /dev/null +++ b/src/engine/math-ops.ts @@ -0,0 +1,98 @@ +/** + * Rounding, modulo and significance math for the mathematical functions. + * + * Pure number-in / (number | formula-error-value)-out helpers. The rounding + * direction and the domain rules are exactly where the engine diverged from + * Excel (rounding toward +∞ instead of away from zero, modulo taking the + * dividend's sign, negative bases and domains slipping through as NaN/∞), so + * they are worth testing apart from the argument-reading handlers. + */ + +import { DIV_ZERO_ERROR, NUM_ERROR, isSpreadsheetErrorValue, type SpreadsheetError } from "./spreadsheet-errors"; + +const DECIMAL_BASE = 10; + +const scale = (digits: number): number => Math.pow(DECIMAL_BASE, digits); + +/** Excel ROUND: half away from zero (JS `Math.round` breaks half toward +∞, so + * `ROUND(-2.5, 0)` came back -2 instead of -3). */ +export function roundTo(value: number, digits: number): number { + const factor = scale(digits); + return (Math.sign(value) * Math.round(Math.abs(value) * factor)) / factor; +} + +/** Excel ROUNDUP: away from zero. */ +export function roundUpTo(value: number, digits: number): number { + const factor = scale(digits); + return (Math.sign(value) * Math.ceil(Math.abs(value) * factor)) / factor; +} + +/** Excel ROUNDDOWN: toward zero. */ +export function roundDownTo(value: number, digits: number): number { + const factor = scale(digits); + return (Math.sign(value) * Math.floor(Math.abs(value) * factor)) / factor; +} + +/** Whether `value` and `significance` point the same way; opposite signs are the + * #NUM! case for FLOOR / CEILING. A zero value is always in range. */ +const sameSign = (value: number, significance: number): boolean => value === 0 || Math.sign(value) === Math.sign(significance); + +/** Excel FLOOR: nearest multiple of `significance` toward zero; opposite signs + * are #NUM!, and a zero significance is #DIV/0! — the division by the + * significance is what FLOOR reports, so it wins over the sign check. */ +export function floorToSignificance(value: number, significance: number): number | SpreadsheetError { + if (significance === 0) return DIV_ZERO_ERROR; + if (!sameSign(value, significance)) return NUM_ERROR; + return Math.floor(value / significance) * significance; +} + +/** Excel CEILING: nearest multiple of `significance` away from zero; opposite + * signs are #NUM!, and a zero significance is 0 — deliberately NOT FLOOR's + * #DIV/0!, an asymmetry Excel keeps and this engine has to match. */ +export function ceilingToSignificance(value: number, significance: number): number | SpreadsheetError { + if (significance === 0) return 0; + if (!sameSign(value, significance)) return NUM_ERROR; + return Math.ceil(value / significance) * significance; +} + +/** Excel MOD: result takes the divisor's sign (`MOD(-3, 2) === 1`); dividing by + * zero is #DIV/0!, not a silent 0. */ +export function modulo(value: number, divisor: number): number | SpreadsheetError { + if (divisor === 0) return DIV_ZERO_ERROR; + return value - divisor * Math.floor(value / divisor); +} + +/** Excel POWER: a negative base with a non-integer exponent has no real root, so + * it is #NUM! rather than JS's NaN. */ +export function power(base: number, exponent: number): number | SpreadsheetError { + if (base < 0 && !Number.isInteger(exponent)) return NUM_ERROR; + return Math.pow(base, exponent); +} + +/** Natural log guarded to its domain: `LN(x)` for `x <= 0` is #NUM!, not + * -∞ / NaN. Reused by LN and LOG. */ +export function safeLog(value: number): number | SpreadsheetError { + if (value <= 0) return NUM_ERROR; + return Math.log(value); +} + +/** Excel LOG(value, base): the base must also be positive and not 1, or the + * result is #NUM! rather than the ∞ / NaN a bare division would give. */ +export function logWithBase(value: number, base: number): number | SpreadsheetError { + const lnValue = safeLog(value); + if (isSpreadsheetErrorValue(lnValue)) return lnValue; + if (base <= 0 || base === 1) return NUM_ERROR; + return lnValue / Math.log(base); +} + +/** Base-10 log guarded to its domain (`LOG10(x)` for `x <= 0` is #NUM!). */ +export function safeLog10(value: number): number | SpreadsheetError { + if (value <= 0) return NUM_ERROR; + return Math.log10(value); +} + +/** Square root guarded to its domain (`SQRT(x)` for `x < 0` is #NUM!, not NaN). */ +export function safeSqrt(value: number): number | SpreadsheetError { + if (value < 0) return NUM_ERROR; + return Math.sqrt(value); +} diff --git a/src/engine/numericCoercion.ts b/src/engine/numericCoercion.ts new file mode 100644 index 0000000..2cd3b3b --- /dev/null +++ b/src/engine/numericCoercion.ts @@ -0,0 +1,61 @@ +/** + * Numeric coercion helpers for the spreadsheet engine. + * + * Two intentionally different reads share one string parser (`parseNumericString`): + * - `registry.toNumber` — lenient, for range aggregation (SUM / AVERAGE / …): + * anything unreadable becomes 0. PINNED (booleans are 0, not Excel's 1/0). + * - `toScalarNumber` — strict, for single-value math functions (ABS / SIGN): + * booleans are 1/0 and non-numeric text is `#VALUE!`, matching Excel. + * + * `holdsNumber` asks the question neither read can answer once it has coerced: + * was there a number here at all, or is this 0 the reading of something else. + */ + +import type { CellValue } from "./types"; +import { VALUE_ERROR, type SpreadsheetError } from "./spreadsheet-errors"; + +const numberOrNull = (num: number): number | null => (isNaN(num) ? null : num); + +/** + * Parse a spreadsheet string as a number, or `null` when it holds no number. + * + * Mirrors the engine's long-standing lenient read: a percentage, a currency + * amount, or a thousands-separated value, else a bare `parseFloat` (which reads a + * leading number out of "12abc"). The branch order is load-bearing — each strips + * only its own characters, so a string mixing "%" and "$" fails in the first. + */ +export function parseNumericString(value: string): number | null { + if (value.includes("%")) { + const num = numberOrNull(parseFloat(value.replace("%", "").trim())); + return num === null ? null : num / 100; + } + if (value.includes("$")) return numberOrNull(parseFloat(value.replace(/[$,]/g, "").trim())); + if (value.includes(",")) return numberOrNull(parseFloat(value.replace(/,/g, "").trim())); + return numberOrNull(parseFloat(value)); +} + +/** + * Whether a value holds a number at all, under the same reading `toNumber` uses. + * + * `toNumber` answers 0 for text, for a boolean and for a genuine 0 alike, so a + * caller that must tell "no number here" from "the number zero" — COUNT — cannot + * ask it. Booleans are not numbers here, matching `toNumber`'s PINNED behaviour. + */ +export function holdsNumber(value: CellValue): boolean { + if (typeof value === "number") return !isNaN(value); + if (typeof value !== "string") return false; + return parseNumericString(value) !== null; +} + +/** + * Strict scalar coercion for single-value math functions (ABS, SIGN). Follows + * Excel's scalar rules: a boolean is 1 / 0 and genuinely non-numeric text is + * `#VALUE!` rather than a silent 0. A partly-numeric string still yields its + * leading number, matching the engine's other numeric reads. + */ +export function toScalarNumber(value: CellValue): number | SpreadsheetError { + if (typeof value === "number") return value; + if (typeof value === "boolean") return value ? 1 : 0; + if (typeof value !== "string") return value; + return parseNumericString(value) ?? VALUE_ERROR; +} diff --git a/src/engine/parser.ts b/src/engine/parser.ts index 504eef1..d040eb7 100644 --- a/src/engine/parser.ts +++ b/src/engine/parser.ts @@ -4,7 +4,6 @@ * Handles Excel A1 notation parsing and conversion */ -import type { CellRef, RangeRef } from "./types"; /** * Convert Excel column letters to 0-based index @@ -38,106 +37,3 @@ export function indexToColumn(index: number): string { } return col; } - -/** - * Parse a cell reference to its components - * Supports: A1, $A$1, $A1, A$1, Sheet1!A1, 'My Sheet'!A1 - * - * @param ref - Cell reference string - * @returns Parsed cell reference object - */ -export function parseCellRef(ref: string): CellRef { - let cellRef = ref; - let sheetName: string | undefined; - - // Check for cross-sheet reference (e.g., 'Sheet Name'!B2 or Sheet1!B2) - const sheetMatch = ref.match(/^(?:'([^']+)'|([^!]+))!(.+)$/); - if (sheetMatch) { - sheetName = sheetMatch[1] || sheetMatch[2]; // Quoted or unquoted sheet name - cellRef = sheetMatch[3]; // Cell reference part - } - - // Parse absolute references ($A$1) - const absoluteRow = cellRef.includes("$") && cellRef.match(/\$\d+/); - const absoluteCol = cellRef.includes("$") && cellRef.match(/\$[A-Z]+/); - - // Remove $ symbols - const cleanRef = cellRef.replace(/\$/g, ""); - const match = cleanRef.match(/^([A-Z]+)(\d+)$/); - - if (!match) { - throw new Error(`Invalid cell reference: ${ref}`); - } - - const col = columnToIndex(match[1]); - const row = parseInt(match[2]) - 1; // 1-indexed to 0-indexed - - const result: CellRef = { row, col }; - - if (sheetName) { - result.sheet = sheetName; - } - - if (absoluteRow || absoluteCol) { - result.absolute = { - row: !!absoluteRow, - col: !!absoluteCol, - }; - } - - return result; -} - -/** - * Parse a range reference to its components - * Supports: A1:B10, $A$1:$B$10, Sheet1!A1:B10 - * - * @param range - Range reference string - * @returns Parsed range reference object - */ -export function parseRangeRef(range: string): RangeRef { - // Use non-greedy match to improve performance - const colonIndex = range.lastIndexOf(":"); - if (colonIndex === -1) { - throw new Error(`Invalid range reference: ${range}`); - } - - const start = parseCellRef(range.substring(0, colonIndex)); - const end = parseCellRef(range.substring(colonIndex + 1)); - - return { start, end }; -} - -/** - * Convert a cell reference object back to A1 notation - * - * @param ref - Cell reference object - * @returns A1 notation string - */ -export function cellRefToA1(ref: CellRef): string { - const col = indexToColumn(ref.col); - const row = ref.row + 1; // 0-based to 1-based - - let result = ""; - - if (ref.absolute?.col) { - result += "$"; - } - result += col; - - if (ref.absolute?.row) { - result += "$"; - } - result += row; - - if (ref.sheet) { - // Quote sheet name if it contains spaces - if (ref.sheet.includes(" ")) { - result = `'${ref.sheet}'!${result}`; - } else { - result = `${ref.sheet}!${result}`; - } - } - - return result; -} diff --git a/src/engine/registry.ts b/src/engine/registry.ts index f332ea4..6242674 100644 --- a/src/engine/registry.ts +++ b/src/engine/registry.ts @@ -6,22 +6,54 @@ */ import type { CellValue } from "./types"; +import { parseNumericString } from "./numericCoercion"; export type { CellValue }; export type CellGetter = (ref: string) => CellValue; export type RangeGetter = (range: string) => CellValue[]; export type RawRangeGetter = (range: string) => CellValue[]; export interface FunctionContext { + /** The registry name the formula invoked, so a handler can report a failure + * the way the evaluator names it. */ + functionName: string; getCellValue: CellGetter; getRangeValues: RangeGetter; - getRangeValuesRaw?: RawRangeGetter; + getRangeValuesRaw?: RawRangeGetter | undefined; evaluateFormula: (formula: string) => CellValue; } -export type FunctionHandler = ( - args: string[], - context: FunctionContext, -) => CellValue; +/** The too-few-arguments error, raised by the evaluator from the registry's + * `minArgs` before any handler runs — and by `requiredArg` for the argument a + * handler actually reads, so both report one wording. */ +export const tooFewArgumentsError = (funcName: string, minArgs: number): Error => + new Error(`${funcName} requires at least ${minArgs} argument${minArgs !== 1 ? "s" : ""}`); + +/** The argument at `index`, which the caller has already established is there — + * the registry's `minArgs` for a mandatory argument, an `args.length` branch for + * an optional one. An absent one means that guarantee is wrong, so it raises the + * arity error the guarantee should have raised. Never a default value: a + * substituted 0 or "" computes a plausible wrong answer, which is worse than + * the error the formula deserves. */ +export const requiredArg = (context: FunctionContext, args: string[], index: number): string => { + const arg = args[index]; + if (arg === undefined) throw tooFewArgumentsError(context.functionName, index + 1); + return arg; +}; + +/** The range reader that keeps every cell, for functions whose answer depends on + * a range's POSITIONS rather than its numbers. + * + * `getRangeValues` drops non-numeric cells, which silently renumbers the rows + * underneath: a criteria/value pair read that way falls out of alignment and + * aggregates a different row, and an index returned from it points at the wrong + * one. SUMIF/AVERAGEIF were moved here in #2358; MATCH/XLOOKUP in #2765. + * + * Falls back to the numeric reader because `getRangeValuesRaw` is optional on + * `FunctionContext` — a host that predates it keeps its old behaviour rather + * than throwing. */ +export const rawRangeReader = (context: FunctionContext): RangeGetter => context.getRangeValuesRaw ?? context.getRangeValues; + +export type FunctionHandler = (args: string[], context: FunctionContext) => CellValue; export interface FunctionDefinition { name: string; @@ -70,35 +102,13 @@ class FunctionRegistry { export const functionRegistry = new FunctionRegistry(); /** - * Helper function to convert a value to a number + * Lenient numeric coercion for range aggregation: anything unreadable is 0. + * PINNED behaviour (booleans are 0, not Excel's 1/0) — the string parsing lives + * in numericCoercion.parseNumericString, shared with the strict scalar read. */ export function toNumber(value: CellValue): number { if (typeof value === "number") return value; - - // Handle percentage strings like "5%" or "0.4167%" - if (typeof value === "string" && value.includes("%")) { - const numericPart = value.replace("%", "").trim(); - const num = parseFloat(numericPart); - return isNaN(num) ? 0 : num / 100; - } - - // Handle currency strings like "$1,000" or "$1,000.00" - if (typeof value === "string" && value.includes("$")) { - const numericPart = value.replace(/[$,]/g, "").trim(); - const num = parseFloat(numericPart); - return isNaN(num) ? 0 : num; - } - - // Handle comma-separated numbers like "1,000" - if (typeof value === "string" && value.includes(",")) { - const numericPart = value.replace(/,/g, "").trim(); - const num = parseFloat(numericPart); - return isNaN(num) ? 0 : num; - } - - // Handle regular numeric strings - const num = parseFloat(String(value)); - return isNaN(num) ? 0 : num; + return parseNumericString(String(value)) ?? 0; } /** @@ -108,6 +118,34 @@ export function toString(value: CellValue): string { return String(value); } +const REGEXP_METACHARACTERS = /[.*+?^${}()|[\]\\]/; + +const escapeRegExpChar = (char: string): string => (REGEXP_METACHARACTERS.test(char) ? `\\${char}` : char); + +// One criteria token: `~x` (an escaped character) or any single character. The +// escape branch needs a following character, so a TRAILING `~` falls through to +// the single-character branch and stands for itself. +const CRITERIA_TOKEN = /~([\s\S])|[\s\S]/g; + +const criteriaRegexSource = (pattern: string): string => + pattern.replace(CRITERIA_TOKEN, (token, escaped: string | undefined) => { + if (escaped !== undefined) return escapeRegExpChar(escaped); + if (token === "*") return ".*"; + if (token === "?") return "."; + return escapeRegExpChar(token); + }); + +/** + * Match text the way a spreadsheet criteria does: case-insensitively, with `*` + * standing for any run of characters, `?` for exactly one, and `~` escaping the + * next character. A plain `String(v) === criteria` missed both — `COUNTIF(range, + * "yes")` skipped a cell holding `Yes`, and `"A*"` was compared literally. + */ +function textMatcher(pattern: string): (text: string) => boolean { + const regex = new RegExp(`^${criteriaRegexSource(pattern)}$`, "iu"); + return (text) => regex.test(text); +} + /** * Helper to parse criteria for conditional functions like COUNTIF, SUMIF * Returns a comparison function that tests if a value matches the criteria @@ -118,11 +156,13 @@ export function parseCriteria(criteria: string): (value: CellValue) => boolean { // Check for comparison operators // eslint-disable -- sonarjs/slow-regex - const opMatch = trimmedCriteria.match(/^([><=!]+)(.+)$/); - if (opMatch) { - const [, op, value] = opMatch; + const [, op, value] = trimmedCriteria.match(/^([><=!]+)(.+)$/) ?? []; + if (op !== undefined && value !== undefined) { const numValue = parseFloat(value); + // `=` / `<>` compare like the bare criteria does — case-insensitive text + // with wildcards, or the number. `<>` is exactly its negation. + const equals = matchesTextOrNumber(value); switch (op) { case ">": return (v) => toNumber(v) > numValue; @@ -134,20 +174,23 @@ export function parseCriteria(criteria: string): (value: CellValue) => boolean { return (v) => toNumber(v) <= numValue; case "=": case "==": - return (v) => String(v) === value || toNumber(v) === numValue; + return equals; case "!=": case "<>": - return (v) => String(v) !== value && toNumber(v) !== numValue; + return (v) => !equals(v); default: return () => false; } } - // Exact match (string or number) - return (v) => { - const strMatch = String(v) === trimmedCriteria; - const numCriteria = parseFloat(trimmedCriteria); - const numMatch = !isNaN(numCriteria) && toNumber(v) === numCriteria; - return strMatch || numMatch; - }; + return matchesTextOrNumber(trimmedCriteria); +} + +/** A value matches when its text matches the criteria (case-insensitively, with + * wildcards) or, for a numeric criteria, when its number is equal. */ +function matchesTextOrNumber(criteria: string): (value: CellValue) => boolean { + const matchesText = textMatcher(criteria); + const numCriteria = parseFloat(criteria); + const hasNumber = !isNaN(numCriteria); + return (value) => matchesText(String(value)) || (hasNumber && toNumber(value) === numCriteria); } diff --git a/src/engine/responseDecoder.ts b/src/engine/responseDecoder.ts new file mode 100644 index 0000000..1cae216 --- /dev/null +++ b/src/engine/responseDecoder.ts @@ -0,0 +1,68 @@ +/** + * Decode the `/api/files/content` response for a spreadsheet file + * into an "ok with sheets" / "error with message" discriminated + * union. Extracted from `fetchSheets` in + * `src/plugins/spreadsheet/View.vue` where the decision tree was + * inlined as several nested try/catch + if branches. + * + * Pure — no fetch, no refs. Takes the parsed JSON body and returns + * a result the caller can pattern-match on. Tested in + * `test/plugins/spreadsheet/engine/test_responseDecoder.ts`. + */ + +import type { SheetData } from "./types.js"; +import { errorMessage } from "./guards"; + +/** Shape of the `/api/files/content` response we care about. The + * server returns more fields (kind, size, modifiedMs, …) but this + * decoder only depends on the three that drive branching. */ +export interface FilesContentResponseLike { + kind?: string; + content?: string; + message?: string; +} + +export type DecodeResult = { kind: "ok"; sheets: SheetData[] } | { kind: "error"; message: string }; + +/** + * Turn a parsed `/files/content` body into an OK/error decision: + * + * - `kind` present and not "text" → error with the server's message + * (e.g. "too-large", "binary"). Spreadsheets only live in text + * JSON files. + * - Missing or non-string `content` → error. + * - `content` is not valid JSON → error. + * - Parsed `content` is not an array → error (server should never + * return a non-array but the guard protects downstream render). + * - Otherwise → ok with the sheets array. + */ +export function decodeSpreadsheetResponse(body: FilesContentResponseLike): DecodeResult { + if (body.kind && body.kind !== "text") { + return { + kind: "error", + message: body.message ?? `Cannot load spreadsheet: ${body.kind}`, + }; + } + if (typeof body.content !== "string") { + return { kind: "error", message: "Spreadsheet file has no content" }; + } + let parsed: unknown; + try { + parsed = JSON.parse(body.content); + } catch (err) { + return { + kind: "error", + message: `Spreadsheet JSON is malformed: ${errorMessage(err, "parse error")}`, + }; + } + if (!Array.isArray(parsed)) { + return { + kind: "error", + message: "Spreadsheet content is not an array of sheets", + }; + } + // Array.isArray narrows to unknown[]; we trust the server contract + // and type the local explicitly rather than using an inline `as`. + const sheets: SheetData[] = parsed; + return { kind: "ok", sheets }; +} diff --git a/src/engine/spreadsheet-errors.ts b/src/engine/spreadsheet-errors.ts new file mode 100644 index 0000000..8e2040d --- /dev/null +++ b/src/engine/spreadsheet-errors.ts @@ -0,0 +1,82 @@ +/** + * Formula errors as a distinct VALUE type. + * + * A `#NUM!` produced by `SQRT(-1)` and the text `"#NUM!"` produced by + * `CONCAT("#N","UM!")` used to be the same string, so IFERROR could not tell a + * real error from text that merely spells one (#2451). An error is now its own + * value carrying the code, which is what the error-aware functions check; the + * display pass renders it back to `#NUM!` so cells look unchanged. + */ + +export const SPREADSHEET_ERRORS = ["#NULL!", "#DIV/0!", "#VALUE!", "#REF!", "#NAME?", "#NUM!", "#N/A", "#ERROR!"] as const; + +export type SpreadsheetErrorCode = (typeof SPREADSHEET_ERRORS)[number]; + +/** A formula error, distinct from any string. `toString` yields the code so the + * value renders as `#NUM!` wherever the engine coerces a cell value to text. */ +export class SpreadsheetError { + constructor(readonly code: SpreadsheetErrorCode) {} + + toString(): string { + return this.code; + } + + toJSON(): string { + return this.code; + } +} + +export const NULL_ERROR = new SpreadsheetError("#NULL!"); +export const DIV_ZERO_ERROR = new SpreadsheetError("#DIV/0!"); +export const VALUE_ERROR = new SpreadsheetError("#VALUE!"); +export const REF_ERROR = new SpreadsheetError("#REF!"); +export const NAME_ERROR = new SpreadsheetError("#NAME?"); +export const NUM_ERROR = new SpreadsheetError("#NUM!"); +export const NA_ERROR = new SpreadsheetError("#N/A"); +export const UNKNOWN_ERROR = new SpreadsheetError("#ERROR!"); + +// One instance per code: two errors of the same kind compare equal, which is how +// error strings behaved and what the condition/comparison paths still rely on. +const ERROR_VALUES: Record = { + "#NULL!": NULL_ERROR, + "#DIV/0!": DIV_ZERO_ERROR, + "#VALUE!": VALUE_ERROR, + "#REF!": REF_ERROR, + "#NAME?": NAME_ERROR, + "#NUM!": NUM_ERROR, + "#N/A": NA_ERROR, + "#ERROR!": UNKNOWN_ERROR, +}; + +/** The error value for a code. */ +export const spreadsheetError = (code: SpreadsheetErrorCode): SpreadsheetError => ERROR_VALUES[code]; + +const errorSet: ReadonlySet = new Set(SPREADSHEET_ERRORS); + +/** Whether a value is one of the error CODES as a plain string. Text that spells + * an error is not an error value — see `isSpreadsheetErrorValue`. */ +export function isSpreadsheetError(value: unknown): value is SpreadsheetErrorCode { + return typeof value === "string" && errorSet.has(value); +} + +/** Whether a value IS a formula error, as opposed to text that spells one. */ +export function isSpreadsheetErrorValue(value: unknown): value is SpreadsheetError { + return value instanceof SpreadsheetError; +} + +/** The code behind a value: an error value's own code, or the code a plain + * string spells. Used where a literal `#REF!` written into a cell must still + * poison the arithmetic that reads it, provenance aside. */ +export function errorCodeOf(value: unknown): SpreadsheetErrorCode | null { + if (isSpreadsheetErrorValue(value)) return value.code; + return isSpreadsheetError(value) ? value : null; +} + +/** Whether an evaluated result should be treated as an error: a formula error + * VALUE, a NaN / infinite number, or a missing value. This is what IFERROR + * catches — deliberately NOT a look-alike string, which is ordinary text. */ +export function isErrorResult(value: unknown): boolean { + if (value === null || value === undefined) return true; + if (typeof value === "number") return Number.isNaN(value) || !Number.isFinite(value); + return isSpreadsheetErrorValue(value); +} diff --git a/src/engine/textFormat.ts b/src/engine/textFormat.ts new file mode 100644 index 0000000..c27880e --- /dev/null +++ b/src/engine/textFormat.ts @@ -0,0 +1,108 @@ +/** + * Excel number-format patterns for the TEXT function. + * + * Deliberately separate from `formatter.formatNumber`, which renders CELL + * display values: the two disagree on defaults — Excel's `TEXT(0.5,"0%")` is + * `"50%"`, while a cell carrying the format `0%` has always displayed + * `"50.00%"` — so sharing the whole path would move every formatted cell in + * every stored workbook. Only the digit-grouping primitive is shared. + */ + +import { groupThousands } from "./formatter"; + +const PERCENT_SCALE = 100; + +// Characters that introduce an Excel format feature this module does not +// render (quoted literals, fill/skip, fractions, dates, text placeholder, +// negative/zero sections). Seeing one means the caller keeps its own fallback. +const UNSUPPORTED_LITERAL_CHARS = `?\\*_[]"@;/`; + +// One run of digit placeholders, optionally carrying grouping commas and a +// decimal point: the `#,##0.00` family. +const NUMERIC_CORE_RE = /[#0][#0,.]*/; +const PLACEHOLDER_RE = /[#0]/; +const TRAILING_ZEROS_RE = /0+$/; + +export interface NumberPattern { + prefix: string; + suffix: string; + useGrouping: boolean; + integerMinDigits: number; + minDecimals: number; + maxDecimals: number; + isPercent: boolean; +} + +const countOf = (text: string, chars: string): number => Array.from(text).filter((char) => chars.includes(char)).length; + +const hasUnsupportedLiteral = (text: string): boolean => Array.from(text).some((char) => UNSUPPORTED_LITERAL_CHARS.includes(char)); + +/** + * Split a format code into a leading literal, one numeric core and a trailing + * literal, or `null` when the code uses a feature this module does not render. + */ +export const parseNumberPattern = (pattern: string): NumberPattern | null => { + const core = NUMERIC_CORE_RE.exec(pattern); + if (!core) return null; + // A comma AFTER the last placeholder scales by a thousand in Excel. + if (core[0].endsWith(",")) return null; + + const prefix = pattern.slice(0, core.index); + const suffix = pattern.slice(core.index + core[0].length); + // A second placeholder run (`0.00E+00`, `# ?/?`) is a format of its own kind. + if (PLACEHOLDER_RE.test(suffix)) return null; + if (hasUnsupportedLiteral(prefix) || hasUnsupportedLiteral(suffix)) return null; + + const [integerPart, decimalPart = "", ...extraParts] = core[0].split("."); + if (integerPart === undefined || extraParts.length > 0) return null; + + return { + prefix, + suffix, + useGrouping: integerPart.includes(","), + integerMinDigits: countOf(integerPart, "0"), + minDecimals: countOf(decimalPart, "0"), + maxDecimals: countOf(decimalPart, "#0"), + isPercent: prefix.includes("%") || suffix.includes("%"), + }; +}; + +// `0` keeps a decimal digit, `#` drops it once it is a trailing zero — so +// `0.0#` renders 0.5 as "0.5" but 0.25 as "0.25". +const trimOptionalZeros = (decimals: string, minDecimals: number): string => { + const kept = decimals.replace(TRAILING_ZEROS_RE, ""); + return kept.length >= minDecimals ? kept : decimals.slice(0, minDecimals); +}; + +const renderDigits = (absValue: number, pattern: NumberPattern): string => { + const fixed = absValue.toFixed(pattern.maxDecimals); + const point = fixed.indexOf("."); + const wholeDigits = point === -1 ? fixed : fixed.slice(0, point); + const decimalDigits = point === -1 ? "" : fixed.slice(point + 1); + const padded = wholeDigits.padStart(pattern.integerMinDigits, "0"); + const whole = pattern.useGrouping ? groupThousands(padded) : padded; + const decimals = trimOptionalZeros(decimalDigits, pattern.minDecimals); + return decimals ? `${whole}.${decimals}` : whole; +}; + +const isZeroText = (digits: string): boolean => Array.from(digits).every((char) => char === "0" || char === "," || char === "."); + +/** + * Render a number the way Excel's TEXT does for the common numeric format + * codes: `#,##0` grouping, `0.00` fixed decimals, `0.##` optional decimals, a + * literal prefix/suffix such as `$`, and a `%` that scales by 100. + * + * Returns `null` for a format code it does not render, so the caller can keep + * its own fallback rather than inventing a plausible wrong string. + */ +export const formatWithPattern = (value: number, pattern: string): string | null => { + if (!Number.isFinite(value)) return null; + const spec = parseNumberPattern(pattern); + if (!spec) return null; + + const scaled = spec.isPercent ? value * PERCENT_SCALE : value; + const digits = renderDigits(Math.abs(scaled), spec); + // The sign follows the ROUNDED digits: -0.001 under "0.00" reads "0.00", not "-0.00". + const sign = scaled < 0 && !isZeroText(digits) ? "-" : ""; + return `${sign}${spec.prefix}${digits}${spec.suffix}`; +}; diff --git a/src/engine/translateFormula.ts b/src/engine/translateFormula.ts new file mode 100644 index 0000000..9dfc0d3 --- /dev/null +++ b/src/engine/translateFormula.ts @@ -0,0 +1,69 @@ +/** + * Excel-formula → JS-expression translation + * + * Pure string transforms that turn the Excel-operator form of an + * already-substituted expression (cell references resolved, functions replaced) + * into the JavaScript form handed to `new Function`, plus the character + * allowlists that gate that evaluation. No engine or evaluator state is + * captured — every function here is input → output only. + */ + +/** Excel `^` exponentiation → JS `**`. + * + * Excel's `^` is left-associative (`2^3^2` = `(2^3)^2` = 64); JS `**` is + * right-associative (`2**3**2` = `2**(3**2)` = 512). This transform does not + * bridge that difference, so a chained `^` still evaluates the JS way — a known + * limitation tracked separately (#2359), pinned by the tests. */ +export function caretToPow(expr: string): string { + return expr.replace(/\^/g, "**"); +} + +/** Move the "inside a string literal" marker across one quote character. + * Empty marker means "not in a literal"; otherwise it holds the opening quote. */ +function toggleQuote(openQuote: string, char: string): string { + if (openQuote === "") return char; + return char === openQuote ? "" : openQuote; +} + +/** Excel `&` string concatenation → JS `+`, leaving any `&` inside a quoted + * string literal untouched. A quote toggles the "inside a literal" state only + * when it is not backslash-escaped. */ +export function replaceConcatOperator(expr: string): string { + const chars = expr.split(""); + const out: string[] = []; + let openQuote = ""; + chars.forEach((char, index) => { + const escaped = index > 0 && chars[index - 1] === "\\"; + if (!escaped && (char === '"' || char === "'")) { + openQuote = toggleQuote(openQuote, char); + } + out.push(char === "&" && openQuote === "" ? "+" : char); + }); + return out.join(""); +} + +/** Excel `=` equality → JS `==`, without disturbing `<=`, `>=` or `!=`. A single + * `=` is rewritten only when flanked by a non-`<>!` char on the left and a + * non-`=` on the right. + * + * Known limitations, pinned by the tests and tracked separately (#2359): the + * match consumes both flanking characters so replacements do not overlap + * (`5=6=7` → `5==6=7`, only the first rewritten); a `=` at either end of the + * expression has no left/right neighbour and is left alone (`5=` → `5=`); and a + * hand-typed `==` becomes `===` (its second `=` matches). Excel never emits the + * last two, so they only bite malformed input. */ +export function rewriteComparisonEq(expr: string): string { + return expr.replace(/([^<>!])=([^=])/g, "$1==$2"); +} + +/** Whether an expression contains only the characters an arithmetic evaluation + * may see: digits, ` + - * / ( ) . ` and spaces. Gates `new Function`. */ +export function isSafeArithmetic(expr: string): boolean { + return /^[\d+\-*/(). ]+$/.test(expr); +} + +/** Whether an expression contains only the characters a comparison evaluation + * may see: the arithmetic set plus ` < > ! = `. Gates `new Function`. */ +export function isSafeComparison(expr: string): boolean { + return /^[\d+\-*/(). <>!=]+$/.test(expr); +} diff --git a/src/engine/types.ts b/src/engine/types.ts index 747f60a..e530812 100644 --- a/src/engine/types.ts +++ b/src/engine/types.ts @@ -2,10 +2,17 @@ * Spreadsheet Engine Type Definitions */ -export type CellValue = number | string | boolean; +import type { SpreadsheetError } from "./spreadsheet-errors"; + +/** A value as it is STORED in a cell and serialized to the workbook JSON. A + * formula error is never stored — it only exists as a computed result. */ +export type StoredCellValue = number | string | boolean; + +/** A value as the engine COMPUTES it: a stored value, or a formula error. */ +export type CellValue = StoredCellValue | SpreadsheetError; export interface SpreadsheetCell { - v: CellValue; // Value or formula (formulas start with "=") + v: StoredCellValue; // Value or formula (formulas start with "=") f?: string; // Format code (e.g., "$#,##0.00") } @@ -57,7 +64,17 @@ export interface CalculationError { type: "circular" | "invalid_ref" | "div_zero" | "syntax" | "unknown"; } -export interface EngineOptions { +/** Per-calculation settings that change how cell CONTENT is read, as opposed + * to how the engine runs. Passed down explicitly so the engine stays pure — + * no ambient locale, no import-time state. */ +export interface CalculateOptions { + /** Read an ambiguous `A/B/YYYY` date as day-first. Only affects dates whose + * two leading numbers are both 12 or under; anything else decides itself. + * See engine/date-locale.ts for how a locale maps onto this. */ + preferDDMMYYYY?: boolean; +} + +export interface EngineOptions extends CalculateOptions { maxIterations?: number; // For circular reference detection enableCrossSheetRefs?: boolean; // Default: true strictMode?: boolean; // Throw on errors vs. return 0 diff --git a/tests/engine/cellAccess.ts b/tests/engine/cellAccess.ts new file mode 100644 index 0000000..4c90f2d --- /dev/null +++ b/tests/engine/cellAccess.ts @@ -0,0 +1,18 @@ +// Reading a cell out of a calculated grid. `0`, `""` and `false` are all +// legitimate cell values, so absence has to be tested explicitly — a truthiness +// check would reject a correct answer, and an assertion on `undefined` would let +// a missing row pass unnoticed. + +import type { CellValue } from "../../src/engine/types.ts"; + +export const rowAt = (grid: CellValue[][], row: number): CellValue[] => { + const line = grid[row]; + if (line === undefined) throw new Error(`calculated sheet has no row ${row}`); + return line; +}; + +export const cellAt = (grid: CellValue[][], row: number, col: number): CellValue => { + const cell = rowAt(grid, row)[col]; + if (cell === undefined) throw new Error(`calculated sheet has no cell (${row}, ${col})`); + return cell; +}; diff --git a/tests/engine/fixtures/README.md b/tests/engine/fixtures/README.md index b66c814..7f154ac 100644 --- a/tests/engine/fixtures/README.md +++ b/tests/engine/fixtures/README.md @@ -464,3 +464,12 @@ This creates one test case per fixture file automatically! ``` Done! No code changes, no configuration - just add two JSON files. + +## A note on the financial fixture + +`financial-functions` expects the sign convention Excel itself uses: a payment +leaves the account, so `PMT`, `IPMT` and `PPMT` are negative when the present +value is positive. Two cells in `expected/financial-functions.json` disagreed +with that and were corrected when the engine was replaced — the interest payment +had the wrong sign, and the principal payment was not `PMT - IPMT`. If a change +makes either go positive again, it is the change that is wrong. diff --git a/tests/engine/fixtures/expected/financial-functions.json b/tests/engine/fixtures/expected/financial-functions.json index 1429d2f..d4b338c 100644 --- a/tests/engine/fixtures/expected/financial-functions.json +++ b/tests/engine/fixtures/expected/financial-functions.json @@ -3,8 +3,8 @@ ["Monthly Payment", "PMT(6%/12, 30*12, 250000)", "-$1,498.88", "Mortgage"], ["Future Value", "FV(5%/12, 12*5, -200, -5000, 0)", "$20,018.01", "Savings"], ["Present Value", "PV(4%/12, 10*12, -300)", "$29,631.05", "Annuity"], - ["Interest Payment P1", "IPMT(5%/12,1,24,10000)", "$41.67", "Loan"], - ["Principal Payment P1", "PPMT(5%/12,1,24,10000)", "-$480.38", "Loan"], + ["Interest Payment P1", "IPMT(5%/12,1,24,10000)", "-$41.67", "Loan"], + ["Principal Payment P1", "PPMT(5%/12,1,24,10000)", "-$397.05", "Loan"], ["Periods Needed", "NPER(7%/12,-500,15000)", "33.07", "Goal"], ["Loan Rate", "RATE(36,-300,9000)", "0.0102", "Guess 10%"], ["", "", "", ""], diff --git a/tests/engine/parser.test.ts b/tests/engine/parser.test.ts index fd862c9..7e626d0 100644 --- a/tests/engine/parser.test.ts +++ b/tests/engine/parser.test.ts @@ -1,15 +1,17 @@ /** * Parser Unit Tests + * + * The A1-reference PARSING tests that used to live here went with + * parseCellRef / parseRangeRef / cellRefToA1: the engine resolves references + * through formulaRefs (expandRange / expandRangeOrCell) now, and those are + * covered in tests/engine/test_expandRangeOrCell.ts and + * test_cellRefSubstitution.ts. Column conversion stays because it is still the + * engine's own public helper, used by the Vue view. */ import { describe, test, expect } from "vitest"; -import { - columnToIndex, - indexToColumn, - parseCellRef, - parseRangeRef, - cellRefToA1, -} from "../../src/engine/parser"; +import { columnToIndex, indexToColumn } from "../../src/engine/parser"; + describe("Parser - Column Conversion", () => { describe("columnToIndex", () => { @@ -68,156 +70,3 @@ describe("Parser - Column Conversion", () => { } }); }); - -describe("Parser - Cell References", () => { - describe("parseCellRef - basic references", () => { - test("parses simple cell reference", () => { - expect(parseCellRef("A1")).toEqual({ row: 0, col: 0 }); - expect(parseCellRef("B2")).toEqual({ row: 1, col: 1 }); - expect(parseCellRef("Z26")).toEqual({ row: 25, col: 25 }); - }); - - test("parses double letter columns", () => { - expect(parseCellRef("AA1")).toEqual({ row: 0, col: 26 }); - expect(parseCellRef("AB10")).toEqual({ row: 9, col: 27 }); - }); - }); - - describe("parseCellRef - absolute references", () => { - test("parses fully absolute reference ($A$1)", () => { - expect(parseCellRef("$A$1")).toEqual({ - row: 0, - col: 0, - absolute: { row: true, col: true }, - }); - }); - - test("parses column-absolute reference ($A1)", () => { - expect(parseCellRef("$A1")).toEqual({ - row: 0, - col: 0, - absolute: { row: false, col: true }, - }); - }); - - test("parses row-absolute reference (A$1)", () => { - expect(parseCellRef("A$1")).toEqual({ - row: 0, - col: 0, - absolute: { row: true, col: false }, - }); - }); - - test("parses mixed absolute reference ($B$5)", () => { - expect(parseCellRef("$B$5")).toEqual({ - row: 4, - col: 1, - absolute: { row: true, col: true }, - }); - }); - }); - - describe("parseCellRef - cross-sheet references", () => { - test("parses simple cross-sheet reference", () => { - expect(parseCellRef("Sheet1!A1")).toEqual({ - row: 0, - col: 0, - sheet: "Sheet1", - }); - }); - - test("parses quoted sheet name with spaces", () => { - expect(parseCellRef("'My Sheet'!B2")).toEqual({ - row: 1, - col: 1, - sheet: "My Sheet", - }); - }); - - test("parses cross-sheet with absolute reference", () => { - expect(parseCellRef("Sheet1!$A$1")).toEqual({ - row: 0, - col: 0, - sheet: "Sheet1", - absolute: { row: true, col: true }, - }); - }); - }); - - describe("parseCellRef - error handling", () => { - test("throws on invalid reference", () => { - expect(() => parseCellRef("invalid")).toThrow("Invalid cell reference"); - expect(() => parseCellRef("123")).toThrow("Invalid cell reference"); - expect(() => parseCellRef("A")).toThrow("Invalid cell reference"); - }); - }); -}); - -describe("Parser - Range References", () => { - test("parses simple range", () => { - expect(parseRangeRef("A1:B2")).toEqual({ - start: { row: 0, col: 0 }, - end: { row: 1, col: 1 }, - }); - }); - - test("parses large range", () => { - expect(parseRangeRef("A1:Z100")).toEqual({ - start: { row: 0, col: 0 }, - end: { row: 99, col: 25 }, - }); - }); - - test("parses range with absolute references", () => { - expect(parseRangeRef("$A$1:$B$10")).toEqual({ - start: { row: 0, col: 0, absolute: { row: true, col: true } }, - end: { row: 9, col: 1, absolute: { row: true, col: true } }, - }); - }); - - test("throws on invalid range", () => { - expect(() => parseRangeRef("A1")).toThrow("Invalid range reference"); - expect(() => parseRangeRef("invalid:range")).toThrow( - "Invalid cell reference", - ); - }); -}); - -describe("Parser - Cell Reference to A1", () => { - test("converts basic cell ref to A1", () => { - expect(cellRefToA1({ row: 0, col: 0 })).toBe("A1"); - expect(cellRefToA1({ row: 1, col: 1 })).toBe("B2"); - expect(cellRefToA1({ row: 25, col: 25 })).toBe("Z26"); - }); - - test("converts absolute references to A1", () => { - expect( - cellRefToA1({ row: 0, col: 0, absolute: { row: true, col: true } }), - ).toBe("$A$1"); - expect( - cellRefToA1({ row: 0, col: 0, absolute: { row: false, col: true } }), - ).toBe("$A1"); - expect( - cellRefToA1({ row: 0, col: 0, absolute: { row: true, col: false } }), - ).toBe("A$1"); - }); - - test("converts cross-sheet references to A1", () => { - expect(cellRefToA1({ row: 0, col: 0, sheet: "Sheet1" })).toBe("Sheet1!A1"); - expect(cellRefToA1({ row: 1, col: 1, sheet: "My Sheet" })).toBe( - "'My Sheet'!B2", - ); - }); - - test("round-trip conversion works", () => { - const testRefs = ["A1", "$A$1", "$A1", "A$1", "Sheet1!A1", "'My Sheet'!B2"]; - - for (const ref of testRefs) { - const parsed = parseCellRef(ref); - const converted = cellRefToA1(parsed); - // Re-parse to normalize (e.g., Sheet1!A1 vs 'Sheet1'!A1) - const reparsed = parseCellRef(converted); - expect(reparsed).toEqual(parsed); - } - }); -}); diff --git a/tests/engine/run-evaluator-tests.ts b/tests/engine/run-evaluator-tests.ts index 7bcd9bd..923fbc8 100644 --- a/tests/engine/run-evaluator-tests.ts +++ b/tests/engine/run-evaluator-tests.ts @@ -50,12 +50,17 @@ function createContext( ranges: Record = {}, rawRanges?: Record, ): EvaluatorContext { + // A single cell answers the RANGE readers too, the way the real context does: + // it resolves every reference through expandRangeOrCell, so an aggregate like + // MAX(B1, A1:A3, 5) reads B1 via getRangeValues. A fake that answered nothing + // for "B1" dropped that argument without saying so. + const oneCell = (ref: string): (number | string)[] => (ref in cells ? [cells[ref]] : []); const context: EvaluatorContext = { getCellValue: (ref: string) => cells[ref] ?? 0, getRangeValues: (range: string) => - ranges[range] ?? rawRanges?.[range] ?? [], + ranges[range] ?? rawRanges?.[range] ?? oneCell(range), getRangeValuesRaw: (range: string) => - rawRanges?.[range] ?? ranges[range] ?? [], + rawRanges?.[range] ?? ranges[range] ?? oneCell(range), evaluateFormula: (formula: string) => evaluateFormula(formula, context), }; return context; diff --git a/tests/engine/test_argCountValidation.ts b/tests/engine/test_argCountValidation.ts new file mode 100644 index 0000000..383702c --- /dev/null +++ b/tests/engine/test_argCountValidation.ts @@ -0,0 +1,128 @@ +// #2397: the handler-side `if (args.length ...) throw` guards were removed for +// every function whose arity is fully expressed by the registry minArgs/maxArgs. +// The evaluator (evaluator.ts) validates arity from the registry BEFORE calling +// the handler, so those guards were unreachable dead code with a divergent +// message ("requires N" vs the evaluator's "requires at least N"). +// +// These tests pin that (a) invalid arity now surfaces the evaluator's single +// consistent message for the removed-guard functions — including the financial +// handlers, whose only remaining validation is the evaluator after #2394/#2442 — +// and (b) the shapes the registry CANNOT express keep their handler guard: IFS's +// even-count requirement and IRR's empty-range check. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { SpreadsheetEngine, type SheetData, type CellValue } from "../../src/engine/index.ts"; +import { cellAt } from "./cellAccess.ts"; + +/** Calculate `sheet` and report the value, error type, and recorded message for + * the cell at (row, col). */ +function cellResult(sheet: SheetData, row: number, col: number): { value: CellValue; type?: string | undefined; message?: string | undefined } { + const result = new SpreadsheetEngine().calculate(sheet); + const entry = result.errors.find((err) => err.cell.row === row && err.cell.col === col); + return { value: cellAt(result.data, row, col), type: entry?.type, message: entry?.error }; +} + +/** A one-cell sheet holding `formula` at A1 (for formulas that need no other cells). */ +const soleFormula = (formula: string) => cellResult({ name: "S", data: [[{ v: formula }]] }, 0, 0); + +describe("#2397 removed handler guards — arity now enforced by the evaluator", () => { + it("too few args surfaces the evaluator's 'requires at least N' message (not the handler's)", () => { + const { value, type, message } = soleFormula("=ROUND(1)"); + assert.equal(value, "#ERROR!"); + assert.equal(type, "unknown"); + assert.equal(message, "ROUND requires at least 2 arguments"); + }); + + it("too many args surfaces the evaluator's 'accepts at most N' message", () => { + const { value, type, message } = soleFormula("=ROUND(1, 2, 3)"); + assert.equal(value, "#ERROR!"); + assert.equal(type, "unknown"); + assert.equal(message, "ROUND accepts at most 2 arguments"); + }); + + it("singular wording for a one-argument bound (ABS accepts at most 1 argument)", () => { + assert.equal(soleFormula("=ABS(1, 2)").message, "ABS accepts at most 1 argument"); + }); + + it("singular wording for a one-argument minimum (UPPER requires at least 1 argument)", () => { + assert.equal(soleFormula("=UPPER()").message, "UPPER requires at least 1 argument"); + }); + + it("a zero-arg function rejects any argument (PI accepts at most 0 arguments)", () => { + assert.equal(soleFormula("=PI(1)").message, "PI accepts at most 0 arguments"); + }); + + // Financial handlers used to carry the ONLY validation (IPMT/PPMT once called + // pmtHandler/fvHandler directly). After #2394/#2442 they call the pure + // computeIpmt/computePpmt and are reached only through the evaluator, so the + // evaluator is now their sole arity gate. + it("financial: FV too few args is the evaluator error (handler no longer guards)", () => { + assert.equal(soleFormula("=FV(0.05, 10)").message, "FV requires at least 3 arguments"); + }); + + it("financial: IPMT too many args is the evaluator error", () => { + assert.equal(soleFormula("=IPMT(0.05, 1, 10, 1000, 0, 0, 9)").message, "IPMT accepts at most 6 arguments"); + }); +}); + +describe("#2397 SUMIF / AVERAGEIF — arity fully expressed by registry [2,3], enforced by evaluator", () => { + // Formula in C1 so it never self-references the A/B ranges it reads. + const withData = (formula: string): SheetData => ({ + name: "S", + data: [ + [{ v: 1 }, { v: 10 }, { v: formula }], + [{ v: 2 }, { v: 20 }], + [{ v: 3 }, { v: 30 }], + ], + }); + + it("SUMIF with 1 arg is rejected with the consistent 'at least 2' message", () => { + const { value, message } = cellResult(withData("=SUMIF(A1:A3)"), 0, 2); + assert.equal(value, "#ERROR!"); + assert.equal(message, "SUMIF requires at least 2 arguments"); + }); + + it("SUMIF with 4 args is rejected with the consistent 'at most 3' message", () => { + const { value, message } = cellResult(withData('=SUMIF(A1:A3, ">0", B1:B3, C1)'), 0, 2); + assert.equal(value, "#ERROR!"); + assert.equal(message, "SUMIF accepts at most 3 arguments"); + }); + + it("AVERAGEIF with 4 args is rejected by the evaluator", () => { + assert.equal(cellResult(withData('=AVERAGEIF(A1:A3, ">0", B1:B3, C1)'), 0, 2).message, "AVERAGEIF accepts at most 3 arguments"); + }); + + it("a valid 3-arg SUMIF still computes", () => { + assert.equal(cellResult(withData('=SUMIF(A1:A3, ">1", B1:B3)'), 0, 2).value, 50); + }); +}); + +describe("#2397 kept guards — shapes the registry cannot express", () => { + // IFS is minArgs:2 with no maxArgs; the EVEN-count requirement is inexpressible + // by min/max, so the handler guard stays. This is the red-on-break target: if + // the guard is removed, a 3-arg IFS no longer reports this arity error. + it("IFS with an odd number of args is rejected by the kept handler guard", () => { + const sheet: SheetData = { name: "S", data: [[{ v: 5 }, { v: 0 }, { v: "=IFS(A1>0, 1, A1>5)" }]] }; + const { value, type, message } = cellResult(sheet, 0, 2); + assert.equal(value, "#ERROR!"); + assert.equal(type, "unknown"); + assert.equal(message, "IFS requires an even number of arguments (condition-value pairs)"); + }); + + it("a valid even-arg IFS still returns the matched value", () => { + const sheet: SheetData = { name: "S", data: [[{ v: 5 }, { v: '=IFS(A1>0, "yes")' }]] }; + assert.equal(cellResult(sheet, 0, 1).value, "yes"); + }); + + // IRR is minArgs:1/maxArgs:2 — its arg-count guard was removed — but the + // "at least one numeric value in the range" rule is not an arg count and is + // kept. D1:D3 is an empty range, so the kept guard fires. + it("IRR over an empty range is rejected by the kept empty-range guard", () => { + const sheet: SheetData = { name: "S", data: [[{ v: 1 }, { v: 2 }, { v: 3 }, { v: "=IRR(F1:F3)" }]] }; + const { value, type, message } = cellResult(sheet, 0, 3); + assert.equal(value, "#ERROR!"); + assert.equal(type, "unknown"); + assert.equal(message, "IRR requires at least one value"); + }); +}); diff --git a/tests/engine/test_calculateDateLocale.ts b/tests/engine/test_calculateDateLocale.ts new file mode 100644 index 0000000..8782039 --- /dev/null +++ b/tests/engine/test_calculateDateLocale.ts @@ -0,0 +1,98 @@ +// The date-order setting reaching the cells. `prefersDayFirst` and `parseDate` +// are each covered on their own; what this file checks is that the flag +// actually travels from `EngineOptions` down to every place that reads a date — +// including the FORMAT the cell is given, because parsing day-first while +// rendering month-first would just move the confusion rather than fix it. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +// Imported through the barrel, not the class module: `engine/index.ts` is what +// registers the built-in functions, so a direct import leaves DAY() unknown. +import { SpreadsheetEngine, type SheetData } from "../../src/engine/index.ts"; +import { cellAt, rowAt } from "./cellAccess.ts"; + +const sheetWith = (value: string): SheetData => ({ name: "S", data: [[{ v: value }]] }); + +const renderedCell = (value: string, preferDDMMYYYY: boolean): unknown => + cellAt(new SpreadsheetEngine({ preferDDMMYYYY }).calculate(sheetWith(value)).data, 0, 0); + +describe("date order reaches the cell", () => { + // The whole point: the same text means different days in different places, + // and neither reading raises. + it("reads an ambiguous date month-first by default and day-first when asked", () => { + assert.equal(renderedCell("03/04/2025", false), "03/04/2025"); + assert.equal(renderedCell("03/04/2025", true), "03/04/2025"); + }); + + // Rendering has to follow the reading, or a day-first user sees their April 3 + // written back as "04/03" under an MM/DD label. + it("renders in the order it read, so the cell round-trips", () => { + const monthFirst = new SpreadsheetEngine({ preferDDMMYYYY: false }).calculate(sheetWith("03/04/2025")); + const dayFirst = new SpreadsheetEngine({ preferDDMMYYYY: true }).calculate(sheetWith("03/04/2025")); + // Same displayed text, different underlying dates — which is exactly what a + // user in each locale expects to see. + assert.equal(cellAt(monthFirst.data, 0, 0), "03/04/2025"); + assert.equal(cellAt(dayFirst.data, 0, 0), "03/04/2025"); + }); + + // Month-name formats are exercised in test_formatter.ts; this only needs the + // shapes the day/month decision can reach. + it("leaves unambiguous dates alone under either setting", () => { + for (const prefer of [false, true]) { + assert.equal(renderedCell("13/04/2025", prefer), "13/04/2025", "13 can only be a day"); + assert.equal(renderedCell("2025-03-04", prefer), "2025-03-04", "ISO is not affected"); + } + }); + + it("leaves non-dates alone", () => { + for (const prefer of [false, true]) { + assert.equal(renderedCell("hello", prefer), "hello"); + } + }); + + // The default must stay month-first, or every existing sheet silently + // reinterprets on the next render. + it("defaults to month-first when the option is omitted", () => { + const engine = new SpreadsheetEngine(); + assert.equal(engine.getOptions().preferDDMMYYYY, false); + }); + + it("can be changed after construction", () => { + const engine = new SpreadsheetEngine(); + engine.setOptions({ preferDDMMYYYY: true }); + assert.equal(engine.getOptions().preferDDMMYYYY, true); + }); +}); + +describe("date order reaches formulas", () => { + const formulaResult = (cells: string[], preferDDMMYYYY: boolean): unknown => + rowAt(new SpreadsheetEngine({ preferDDMMYYYY }).calculate({ name: "S", data: [cells.map((cell) => ({ v: cell }))] }).data, 0).at(-1); + + // `DAY()` reads the serial, so it reports which number the parser took as the + // day — the clearest observable difference between the two settings. + it("applies the setting to a date held in a cell", () => { + assert.equal(formulaResult(["03/04/2025", "=DAY(A1)"], false), 4, "month-first: the 4 is the day"); + assert.equal(formulaResult(["03/04/2025", "=DAY(A1)"], true), 3, "day-first: the 3 is the day"); + }); + + // A date written INSIDE a formula never passes through the cell + // preprocessing — the evaluator parses it on its own, so the setting has to + // reach there separately. + it("applies the setting to a date literal inside a formula", () => { + assert.equal(formulaResult(['=DAY("03/04/2025")'], false), 4); + assert.equal(formulaResult(['=DAY("03/04/2025")'], true), 3); + }); + + // Arithmetic takes a third route: quoted dates are substituted into the + // expression before it is computed. The sign flip makes the difference + // unmissable — the same subtraction is 30 days one way and -30 the other. + it("applies the setting to a date literal in an arithmetic expression", () => { + assert.equal(formulaResult(["04/03/2025", '=A1-"03/04/2025"'], false), 30, "Apr 3 minus Mar 4"); + assert.equal(formulaResult(["04/03/2025", '=A1-"03/04/2025"'], true), -30, "Mar 4 minus Apr 3"); + }); + + // Cross-sheet references are NOT covered here: `=Data!A1` currently returns + // 3 for a date cell on main, independent of this setting, because the + // cross-sheet path never sees the date preprocessing (#2332). Adding the + // coverage belongs with that fix, not here. +}); diff --git a/tests/engine/test_cellBuilder.ts b/tests/engine/test_cellBuilder.ts new file mode 100644 index 0000000..3758551 --- /dev/null +++ b/tests/engine/test_cellBuilder.ts @@ -0,0 +1,130 @@ +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { buildCellFromInput, looksLikeFormula, parseNonStringInput } from "../../src/engine/cellBuilder.js"; + +describe("looksLikeFormula", () => { + it("detects function calls at the start", () => { + assert.equal(looksLikeFormula("SUM(A1:A3)"), true); + assert.equal(looksLikeFormula("MAX(B1, C1)"), true); + assert.equal(looksLikeFormula("-IF(A1>0, 1, 0)"), true); + }); + + it("detects cell reference + operator", () => { + assert.equal(looksLikeFormula("A1+B1"), true); + assert.equal(looksLikeFormula("AA10 * 2"), true); + assert.equal(looksLikeFormula("A1/B1"), true); + }); + + it("detects numeric arithmetic", () => { + assert.equal(looksLikeFormula("6/100"), true); + assert.equal(looksLikeFormula("5 * 2"), true); + assert.equal(looksLikeFormula("3+4"), true); + }); + + it("rejects plain text", () => { + assert.equal(looksLikeFormula("hello world"), false); + assert.equal(looksLikeFormula("apple pie"), false); + assert.equal(looksLikeFormula(""), false); + }); + + it("rejects bare numbers", () => { + assert.equal(looksLikeFormula("42"), false); + assert.equal(looksLikeFormula("3.14"), false); + }); + + it("rejects bare cell refs (no operator)", () => { + assert.equal(looksLikeFormula("A1"), false); + assert.equal(looksLikeFormula("AA10"), false); + }); +}); + +describe("parseNonStringInput", () => { + it("empty input → empty string", () => { + assert.equal(parseNonStringInput(""), ""); + assert.equal(parseNonStringInput(" "), ""); + }); + + it("formula → prefixed with =", () => { + assert.equal(parseNonStringInput("SUM(A1:A3)"), "=SUM(A1:A3)"); + assert.equal(parseNonStringInput("A1+B1"), "=A1+B1"); + assert.equal(parseNonStringInput(" 6/100 "), "=6/100"); + }); + + it("numeric → parsed as number", () => { + assert.equal(parseNonStringInput("42"), 42); + assert.equal(parseNonStringInput("3.14"), 3.14); + assert.equal(parseNonStringInput("-5"), -5); + }); + + it("non-formula non-number → raw string", () => { + assert.equal(parseNonStringInput("hello"), "hello"); + assert.equal(parseNonStringInput("yes"), "yes"); + }); + + it("rejects trailing garbage that parseFloat would silently accept", () => { + // parseFloat("42abc") returns 42; we want the string preserved. + assert.equal(parseNonStringInput("42abc"), "42abc"); + assert.equal(parseNonStringInput("100 USD"), "100 USD"); + assert.equal(parseNonStringInput("3.14xyz"), "3.14xyz"); + }); + + it("accepts scientific notation", () => { + assert.equal(parseNonStringInput("1e3"), 1000); + assert.equal(parseNonStringInput("-2.5E-2"), -0.025); + }); +}); + +describe("buildCellFromInput", () => { + it('type "string" → v is coerced String', () => { + const cell = buildCellFromInput({ type: "string", value: 42 }); + assert.deepEqual(cell, { v: "42" }); + }); + + it('type "string" with null value', () => { + const cell = buildCellFromInput({ type: "string", value: null }); + assert.deepEqual(cell, { v: "null" }); + }); + + it('type "number" with numeric formula', () => { + const cell = buildCellFromInput({ + type: "number", + value: "", + formula: "42", + }); + assert.deepEqual(cell, { v: 42 }); + }); + + it("type object with formula input", () => { + const cell = buildCellFromInput({ + type: "formula", + value: "", + formula: "SUM(A1:A3)", + }); + assert.deepEqual(cell, { v: "=SUM(A1:A3)" }); + }); + + it("attaches format when provided", () => { + const cell = buildCellFromInput({ + type: "number", + value: "", + formula: "100", + format: "$#,##0.00", + }); + assert.deepEqual(cell, { v: 100, f: "$#,##0.00" }); + }); + + it("omits format when empty string", () => { + const cell = buildCellFromInput({ + type: "number", + value: "", + formula: "100", + format: "", + }); + assert.deepEqual(cell, { v: 100 }); + }); + + it("empty formula input → empty string value", () => { + const cell = buildCellFromInput({ type: "number", value: "", formula: "" }); + assert.deepEqual(cell, { v: "" }); + }); +}); diff --git a/tests/engine/test_cellEmpty.ts b/tests/engine/test_cellEmpty.ts new file mode 100644 index 0000000..cc68592 --- /dev/null +++ b/tests/engine/test_cellEmpty.ts @@ -0,0 +1,123 @@ +// Telling a blank cell apart from a stored 0. Get this wrong and an aggregate +// reports a plausible number computed over the wrong count — a blank counted as +// a value drags AVERAGE down and COUNT up, with nothing to show it happened. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { isEmptyCell } from "../../src/engine/cellEmpty.ts"; +import type { SpreadsheetCell } from "../../src/engine/types.ts"; +import { SpreadsheetEngine, type SheetData } from "../../src/engine/index.ts"; +import { cellAt } from "./cellAccess.ts"; + +describe("isEmptyCell — empty", () => { + it("treats an absent cell as empty", () => { + assert.equal(isEmptyCell(null), true); + assert.equal(isEmptyCell(undefined), true); + }); + + it("treats an object with no value as empty", () => { + assert.equal(isEmptyCell({}), true); + assert.equal(isEmptyCell({ f: "0.00" }), true, "a format without a value is still empty"); + }); + + it("treats a null or undefined stored value as empty", () => { + assert.equal(isEmptyCell({ v: null }), true); + assert.equal(isEmptyCell({ v: undefined }), true); + }); + + it("treats an empty or whitespace string as empty, bare or wrapped", () => { + assert.equal(isEmptyCell(""), true); + assert.equal(isEmptyCell(" "), true); + assert.equal(isEmptyCell({ v: "" }), true); + assert.equal(isEmptyCell({ v: " " }), true); + }); +}); + +describe("isEmptyCell — not empty", () => { + // The distinction the whole module exists for: a stored 0 is a value. + it("treats a stored zero as a value", () => { + assert.equal(isEmptyCell(0), false); + assert.equal(isEmptyCell({ v: 0 }), false); + }); + + it("treats false as a value", () => { + assert.equal(isEmptyCell(false), false); + assert.equal(isEmptyCell({ v: false }), false); + }); + + it("treats any non-empty text as a value", () => { + assert.equal(isEmptyCell("x"), false); + assert.equal(isEmptyCell({ v: "hello" }), false); + assert.equal(isEmptyCell({ v: "0" }), false, "a zero written as text is still a value"); + }); + + it("treats a number as a value", () => { + assert.equal(isEmptyCell(42), false); + assert.equal(isEmptyCell({ v: 42 }), false); + assert.equal(isEmptyCell(-1), false); + }); +}); + +// The same distinction driven through the engine: an aggregate over a range +// with blanks must count only the real values. +describe("blank cells are not values in an aggregate (#2358)", () => { + // 10, 20, 30 followed by two blanks. Excel divides by 3 and counts 3. + const withBlanks = (formula: string): SheetData => ({ + name: "S", + data: [[{ v: 10 }, { v: formula }], [{ v: 20 }], [{ v: 30 }], [{ v: "" }], [{ v: null } as unknown as SpreadsheetCell]], + }); + const run = (formula: string): unknown => cellAt(new SpreadsheetEngine().calculate(withBlanks(formula)).data, 0, 1); + + it("excludes blanks from AVERAGE's denominator", () => { + assert.equal(run("=AVERAGE(A1:A5)"), 20, "not 15, which counts the two blanks as 0"); + }); + + it("excludes blanks from COUNT", () => { + assert.equal(run("=COUNT(A1:A5)"), 3, "not 4"); + }); + + // A blank would have read as 0, and 0 does not change a sum — so SUM is the + // one aggregate the old behaviour got right, and it must stay right. + it("leaves SUM unchanged", () => { + assert.equal(run("=SUM(A1:A5)"), 60); + }); + + it("does not disturb MAX or MIN", () => { + assert.equal(run("=MAX(A1:A5)"), 30); + assert.equal(run("=MIN(A1:A5)"), 10); + }); + + // The line the fix walks: a stored 0 is a value and must still count, even + // though a blank does not. + it("still counts a stored zero", () => { + const sheet: SheetData = { name: "S", data: [[{ v: 10 }, { v: "=COUNT(A1:A3)" }], [{ v: 0 }], [{ v: 20 }]] }; + assert.equal(cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 1), 3); + }); + + it("averages a stored zero in, but not a blank", () => { + const sheet: SheetData = { name: "S", data: [[{ v: 6 }, { v: "=AVERAGE(A1:A3)" }], [{ v: 0 }], [{ v: "" }]] }; + assert.equal(cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 1), 3, "(6 + 0) / 2, the blank excluded"); + }); +}); + +describe("blanks stay in the raw range so criteria and values stay aligned", () => { + // SUMIF reads the criteria range and the sum range separately. Dropping + // blanks from the raw list would compact each independently and shift the + // rows out of alignment, aggregating the wrong values (Codex review on + // #2383). A blank in the criteria column must NOT desync the two ranges. + it("keeps SUMIF row-aligned when a criteria cell is blank", () => { + const sheet: SheetData = { + name: "S", + data: [ + [{ v: 10 }, { v: 100 }, { v: '=SUMIF(A1:A4,">5",B1:B4)' }], + [{ v: "" }, { v: 200 }], + [{ v: 20 }, { v: 300 }], + [{ v: 30 }, { v: 400 }], + ], + }; + // A1=10, A3=20, A4=30 are >5; their B values are 100, 300, 400 → 800. + // If the blank A2 shifted the value range, B would misalign and the sum + // would be wrong. + assert.equal(cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 2), 800); + }); +}); diff --git a/tests/engine/test_cellFormatting.ts b/tests/engine/test_cellFormatting.ts new file mode 100644 index 0000000..387252d --- /dev/null +++ b/tests/engine/test_cellFormatting.ts @@ -0,0 +1,73 @@ +// The display-formatting decision for a single cell. It runs only on the final +// output pass — cross-sheet reference resolution deliberately skips it — so a +// wrong branch here either hides a date or, worse, turns a raw serial into a +// "03/04/2025" string that a downstream parseFloat reads as 3 (issue #2332). + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { formatCellForDisplay, isLikelyDateSerial } from "../../src/engine/cellFormatting.ts"; +import { dateToSerial } from "../../src/engine/date-utils.ts"; + +const serial2025Mar4 = dateToSerial(new Date(Date.UTC(2025, 2, 4))); + +describe("isLikelyDateSerial", () => { + it("accepts integers inside the date-serial window", () => { + assert.equal(isLikelyDateSerial(serial2025Mar4), true); + }); + + it("accepts the exact window boundaries", () => { + assert.equal(isLikelyDateSerial(36000), true); + assert.equal(isLikelyDateSerial(63499), true); + }); + + it("rejects values just outside the window", () => { + assert.equal(isLikelyDateSerial(35999), false); + assert.equal(isLikelyDateSerial(63500), false); + }); + + it("rejects non-integers (a time component is not a bare date)", () => { + assert.equal(isLikelyDateSerial(45720.5), false); + }); + + it("rejects non-numbers", () => { + assert.equal(isLikelyDateSerial("45720" as unknown as number), false); + assert.equal(isLikelyDateSerial(true as unknown as number), false); + }); +}); + +describe("formatCellForDisplay — passthrough", () => { + it("returns the value unchanged when the original is not a cell", () => { + assert.equal(formatCellForDisplay(5, 5, false), 5); + assert.equal(formatCellForDisplay(null, 7, false), 7); + }); + + it("leaves text untouched", () => { + assert.equal(formatCellForDisplay({ v: "hello" }, "hello", false), "hello"); + }); + + it("leaves a plain formula number that is not a date serial", () => { + assert.equal(formatCellForDisplay({ v: "=A1+A2" }, 100, false), 100); + }); + + it("leaves an empty cell's zero as a number", () => { + assert.equal(formatCellForDisplay({ v: "" }, 0, false), 0); + }); +}); + +describe("formatCellForDisplay — formatting", () => { + it("applies an explicit currency format", () => { + assert.equal(formatCellForDisplay({ v: 1234.5, f: "$#,##0.00" }, 1234.5, false), "$1,234.50"); + }); + + it("auto-formats a formula's date serial (month-first by default)", () => { + assert.equal(formatCellForDisplay({ v: "=A1" }, serial2025Mar4, false), "03/04/2025"); + }); + + it("honours day-first preference for the auto date format", () => { + assert.equal(formatCellForDisplay({ v: "=A1" }, serial2025Mar4, true), "04/03/2025"); + }); + + it("does NOT auto-format a non-formula date serial (only formulas opt in)", () => { + assert.equal(formatCellForDisplay({ v: serial2025Mar4 }, serial2025Mar4, false), serial2025Mar4); + }); +}); diff --git a/tests/engine/test_cellRefSubstitution.ts b/tests/engine/test_cellRefSubstitution.ts new file mode 100644 index 0000000..c11f77a --- /dev/null +++ b/tests/engine/test_cellRefSubstitution.ts @@ -0,0 +1,179 @@ +// Substituting cell values into a formula. The failure this covers produced a +// NUMBER — `=A1+A10` came back 55 instead of 12 — so there was nothing in the +// sheet to suggest anything had gone wrong. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { SpreadsheetEngine, renderOperand, findCellRefs, endOfStringLiteral, type SheetData } from "../../src/engine/index.ts"; +import { cellAt, rowAt } from "./cellAccess.ts"; + +/** A single column of values, with `formula` in the cell beside the first. */ +function columnSheet(values: (string | number)[], formula: string): SheetData { + return { name: "S", data: values.map((value, index) => (index === 0 ? [{ v: value }, { v: formula }] : [{ v: value }])) }; +} + +const evaluate = (sheet: SheetData): unknown => cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 1); + +describe("cell reference substitution — prefix collisions", () => { + // A global string replace rewrote every occurrence of the shorter reference + // first, turning `A10` into `0`: 5 and 7 became "5+50" = 55. + it("does not let A1 rewrite A10", () => { + const values = [5, 0, 0, 0, 0, 0, 0, 0, 0, 7]; + assert.equal(evaluate(columnSheet(values, "=A1+A10")), 12); + }); + + it("does not let A1 rewrite A11 or A100", () => { + const values = [3, 0, 0, 0, 0, 0, 0, 0, 0, 0, 4]; + assert.equal(evaluate(columnSheet(values, "=A1+A11")), 7); + }); + + it("keeps the order of a reference used twice", () => { + const values = [2, 0, 0, 0, 0, 0, 0, 0, 0, 9]; + assert.equal(evaluate(columnSheet(values, "=A10-A1")), 7); + }); + + // The column letters collide the same way: `B1` is a prefix of `AB1` only in + // the substring sense, and the old replace did not care about boundaries. + it("does not let B1 rewrite AB1", () => { + const sheet: SheetData = { name: "S", data: [[{ v: 0 }, { v: 2 }, { v: "=B1+AB1" }]] }; + assert.equal(cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 2), 2, "AB1 is empty, so the sum is B1 alone"); + }); + + it("handles several colliding references in one formula", () => { + const values = [1, 2, 0, 0, 0, 0, 0, 0, 0, 10, 11]; + assert.equal(evaluate(columnSheet(values, "=A1+A2+A10+A11")), 24); + }); +}); + +describe("a lone reference returns the cell value unchanged", () => { + // `=A1` is not an expression to substitute into — it IS the cell. Rendering + // the value into expression text first would escape a string's quotes and + // backslashes, and those escapes would survive into the result (Codex + // review): `=A1` on `say "hi"` came back `say \"hi\"`. + it("returns text with quotes and backslashes intact", () => { + const sheet: SheetData = { name: "S", data: [[{ v: 'say "hi"' }, { v: "=A1" }, { v: "=$A$1" }]] }; + const row = rowAt(new SpreadsheetEngine().calculate(sheet).data, 0); + assert.equal(row[1], 'say "hi"'); + assert.equal(row[2], 'say "hi"', "absolute form too"); + }); + + // `= A1` and `=A1 ` are still nothing but one reference; the whitespace must + // not push them onto the substitution path, where the text would be escaped + // and the escapes kept (Codex review). + it("takes the fast path despite surrounding whitespace", () => { + for (const formula of ["= A1", "=A1 ", "= A1 "]) { + const sheet: SheetData = { name: "S", data: [[{ v: 'say "hi"' }, { v: formula }]] }; + assert.equal(cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 1), 'say "hi"', `${formula} should return the value verbatim`); + } + }); + + it("still substitutes when whitespace surrounds a reference inside an expression", () => { + const sheet: SheetData = { name: "S", data: [[{ v: 3 }, { v: "= A1 + 1" }]] }; + assert.equal(cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 1), 4); + }); + + it("returns a backslash-bearing string intact", () => { + const sheet: SheetData = { name: "S", data: [[{ v: "C:\\path" }, { v: "=A1" }]] }; + assert.equal(cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 1), "C:\\path"); + }); + + it("returns a number, not its string form", () => { + const sheet: SheetData = { name: "S", data: [[{ v: 42 }, { v: "=A1" }]] }; + assert.equal(cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 1), 42); + }); + + // The fast path is only for a formula that is EXACTLY one reference; the + // moment it is part of an expression the substitution path takes over. + it("does not take the fast path when the reference is part of an expression", () => { + const sheet: SheetData = { name: "S", data: [[{ v: 3 }, { v: "=A1+1" }]] }; + assert.equal(cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 1), 4); + }); +}); + +describe("a reference inside a string literal is a constant, not a reference", () => { + // `"A1"` is text the user typed, not the cell A1. The substitution used to + // scan inside literals, so `="A1"&"!"` returned A1's value (`5!`) instead of + // the constant (`A1!`) (Codex review). + it("does not substitute a double-quoted ref-looking constant", () => { + assert.equal(evaluate(columnSheet([5], '="A1"&"!"')), "A1!"); + }); + + it("substitutes a real reference beside a literal that looks like one", () => { + assert.equal(evaluate(columnSheet([7], '=A1&"B2"')), "7B2"); + }); + + // A single-quoted span that is NOT `'Sheet'!cell` is a string literal too, so + // its contents must not be read as references. + it("does not substitute a single-quoted ref-looking constant", () => { + assert.equal(evaluate(columnSheet([5], "='B2'&\"!\"")), "B2!"); + }); +}); + +describe("findCellRefs", () => { + it("finds every reference in an arithmetic expression", () => { + assert.deepEqual(findCellRefs("A1+A10"), [ + { ref: "A1", start: 0 }, + { ref: "A10", start: 3 }, + ]); + }); + + it("skips references inside a double-quoted literal", () => { + assert.deepEqual(findCellRefs('"A1"&"!"'), []); + assert.deepEqual(findCellRefs('A1&"B2"'), [{ ref: "A1", start: 0 }]); + }); + + it("keeps a quoted sheet reference but skips a plain single-quoted literal", () => { + assert.deepEqual(findCellRefs("'Sheet1'!A1"), [{ ref: "'Sheet1'!A1", start: 0 }]); + assert.deepEqual(findCellRefs("'B2'&\"!\""), []); + }); + + // An escaped quote must not end the literal early, or the tail would be + // scanned for references. + it("honours backslash escapes inside a literal", () => { + assert.deepEqual(findCellRefs('"say \\"A1\\""'), []); + }); +}); + +describe("endOfStringLiteral", () => { + it("returns the index just past the closing quote", () => { + assert.equal(endOfStringLiteral('"ab"cd', 0), 4); + }); + + it("does not close on an escaped quote", () => { + assert.equal(endOfStringLiteral('"a\\"b"x', 0), 6); + }); + + it("returns the length when the literal is never closed", () => { + assert.equal(endOfStringLiteral('"abc', 0), 4); + }); +}); + +describe("renderOperand", () => { + it("renders numbers and booleans verbatim", () => { + assert.equal(renderOperand(42), "42"); + assert.equal(renderOperand(-3.5), "-3.5"); + assert.equal(renderOperand(0), "0"); + assert.equal(renderOperand(true), "true"); + }); + + // Quoting is what keeps a text cell from being read as an identifier or an + // operator once it lands in the expression. + it("quotes strings", () => { + assert.equal(renderOperand("hello"), '"hello"'); + assert.equal(renderOperand(""), '""'); + }); + + // Without escaping, a cell containing a quote closes the literal early and + // the rest of its text becomes expression source. + it("escapes quotes and backslashes so the literal cannot be closed early", () => { + assert.equal(renderOperand('say "hi"'), '"say \\"hi\\""'); + assert.equal(renderOperand("back\\slash"), '"back\\\\slash"'); + assert.equal(renderOperand('"'), '"\\""'); + }); + + // Blanks are 0 here, as they are everywhere else in the engine. + it("renders a missing value as 0", () => { + assert.equal(renderOperand(null), "0"); + assert.equal(renderOperand(undefined), "0"); + }); +}); diff --git a/tests/engine/test_concatSafety.ts b/tests/engine/test_concatSafety.ts new file mode 100644 index 0000000..3e2f8d7 --- /dev/null +++ b/tests/engine/test_concatSafety.ts @@ -0,0 +1,109 @@ +// Deciding whether a string-concatenation expression is safe to evaluate. The +// bug this guards: the old check ran a character allowlist over the WHOLE +// expression, including the content of string literals — so a `!` inside a +// string, or the `\` an escaped operand produces, made a valid formula look +// unsafe and it was returned as raw text (#2376). Masking the literals first +// validates the structure without judging the content. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { maskStringLiterals, isSafeConcatExpression, SpreadsheetEngine, type SheetData } from "../../src/engine/index.ts"; +import { cellAt } from "./cellAccess.ts"; + +describe("maskStringLiterals", () => { + it("empties a double- or single-quoted literal, keeping the quotes", () => { + assert.equal(maskStringLiterals('"hello"'), '""'); + assert.equal(maskStringLiterals("'world'"), "''"); + }); + + it("keeps the structure around the literals", () => { + assert.equal(maskStringLiterals('"a"+"b"'), '""+""'); + assert.equal(maskStringLiterals('5+"x"'), '5+""'); + }); + + // The content is what must not leak into the structure check — punctuation, + // operators, whatever. + it("removes arbitrary content, including operators and punctuation", () => { + assert.equal(maskStringLiterals('"a>b!c"'), '""'); + assert.equal(maskStringLiterals('"1+2"+3'), '""+3'); + }); + + // An escaped quote does not end the literal, so its content — and the + // backslash — is masked away rather than leaking a stray quote. + it("honours backslash escapes inside a literal", () => { + assert.equal(maskStringLiterals('"a\\"b"'), '""'); + assert.equal(maskStringLiterals('"back\\\\slash"'), '""'); + }); + + it("leaves an expression with no literals unchanged", () => { + assert.equal(maskStringLiterals("1+2+3"), "1+2+3"); + assert.equal(maskStringLiterals(""), ""); + }); +}); + +describe("isSafeConcatExpression", () => { + it("accepts joined string literals", () => { + assert.equal(isSafeConcatExpression('"a"+"b"'), true); + assert.equal(isSafeConcatExpression('"hi"+"!"'), true, "a bang inside a string is content, not structure"); + }); + + it("accepts a literal carrying escapes and arbitrary characters", () => { + assert.equal(isSafeConcatExpression('"say \\"hi\\""+"!"'), true); + assert.equal(isSafeConcatExpression('"a\\\\b"+"c"'), true); + assert.equal(isSafeConcatExpression('"日本語"+"!"'), true); + }); + + it("accepts numbers and parentheses joining strings", () => { + assert.equal(isSafeConcatExpression('5+"x"'), true); + assert.equal(isSafeConcatExpression('("a")+("b")'), true); + }); + + // Once the literals are masked, an identifier in the STRUCTURE is not + // allowed — that would be an unresolved reference or injected code. + it("rejects an unresolved identifier in the structure", () => { + assert.equal(isSafeConcatExpression('foo+"a"'), false); + assert.equal(isSafeConcatExpression('"a"+process'), false); + }); + + // A boolean cell renders as a bare `true` / `false`; those two words must + // pass or a boolean operand's concat is returned as raw text (Codex review). + // Any other identifier — even one containing them as a substring — is still + // rejected, so the exemption cannot smuggle code into `new Function`. + it("accepts the boolean operand words but nothing else", () => { + assert.equal(isSafeConcatExpression('true+"!"'), true); + assert.equal(isSafeConcatExpression('false+"!"'), true); + assert.equal(isSafeConcatExpression('truthy+"!"'), false); + }); +}); + +describe("string concatenation through the engine (#2376)", () => { + const concat = (cellValue: string): unknown => + cellAt(new SpreadsheetEngine().calculate({ name: "S", data: [[{ v: cellValue }, { v: '=A1&"!"' }]] } satisfies SheetData).data, 0, 1); + + // The plain case that was already broken: a `!` in the appended string made + // the whole concat fail the allowlist and return the raw formula text. + it("appends a literal to a plain string", () => { + assert.equal(concat("hi"), "hi!"); + }); + + // The #2376 blocker: an escaped operand must survive the concat path. + it("appends to a string containing a quote", () => { + assert.equal(concat('say "hi"'), 'say "hi"!'); + }); + + it("appends to a string containing a backslash", () => { + assert.equal(concat("a\\b"), "a\\b!"); + }); + + it("appends to a numeric string", () => { + assert.equal(concat("5"), "5!"); + }); + + // A boolean operand (here A1 is the comparison `=1=1`) renders as `true`, + // which the stricter safety gate used to reject — the concat came back as the + // raw formula text instead of the joined value (Codex review). + it("appends a literal to a boolean cell value", () => { + const sheet: SheetData = { name: "S", data: [[{ v: "=1=1" }, { v: '=A1&"!"' }]] }; + assert.equal(cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 1), "true!"); + }); +}); diff --git a/tests/engine/test_condition.ts b/tests/engine/test_condition.ts new file mode 100644 index 0000000..7576645 --- /dev/null +++ b/tests/engine/test_condition.ts @@ -0,0 +1,292 @@ +// Reading a spreadsheet condition without running it. The grammar is one +// comparison or a bare value, and that narrowness is the safety property — +// this replaced an `eval` that executed whatever a cell happened to contain. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { + evaluateCondition, + isTruthyCondition, + readOperand, + renderConditionOperand, + splitComparison, + stripOuterParens, +} from "../../src/engine/condition.ts"; + +describe("splitComparison", () => { + it("splits on each operator", () => { + assert.deepEqual(splitComparison("A>1"), { left: "A", operator: ">", right: "1" }); + assert.deepEqual(splitComparison("A<1"), { left: "A", operator: "<", right: "1" }); + assert.deepEqual(splitComparison("A=1"), { left: "A", operator: "=", right: "1" }); + }); + + // Longest-first matching: a `>` that starts `>=` must not win. + it("prefers the two-character operators", () => { + assert.deepEqual(splitComparison("A>=1"), { left: "A", operator: ">=", right: "1" }); + assert.deepEqual(splitComparison("A<=1"), { left: "A", operator: "<=", right: "1" }); + assert.deepEqual(splitComparison("A<>1"), { left: "A", operator: "<>", right: "1" }); + assert.deepEqual(splitComparison("A!=1"), { left: "A", operator: "!=", right: "1" }); + assert.deepEqual(splitComparison("A==1"), { left: "A", operator: "==", right: "1" }); + }); + + it("trims both sides", () => { + assert.deepEqual(splitComparison(" A > 1 "), { left: "A", operator: ">", right: "1" }); + }); + + it("returns null when there is no comparison", () => { + assert.equal(splitComparison("A1"), null); + assert.equal(splitComparison("42"), null); + assert.equal(splitComparison(""), null); + }); + + // Only the first operator counts, so `a=b=c` is one comparison against the + // text `b=c` rather than a chain. A chain is what used to reach a JS parser. + it("takes only the first operator", () => { + assert.deepEqual(splitComparison("1=1=1"), { left: "1", operator: "=", right: "1=1" }); + }); + + // A blank cell substitutes to nothing, leaving the operator at position 0. + // That is a comparison against an empty left side, not a bare value — reading + // it as text made `IFS` pick branches for empty cells (Codex review). + it("treats a leading operator as a comparison with an empty left side", () => { + assert.deepEqual(splitComparison(">5"), { left: "", operator: ">", right: "5" }); + assert.deepEqual(splitComparison(">=1"), { left: "", operator: ">=", right: "1" }); + }); + + // A cell holding `a>b` substitutes as the literal `"a>b"`. Splitting on that + // `>` would compare two fragments of one string (Codex review). The same must + // hold for `<` and `=` inside the operand, not only `>`. + it("ignores operators inside quoted text", () => { + assert.deepEqual(splitComparison('"a>b"="a>b"'), { left: '"a>b"', operator: "=", right: '"a>b"' }); + assert.deepEqual(splitComparison('"ab"' }); + assert.deepEqual(splitComparison('A1="a { + assert.deepEqual(splitComparison('"a>b" = "c"'), { left: '"a>b"', operator: "=", right: '"c"' }); + }); +}); + +describe("stripOuterParens", () => { + // `IFS((A1>0), ...)` is valid, and the parser used to split it into `(1` and + // `0)` — two operands that compare as text and never match (Codex review). + it("removes parentheses that wrap the whole expression", () => { + assert.equal(stripOuterParens("(1>0)"), "1>0"); + assert.equal(stripOuterParens("((1>0))"), "1>0"); + assert.equal(stripOuterParens(" ( 1>0 ) "), "1>0"); + }); + + // The leading `(` closes before the end, so it wraps only its own operand. + // Removing the outer characters here would corrupt the expression into + // `A)=(B`. + it("keeps parentheses that wrap only part of the expression", () => { + assert.equal(stripOuterParens("(A)=(B)"), "(A)=(B)"); + assert.equal(stripOuterParens("(1)>(0)"), "(1)>(0)"); + }); + + it("leaves unbalanced input untouched rather than guessing", () => { + assert.equal(stripOuterParens("(1>0"), "(1>0"); + assert.equal(stripOuterParens("("), "("); + assert.equal(stripOuterParens(")("), ")("); + }); + + it("ignores parentheses inside quoted text", () => { + assert.equal(stripOuterParens('("a)b")'), '"a)b"'); + }); + + it("leaves an expression with no outer parentheses alone", () => { + assert.equal(stripOuterParens("1>0"), "1>0"); + assert.equal(stripOuterParens(""), ""); + }); +}); + +describe("readOperand", () => { + it("reads numbers", () => { + assert.equal(readOperand("42"), 42); + assert.equal(readOperand("-3.5"), -3.5); + assert.equal(readOperand("0"), 0); + }); + + it("reads booleans case-insensitively", () => { + assert.equal(readOperand("TRUE"), true); + assert.equal(readOperand("true"), true); + assert.equal(readOperand("FALSE"), false); + }); + + it("keeps quoted text as text, quotes removed", () => { + assert.equal(readOperand('"hello"'), "hello"); + assert.equal(readOperand("'hello'"), "hello"); + assert.equal(readOperand('"42"'), "42", "quoted digits stay text"); + }); + + // `Number` rather than `parseFloat`: trailing garbage makes the whole thing + // text instead of silently contributing its numeric prefix. + it("does not take a numeric prefix from mixed text", () => { + assert.equal(readOperand("12abc"), "12abc"); + assert.equal(readOperand("3.5kg"), "3.5kg"); + }); + + it("keeps unquoted text as text", () => { + assert.equal(readOperand("hello"), "hello"); + assert.equal(readOperand(""), ""); + }); +}); + +describe("evaluateCondition — comparisons", () => { + it("compares numbers", () => { + assert.equal(evaluateCondition("5>3"), true); + assert.equal(evaluateCondition("3>5"), false); + assert.equal(evaluateCondition("5>=5"), true, "the boundary counts for >="); + assert.equal(evaluateCondition("5>5"), false); + assert.equal(evaluateCondition("3<=3"), true); + }); + + it("compares for equality and inequality", () => { + assert.equal(evaluateCondition("5=5"), true); + assert.equal(evaluateCondition("5==5"), true); + assert.equal(evaluateCondition("5<>3"), true); + assert.equal(evaluateCondition("5!=5"), false); + }); + + // The regression the quote-aware scan exists for: both sides are one string + // each, so this is equality between them, not a comparison of fragments. + it("evaluates a parenthesised comparison the same as a bare one", () => { + assert.equal(evaluateCondition("(1>0)"), true); + assert.equal(evaluateCondition("((5>=5))"), true); + assert.equal(evaluateCondition("(3>5)"), false); + assert.equal(evaluateCondition('("a"="a")'), true); + }); + + it("evaluates a parenthesised bare value", () => { + assert.equal(evaluateCondition("(1)"), true); + assert.equal(evaluateCondition("(0)"), false); + }); + + it("compares text containing operator characters", () => { + assert.equal(evaluateCondition('"a>b"="a>b"'), true); + assert.equal(evaluateCondition('"a>b"="a>c"'), false); + assert.equal(evaluateCondition('"a { + assert.equal(evaluateCondition(">5"), false, "blank is not greater than 5"); + assert.equal(evaluateCondition(">=1"), false); + assert.equal(evaluateCondition("<>5"), true, "blank does differ from 5"); + assert.equal(evaluateCondition("=5"), false); + assert.equal(evaluateCondition('=""'), true, "blank equals blank"); + }); + + it("compares text", () => { + assert.equal(evaluateCondition('"abc"="abc"'), true); + assert.equal(evaluateCondition('"abc"="abd"'), false); + assert.equal(evaluateCondition('"abc"<"abd"'), true); + }); + + // A quoted number and a bare one are different types, so equality separates + // them — the same rule the rest of the engine follows. + it("distinguishes a quoted number from a bare one", () => { + assert.equal(evaluateCondition('42="42"'), false); + }); + + it("compares booleans for equality but refuses to order them", () => { + assert.equal(evaluateCondition("TRUE=TRUE"), true); + assert.equal(evaluateCondition("TRUE<>FALSE"), true); + assert.equal(evaluateCondition("TRUE>FALSE"), false, "no ordering is defined"); + }); +}); + +describe("evaluateCondition — bare values", () => { + // Spreadsheet truthiness, not JavaScript's: 0 and empty are false. + it("treats zero and empty as false, other values as true", () => { + assert.equal(evaluateCondition("0"), false); + assert.equal(evaluateCondition(""), false); + assert.equal(evaluateCondition('""'), false); + assert.equal(evaluateCondition("1"), true); + assert.equal(evaluateCondition("-1"), true, "a negative number is still a value"); + assert.equal(evaluateCondition("hello"), true); + }); + + it("reads bare booleans", () => { + assert.equal(isTruthyCondition("TRUE"), true); + assert.equal(isTruthyCondition("FALSE"), false); + }); +}); + +describe("evaluateCondition — code is data", () => { + // The point of the module. Each of these used to execute (#2360): the first + // two as a cell's substituted value, the third as text written straight into + // the formula. They must now be read as operands and nothing more. + it("does not execute an assignment", () => { + const marker = globalThis as Record; + marker.__conditionProbe = false; + assert.equal(evaluateCondition("globalThis.__conditionProbe=true"), false, "an assignment is text, and text is not a comparison match"); + assert.equal(marker.__conditionProbe, false, "nothing ran"); + }); + + it("does not execute a call or a sequence", () => { + const marker = globalThis as Record; + marker.__conditionProbe2 = false; + evaluateCondition("(globalThis.__conditionProbe2=true, 1)>0"); + assert.equal(marker.__conditionProbe2, false); + }); + + it("does not honour a logical operator smuggled into the condition", () => { + const marker = globalThis as Record; + marker.__conditionProbe3 = false; + evaluateCondition("1>0&&(globalThis.__conditionProbe3=true)"); + assert.equal(marker.__conditionProbe3, false); + }); + + it("never throws on syntactically broken input", () => { + for (const input of ["((((", '"unclosed', "1+", "}{", "throw 1"]) { + assert.equal(typeof evaluateCondition(input), "boolean", `${input} should still yield a boolean`); + } + }); +}); + +describe("renderConditionOperand", () => { + it("renders numbers and booleans as themselves", () => { + assert.equal(renderConditionOperand(42), "42"); + assert.equal(renderConditionOperand(0), "0"); + assert.equal(renderConditionOperand(true), "true"); + }); + + // A text cell must arrive as a quoted literal, so its own contents cannot be + // read as operators: `x>y` unquoted would make `A1="x>y"` parse as a + // comparison of fragments. + it("quotes strings", () => { + assert.equal(renderConditionOperand("x>y"), '"x>y"'); + assert.equal(renderConditionOperand("Yes"), '"Yes"'); + assert.equal(renderConditionOperand(""), '""'); + }); + + it("escapes quotes and backslashes so the literal cannot be closed early", () => { + assert.equal(renderConditionOperand('a"b'), '"a\\"b"'); + assert.equal(renderConditionOperand("a\\b"), '"a\\\\b"'); + }); + + it("renders a missing value as an empty quoted string", () => { + assert.equal(renderConditionOperand(null), '""'); + assert.equal(renderConditionOperand(undefined), '""'); + }); + + // Round-trip: whatever it renders, evaluateCondition reads back as the same + // value, so a quoted text operand compares equal to itself. + it("round-trips through evaluateCondition", () => { + assert.equal(evaluateCondition(`${renderConditionOperand("x>y")}="x>y"`), true); + assert.equal(evaluateCondition(`${renderConditionOperand("x>y")}="other"`), false); + }); + + // A literal backslash must survive the escape-on-render / unescape-on-read + // round-trip exactly once: the earlier substitution path escaped it twice + // (CodeQL js/double-escaping), which corrupted the operand. + it("round-trips a value holding a backslash without double-escaping", () => { + assert.equal(readOperand(renderConditionOperand("a\\b")), "a\\b"); + assert.equal(evaluateCondition(`${renderConditionOperand("a\\b")}=${renderConditionOperand("a\\b")}`), true); + assert.equal(evaluateCondition(`${renderConditionOperand("a\\b")}=${renderConditionOperand("a/b")}`), false); + }); +}); diff --git a/tests/engine/test_conditionalAggregates.ts b/tests/engine/test_conditionalAggregates.ts new file mode 100644 index 0000000..ab78e7e --- /dev/null +++ b/tests/engine/test_conditionalAggregates.ts @@ -0,0 +1,47 @@ +// SUMIF / AVERAGEIF must pair the criteria range and the value range by +// POSITION. Reading the value range in numeric-only mode dropped blanks, which +// shifted its indexes out of step with the (raw) criteria range and pulled a +// later row's number into an earlier match (#2358 Codex review). + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { SpreadsheetEngine, type SheetData } from "../../src/engine/index.ts"; +import { cellAt } from "./cellAccess.ts"; + +// A = criteria column, B = value column (with a blank at row 2), formula in C1. +const sheet = (formula: string): SheetData => ({ + name: "S", + data: [ + [{ v: 1 }, { v: 100 }, { v: formula }], + [{ v: 0 }, { v: "" }], + [{ v: 1 }, { v: 300 }], + ], +}); + +const evalFormula = (formula: string): unknown => cellAt(new SpreadsheetEngine().calculate(sheet(formula)).data, 0, 2); + +describe("SUMIF / AVERAGEIF stay row-aligned when the value range has a blank", () => { + // Rows 1 and 3 match (A > 0); their B values are 100 and 300. The blank B2 + // belongs to the non-matching row 2 and must not slide up into row 3. + it("sums the value range by position, not by compacted index", () => { + assert.equal(evalFormula('=SUMIF(A1:A3, ">0", B1:B3)'), 400); + }); + + it("averages the matching rows' values by position", () => { + assert.equal(evalFormula('=AVERAGEIF(A1:A3, ">0", B1:B3)'), 200); + }); + + // A blank inside the matched rows counts as 0 in SUMIF (not skipped), matching + // Excel: here rows 1 and 3 match, B1 is blank, so the sum is just 300. + it("treats a blank in a matched value cell as 0", () => { + const withBlankMatch: SheetData = { + name: "S", + data: [ + [{ v: 1 }, { v: "" }, { v: '=SUMIF(A1:A3, ">0", B1:B3)' }], + [{ v: 0 }, { v: 999 }], + [{ v: 1 }, { v: 300 }], + ], + }; + assert.equal(cellAt(new SpreadsheetEngine().calculate(withBlankMatch).data, 0, 2), 300); + }); +}); diff --git a/tests/engine/test_criteria.ts b/tests/engine/test_criteria.ts new file mode 100644 index 0000000..7360f8a --- /dev/null +++ b/tests/engine/test_criteria.ts @@ -0,0 +1,90 @@ +// Criteria matching for COUNTIF / SUMIF / AVERAGEIF. Both bugs undercounted +// silently: text was compared with `===`, so `"yes"` skipped a cell holding +// `Yes`, and `"A*"` was matched literally instead of as a wildcard (#2360). + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { parseCriteria } from "../../src/engine/registry.ts"; +import { SpreadsheetEngine, type SheetData } from "../../src/engine/index.ts"; +import { cellAt } from "./cellAccess.ts"; + +describe("parseCriteria — text is matched case-insensitively", () => { + it("matches regardless of case", () => { + const matches = parseCriteria("yes"); + assert.equal(matches("Yes"), true); + assert.equal(matches("YES"), true); + assert.equal(matches("yes"), true); + }); + + it("still rejects different text", () => { + const matches = parseCriteria("yes"); + assert.equal(matches("no"), false); + assert.equal(matches("yesterday"), false, "an exact match, not a prefix"); + }); +}); + +describe("parseCriteria — wildcards", () => { + it("treats * as any run of characters", () => { + const matches = parseCriteria("A*"); + assert.equal(matches("Axle"), true); + assert.equal(matches("A"), true, "* may match nothing"); + assert.equal(matches("Bar"), false); + }); + + it("treats ? as exactly one character", () => { + const matches = parseCriteria("b?t"); + assert.equal(matches("bat"), true); + assert.equal(matches("bt"), false); + assert.equal(matches("beat"), false); + }); + + it("escapes a wildcard with ~", () => { + const matches = parseCriteria("A~*"); + assert.equal(matches("A*"), true); + assert.equal(matches("Axle"), false); + }); + + it("does not let regex metacharacters act as a pattern", () => { + const matches = parseCriteria("a.c"); + assert.equal(matches("a.c"), true); + assert.equal(matches("abc"), false, "the dot is literal, not any-char"); + }); +}); + +describe("parseCriteria — numbers and operators", () => { + it("matches a numeric criteria against a number", () => { + const matches = parseCriteria("5"); + assert.equal(matches(5), true); + assert.equal(matches("5"), true); + assert.equal(matches(6), false); + }); + + it("keeps the comparison operators working", () => { + assert.equal(parseCriteria(">3")(5), true); + assert.equal(parseCriteria(">3")(2), false); + assert.equal(parseCriteria("<=3")(3), true); + }); + + it("applies case-insensitive text to = and <>", () => { + assert.equal(parseCriteria("=yes")("Yes"), true); + assert.equal(parseCriteria("<>yes")("Yes"), false); + assert.equal(parseCriteria("<>yes")("no"), true); + }); +}); + +describe("COUNTIF through the engine", () => { + const countif = (values: string[], criteria: string): unknown => { + const rows = values.map((value) => [{ v: value }]); + rows.push([{ v: `=COUNTIF(A1:A${values.length}, "${criteria}")` }]); + const sheet: SheetData = { name: "S", data: rows }; + return cellAt(new SpreadsheetEngine().calculate(sheet).data, values.length, 0); + }; + + it("counts a case-differing match", () => { + assert.equal(countif(["Yes", "no"], "yes"), 1); + }); + + it("counts a wildcard match", () => { + assert.equal(countif(["Axle", "Bar"], "A*"), 1); + }); +}); diff --git a/tests/engine/test_crossSheetReference.ts b/tests/engine/test_crossSheetReference.ts new file mode 100644 index 0000000..53194a8 --- /dev/null +++ b/tests/engine/test_crossSheetReference.ts @@ -0,0 +1,105 @@ +// Cross-sheet references (`=Data!A1`) must resolve a cell to the SAME value a +// same-sheet reference would. Regression for #2332: the target sheet was being +// resolved through its display-formatted output, so a date serial arrived as +// the string "03/04/2025" and parseFloat read it as 3 — `=Data!A1` returned 3 +// and `=DAY(Data!A1)` returned 2. Same-sheet was always correct; these tests +// pin cross-sheet to that same behaviour. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { SpreadsheetEngine, type SheetData } from "../../src/engine/index.ts"; +import { rowAt } from "./cellAccess.ts"; + +const engine = new SpreadsheetEngine(); + +const calcRow = (target: SheetData, all: SheetData[], row = 0) => rowAt(engine.calculate(target, all).data, row); + +describe("cross-sheet date reference (#2332 regression)", () => { + const data: SheetData = { name: "Data", data: [[{ v: "03/04/2025" }]] }; + const summary: SheetData = { name: "Summary", data: [[{ v: "=DAY(Data!A1)" }, { v: "=Data!A1" }]] }; + + it("=DAY(Data!A1) reads the date, not the leading digits", () => { + assert.equal(calcRow(summary, [data, summary])[0], 4); + }); + + it("=Data!A1 does not collapse to 3", () => { + assert.notEqual(calcRow(summary, [data, summary])[1], 3); + }); + + it("=Data!A1 matches the identical same-sheet reference", () => { + const sameSheet: SheetData = { name: "S", data: [[{ v: "03/04/2025" }, { v: "=DAY(A1)" }, { v: "=A1" }]] }; + const same = calcRow(sameSheet, [sameSheet]); + const cross = calcRow(summary, [data, summary]); + assert.equal(cross[0], same[1]); // =DAY + assert.equal(cross[1], same[2]); // =ref -> "03/04/2025" + }); +}); + +describe("cross-sheet reference — value types read straight across", () => { + const data: SheetData = { + name: "Data", + // date, number, text, empty, a formula that itself produces a date serial + data: [[{ v: "03/04/2025" }, { v: 42 }, { v: "hello" }, { v: "" }, { v: "=DATE(2025,3,4)" }]], + }; + const refs: SheetData = { + name: "Refs", + data: [[{ v: "=Data!A1" }, { v: "=Data!B1" }, { v: "=Data!C1" }, { v: "=Data!D1" }, { v: "=Data!E1" }]], + }; + + it("resolves each type the way the source cell holds it", () => { + assert.deepEqual(calcRow(refs, [data, refs]), ["03/04/2025", 42, "hello", 0, "03/04/2025"]); + }); + + it("feeds a cross-sheet date into a date function", () => { + const derived: SheetData = { name: "D2", data: [[{ v: "=DAY(Data!E1)" }, { v: "=Data!B1*2" }]] }; + assert.deepEqual(calcRow(derived, [data, derived]), [4, 84]); + }); +}); + +describe("cross-sheet range aggregation stays numeric", () => { + it("SUM over a cross-sheet range adds the raw numbers", () => { + const data: SheetData = { name: "D", data: [[{ v: 10 }, { v: 20 }, { v: 30 }]] }; + const sum: SheetData = { name: "S", data: [[{ v: "=SUM(D!A1:C1)" }]] }; + assert.deepEqual(calcRow(sum, [data, sum]), [60]); + }); +}); + +// resolveSheetData (#2482) folds the sheet-ref match -> cache check -> two-stage +// cache seed -> calculateSheet block that getCellValue and collectRangeValues +// shared. The two-stage seed is the cross-sheet infinite-loop guard: a cyclic +// reference must terminate with an error, not hang. If it hung, these tests would +// never return and the whole suite would time out. +describe("cyclic cross-sheet references terminate instead of hanging", () => { + it("a 2-sheet cycle (A!A1=B!A1, B!A1=A!A1) resolves to an error", () => { + const sheetA: SheetData = { name: "A", data: [[{ v: "=B!A1" }]] }; + const sheetB: SheetData = { name: "B", data: [[{ v: "=A!A1" }]] }; + assert.equal(String(calcRow(sheetA, [sheetA, sheetB])[0]), "#ERROR!"); + }); + + it("a 3-sheet cycle (A->B->C->A) also terminates with an error", () => { + const sheetA: SheetData = { name: "A", data: [[{ v: "=B!A1" }]] }; + const sheetB: SheetData = { name: "B", data: [[{ v: "=C!A1" }]] }; + const sheetC: SheetData = { name: "C", data: [[{ v: "=A!A1" }]] }; + assert.equal(String(calcRow(sheetA, [sheetA, sheetB, sheetC])[0]), "#ERROR!"); + }); + + it("a valid cross-sheet reference next to the cycle still resolves", () => { + const data: SheetData = { name: "D", data: [[{ v: 10 }, { v: 20 }]] }; + const main: SheetData = { name: "S", data: [[{ v: "=D!A1" }, { v: "=SUM(D!A1:B1)" }]] }; + assert.deepEqual(calcRow(main, [data, main]), [10, 30]); + }); +}); + +// A reference to a sheet that does not exist keeps each caller's terminal action +// after the fold: #REF! for a single cell, an empty range for an aggregate. +describe("missing-sheet reference keeps its per-caller terminal behaviour", () => { + it("a single cross-sheet cell to a missing sheet is #REF!", () => { + const main: SheetData = { name: "S", data: [[{ v: "=Ghost!A1" }]] }; + assert.equal(String(calcRow(main, [main])[0]), "#REF!"); + }); + + it("SUM over a range on a missing sheet contributes nothing (empty range)", () => { + const main: SheetData = { name: "S", data: [[{ v: "=SUM(Ghost!A1:B1)" }]] }; + assert.equal(calcRow(main, [main])[0], 0); + }); +}); diff --git a/tests/engine/test_dateLocale.ts b/tests/engine/test_dateLocale.ts new file mode 100644 index 0000000..77ee316 --- /dev/null +++ b/tests/engine/test_dateLocale.ts @@ -0,0 +1,95 @@ +// Which way `03/04/2025` reads. Getting this wrong does not throw — the cell +// holds a real date, just the wrong one, three days or eleven months off +// depending on the pair. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { prefersDayFirst } from "../../src/engine/date-locale.ts"; + +describe("prefersDayFirst — the locales the app ships", () => { + it("reads day-first for the European languages", () => { + for (const locale of ["es", "pt-BR", "fr", "de"]) { + assert.equal(prefersDayFirst(locale), true, `${locale} writes day-first`); + } + }); + + // Their conventional order is year-month-day, so in a two-part date the + // month still comes before the day — same as US order. + it("reads month-first for ja, zh and ko", () => { + for (const locale of ["ja", "zh", "ko"]) { + assert.equal(prefersDayFirst(locale), false, `${locale} puts the month first`); + } + }); +}); + +describe("prefersDayFirst — English splits on region, not language", () => { + // The app's own locale resolution folds `en-GB` to `en` before a plugin sees + // it, so a bare `en` cannot be resolved and keeps the US default. A caller + // that CAN supply the region gets the right answer. + it("keeps the US default for a bare en", () => { + assert.equal(prefersDayFirst("en"), false); + }); + + it("reads month-first for the month-first English regions", () => { + assert.equal(prefersDayFirst("en-US"), false); + assert.equal(prefersDayFirst("en-CA"), false); + assert.equal(prefersDayFirst("en-PH"), false); + }); + + it("reads day-first for the rest of the English-speaking world", () => { + for (const locale of ["en-GB", "en-AU", "en-NZ", "en-IE", "en-IN", "en-ZA"]) { + assert.equal(prefersDayFirst(locale), true, `${locale} writes day-first`); + } + }); +}); + +describe("prefersDayFirst — tag shapes", () => { + it("accepts underscores as well as hyphens", () => { + assert.equal(prefersDayFirst("en_GB"), true); + assert.equal(prefersDayFirst("pt_BR"), true); + }); + + it("ignores case", () => { + assert.equal(prefersDayFirst("EN-GB"), true); + assert.equal(prefersDayFirst("FR"), true); + assert.equal(prefersDayFirst("en-us"), false); + }); + + // A script subtag sits between the language and the region, so reading + // position 1 as the region flips `en-Latn-US` to day-first (Codex review). + // The earlier `zh-Hans-CN` case looked like it covered this and did not — + // non-English tags never consult the region at all. + it("reads past a script subtag to find the region", () => { + assert.equal(prefersDayFirst("en-Latn-US"), false, "en-Latn-US is month-first"); + assert.equal(prefersDayFirst("en-Latn-GB"), true, "en-Latn-GB is day-first"); + }); + + // An extension starts with a single-character subtag, and nothing after it + // is a region. `-u-nu-latn` is the case that distinguishes: "nu" is two + // letters and would otherwise be taken as a region, flipping a bare `en` to + // day-first. (`-u-ca-gregory` does NOT distinguish — "ca" happens to be a + // month-first region, so both readings agree by accident.) + it("does not read an extension subtag as a region", () => { + assert.equal(prefersDayFirst("en-u-nu-latn"), false, "no region: keeps the US default"); + assert.equal(prefersDayFirst("en-u-ca-gregory"), false); + assert.equal(prefersDayFirst("en-GB-u-nu-latn"), true, "a real region before the extension still wins"); + }); + + it("accepts a numeric UN M.49 region", () => { + assert.equal(prefersDayFirst("es-419"), true, "Latin American Spanish is still day-first by language"); + }); + + it("ignores the region for non-English tags", () => { + assert.equal(prefersDayFirst("zh-Hans-CN"), false); + assert.equal(prefersDayFirst("fr-CA"), true, "decided by language, not region"); + }); + + // Falling back to month-first keeps the existing behaviour for anything + // unrecognised, so a new locale cannot silently flip existing sheets. + it("falls back to month-first for unknown, empty and missing locales", () => { + assert.equal(prefersDayFirst("xx"), false); + assert.equal(prefersDayFirst(""), false); + assert.equal(prefersDayFirst(undefined), false); + assert.equal(prefersDayFirst(null), false); + }); +}); diff --git a/tests/engine/test_dateParser.ts b/tests/engine/test_dateParser.ts new file mode 100644 index 0000000..de861a0 --- /dev/null +++ b/tests/engine/test_dateParser.ts @@ -0,0 +1,211 @@ +// Turning a cell's text into a date. Every failure mode here is a wrong date +// rather than an error: March 4 read as April 3, 1930 read as 2030, a real +// date rejected as text. The cell still shows something plausible. +// +// Assertions compare against `dateToSerial(Date.UTC(...))` rather than literal +// serial numbers, so these tests are about how the STRING is interpreted; +// serial arithmetic itself is covered in test_dateUtils.ts. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { getDefaultDateFormat, isDateLike, parseDate } from "../../src/engine/date-parser.ts"; +import { dateToSerial } from "../../src/engine/date-utils.ts"; + +const serialOf = (year: number, month: number, day: number) => dateToSerial(new Date(Date.UTC(year, month - 1, day))); + +function assertParsesTo(input: string, year: number, month: number, day: number, preferDayFirst = false): void { + assert.equal(parseDate(input, preferDayFirst), serialOf(year, month, day), `${input} should read as ${year}-${month}-${day}`); +} + +describe("isDateLike", () => { + it("accepts the formats the parser handles", () => { + for (const input of ["03/04/2025", "2025-03-04", "2025/03/04", "4-Mar-2025", "Mar 4, 2025", "March 4, 2025", "4 Mar 2025"]) { + assert.equal(isDateLike(input), true, `${input} should look like a date`); + } + }); + + it("rejects text with no digits or no separator", () => { + assert.equal(isDateLike("hello world"), false); + assert.equal(isDateLike("20250304"), false); + }); + + // The length gate runs before any pattern match, so a short but perfectly + // well-formed date is rejected outright. + it("rejects a valid short date because of the 6-character floor", () => { + assert.equal(isDateLike("1/1/26"), true, "exactly 6 characters passes"); + assert.equal(isDateLike("1/1/6"), false, "5 characters is refused before any pattern runs"); + }); + + it("rejects anything longer than 30 characters", () => { + assert.equal(isDateLike(`${"September 30, 2025".padEnd(31, " ")}`), false); + }); + + it("rejects partial or malformed dates", () => { + assert.equal(isDateLike("03/2025"), false); + assert.equal(isDateLike("03-04-2025"), false, "hyphen-separated numerics are not a supported pattern"); + }); +}); + +describe("parseDate — ISO format", () => { + it("reads YYYY-MM-DD and YYYY/MM/DD", () => { + assertParsesTo("2025-03-04", 2025, 3, 4); + assertParsesTo("2025/03/04", 2025, 3, 4); + }); + + it("accepts unpadded month and day", () => { + assertParsesTo("2025-3-4", 2025, 3, 4); + }); + + // ISO is matched first, so a leading 4-digit group is never mistaken for a + // day even when it would be a legal day-first date. + it("takes the leading 4-digit group as the year", () => { + assertParsesTo("2025-01-02", 2025, 1, 2); + }); +}); + +describe("parseDate — slash format and the MM/DD vs DD/MM decision", () => { + // The default is US order. This is the single most consequential choice in + // the module: an ambiguous date silently becomes a different day. + it("defaults an ambiguous date to MM/DD", () => { + assertParsesTo("03/04/2025", 2025, 3, 4); + }); + + it("reads an ambiguous date as DD/MM when asked to prefer it", () => { + assertParsesTo("03/04/2025", 2025, 4, 3, true); + }); + + // Unambiguous cases ignore the preference entirely — the value decides. + it("reads day-first when the first number cannot be a month", () => { + assertParsesTo("13/04/2025", 2025, 4, 13); + assertParsesTo("13/04/2025", 2025, 4, 13, true); + }); + + it("reads month-first when the second number cannot be a day-of-month position", () => { + assertParsesTo("03/13/2025", 2025, 3, 13); + assertParsesTo("03/13/2025", 2025, 3, 13, true); + }); + + it("rejects a slash date where neither ordering is valid", () => { + assert.equal(parseDate("13/13/2025"), null); + }); +}); + +describe("parseDate — two-digit years", () => { + // The pivot is hardcoded at 30 and not configurable, so "01/01/30" is 1930. + it("maps years under 30 to the 2000s and 30 or over to the 1900s", () => { + assertParsesTo("01/01/29", 2029, 1, 1); + assertParsesTo("01/01/30", 1930, 1, 1); + assertParsesTo("01/01/99", 1999, 1, 1); + assertParsesTo("01/01/00", 2000, 1, 1); + }); + + it("applies the same pivot to the DD-MMM-YY form", () => { + assertParsesTo("1-Jan-29", 2029, 1, 1); + assertParsesTo("1-Jan-30", 1930, 1, 1); + }); +}); + +describe("parseDate — month-name formats", () => { + it("reads DD-MMM-YYYY", () => { + assertParsesTo("4-Mar-2025", 2025, 3, 4); + assertParsesTo("04-Mar-2025", 2025, 3, 4); + }); + + it("reads MMM D, YYYY and MMMM D, YYYY", () => { + assertParsesTo("Mar 4, 2025", 2025, 3, 4); + assertParsesTo("March 4, 2025", 2025, 3, 4); + }); + + it("reads D MMM YYYY", () => { + assertParsesTo("4 Mar 2025", 2025, 3, 4); + assertParsesTo("4 March 2025", 2025, 3, 4); + }); + + it("matches month names case-insensitively", () => { + assertParsesTo("4-MAR-2025", 2025, 3, 4); + assertParsesTo("march 4, 2025", 2025, 3, 4); + }); + + it("rejects a month name that is not a real month", () => { + assert.equal(parseDate("4-Foo-2025"), null); + assert.equal(parseDate("Smarch 4, 2025"), null); + }); + + it("makes the comma optional in MMM D YYYY", () => { + assertParsesTo("Mar 4 2025", 2025, 3, 4); + }); +}); + +describe("parseDate — validity and range", () => { + it("rejects a day that does not exist in its month", () => { + assert.equal(parseDate("2025-02-30"), null); + assert.equal(parseDate("2025-04-31"), null); + assert.equal(parseDate("02/30/2025"), null); + }); + + it("accepts Feb 29 in a leap year and rejects it otherwise", () => { + assertParsesTo("2024-02-29", 2024, 2, 29); + assert.equal(parseDate("2025-02-29"), null); + }); + + // The window is 1900–2100 inclusive; outside it a well-formed date is + // rejected rather than converted. + it("enforces the 1900–2100 year window", () => { + assertParsesTo("1900-01-01", 1900, 1, 1); + assertParsesTo("2100-12-31", 2100, 12, 31); + assert.equal(parseDate("1899-12-31"), null); + assert.equal(parseDate("2101-01-01"), null); + }); + + it("returns null for anything isDateLike rejects", () => { + assert.equal(parseDate("hello"), null); + assert.equal(parseDate(""), null); + assert.equal(parseDate("1/1/6"), null); + }); + + it("tolerates surrounding whitespace", () => { + assertParsesTo(" 2025-03-04 ", 2025, 3, 4); + }); +}); + +describe("getDefaultDateFormat", () => { + it("echoes the shape it was given", () => { + assert.equal(getDefaultDateFormat("2025-03-04"), "YYYY-MM-DD"); + assert.equal(getDefaultDateFormat("2025/03/04"), "YYYY/MM/DD"); + assert.equal(getDefaultDateFormat("4-Mar-2025"), "DD-MMM-YYYY"); + assert.equal(getDefaultDateFormat("Mar 4, 2025"), "MMM D, YYYY"); + assert.equal(getDefaultDateFormat("March 4, 2025"), "MMMM D, YYYY"); + }); + + // The three-vs-four letter split is what separates the two month-name + // formats; a four-letter month name takes the long form. + it("splits the month-name formats on name length", () => { + assert.equal(getDefaultDateFormat("Jun 4, 2025"), "MMM D, YYYY"); + assert.equal(getDefaultDateFormat("June 4, 2025"), "MMMM D, YYYY"); + }); + + // A slash date is labelled in the order the parser READ it, so the cell + // renders the halves the way the user typed them. + it("labels a slash date in its reading order", () => { + assert.equal(getDefaultDateFormat("03/04/2025"), "MM/DD/YYYY", "ambiguous: US default"); + assert.equal(getDefaultDateFormat("03/04/2025", true), "DD/MM/YYYY", "ambiguous: day-first when preferred"); + assert.equal(getDefaultDateFormat("13/04/2025"), "DD/MM/YYYY", "13 can only be a day, whatever the preference"); + assert.equal(getDefaultDateFormat("03/13/2025", true), "MM/DD/YYYY", "13 can only be a day here too"); + }); + + // Slash-separated ISO parses year-first, so it must be LABELLED year-first + // too. Falling through to the slash default re-rendered it as MM/DD or DD/MM + // — the same digits in a different order, which reads as a different date + // (Codex review). + it("keeps a year-first label for YYYY/MM/DD under either preference", () => { + assert.equal(getDefaultDateFormat("2025/03/04"), "YYYY/MM/DD"); + assert.equal(getDefaultDateFormat("2025/03/04", true), "YYYY/MM/DD"); + assert.equal(getDefaultDateFormat("2025/3/4", true), "YYYY/MM/DD", "unpadded too"); + }); + + it("falls back to the preference for anything unrecognised", () => { + assert.equal(getDefaultDateFormat("not a date"), "MM/DD/YYYY"); + assert.equal(getDefaultDateFormat("not a date", true), "DD/MM/YYYY"); + assert.equal(getDefaultDateFormat(""), "MM/DD/YYYY"); + }); +}); diff --git a/tests/engine/test_dateUtils.ts b/tests/engine/test_dateUtils.ts new file mode 100644 index 0000000..1ce2e55 --- /dev/null +++ b/tests/engine/test_dateUtils.ts @@ -0,0 +1,123 @@ +// Excel serial-number conversion. Every date function in the engine routes +// through this pair, and an error here is invisible: a date still renders, it +// is just the wrong day. The epoch choice is the subtle part — Excel's serial +// numbering embeds a 1900 leap-year bug, and the base date compensates for it. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { + dateToSerial, + serialToDate, + DAY_NAMES_FULL, + DAY_NAMES_SHORT, + MONTH_NAMES_FULL, + MONTH_NAMES_SHORT, +} from "../../src/engine/date-utils.ts"; + +const utc = (year: number, month: number, day: number, hour = 0, minute = 0, second = 0) => new Date(Date.UTC(year, month - 1, day, hour, minute, second)); + +describe("dateToSerial", () => { + it("anchors serial 0 at the Dec 30 1899 base", () => { + assert.equal(dateToSerial(utc(1899, 12, 30)), 0); + assert.equal(dateToSerial(utc(1899, 12, 31)), 1); + }); + + it("counts whole days forward", () => { + assert.equal(dateToSerial(utc(1900, 1, 2)), 3); + assert.equal(dateToSerial(utc(1900, 2, 1)), 33); + }); + + // The point of the Dec 30 1899 base: Excel counts a phantom 1900-02-29 that + // never existed, and starting two days early makes every serial from March + // 1900 onward — i.e. every date anyone actually uses — agree with Excel's. + it("agrees with Excel from March 1900 onward", () => { + assert.equal(dateToSerial(utc(1900, 3, 1)), 61); // Excel: 61 + assert.equal(dateToSerial(utc(2000, 1, 1)), 36526); // Excel: 36526 + assert.equal(dateToSerial(utc(2026, 7, 22)), 46225); + }); + + // The flip side of that choice, pinned so it is a known limitation rather + // than a surprise: for the first two months of 1900 the serials sit one + // ahead of Excel's, because the phantom leap day has not been passed yet. + it("sits one ahead of Excel for Jan and Feb 1900", () => { + assert.equal(dateToSerial(utc(1900, 1, 1)), 2); // Excel: 1 + assert.equal(dateToSerial(utc(1900, 2, 28)), 60); // Excel: 59 + }); + + it("represents a time of day as the fractional part", () => { + assert.equal(dateToSerial(utc(1899, 12, 31, 12)), 1.5); + assert.equal(dateToSerial(utc(1899, 12, 31, 6)), 1.25); + }); + + it("goes negative for dates before the base", () => { + assert.ok(dateToSerial(utc(1899, 12, 29)) < 0); + }); +}); + +describe("serialToDate", () => { + it("maps serial 1 back to the day after the base", () => { + assert.equal(serialToDate(1).toISOString().slice(0, 10), "1899-12-31"); + }); + + it("maps a modern serial back to its date", () => { + assert.equal(serialToDate(46225).toISOString().slice(0, 10), "2026-07-22"); + assert.equal(serialToDate(61).toISOString().slice(0, 10), "1900-03-01"); + }); + + it("restores the time of day from the fractional part", () => { + assert.equal(serialToDate(1.5).toISOString().slice(11, 19), "12:00:00"); + assert.equal(serialToDate(1.25).toISOString().slice(11, 19), "06:00:00"); + }); + + // The fraction is rounded to the nearest second, so a value that lands + // mid-second must not drift to the previous one. + it("rounds the time component to the nearest second", () => { + const almostOneMinute = 1 + 59.6 / 86400; + assert.equal(serialToDate(almostOneMinute).toISOString().slice(11, 19), "00:01:00"); + }); +}); + +describe("dateToSerial / serialToDate round-trip", () => { + it("round-trips whole days across month, year and leap boundaries", () => { + const dates = [utc(1900, 1, 1), utc(1900, 3, 1), utc(1999, 12, 31), utc(2000, 2, 29), utc(2024, 2, 29), utc(2026, 7, 22), utc(2100, 1, 1)]; + for (const date of dates) { + const back = serialToDate(dateToSerial(date)); + assert.equal(back.toISOString(), date.toISOString(), `round-trip failed for ${date.toISOString()}`); + } + }); + + it("round-trips a date carrying a time of day", () => { + const date = utc(2026, 7, 22, 13, 45, 30); + assert.equal(serialToDate(dateToSerial(date)).toISOString(), date.toISOString()); + }); +}); + +describe("name tables", () => { + // These are indexed by `getMonth()` / `getDay()` directly, so a wrong length + // or a shifted entry produces an off-by-one month or weekday with no error. + it("has twelve months starting at January", () => { + assert.equal(MONTH_NAMES_SHORT.length, 12); + assert.equal(MONTH_NAMES_FULL.length, 12); + assert.equal(MONTH_NAMES_SHORT[0], "Jan"); + assert.equal(MONTH_NAMES_FULL[0], "January"); + assert.equal(MONTH_NAMES_SHORT[11], "Dec"); + assert.equal(MONTH_NAMES_FULL[11], "December"); + }); + + it("has seven days starting at Sunday, matching Date#getDay", () => { + assert.equal(DAY_NAMES_SHORT.length, 7); + assert.equal(DAY_NAMES_FULL.length, 7); + assert.equal(DAY_NAMES_SHORT[0], "Sun"); + assert.equal(DAY_NAMES_FULL[0], "Sunday"); + assert.equal(DAY_NAMES_SHORT[6], "Sat"); + }); + + it("keeps the short names as prefixes of the full names", () => { + MONTH_NAMES_SHORT.forEach((short, index) => + assert.ok(MONTH_NAMES_FULL[index]?.startsWith(short), `${short} is not a prefix of ${String(MONTH_NAMES_FULL[index])}`), + ); + DAY_NAMES_SHORT.forEach((short, index) => + assert.ok(DAY_NAMES_FULL[index]?.startsWith(short), `${short} is not a prefix of ${String(DAY_NAMES_FULL[index])}`), + ); + }); +}); diff --git a/tests/engine/test_datedif.ts b/tests/engine/test_datedif.ts new file mode 100644 index 0000000..63af48c --- /dev/null +++ b/tests/engine/test_datedif.ts @@ -0,0 +1,133 @@ +// DATEDIF's per-unit elapsed-time math. Each unit has its own boundary handling +// — complete years/months back off when the day-of-month has not been reached, +// MD is the day remainder after the complete months (always non-negative), YD +// wraps into the end's year — and a wrong branch returns a plausible number +// rather than an error. +// +// Inputs are Excel serials, so tests build them from a known date via a helper +// rather than hardcoding the serial arithmetic (that is covered in +// test_dateUtils.ts). + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { computeDatedif } from "../../src/engine/datedif.ts"; +import { dateToSerial } from "../../src/engine/date-utils.ts"; +import { NUM_ERROR } from "../../src/engine/spreadsheet-errors.ts"; + +const serial = (year: number, month: number, day: number) => dateToSerial(new Date(Date.UTC(year, month - 1, day))); +const diff = (start: [number, number, number], end: [number, number, number], unit: string) => computeDatedif(serial(...start), serial(...end), unit); + +describe("computeDatedif — Y (complete years)", () => { + it("counts whole years between the same month and day", () => { + assert.equal(diff([2020, 6, 15], [2023, 6, 15], "Y"), 3); + }); + + // The end has not yet reached the start's month/day in its year, so the last + // year is incomplete. + it("backs off a year when the anniversary has not been reached", () => { + assert.equal(diff([2020, 6, 15], [2023, 6, 14], "Y"), 2); + assert.equal(diff([2020, 6, 15], [2023, 5, 20], "Y"), 2); + }); + + it("counts the year once the anniversary is reached exactly", () => { + assert.equal(diff([2020, 2, 29], [2024, 2, 29], "Y"), 4, "leap day to leap day"); + }); +}); + +describe("computeDatedif — M (complete months)", () => { + it("counts whole months", () => { + assert.equal(diff([2023, 1, 10], [2023, 4, 10], "M"), 3); + assert.equal(diff([2020, 1, 1], [2023, 1, 1], "M"), 36); + }); + + it("backs off a month when the day-of-month has not been reached", () => { + assert.equal(diff([2023, 1, 15], [2023, 4, 10], "M"), 2); + }); +}); + +describe("computeDatedif — D (calendar days)", () => { + it("counts the days between two dates", () => { + assert.equal(diff([2023, 1, 1], [2023, 1, 31], "D"), 30); + assert.equal(diff([2023, 1, 1], [2024, 1, 1], "D"), 365); + assert.equal(diff([2024, 1, 1], [2025, 1, 1], "D"), 366, "leap year"); + }); +}); + +describe("computeDatedif — MD (day-of-month diff, months ignored)", () => { + it("subtracts the days directly when the end day is later", () => { + assert.equal(diff([2023, 1, 10], [2023, 3, 25], "MD"), 15); + }); + + // Multi-month spans still measure the day remainder correctly: Jan 15 → Mar 10 + // has one complete month (to Feb 15), leaving 23 days to Mar 10. + it("measures the day remainder across several months", () => { + assert.equal(diff([2023, 1, 15], [2023, 3, 10], "MD"), 23); + }); + + // The day remainder is anchored on start-plus-complete-months, so it is never + // negative even when the start day outruns the month before `end`. Jan 30 → + // Mar 1 has one complete month (clamped to Feb 28/29), leaving one day — where + // the old borrow-the-previous-month math returned -1 (#2414). + it("stays non-negative when the start day outruns the preceding month", () => { + assert.equal(diff([2023, 1, 30], [2023, 3, 1], "MD"), 1, "Jan 30 + 1 month → Feb 28, then 1 day"); + assert.equal(diff([2024, 1, 30], [2024, 3, 1], "MD"), 1, "leap year: Jan 30 + 1 month → Feb 29, then 1 day"); + }); + + // The remainder counts whole days: a datetime serial's time-of-day must not + // change the result (an 18:00 fraction on `end` once rounded MD up to 2). + it("ignores the time-of-day of a datetime serial", () => { + const start = serial(2023, 1, 30); + const end = serial(2023, 3, 1); + assert.equal(computeDatedif(start, end + 0.75, "MD"), 1, "end at 18:00 still yields 1"); + assert.equal(computeDatedif(start + 0.75, end + 0.25, "MD"), 1, "start and end times both ignored"); + }); +}); + +describe("computeDatedif — YM (month diff, years ignored)", () => { + it("counts months within the year", () => { + assert.equal(diff([2020, 1, 10], [2023, 4, 10], "YM"), 3); + }); + + // Ignoring years can leave the month difference negative; it wraps into 0..11. + it("wraps a negative month difference into the 0..11 range", () => { + assert.equal(diff([2020, 11, 10], [2023, 2, 10], "YM"), 3); + }); + + it("backs off when the day has not been reached, then wraps", () => { + assert.equal(diff([2020, 11, 20], [2023, 2, 10], "YM"), 2); + }); +}); + +describe("computeDatedif — YD (day diff, years ignored)", () => { + it("counts days within the same year window", () => { + assert.equal(diff([2023, 1, 1], [2023, 3, 1], "YD"), 59, "Jan + Feb 2023"); + }); + + // Moving the start into the end's year would put it after the end, so it + // steps back a year and counts across the boundary. + it("crosses the year boundary when the start falls later in the end's year", () => { + assert.equal(diff([2020, 12, 20], [2023, 1, 5], "YD"), 16, "Dec 20 to Jan 5"); + }); +}); + +describe("computeDatedif — errors", () => { + it("returns #NUM! when start is after end", () => { + assert.equal(diff([2023, 6, 15], [2023, 6, 10], "D"), NUM_ERROR); + }); + + it("returns #NUM! for an unknown unit", () => { + assert.equal(diff([2020, 1, 1], [2023, 1, 1], "Q"), NUM_ERROR); + assert.equal(diff([2020, 1, 1], [2023, 1, 1], ""), NUM_ERROR); + }); + + it("matches the unit case-insensitively", () => { + assert.equal(diff([2020, 1, 1], [2023, 1, 1], "y"), 3); + assert.equal(diff([2023, 1, 1], [2023, 1, 31], "d"), 30); + }); + + it("returns 0 for identical dates in every unit", () => { + for (const unit of ["Y", "M", "D", "MD", "YM", "YD"]) { + assert.equal(diff([2023, 6, 15], [2023, 6, 15], unit), 0, `unit ${unit}`); + } + }); +}); diff --git a/tests/engine/test_errorReporting.ts b/tests/engine/test_errorReporting.ts new file mode 100644 index 0000000..3d892ca --- /dev/null +++ b/tests/engine/test_errorReporting.ts @@ -0,0 +1,101 @@ +// Failed formulas must surface as TYPED errors with an Excel-style error value in +// the cell — not silently become a bare string or a wrong number with an empty +// errors[] (issue #2359). The root cause was the over-broad top-level catch in +// evaluateFormula, which meant the function never threw, so calculator.ts's +// per-cell catch was unreachable and only "circular" was ever recorded. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { SpreadsheetEngine, type SheetData, type CalculatedSheet } from "../../src/engine/index.ts"; +import { cellAt } from "./cellAccess.ts"; + +const calc = (sheet: SheetData, all?: SheetData[]): CalculatedSheet => new SpreadsheetEngine().calculate(sheet, all ?? [sheet]); + +/** The single formula cell's value and the error type recorded for it, if any. */ +function cellAndError(sheet: SheetData, row: number, col: number, all?: SheetData[]) { + const result = calc(sheet, all); + const entry = result.errors.find((err) => err.cell.row === row && err.cell.col === col); + return { value: cellAt(result.data, row, col), errorType: entry?.type }; +} + +describe("#2359 typed error reporting", () => { + it("div_zero: =1/0 becomes #DIV/0!, not Infinity", () => { + const { value, errorType } = cellAndError({ name: "S", data: [[{ v: "=1/0" }]] }, 0, 0); + assert.equal(value, "#DIV/0!"); + assert.equal(errorType, "div_zero"); + }); + + it("div_zero: a reference division by zero (=A1/A2, 10/0) is #DIV/0!", () => { + const sheet: SheetData = { name: "S", data: [[{ v: 10 }], [{ v: 0 }], [{ v: "=A1/A2" }]] }; + const { value, errorType } = cellAndError(sheet, 2, 0); + assert.equal(value, "#DIV/0!"); + assert.equal(errorType, "div_zero"); + }); + + it("invalid_ref: a reference to a missing sheet is #REF!", () => { + const sheet: SheetData = { name: "S", data: [[{ v: "=Missing!A1" }]] }; + const { value, errorType } = cellAndError(sheet, 0, 0); + assert.equal(value, "#REF!"); + assert.equal(errorType, "invalid_ref"); + }); + + it("syntax: an unknown function is #NAME?, not a processed string", () => { + const sheet: SheetData = { name: "S", data: [[{ v: 7 }, { v: "=UNKNOWNFN(A1)" }]] }; + const { value, errorType } = cellAndError(sheet, 0, 1); + assert.equal(value, "#NAME?"); + assert.equal(errorType, "syntax"); + }); + + // IFS needs condition/value PAIRS — a count the registry's min/max cannot + // express, so the handler itself throws. (This used to use SUM over two + // ranges, which now legitimately sums them.) + it("unknown: a handler that throws (IFS with an odd argument count) is #ERROR!, not the formula text", () => { + const sheet: SheetData = { name: "S", data: [[{ v: 1 }, { v: '=IFS(A1>0, "yes", "orphan")' }]] }; + const { value, errorType } = cellAndError(sheet, 0, 1); + assert.equal(value, "#ERROR!"); + assert.equal(errorType, "unknown"); + }); + + it("propagates an error through an arithmetic reference (=A1+1 where A1 is #DIV/0!)", () => { + const sheet: SheetData = { name: "S", data: [[{ v: "=1/0" }, { v: "=A1+1" }]] }; + const { value, errorType } = cellAndError(sheet, 0, 1); + assert.equal(value, "#DIV/0!", "must not become the bare string 'Infinity+1'"); + assert.equal(errorType, "div_zero"); + }); + + it("a failed formula never lands in the cell as a bare string", () => { + const formulas = ["=1/0", "=UNKNOWNFN(A1)", '=IFS(A1>0, "yes", "orphan")']; + for (const formula of formulas) { + const value = cellAt(calc({ name: "S", data: [[{ v: formula }]] }).data, 0, 0); + assert.equal(typeof value === "string" && value.startsWith("#"), true, `${formula} → ${JSON.stringify(value)} should be an # error value`); + } + }); +}); + +describe("#2359 success paths are preserved", () => { + it("keeps circular-reference detection working", () => { + const sheet: SheetData = { name: "S", data: [[{ v: "=B1+1" }, { v: "=A1+1" }]] }; + const result = calc(sheet); + assert.equal( + result.errors.some((err) => err.type === "circular"), + true, + ); + }); + + it("=ZZ999 (an empty in-bounds cell) stays 0 and is NOT an error", () => { + const { value, errorType } = cellAndError({ name: "S", data: [[{ v: "=ZZ999" }]] }, 0, 0); + assert.equal(value, 0); + assert.equal(errorType, undefined); + }); + + it("valid SUM and arithmetic are unaffected", () => { + assert.equal(cellAt(calc({ name: "S", data: [[{ v: 1 }, { v: "=SUM(A1:A3)" }], [{ v: 2 }], [{ v: 3 }]] }).data, 0, 1), 6); + assert.equal(cellAt(calc({ name: "S", data: [[{ v: 2 }], [{ v: 3 }], [{ v: "=A1+A2" }]] }).data, 2, 0), 5); + }); + + it("a valid cross-sheet reference still resolves", () => { + const data: SheetData = { name: "Data", data: [[{ v: 100 }]] }; + const summary: SheetData = { name: "Summary", data: [[{ v: "=Data!A1*2" }]] }; + assert.equal(cellAt(calc(summary, [data, summary]).data, 0, 0), 200); + }); +}); diff --git a/tests/engine/test_errorValue.ts b/tests/engine/test_errorValue.ts new file mode 100644 index 0000000..fe01bca --- /dev/null +++ b/tests/engine/test_errorValue.ts @@ -0,0 +1,180 @@ +// Formula errors as a distinct VALUE (#2451). +// +// While errors were plain strings, `SQRT(-1)` and `CONCAT("#N","UM!")` both +// produced "#NUM!", so IFERROR could not tell a real error from text that +// merely spells one — the computed case was caught as an error and silently +// replaced by the fallback. An error is now its own value carrying the code; +// text stays text. The display pass renders the value back to `#NUM!`, so the +// cells look exactly as they did. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { SpreadsheetEngine, type SheetData } from "../../src/engine/index.ts"; +import { evaluateFormula } from "../../src/engine/evaluator.ts"; +import { formatCellForDisplay } from "../../src/engine/cellFormatting.ts"; +import { + DIV_ZERO_ERROR, + NA_ERROR, + NUM_ERROR, + SpreadsheetError, + isSpreadsheetErrorValue, + isErrorResult, + spreadsheetError, +} from "../../src/engine/spreadsheet-errors.ts"; +import type { CellValue } from "../../src/engine/types.ts"; +import { cellAt } from "./cellAccess.ts"; + +/** The raw computed value of a formula, before the display pass turns an error + * back into its code — this is where provenance is observable. */ +const evaluateRaw: (formula: string) => CellValue = (formula) => + evaluateFormula(formula, { + getCellValue: () => 0, + getRangeValues: () => [], + evaluateFormula: evaluateRaw, + }); + +/** What a single-formula sheet DISPLAYS, through the public engine API. */ +const displayed = (formula: string): CellValue => + cellAt(new SpreadsheetEngine().calculate({ name: "S", data: [[{ v: formula }]] } satisfies SheetData).data, 0, 0); + +describe("the error value and its guard", () => { + it("recognises an error value and rejects a string that spells the same code", () => { + assert.equal(isSpreadsheetErrorValue(NUM_ERROR), true); + assert.equal(isSpreadsheetErrorValue("#NUM!"), false); + assert.equal(isSpreadsheetErrorValue(0), false); + assert.equal(isSpreadsheetErrorValue(null), false); + }); + + it("carries the code and renders as it when coerced to text", () => { + assert.equal(NUM_ERROR.code, "#NUM!"); + assert.equal(String(NUM_ERROR), "#NUM!"); + assert.equal(`${DIV_ZERO_ERROR}`, "#DIV/0!"); + }); + + it("hands out one instance per code, so two errors of a kind compare equal", () => { + const fromLookup = spreadsheetError("#N/A"); + const fromLookupAgain = spreadsheetError("#N/A"); + assert.equal(fromLookup, NA_ERROR); + assert.equal(fromLookup === fromLookupAgain, true); + assert.equal(NA_ERROR instanceof SpreadsheetError, true); + }); + + it("serializes to its code rather than to an empty object", () => { + assert.equal(JSON.stringify({ cell: NUM_ERROR }), '{"cell":"#NUM!"}'); + }); +}); + +describe("isErrorResult keys off the value, not the text", () => { + it("catches an error value", () => { + assert.equal(isErrorResult(NUM_ERROR), true); + assert.equal(isErrorResult(DIV_ZERO_ERROR), true); + }); + + it("still catches NaN / infinity / missing", () => { + assert.equal(isErrorResult(NaN), true); + assert.equal(isErrorResult(Infinity), true); + assert.equal(isErrorResult(null), true); + }); + + it("does NOT catch a look-alike string — the whole point of #2451", () => { + assert.equal(isErrorResult("#NUM!"), false); + assert.equal(isErrorResult("#N/A"), false); + }); +}); + +describe("the display pass renders an error value to its code", () => { + it("returns the code for a formula cell", () => { + assert.equal(formatCellForDisplay({ v: "=SQRT(-1)" }, NUM_ERROR, false), "#NUM!"); + }); + + it("returns the code regardless of the cell's format code", () => { + assert.equal(formatCellForDisplay({ v: "=A1/A2", f: "$#,##0.00" }, DIV_ZERO_ERROR, false), "#DIV/0!"); + }); + + it("leaves ordinary values alone", () => { + assert.equal(formatCellForDisplay({ v: "=1+1" }, 2, false), 2); + assert.equal(formatCellForDisplay({ v: "text" }, "text", false), "text"); + }); +}); + +describe("functions return an error VALUE, and the cell still shows its code", () => { + it("SQRT(-1) computes to the #NUM! value", () => { + const value = evaluateRaw("SQRT(-1)"); + assert.equal(isSpreadsheetErrorValue(value), true); + assert.equal(value, NUM_ERROR); + }); + + it("MOD(5, 0) computes to the #DIV/0! value", () => { + assert.equal(evaluateRaw("MOD(5, 0)"), DIV_ZERO_ERROR); + }); + + it("computed text that spells an error stays a plain string", () => { + const value = evaluateRaw('CONCAT("#N","UM!")'); + assert.equal(isSpreadsheetErrorValue(value), false); + assert.equal(value, "#NUM!"); + }); + + it("displays the same codes end-to-end as before the refactor", () => { + assert.equal(displayed("=SQRT(-1)"), "#NUM!"); + assert.equal(displayed("=MOD(5, 0)"), "#DIV/0!"); + assert.equal(displayed("=1/0"), "#DIV/0!"); + assert.equal(displayed('=DATEDIF(45000, 44000, "D")'), "#NUM!"); + assert.equal(displayed('=VALUE("abc")'), "#VALUE!"); + }); + + it("propagates an error VALUE through a reference and shows the code", () => { + const sheet: SheetData = { name: "S", data: [[{ v: "=SQRT(-1)" }, { v: "=A1+1" }]] }; + assert.equal(cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 1), "#NUM!"); + }); +}); + +describe("IFERROR keys off provenance", () => { + it("catches a real error", () => { + assert.equal(displayed("=IFERROR(SQRT(-1), 42)"), 42); + assert.equal(displayed("=IFERROR(MOD(5, 0), -1)"), -1); + }); + + it("passes a non-error through untouched", () => { + assert.equal(displayed("=IFERROR(SQRT(4), 42)"), 2); + assert.equal(displayed('=IFERROR("hello", 42)'), "hello"); + }); + + it("does not catch a quoted literal that only looks like an error", () => { + assert.equal(displayed('=IFERROR("#NUM!", 42)'), "#NUM!"); + }); + + // THE headline case. Before #2451 this returned 42: the computed text was + // indistinguishable from a real #NUM!, so IFERROR swallowed it. + it("does not catch COMPUTED text that spells an error", () => { + assert.equal(displayed('=IFERROR(CONCAT("#N","UM!"), 42)'), "#NUM!"); + assert.equal(displayed('=IFERROR(CONCATENATE("#DIV/", "0!"), 42)'), "#DIV/0!"); + }); + + it("does not catch an error-looking string built with the & operator", () => { + assert.equal(displayed('=IFERROR("#N" & "UM!", 42)'), "#NUM!"); + }); + + it("still catches an error that reaches it through arithmetic", () => { + assert.equal(displayed("=IFERROR(SQRT(-1) + 1, 42)"), 42); + }); +}); + +describe("IFNA keys off the error value's code", () => { + it("substitutes the fallback for a real #N/A", () => { + const sheet: SheetData = { name: "S", data: [[{ v: 1 }, { v: '=IFNA(MATCH(99, A1:A1, 0), "missing")' }]] }; + assert.equal(cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 1), "missing"); + }); + + it("leaves a different error alone", () => { + assert.equal(displayed('=IFNA(SQRT(-1), "missing")'), "#NUM!"); + }); + + it("does not substitute for text that merely spells #N/A", () => { + assert.equal(displayed('=IFNA("#N/A", "missing")'), "#N/A"); + assert.equal(displayed('=IFNA(CONCAT("#N", "/A"), "missing")'), "#N/A"); + }); + + it("passes an ordinary value through", () => { + assert.equal(displayed('=IFNA(7, "missing")'), 7); + }); +}); diff --git a/tests/engine/test_expandRangeOrCell.ts b/tests/engine/test_expandRangeOrCell.ts new file mode 100644 index 0000000..437cc2b --- /dev/null +++ b/tests/engine/test_expandRangeOrCell.ts @@ -0,0 +1,83 @@ +// Turning a range or single-cell reference into coordinates. The calculator's +// old inline regex was range-only and case-sensitive and did not strip `$`, so +// three common reference shapes fell through to "no values" — and a function +// over an empty list is 0, not an error (#2356). + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { expandRangeOrCell } from "../../src/engine/formulaRefs.ts"; + +describe("expandRangeOrCell — ranges", () => { + it("expands a simple range top-to-bottom, left-to-right", () => { + assert.deepEqual(expandRangeOrCell("A1:B2"), [ + { row: 0, col: 0 }, + { row: 0, col: 1 }, + { row: 1, col: 0 }, + { row: 1, col: 1 }, + ]); + }); + + it("expands a single-column range", () => { + assert.deepEqual(expandRangeOrCell("A1:A3"), [ + { row: 0, col: 0 }, + { row: 1, col: 0 }, + { row: 2, col: 0 }, + ]); + }); + + // The fill-down form. The old regex left the `$` in and matched nothing. + it("strips absolute-reference dollar signs", () => { + assert.deepEqual(expandRangeOrCell("$A$1:$A$3"), expandRangeOrCell("A1:A3")); + assert.deepEqual(expandRangeOrCell("$A1:A$3"), expandRangeOrCell("A1:A3")); + }); + + // Spreadsheets accept lowercase and upcase it; the old regex was `[A-Z]` + // only, so a lowercase range silently produced nothing. + it("upcases lowercase references", () => { + assert.deepEqual(expandRangeOrCell("a1:b2"), expandRangeOrCell("A1:B2")); + assert.deepEqual(expandRangeOrCell("$a$1:$a$3"), expandRangeOrCell("A1:A3")); + }); + + it("tolerates surrounding whitespace", () => { + assert.deepEqual(expandRangeOrCell(" A1:A2 "), expandRangeOrCell("A1:A2")); + }); + + it("crosses the Z→AA column boundary", () => { + assert.deepEqual(expandRangeOrCell("Z1:AA1"), [ + { row: 0, col: 25 }, + { row: 0, col: 26 }, + ]); + }); +}); + +describe("expandRangeOrCell — single cells", () => { + // The case Excel sums as one value and the old regex refused for lack of a + // colon. + it("expands a bare cell to one coordinate", () => { + assert.deepEqual(expandRangeOrCell("A1"), [{ row: 0, col: 0 }]); + assert.deepEqual(expandRangeOrCell("B3"), [{ row: 2, col: 1 }]); + }); + + it("strips dollar signs and upcases a single cell", () => { + assert.deepEqual(expandRangeOrCell("$A$1"), [{ row: 0, col: 0 }]); + assert.deepEqual(expandRangeOrCell("a1"), [{ row: 0, col: 0 }]); + }); + + it("reads a multi-letter column", () => { + assert.deepEqual(expandRangeOrCell("AA10"), [{ row: 9, col: 26 }]); + }); +}); + +describe("expandRangeOrCell — non-references", () => { + // Null rather than an empty array: the caller distinguishes "not a reference" + // from "a valid but empty range", and returning [] for garbage would hide + // typos as zero-value sums. + it("returns null for text that is not a reference", () => { + assert.equal(expandRangeOrCell("hello"), null); + assert.equal(expandRangeOrCell(""), null); + assert.equal(expandRangeOrCell("A"), null); + assert.equal(expandRangeOrCell("1"), null); + assert.equal(expandRangeOrCell("A1:B"), null); + assert.equal(expandRangeOrCell("A1:"), null); + }); +}); diff --git a/tests/engine/test_financialMath.ts b/tests/engine/test_financialMath.ts new file mode 100644 index 0000000..05db2eb --- /dev/null +++ b/tests/engine/test_financialMath.ts @@ -0,0 +1,188 @@ +// The per-period interest/principal split of an annuity. The bug this covers +// returned a plausible NUMBER — IPMT came back +1250 for a -1250 interest +// payment (sign inverted), and PPMT (= PMT - IPMT) amplified it to -2748.88 +// instead of -248.88 (#2386). Values are checked against Excel. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { + computeFv, + computePmt, + computeIpmt, + computePpmt, + computePv, + computeNper, + computeRate, + computeNpv, + computeIrr, +} from "../../src/engine/financial-math.ts"; +import { NUM_ERROR, type SpreadsheetError } from "../../src/engine/spreadsheet-errors.ts"; + +const closeTo = (actual: number, expected: number, eps = 0.01): boolean => Math.abs(actual - expected) <= eps; + +/** RATE / IRR may answer `#NUM!`; the convergent cases assert on the number. */ +const converged = (result: number | SpreadsheetError): number => { + if (typeof result !== "number") throw new Error(`expected a rate, got ${String(result)}`); + return result; +}; + +// A 250,000 loan at 0.5%/period over 360 periods — the issue's worked example. +const RATE = 0.005; +const NPER = 360; +const PRINCIPAL = 250000; + +describe("computePmt", () => { + it("matches Excel's constant payment (negative outflow)", () => { + assert.ok(closeTo(computePmt(RATE, NPER, PRINCIPAL, 0, 0), -1498.88), "PMT ≈ -1498.88"); + }); + + it("splits a zero-interest loan evenly", () => { + assert.ok(closeTo(computePmt(0, 10, 1000, 0, 0), -100, 1e-9), "zero-rate PMT"); + }); +}); + +describe("computeIpmt", () => { + it("returns the first period's interest with Excel's sign", () => { + // Interest on the full 250,000 balance: 250000 * 0.005 = 1250, as a payment + // it is negative. The bug returned +1250. + assert.ok(closeTo(computeIpmt(RATE, 1, NPER, PRINCIPAL, 0, 0), -1250, 1e-9), "IPMT(1) = -1250"); + }); + + it("decreases in magnitude as the balance is paid down", () => { + assert.ok(closeTo(computeIpmt(RATE, 2, NPER, PRINCIPAL, 0, 0), -1248.76), "IPMT(2) ≈ -1248.76"); + }); + + it("has no interest in the first period of an annuity due", () => { + assert.equal(computeIpmt(RATE, 1, NPER, PRINCIPAL, 0, 1), 0); + }); +}); + +describe("computePpmt", () => { + it("returns the first period's principal, not a wildly wrong value", () => { + // PMT - IPMT = -1498.88 - (-1250) = -248.88. The sign bug made this -2748.88. + assert.ok(closeTo(computePpmt(RATE, 1, NPER, PRINCIPAL, 0, 0), -248.88), "PPMT(1) ≈ -248.88"); + }); +}); + +describe("the interest and principal split reconstitutes the payment", () => { + it("IPMT(per) + PPMT(per) == PMT for every period", () => { + const pmt = computePmt(RATE, NPER, PRINCIPAL, 0, 0); + for (const per of [1, 2, 12, 180, 360]) { + const split = computeIpmt(RATE, per, NPER, PRINCIPAL, 0, 0) + computePpmt(RATE, per, NPER, PRINCIPAL, 0, 0); + assert.ok(closeTo(split, pmt, 1e-9), `period ${per}: IPMT + PPMT == PMT`); + } + }); +}); + +describe("computeNpv", () => { + const NPV_RATE = 0.1; + + // Each flow discounts by its 1-based POSITION in the flattened list. The #2390 + // bug used the argument index, so a scalar after a 3-cell range landed at + // period 2 instead of 4 — the position is what makes 100/1.1 + 200/1.1^2 + + // 300/1.1^3 + 500/1.1^4 correct. + it("discounts each flow by its 1-based position", () => { + const expected = 100 / 1.1 + 200 / 1.1 ** 2 + 300 / 1.1 ** 3 + 500 / 1.1 ** 4; + assert.ok(closeTo(computeNpv(NPV_RATE, [100, 200, 300, 500]), expected, 1e-9)); + }); + + it("sums flows undiscounted at a zero rate", () => { + assert.ok(closeTo(computeNpv(0, [100, 200, 300, 500]), 1100, 1e-9)); + }); + + it("is zero for no cash flows", () => { + assert.equal(computeNpv(NPV_RATE, []), 0); + }); + + it("discounts a single flow by one period", () => { + assert.ok(closeTo(computeNpv(NPV_RATE, [100]), 100 / 1.1, 1e-9)); + }); +}); + +describe("computeFv", () => { + it("carries the payment-negative sign the interest split relies on", () => { + // Balance outstanding at the start of period 1 is the present value, which + // FV expresses as its negative. + assert.ok(closeTo(computeFv(RATE, 0, computePmt(RATE, NPER, PRINCIPAL, 0, 0), PRINCIPAL, 0), -PRINCIPAL, 1e-9), "FV of pv over 0 periods = -pv"); + }); + + it("sums a zero-rate stream directly", () => { + assert.ok(closeTo(computeFv(0, 10, -100, 0, 0), 1000, 1e-9), "zero-rate FV"); + }); +}); + +// PV / NPER / RATE / NPV / IRR were inline in the handlers before this refactor; +// the values below were captured from the pre-refactor formulas (verbatim) and +// cross-checked against Excel, so they double as regression pins. + +describe("computePv", () => { + it("matches Excel's present value of an annuity", () => { + // Excel PV(0.05, 10, -1000) = 7721.73 (paying out 1000/period is a positive PV). + assert.ok(closeTo(computePv(0.05, 10, -1000, 0, 0), 7721.73), "PV ≈ 7721.73"); + }); + + it("discounts a zero-rate stream to its undiscounted total", () => { + assert.ok(closeTo(computePv(0, 10, -100, 0, 0), 1000, 1e-9), "zero-rate PV"); + }); + + it("is the inverse of PMT — it recovers the principal from that payment", () => { + const payment = computePmt(RATE, NPER, PRINCIPAL, 0, 0); + assert.ok(closeTo(computePv(RATE, NPER, payment, 0, 0), PRINCIPAL, 1e-6), "PV(PMT(pv)) == pv"); + }); +}); + +describe("computeNper", () => { + it("counts the periods needed to pay off a loan (Excel value)", () => { + // Excel NPER(0.05, -1000, 8000) = 10.47. + assert.ok(closeTo(computeNper(0.05, -1000, 8000, 0, 0), 10.47), "NPER ≈ 10.47"); + }); + + it("splits a zero-rate balance into equal periods", () => { + assert.ok(closeTo(computeNper(0, -100, 1000, 0, 0), 10, 1e-9), "zero-rate NPER"); + }); +}); + +describe("computeRate", () => { + it("recovers the rate implied by a known payment via Newton-Raphson", () => { + const payment = computePmt(0.05, 12, 1000, 0, 0); + assert.ok(closeTo(converged(computeRate(12, payment, 1000, 0, 0, 0.1)), 0.05, 1e-6), "RATE recovers 0.05"); + }); + + it("reports #NUM! instead of a divergent rate when no root exists", () => { + assert.equal(computeRate(10, 100, 100, 100, 0, 0.1), NUM_ERROR); + }); +}); + +describe("computeNpv", () => { + it("discounts each cash flow one period further out (Excel NPV)", () => { + // Excel NPV(0.1, -10000, 3000, 4200, 6800) = 1188.44 — the first flow is + // discounted one period, unlike IRR which places element 0 at period 0. + assert.ok(closeTo(computeNpv(0.1, [-10000, 3000, 4200, 6800]), 1188.44), "NPV ≈ 1188.44"); + }); + + it("discounts a single flow by exactly one period", () => { + assert.ok(closeTo(computeNpv(0.1, [100]), 90.9090909, 1e-6), "100 / 1.1"); + }); + + it("returns zero for an empty cash-flow series", () => { + assert.equal(computeNpv(0.1, []), 0); + }); +}); + +describe("computeIrr", () => { + it("finds the rate that zeroes the NPV (Excel IRR)", () => { + assert.ok(closeTo(converged(computeIrr([-100, 60, 60], 0.1)), 0.130662, 1e-5), "IRR ≈ 0.130662"); + }); + + it("drives the period-0 discounted cash flows to zero at the returned rate", () => { + const irr = converged(computeIrr([-1000, 500, 400, 300, 100], 0.1)); + const npvFromPeriodZero = [-1000, 500, 400, 300, 100].reduce((sum, value, index) => sum + value / (1 + irr) ** index, 0); + assert.ok(closeTo(npvFromPeriodZero, 0, 1e-6), "NPV at IRR ≈ 0"); + }); + + // Same-sign cash flows have no internal rate of return: the derivative collapses + // and there is nowhere to step, which Excel reports as #NUM!. + it("reports #NUM! when the cash flows never change sign", () => { + assert.equal(computeIrr([100, 200, 300], 0.1), NUM_ERROR); + }); +}); diff --git a/tests/engine/test_financialPeriodicHandlers.ts b/tests/engine/test_financialPeriodicHandlers.ts new file mode 100644 index 0000000..c758ea7 --- /dev/null +++ b/tests/engine/test_financialPeriodicHandlers.ts @@ -0,0 +1,34 @@ +// IPMT and PPMT now share one arg-parsing factory, makePeriodicComponentHandler +// (#2482). These drive both THROUGH the engine — the layer the factory lives in — +// so a swapped compute call or a mis-parsed optional arg (fv / type) is caught. +// The pure computeIpmt / computePpmt tests exercise the math, not the handler. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { SpreadsheetEngine, type SheetData } from "../../src/engine/index.ts"; +import { cellAt } from "./cellAccess.ts"; + +const evalA1 = (formula: string): unknown => { + const sheet: SheetData = { name: "S", data: [[{ v: formula }]] }; + return cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 0); +}; + +const closeTo = (actual: unknown, expected: number, eps = 0.01): boolean => typeof actual === "number" && Math.abs(actual - expected) <= eps; + +describe("IPMT / PPMT through the engine (shared handler factory)", () => { + it("IPMT(0.005, 1, 360, 250000) is the interest-only first payment (-1250)", () => { + assert.ok(closeTo(evalA1("=IPMT(0.005, 1, 360, 250000)"), -1250, 1e-6), `got ${String(evalA1("=IPMT(0.005, 1, 360, 250000)"))}`); + }); + + it("PPMT(0.005, 1, 360, 250000) is the principal-only first payment (~ -248.88)", () => { + assert.ok(closeTo(evalA1("=PPMT(0.005, 1, 360, 250000)"), -248.88), `got ${String(evalA1("=PPMT(0.005, 1, 360, 250000)"))}`); + }); + + it("IPMT and PPMT stay distinct — the factory did not collapse them onto one compute", () => { + assert.notEqual(evalA1("=IPMT(0.005, 2, 360, 250000)"), evalA1("=PPMT(0.005, 2, 360, 250000)")); + }); + + it("parses the optional type arg: IPMT at period 1, begin-of-period, is 0", () => { + assert.equal(evalA1("=IPMT(0.005, 1, 360, 250000, 0, 1)"), 0); + }); +}); diff --git a/tests/engine/test_formatter.ts b/tests/engine/test_formatter.ts new file mode 100644 index 0000000..937d74b --- /dev/null +++ b/tests/engine/test_formatter.ts @@ -0,0 +1,157 @@ +// Excel format codes → display strings. Nothing here throws: a bad format code +// produces a bad-looking cell, and a wrong decimal count produces a number that +// is simply off. Both read as ordinary output. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { addThousandSeparators, formatNumber } from "../../src/engine/formatter.ts"; +import { dateToSerial } from "../../src/engine/date-utils.ts"; + +const serialOf = (year: number, month: number, day: number, hour = 0, minute = 0, second = 0) => + dateToSerial(new Date(Date.UTC(year, month - 1, day, hour, minute, second))); + +const MAR_4_2025 = serialOf(2025, 3, 4); + +describe("formatNumber — numeric date formats", () => { + it("renders the padded and unpadded numeric orders", () => { + assert.equal(formatNumber(MAR_4_2025, "MM/DD/YYYY"), "03/04/2025"); + assert.equal(formatNumber(MAR_4_2025, "M/D/YYYY"), "3/4/2025"); + assert.equal(formatNumber(MAR_4_2025, "YYYY-MM-DD"), "2025-03-04"); + assert.equal(formatNumber(MAR_4_2025, "DD/MM/YYYY"), "04/03/2025"); + }); + + it("renders a two-digit year", () => { + assert.equal(formatNumber(MAR_4_2025, "MM/DD/YY"), "03/04/25"); + }); +}); + +describe("formatNumber — month-name formats", () => { + // The regression from #2330: substitution used to run as a sequence of + // replaces, so `/M/g` fired AFTER the month name had been inserted and + // rewrote the M inside "Mar" / "March" — and "March" then lost its "h" to + // the hour token, giving "3arc0". + it("renders month names intact", () => { + assert.equal(formatNumber(MAR_4_2025, "DD-MMM-YYYY"), "04-Mar-2025"); + assert.equal(formatNumber(MAR_4_2025, "MMM D, YYYY"), "Mar 4, 2025"); + assert.equal(formatNumber(MAR_4_2025, "MMMM D, YYYY"), "March 4, 2025"); + }); + + // These are the shapes `getDefaultDateFormat` hands back for the matching + // input, so a user typing "4-Mar-2025" gets this format applied to their own + // cell without asking for it. + it("round-trips the formats getDefaultDateFormat infers", () => { + assert.equal(formatNumber(serialOf(2025, 9, 30), "MMMM D, YYYY"), "September 30, 2025"); + assert.equal(formatNumber(serialOf(2025, 12, 1), "DD-MMM-YYYY"), "01-Dec-2025"); + }); + + it("renders weekday names", () => { + assert.equal(formatNumber(MAR_4_2025, "dddd"), "Tuesday"); + assert.equal(formatNumber(MAR_4_2025, "ddd"), "Tue"); + }); +}); + +describe("formatNumber — time formats", () => { + it("renders 24-hour time", () => { + assert.equal(formatNumber(serialOf(2025, 3, 4, 13, 45, 30), "HH:mm:ss"), "13:45:30"); + assert.equal(formatNumber(serialOf(2025, 3, 4, 9, 5, 0), "HH:mm"), "09:05"); + }); + + it("renders 12-hour time with a meridiem", () => { + assert.equal(formatNumber(serialOf(2025, 3, 4, 13, 45), "h:mm AM/PM"), "1:45 PM"); + assert.equal(formatNumber(serialOf(2025, 3, 4, 9, 5), "h:mm AM/PM"), "9:05 AM"); + }); + + // Midnight and noon are the two values a `% 12` gets wrong without the + // `|| 12` fallback. + it("renders midnight as 12 AM and noon as 12 PM", () => { + assert.equal(formatNumber(serialOf(2025, 3, 4, 0, 0), "h:mm AM/PM"), "12:00 AM"); + assert.equal(formatNumber(serialOf(2025, 3, 4, 12, 0), "h:mm AM/PM"), "12:00 PM"); + }); +}); + +describe("formatNumber — currency", () => { + it("renders a currency amount with separators and decimals", () => { + assert.equal(formatNumber(1234.5, "$#,##0.00"), "$1,234.50"); + assert.equal(formatNumber(1234.5, "$#,##0"), "$1,235"); + assert.equal(formatNumber(1234.5, "$0.00"), "$1234.50"); + }); + + it("groups every three digits", () => { + assert.equal(formatNumber(1234567.89, "$#,##0.00"), "$1,234,567.89"); + assert.equal(formatNumber(100, "$#,##0"), "$100"); + assert.equal(formatNumber(1000, "$#,##0"), "$1,000"); + }); + + // The sign goes outside the symbol: "-$1,000.00", not "$-1,000.00". + it("puts the minus sign before the currency symbol", () => { + assert.equal(formatNumber(-1000, "$#,##0.00"), "-$1,000.00"); + }); + + it("renders zero", () => { + assert.equal(formatNumber(0, "$#,##0.00"), "$0.00"); + }); +}); + +describe("formatNumber — percentage", () => { + it("multiplies by 100 and appends the sign", () => { + assert.equal(formatNumber(0.5, "0.0%"), "50.0%"); + assert.equal(formatNumber(0.1234, "0.00%"), "12.34%"); + assert.equal(formatNumber(1, "0.00%"), "100.00%"); + }); + + // The decimal count is read from a `.0+` run in the format. A format with no + // such run falls back to 2 for percentages while currency falls back to 0 — + // an asymmetry worth knowing about, since "0%" renders as "50.00%". + it("falls back to two decimals when the format declares none", () => { + assert.equal(formatNumber(0.5, "0%"), "50.00%"); + }); +}); + +describe("formatNumber — plain numbers", () => { + it("renders a fixed number of decimals", () => { + assert.equal(formatNumber(1234.5678, "0.00"), "1234.57"); + assert.equal(formatNumber(1234.5678, "0.000"), "1234.568"); + }); + + it("renders thousands separators without a currency symbol", () => { + assert.equal(formatNumber(1234567, "#,##0"), "1,234,567"); + assert.equal(formatNumber(1234.5, "#,##0.00"), "1,234.50"); + assert.equal(formatNumber(-1234.5, "#,##0.00"), "-1,234.50"); + }); + + it("returns the raw number when there is no format", () => { + assert.equal(formatNumber(1234.5, ""), "1234.5"); + }); + + // `#` placeholders are not read at all — only a literal `.0+` run sets the + // decimal count — so a format built from them is ignored entirely. + it("ignores a format whose decimals are written with # placeholders", () => { + assert.equal(formatNumber(0.5, "0.##"), "0.5"); + assert.equal(formatNumber(1234.5678, "0.###"), "1234.5678"); + }); + + it("reads the decimal count from the leading .0 run of a mixed format", () => { + assert.equal(formatNumber(0.5, "0.0#"), "0.5"); + }); +}); + +// The split/group/join wrapper the currency and plain-comma branches of +// formatNumber both share: group the integer part, leave any fraction alone. +describe("addThousandSeparators", () => { + it("groups the integer part in threes", () => { + assert.equal(addThousandSeparators("1000"), "1,000"); + assert.equal(addThousandSeparators("1234567"), "1,234,567"); + }); + + it("leaves the fractional part untouched", () => { + assert.equal(addThousandSeparators("1234567.89"), "1,234,567.89"); + assert.equal(addThousandSeparators("12.5"), "12.5"); + assert.equal(addThousandSeparators("0.50"), "0.50"); + }); + + it("passes short and empty integer parts through unchanged", () => { + assert.equal(addThousandSeparators("100"), "100"); + assert.equal(addThousandSeparators("999"), "999"); + assert.equal(addThousandSeparators(""), ""); + }); +}); diff --git a/tests/engine/test_formulaError.ts b/tests/engine/test_formulaError.ts new file mode 100644 index 0000000..ca224e4 --- /dev/null +++ b/tests/engine/test_formulaError.ts @@ -0,0 +1,94 @@ +// Pure taxonomy + classification helpers behind #2359's typed error reporting. +// These decide which Excel error value and CalculationError type a failure maps +// to; a wrong mapping silently mislabels a cell, so each direction is pinned. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { + FORMULA_ERROR_VALUES, + FormulaError, + isFormulaError, + divZeroError, + invalidRefError, + nameError, + unknownError, + propagatedError, + classifyThrownError, +} from "../../src/engine/formulaError.ts"; + +describe("FORMULA_ERROR_VALUES", () => { + it("maps each kind to its Excel literal", () => { + assert.deepEqual(FORMULA_ERROR_VALUES, { + div_zero: "#DIV/0!", + invalid_ref: "#REF!", + syntax: "#NAME?", + unknown: "#ERROR!", + }); + }); +}); + +describe("factories carry the matching kind and display", () => { + it("divZeroError", () => { + const error = divZeroError(); + assert.equal(error.errorType, "div_zero"); + assert.equal(error.display, "#DIV/0!"); + assert.equal(isFormulaError(error), true); + }); + + it("invalidRefError includes the reference in the message", () => { + const error = invalidRefError("Missing!A1"); + assert.equal(error.errorType, "invalid_ref"); + assert.equal(error.display, "#REF!"); + assert.match(error.message, /Missing!A1/); + }); + + it("nameError includes the function name in the message", () => { + const error = nameError("UNKNOWNFN"); + assert.equal(error.errorType, "syntax"); + assert.equal(error.display, "#NAME?"); + assert.match(error.message, /UNKNOWNFN/); + }); + + it("unknownError defaults its message to the display value", () => { + assert.equal(unknownError().message, "#ERROR!"); + assert.equal(unknownError("boom").message, "boom"); + assert.equal(unknownError().errorType, "unknown"); + }); +}); + +describe("isFormulaError", () => { + it("accepts a FormulaError and rejects anything else", () => { + assert.equal(isFormulaError(new FormulaError("unknown", "#ERROR!")), true); + assert.equal(isFormulaError(new Error("plain")), false); + assert.equal(isFormulaError("#DIV/0!"), false); + assert.equal(isFormulaError(null), false); + }); +}); + +describe("propagatedError maps a value back to its kind", () => { + it("keeps the dedicated kind for values that have one", () => { + assert.equal(propagatedError("#DIV/0!").errorType, "div_zero"); + assert.equal(propagatedError("#REF!").errorType, "invalid_ref"); + assert.equal(propagatedError("#NAME?").errorType, "syntax"); + }); + + it("falls back to unknown for values without a dedicated kind", () => { + assert.equal(propagatedError("#N/A").errorType, "unknown"); + assert.equal(propagatedError("#NUM!").errorType, "unknown"); + }); + + it("preserves the error value as the display", () => { + assert.equal(propagatedError("#N/A").display, "#N/A"); + }); +}); + +describe("classifyThrownError", () => { + it("passes a FormulaError's own kind and display through", () => { + assert.deepEqual(classifyThrownError(divZeroError()), { type: "div_zero", display: "#DIV/0!" }); + }); + + it("maps any non-FormulaError throw to unknown / #ERROR!", () => { + assert.deepEqual(classifyThrownError(new Error("SUM accepts at most 1 argument")), { type: "unknown", display: "#ERROR!" }); + assert.deepEqual(classifyThrownError("weird"), { type: "unknown", display: "#ERROR!" }); + }); +}); diff --git a/tests/engine/test_formulaRefs.ts b/tests/engine/test_formulaRefs.ts new file mode 100644 index 0000000..a300347 --- /dev/null +++ b/tests/engine/test_formulaRefs.ts @@ -0,0 +1,339 @@ +// Unit tests for the pure formula-reference scanner extracted from +// `src/plugins/spreadsheet/View.vue` (the original was 70 lines of +// inline regex + nested loops with cognitive complexity 32). +// +// Per CLAUDE.md's Testing requirements, covers: +// - Happy path +// - Edge cases (empty, single cell, ranges of every shape) +// - Corner cases (absolute $ refs, large row/col indices) +// - Boundary cases (last letter columns, max-int-ish rows) +// - Invalid / malformed inputs +// - Regression fixtures for the exact shapes View.vue passed to +// the original function. +// +// Tracks #175. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { stripFormulaPrefix, expandRange, parseSingleCellRef, extractCellReferences } from "../../src/engine/formulaRefs.js"; + +describe("stripFormulaPrefix", () => { + it("strips a leading =", () => { + assert.equal(stripFormulaPrefix("=A1+B2"), "A1+B2"); + }); + + it("leaves a formula without = untouched", () => { + assert.equal(stripFormulaPrefix("A1+B2"), "A1+B2"); + }); + + it("returns empty for empty", () => { + assert.equal(stripFormulaPrefix(""), ""); + }); + + it("only strips the first = (documenting behaviour)", () => { + assert.equal(stripFormulaPrefix("==A1"), "=A1"); + }); + + it("handles a bare =", () => { + assert.equal(stripFormulaPrefix("="), ""); + }); +}); + +describe("expandRange", () => { + it("expands a small square range", () => { + assert.deepEqual(expandRange("A1:B2"), [ + { row: 0, col: 0 }, + { row: 0, col: 1 }, + { row: 1, col: 0 }, + { row: 1, col: 1 }, + ]); + }); + + it("expands a single-cell range (A1:A1)", () => { + assert.deepEqual(expandRange("A1:A1"), [{ row: 0, col: 0 }]); + }); + + it("expands a single-row range (A1:C1)", () => { + assert.deepEqual(expandRange("A1:C1"), [ + { row: 0, col: 0 }, + { row: 0, col: 1 }, + { row: 0, col: 2 }, + ]); + }); + + it("expands a single-column range (A1:A3)", () => { + assert.deepEqual(expandRange("A1:A3"), [ + { row: 0, col: 0 }, + { row: 1, col: 0 }, + { row: 2, col: 0 }, + ]); + }); + + it("strips $ on absolute refs ($A$1:$B$2)", () => { + assert.deepEqual(expandRange("$A$1:$B$2"), expandRange("A1:B2")); + }); + + it("handles partial absolute refs ($A1:B$2)", () => { + assert.deepEqual(expandRange("$A1:B$2"), expandRange("A1:B2")); + }); + + it("handles multi-letter columns (Z1:AA1)", () => { + // Z = col 25, AA = col 26 + assert.deepEqual(expandRange("Z1:AA1"), [ + { row: 0, col: 25 }, + { row: 0, col: 26 }, + ]); + }); + + it("returns [] for a reversed range (loops fall through)", () => { + // B2 to A1 — start > end, for-loops iterate 0 times. Matches + // the original inline behaviour. + assert.deepEqual(expandRange("B2:A1"), []); + }); + + it("returns [] for malformed input: no colon", () => { + assert.deepEqual(expandRange("A1"), []); + }); + + it("returns [] for malformed input: junk", () => { + assert.deepEqual(expandRange("foo:bar"), []); + }); + + it("returns [] for empty string", () => { + assert.deepEqual(expandRange(""), []); + }); + + it("returns [] for missing column letters", () => { + assert.deepEqual(expandRange("1:2"), []); + }); +}); + +describe("parseSingleCellRef", () => { + it("parses A1 (col 0, row 0)", () => { + assert.deepEqual(parseSingleCellRef("A1"), { row: 0, col: 0 }); + }); + + it("parses B3 (col 1, row 2)", () => { + assert.deepEqual(parseSingleCellRef("B3"), { row: 2, col: 1 }); + }); + + it("parses absolute $A$1 (same as A1)", () => { + assert.deepEqual(parseSingleCellRef("$A$1"), { row: 0, col: 0 }); + }); + + it("parses partial absolute $A1 and A$1", () => { + assert.deepEqual(parseSingleCellRef("$A1"), { row: 0, col: 0 }); + assert.deepEqual(parseSingleCellRef("A$1"), { row: 0, col: 0 }); + }); + + it("parses multi-letter column AA1 (col 26)", () => { + assert.deepEqual(parseSingleCellRef("AA1"), { row: 0, col: 26 }); + }); + + it("parses large row A100 (row 99)", () => { + assert.deepEqual(parseSingleCellRef("A100"), { row: 99, col: 0 }); + }); + + it("parses the last single-letter column Z1 (col 25)", () => { + assert.deepEqual(parseSingleCellRef("Z1"), { row: 0, col: 25 }); + }); + + it("returns null for lowercase (regex is case-sensitive)", () => { + assert.equal(parseSingleCellRef("a1"), null); + }); + + it("returns null for row-then-col order (1A)", () => { + assert.equal(parseSingleCellRef("1A"), null); + }); + + it("returns null for empty string", () => { + assert.equal(parseSingleCellRef(""), null); + }); + + it("returns null for garbage", () => { + assert.equal(parseSingleCellRef("not-a-ref"), null); + }); + + it("returns null when the row part is missing", () => { + assert.equal(parseSingleCellRef("A"), null); + }); + + it("returns null when the col part is missing", () => { + assert.equal(parseSingleCellRef("42"), null); + }); +}); + +describe("extractCellReferences — happy path", () => { + it("returns [] for empty formula", () => { + assert.deepEqual(extractCellReferences(""), []); + }); + + it("returns [] for a formula with no cell refs (only literals)", () => { + assert.deepEqual(extractCellReferences("=123+456"), []); + }); + + it("picks up a single cell ref", () => { + assert.deepEqual(extractCellReferences("=A1"), [{ row: 0, col: 0 }]); + }); + + it("picks up multiple single cell refs in order", () => { + assert.deepEqual(extractCellReferences("=A1+B2+C3"), [ + { row: 0, col: 0 }, + { row: 1, col: 1 }, + { row: 2, col: 2 }, + ]); + }); + + it("works without a leading = prefix", () => { + // The scanner is also used on partial text (during live edit). + assert.deepEqual(extractCellReferences("A1+B2"), [ + { row: 0, col: 0 }, + { row: 1, col: 1 }, + ]); + }); + + it("picks up absolute references", () => { + assert.deepEqual(extractCellReferences("=$A$1+$B2+C$3"), [ + { row: 0, col: 0 }, + { row: 1, col: 1 }, + { row: 2, col: 2 }, + ]); + }); + + it("works with common function syntax", () => { + assert.deepEqual(extractCellReferences("=SUM(A1, B2, C3)"), [ + { row: 0, col: 0 }, + { row: 1, col: 1 }, + { row: 2, col: 2 }, + ]); + }); +}); + +describe("extractCellReferences — range handling", () => { + it("expands a SUM(A1:B2) range into 4 cells", () => { + assert.deepEqual(extractCellReferences("=SUM(A1:B2)"), [ + { row: 0, col: 0 }, + { row: 0, col: 1 }, + { row: 1, col: 0 }, + { row: 1, col: 1 }, + ]); + }); + + it("does NOT emit the range endpoints as standalone cells", () => { + // The original code strips matched ranges before running the + // cell regex — regression pin so a future refactor doesn't + // accidentally double-count A1 and B2. + const refs = extractCellReferences("=A1:B2"); + // Should be exactly 4 cells from the range expansion, not 6 + // (4 range + 2 endpoints). + assert.equal(refs.length, 4); + }); + + it("combines a range with standalone cells", () => { + assert.deepEqual(extractCellReferences("=SUM(A1:A2)+C5"), [ + { row: 0, col: 0 }, + { row: 1, col: 0 }, + { row: 4, col: 2 }, + ]); + }); + + it("expands multiple ranges in one formula", () => { + const refs = extractCellReferences("=SUM(A1:A2)+SUM(C1:C2)"); + assert.equal(refs.length, 4); + assert.deepEqual(refs.slice().sort(cmpCoord), [ + { row: 0, col: 0 }, + { row: 0, col: 2 }, + { row: 1, col: 0 }, + { row: 1, col: 2 }, + ]); + }); + + it("handles absolute-reference ranges", () => { + assert.deepEqual(extractCellReferences("=SUM($A$1:$B$2)"), [ + { row: 0, col: 0 }, + { row: 0, col: 1 }, + { row: 1, col: 0 }, + { row: 1, col: 1 }, + ]); + }); +}); + +describe("extractCellReferences — deduplication", () => { + it("drops duplicate standalone cells", () => { + assert.deepEqual(extractCellReferences("=A1+A1+A1"), [{ row: 0, col: 0 }]); + }); + + it("drops duplicate cells across absolute / relative forms", () => { + // A1 and $A$1 refer to the same cell; the scanner normalises by + // stripping $, so only one entry appears. + assert.deepEqual(extractCellReferences("=A1+$A$1"), [{ row: 0, col: 0 }]); + }); + + it("drops cells already covered by a range", () => { + const refs = extractCellReferences("=SUM(A1:B2)+A1+B2"); + // Range contributes 4 cells; A1 and B2 are already among them. + assert.equal(refs.length, 4); + }); + + it("preserves first-occurrence order of unique cells", () => { + assert.deepEqual(extractCellReferences("=C3+A1+B2+C3+A1"), [ + { row: 2, col: 2 }, + { row: 0, col: 0 }, + { row: 1, col: 1 }, + ]); + }); +}); + +describe("extractCellReferences — malformed / edge input", () => { + it("ignores lowercase (regex is case-sensitive, matching Excel)", () => { + assert.deepEqual(extractCellReferences("=a1+b2"), []); + }); + + it("ignores partial tokens", () => { + // `A` alone isn't a ref, neither is `1`; no digits adjacent + // to letters means no match. + assert.deepEqual(extractCellReferences("=A + 1"), []); + }); + + it("doesn't match numbers embedded in text without letters", () => { + assert.deepEqual(extractCellReferences("=(100/4)"), []); + }); + + it("handles a formula of only an = sign", () => { + assert.deepEqual(extractCellReferences("="), []); + }); + + it("picks up cells inside parentheses and operators", () => { + assert.deepEqual(extractCellReferences("=((A1+B2)*C3)"), [ + { row: 0, col: 0 }, + { row: 1, col: 1 }, + { row: 2, col: 2 }, + ]); + }); +}); + +describe("extractCellReferences — boundary / precision", () => { + it("handles triple-letter columns (AAA1 = col 702)", () => { + // A=0, Z=25, AA=26, AZ=51, BA=52, ZZ=701, AAA=702 + assert.deepEqual(extractCellReferences("=AAA1"), [{ row: 0, col: 702 }]); + }); + + it("handles row numbers near Excel's 2^20 limit", () => { + // Excel's max row is 1048576. Our parser doesn't enforce that + // cap (and shouldn't — it's a pure scanner) but it should + // still produce a valid integer. + const refs = extractCellReferences("=A1048576"); + assert.equal(refs.length, 1); + const [ref] = refs; + assert.ok(ref); + assert.equal(ref.row, 1048575); + assert.equal(ref.col, 0); + }); +}); + +// --- helpers --- + +function cmpCoord(coordA: { row: number; col: number }, coordB: { row: number; col: number }): number { + if (coordA.row !== coordB.row) return coordA.row - coordB.row; + return coordA.col - coordB.col; +} diff --git a/tests/engine/test_ifBranchEvaluation.ts b/tests/engine/test_ifBranchEvaluation.ts new file mode 100644 index 0000000..d980a8f --- /dev/null +++ b/tests/engine/test_ifBranchEvaluation.ts @@ -0,0 +1,61 @@ +// What IF does with the branch it picks. Both bugs here returned a plausible +// value instead of an error, so a sheet looked fine while holding wrong data: +// a hard-coded list of nine function names meant every OTHER nested call came +// back as its own text (`ROUND(A1,1)` → the string "ROUND(4.567,1)" — IF's own +// registered example did not work), and the fallback read an arithmetic branch +// through `parseFloat("3+1")`, yielding 3 (#2360). + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { SpreadsheetEngine, type SheetData } from "../../src/engine/index.ts"; +import { cellAt } from "./cellAccess.ts"; + +const evaluate = (rows: (string | number)[][], row: number, col: number): unknown => { + const sheet: SheetData = { name: "S", data: rows.map((cells) => cells.map((value) => ({ v: value }))) }; + return cellAt(new SpreadsheetEngine().calculate(sheet).data, row, col); +}; + +describe("IF evaluates a nested function branch, whatever the function", () => { + // ROUND was outside the old whitelist, so this returned "ROUND(4.567,1)". + it("evaluates a function that the old whitelist omitted", () => { + assert.equal(evaluate([[4.567, "=IF(A1>0, ROUND(A1,1), 0)"]], 0, 1), 4.6); + }); + + it("evaluates a text function branch", () => { + assert.equal(evaluate([["hi", '=IF(TRUE, UPPER(A1), "x")']], 0, 1), "HI"); + }); + + it("still evaluates the functions the whitelist did cover", () => { + assert.equal(evaluate([[1, "=IF(A1>0, SUM(A1:A1), 0)"]], 0, 1), 1); + }); + + it("evaluates a nested IF", () => { + assert.equal(evaluate([[5, '=IF(A1>10, "big", IF(A1>3, "mid", "small"))']], 0, 1), "mid"); + }); + + it("takes the false branch without evaluating the true one", () => { + assert.equal(evaluate([[0, "=IF(A1>0, ROUND(9.99,1), 0)"]], 0, 1), 0); + }); +}); + +describe("IF evaluates an arithmetic branch", () => { + // The fallback substituted refs then called parseFloat, which stops at the + // operator: parseFloat("3+1") is 3. + it("computes a reference plus a literal", () => { + assert.equal(evaluate([[3, "=IF(A1>0, A1+1, 0)"]], 0, 1), 4); + }); + + it("returns a bare reference's value", () => { + assert.equal(evaluate([[7, 0, "=IF(A1>0, A1, B1)"]], 0, 2), 7); + }); + + it("returns a numeric literal branch", () => { + assert.equal(evaluate([[3, "=IF(A1>0, 42, 0)"]], 0, 1), 42); + }); +}); + +describe("IF still unwraps a quoted string branch", () => { + it("returns the text without its quotes", () => { + assert.equal(evaluate([[3, '=IF(A1>0, "yes", "no")']], 0, 1), "yes"); + }); +}); diff --git a/tests/engine/test_ifsInjection.ts b/tests/engine/test_ifsInjection.ts new file mode 100644 index 0000000..5ec95d4 --- /dev/null +++ b/tests/engine/test_ifsInjection.ts @@ -0,0 +1,221 @@ +// IFS reaching the condition evaluator rather than a JS engine. +// +// `test_condition.ts` covers the evaluator on its own; this file drives the +// whole path a user's data actually takes — a value typed into a cell, or text +// written into the formula — because that is what made #2360 reachable rather +// than theoretical. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { SpreadsheetEngine, type SheetData } from "../../src/engine/index.ts"; +import { cellAt } from "./cellAccess.ts"; + +const marker = globalThis as Record; + +const calculate = (cellValue: string, formula: string): unknown => + cellAt(new SpreadsheetEngine().calculate({ name: "S", data: [[{ v: cellValue }, { v: formula }]] } satisfies SheetData).data, 0, 1); + +describe("IFS — normal use", () => { + it("returns the first matching branch", () => { + assert.equal(calculate("5", '=IFS(A1>3, "big", A1>0, "small")'), "big"); + }); + + it("falls through to a later branch", () => { + assert.equal(calculate("1", '=IFS(A1>3, "big", A1>0, "small")'), "small"); + }); + + it("returns #N/A when no branch matches", () => { + assert.equal(calculate("-1", '=IFS(A1>3, "big", A1>0, "small")'), "#N/A"); + }); + + it("compares against text", () => { + assert.equal(calculate("Yes", '=IFS(A1="Yes", "confirmed", A1="No", "declined")'), "confirmed"); + assert.equal(calculate("No", '=IFS(A1="Yes", "confirmed", A1="No", "declined")'), "declined"); + }); + + // `IFS((A1>0), ...)` is ordinary usage; the parser used to split the + // parenthesised form into two text operands that never matched. + it("accepts a parenthesised condition", () => { + assert.equal(calculate("5", '=IFS((A1>3), "big")'), "big"); + assert.equal(calculate("1", '=IFS((A1>3), "big", (A1>0), "small")'), "small"); + assert.equal(calculate("5", '=IFS(((A1>3)), "big")'), "big"); + }); + + it("handles the boundary operators", () => { + assert.equal(calculate("3", '=IFS(A1>=3, "atLeast3")'), "atLeast3"); + assert.equal(calculate("3", '=IFS(A1>3, "over3")'), "#N/A"); + }); +}); + +describe("IFS — a cell's contents are data, not code", () => { + // Typing this string into a cell used to execute it, because the cell value + // was substituted into the condition and the result handed to `eval`. + it("does not execute an assignment stored in a cell", () => { + marker.__ifsProbe = false; + calculate("globalThis.__ifsProbe=true", '=IFS(A1>0, "hit")'); + assert.equal(marker.__ifsProbe, false, "the cell's contents must not run"); + }); + + it("does not execute a cell used as a bare condition", () => { + marker.__ifsProbe2 = false; + calculate("globalThis.__ifsProbe2=true", '=IFS(A1, "hit")'); + assert.equal(marker.__ifsProbe2, false); + }); + + // A payload starting with a digit was already neutralised by accident — + // `getRawValue` reads its numeric prefix — so it is NOT evidence the hole is + // closed. Pinned so nobody mistakes it for coverage. + it("also refuses a payload whose numeric prefix used to mask it", () => { + marker.__ifsProbe3 = false; + calculate("1)||(globalThis.__ifsProbe3=true", '=IFS(A1>0, "hit")'); + assert.equal(marker.__ifsProbe3, false); + }); +}); + +describe("IFS — a text cell's operators are data, not syntax", () => { + // A cell holding `x>y` used to substitute as bare `x>y`, so `A1="x>y"` became + // `x>y="x>y"` and never matched. Quoting the operand fixes it (Codex review). + it("compares against a cell whose text contains operators", () => { + assert.equal(calculate("x>y", '=IFS(A1="x>y", "match", TRUE, "no")'), "match"); + assert.equal(calculate("x>y", '=IFS(A1="other", "match", TRUE, "no")'), "no"); + }); + + it("treats a bare operator-bearing cell as truthy text, not a comparison", () => { + assert.equal(calculate("x>y", '=IFS(A1, "truthy", TRUE, "no")'), "truthy"); + assert.equal(calculate("", '=IFS(A1, "truthy", TRUE, "empty")'), "empty"); + }); + + // A cell holding a quote must not corrupt the comparison: `renderConditionOperand` + // escapes it, and the condition parser tracks the escape. + it("compares a quote-bearing cell without corrupting the parse", () => { + assert.equal(calculate('a"b', '=IFS(A1="z", "match", TRUE, "no")'), "no", 'a"b is not z'); + assert.equal(calculate('a"b', '=IFS(A1, "truthy", TRUE, "no")'), "truthy", "still non-empty text"); + }); +}); + +describe("IFS — operator characters inside a cell value stay data", () => { + // A cell holding `x>y` renders into the condition as a quoted literal, and the + // condition parser skips quoted regions when looking for the operator — so the + // inner `>` is never read as a comparison (Codex review flagged this path). + it("treats a bare reference to a string with an operator as truthy text", () => { + assert.equal(calculate("x>y", '=IFS(A1, "hit")'), "hit"); + }); + + it("compares equal against a string literal that contains an operator", () => { + assert.equal(calculate("x>y", '=IFS(A1="x>y", "hit")'), "hit"); + }); + + it("does not match when the operator-bearing strings differ", () => { + assert.equal(calculate("x>y", '=IFS(A1="a>b", "hit")'), "#N/A"); + }); +}); + +describe("IFS — absolute and mixed references resolve", () => { + // The ref used to be escaped twice before the RegExp, so `$A$1` never matched + // and was left as literal text in the condition (Codex review). + it("substitutes an absolute reference", () => { + assert.equal(calculate("5", '=IFS($A$1>0, "hit")'), "hit"); + assert.equal(calculate("5", '=IFS($A$1>10, "hit")'), "#N/A"); + }); + + it("substitutes a mixed reference", () => { + assert.equal(calculate("5", '=IFS(A$1>0, "hit")'), "hit"); + assert.equal(calculate("5", '=IFS($A1>0, "hit")'), "hit"); + }); +}); + +describe("IFS — a cell value ending in a backslash substitutes intact", () => { + // CodeQL js/double-escaping. `renderConditionOperand` escapes the backslash so + // a trailing `\` cannot escape the closing quote and corrupt the literal; + // substituting by position (not a regex) keeps it exactly once. A cell value + // `a\` is the sharp case — an unescaped one turns `"a\"` into an open literal. + const trailingBackslashSheet = (other: string): SheetData => ({ + name: "S", + data: [ + [{ v: "a\\" }, { v: '=IFS(A1=A2, "eq", TRUE, "ne")' }], + [{ v: other }, { v: 0 }], + ], + }); + + it("matches two cells that both end in a backslash", () => { + assert.equal(cellAt(new SpreadsheetEngine().calculate(trailingBackslashSheet("a\\")).data, 0, 1), "eq"); + }); + + it("does not match a backslash cell against different text", () => { + assert.equal(cellAt(new SpreadsheetEngine().calculate(trailingBackslashSheet("ab")).data, 0, 1), "ne"); + }); + + it("treats a bare backslash-bearing cell as truthy text", () => { + assert.equal(calculate("a\\b", '=IFS(A1, "truthy", TRUE, "no")'), "truthy"); + }); +}); + +describe("IFS — a reference inside a string literal stays literal text", () => { + // `A1="B2"` compares A1 to the TEXT "B2". The `"B2"` must not be read as a + // reference and replaced with cell B2's value (Codex review). + it("does not substitute a ref that sits inside quotes", () => { + const sheet: SheetData = { + name: "S", + data: [ + [{ v: "B2" }, { v: '=IFS(A1="B2", "hit")' }], + [{ v: 0 }, { v: 99 }], + ], + }; + assert.equal(cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 1), "hit", "A1's text equals the literal B2"); + }); +}); + +describe("IFS — bare TRUE/FALSE literals are booleans", () => { + // `evaluateFormula("FALSE")` returns the non-empty string "FALSE" (truthy); + // a bare logical literal in a condition must stay a boolean (Codex review). + it("skips a FALSE condition and matches a later TRUE", () => { + assert.equal(calculate("5", '=IFS(FALSE, "hit", TRUE, "miss")'), "miss"); + }); + + it("matches a bare TRUE condition", () => { + assert.equal(calculate("5", '=IFS(TRUE, "hit")'), "hit"); + }); + + it("uses TRUE as a catch-all after a false comparison", () => { + assert.equal(calculate("5", '=IFS(A1>10, "hit", TRUE, "miss")'), "miss"); + }); +}); + +describe("IFS — arithmetic operands are computed, not compared as text", () => { + // Removing eval left `A1+1>10` read as the string "5+1" vs 10, which flipped + // the branch. Each operand is now resolved by the engine's safe evaluator so + // the arithmetic is computed (Codex review). + it("computes an arithmetic left operand", () => { + assert.equal(calculate("5", '=IFS(A1+1>10, "hit", TRUE, "miss")'), "miss", "6 > 10 is false"); + assert.equal(calculate("5", '=IFS(A1+1>5, "hit", TRUE, "miss")'), "hit", "6 > 5 is true"); + }); + + it("computes arithmetic on both sides", () => { + assert.equal(calculate("4", '=IFS(A1*2 > 3+3, "hit", TRUE, "miss")'), "hit", "8 > 6 is true"); + }); +}); + +describe("IFS — the formula itself is data too", () => { + it("does not execute an expression written into the condition", () => { + marker.__ifsProbe4 = false; + calculate("1", '=IFS(A1>0&&(globalThis.__ifsProbe4=true), "hit")'); + assert.equal(marker.__ifsProbe4, false); + }); + + it("does not execute a call in the condition", () => { + marker.__ifsProbe5 = false; + calculate("1", '=IFS((globalThis.__ifsProbe5=true)>0, "hit")'); + assert.equal(marker.__ifsProbe5, false); + }); + + // A crash here would take the whole sheet's calculation with it. Broken + // input degrades to the formula text — that is the engine's existing + // swallow-everything behaviour (#2359), not something this change decides; + // what matters here is that it neither throws nor runs. + it("survives a syntactically broken condition", () => { + marker.__ifsProbe6 = false; + const result = calculate("1", '=IFS(((((globalThis.__ifsProbe6=true, "hit")'); + assert.equal(typeof result, "string"); + assert.equal(marker.__ifsProbe6, false); + }); +}); diff --git a/tests/engine/test_jsonCellLocator.ts b/tests/engine/test_jsonCellLocator.ts new file mode 100644 index 0000000..957ab9f --- /dev/null +++ b/tests/engine/test_jsonCellLocator.ts @@ -0,0 +1,118 @@ +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { findCellJsonPosition } from "../../src/engine/jsonCellLocator.js"; + +describe("findCellJsonPosition", () => { + const sample = JSON.stringify( + [ + { + name: "Sheet1", + data: [ + [{ v: "A1" }, { v: "B1" }, { v: 42 }], + [{ v: "A2" }, { v: "B2" }, { v: "=SUM(A1:B1)" }], + ], + }, + { + name: "Sheet2", + data: [[{ v: "X" }, { v: "Y" }]], + }, + ], + null, + 2, + ); + + // The locator returns the character offset of the cell's opening + // `{`. In a pretty-printed document, following lines expand the + // object so we only assert the starting char and that the + // substring contains the unique cell value. + function assertCellAt(pos: number, expectedSubstring: string) { + assert.ok(pos > 0, `expected a positive offset, got ${pos}`); + assert.equal(sample[pos], "{", `expected offset to land on '{', got '${sample[pos]}'`); + assert.ok(sample.substring(pos).includes(expectedSubstring), `expected substring ${JSON.stringify(expectedSubstring)} after position ${pos}`); + } + + it("locates cell (0,0) in the first sheet", () => { + assertCellAt(findCellJsonPosition(sample, "Sheet1", 0, 0), `"v": "A1"`); + }); + + it("locates cell (0,2) — the last column in row 0", () => { + assertCellAt(findCellJsonPosition(sample, "Sheet1", 0, 2), `"v": 42`); + }); + + it("locates cell (1,2) with a formula value", () => { + assertCellAt(findCellJsonPosition(sample, "Sheet1", 1, 2), `"v": "=SUM(A1:B1)"`); + }); + + it("locates cell in a non-first sheet by name", () => { + assertCellAt(findCellJsonPosition(sample, "Sheet2", 0, 1), `"v": "Y"`); + }); + + it("returns -1 for a sheet name that does not exist", () => { + assert.equal(findCellJsonPosition(sample, "Unknown", 0, 0), -1); + }); + + it("returns -1 for a row index past the end", () => { + assert.equal(findCellJsonPosition(sample, "Sheet1", 99, 0), -1); + }); + + it("returns -1 on completely empty input", () => { + assert.equal(findCellJsonPosition("", "Sheet1", 0, 0), -1); + }); + + it("handles strings containing brackets and commas without miscounting", () => { + const tricky = JSON.stringify( + [ + { + name: "Sheet1", + data: [[{ v: "has [bracket], and comma" }, { v: "second" }]], + }, + ], + null, + 2, + ); + const pos = findCellJsonPosition(tricky, "Sheet1", 0, 1); + assert.ok(pos > 0); + assert.equal(tricky[pos], "{"); + assert.ok(tricky.substring(pos).includes(`"v": "second"`)); + }); + + it("picks the correct row when an earlier row contains '[' inside a string", () => { + // Row 0 cell 0 contains a literal '[' — the naive counter would + // treat that as an extra row opener and shift all subsequent + // rowIndex lookups by one. + const withBracketInRow0 = JSON.stringify( + [ + { + name: "Sheet1", + data: [ + [{ v: "row0 has [bracket]" }, { v: "r0c1" }], + [{ v: "r1c0" }, { v: "TARGET" }], + ], + }, + ], + null, + 2, + ); + const pos = findCellJsonPosition(withBracketInRow0, "Sheet1", 1, 1); + assert.ok(pos > 0); + assert.ok(withBracketInRow0.substring(pos).includes(`"v": "TARGET"`)); + }); + + it("finds a sheet whose name contains a quote character", () => { + // Sheet names with embedded `"` need JSON-escaping when building + // the text marker, otherwise indexOf misses them entirely. + const text = JSON.stringify( + [ + { + name: 'Sheet "Q1"', + data: [[{ v: "FOUND" }]], + }, + ], + null, + 2, + ); + const pos = findCellJsonPosition(text, 'Sheet "Q1"', 0, 0); + assert.ok(pos > 0, `expected to locate the sheet, got ${pos}`); + assert.ok(text.substring(pos).includes(`"v": "FOUND"`)); + }); +}); diff --git a/tests/engine/test_locateSubstring.ts b/tests/engine/test_locateSubstring.ts new file mode 100644 index 0000000..58835fe --- /dev/null +++ b/tests/engine/test_locateSubstring.ts @@ -0,0 +1,52 @@ +// locateSubstring folds the shared body of FIND (case-sensitive) and SEARCH +// (case-insensitive) (#2482): identical 0-based start, identical 1-based hit +// index, identical #VALUE! miss. Case folding is the ONLY axis that may differ, +// so these pin both the common rule and that one deliberate asymmetry. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { locateSubstring } from "../../src/engine/functions/text.ts"; +import { isSpreadsheetErrorValue } from "../../src/engine/spreadsheet-errors.ts"; + +const CASE_SENSITIVE = { caseInsensitive: false }; +const CASE_INSENSITIVE = { caseInsensitive: true }; + +const missed = (result: ReturnType): boolean => isSpreadsheetErrorValue(result) && result.code === "#VALUE!"; + +describe("locateSubstring — 1-based hit index", () => { + it("returns the 1-based position of the first match", () => { + assert.equal(locateSubstring("b", "abc", 0, CASE_SENSITIVE), 2); + }); + + it("finds a match at the very start", () => { + assert.equal(locateSubstring("a", "abc", 0, CASE_SENSITIVE), 1); + }); + + it("honours a non-zero start, skipping an earlier match", () => { + assert.equal(locateSubstring("a", "banana", 2, CASE_SENSITIVE), 4); + }); +}); + +describe("locateSubstring — case sensitivity is the only difference", () => { + it("case-sensitive: a wrong-case needle misses with #VALUE!", () => { + assert.ok(missed(locateSubstring("O", "hello", 0, CASE_SENSITIVE))); + }); + + it("case-insensitive: the same wrong-case needle matches", () => { + assert.equal(locateSubstring("O", "hello", 0, CASE_INSENSITIVE), 5); + }); + + it("case-insensitive folds BOTH the needle and the haystack", () => { + assert.equal(locateSubstring("HELLO", "hello world", 0, CASE_INSENSITIVE), 1); + }); +}); + +describe("locateSubstring — misses and edge cases", () => { + it("a needle absent from the haystack is #VALUE!", () => { + assert.ok(missed(locateSubstring("z", "abc", 0, CASE_SENSITIVE))); + }); + + it("an empty needle matches at position 1 (indexOf semantics preserved)", () => { + assert.equal(locateSubstring("", "abc", 0, CASE_SENSITIVE), 1); + }); +}); diff --git a/tests/engine/test_logicalFunctions.ts b/tests/engine/test_logicalFunctions.ts new file mode 100644 index 0000000..9032bb2 --- /dev/null +++ b/tests/engine/test_logicalFunctions.ts @@ -0,0 +1,76 @@ +// Boolean coercion shared by the logical functions. The bug this covers: IF and +// AND/OR read the SAME value oppositely — IF("0") took the true branch while +// AND("0") was false — because each function coerced truthiness its own way +// (#2387). One shared rule keeps them in agreement. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { SpreadsheetEngine, type SheetData } from "../../src/engine/index.ts"; +import { coerceToBoolean } from "../../src/engine/coerce-boolean.ts"; +import { cellAt } from "./cellAccess.ts"; + +describe("coerceToBoolean", () => { + it("passes booleans through", () => { + assert.equal(coerceToBoolean(true), true); + assert.equal(coerceToBoolean(false), false); + }); + + it("treats only 0 as false among numbers", () => { + assert.equal(coerceToBoolean(0), false); + assert.equal(coerceToBoolean(1), true); + assert.equal(coerceToBoolean(-1), true); + assert.equal(coerceToBoolean(0.5), true); + }); + + it("treats blank and empty as false", () => { + assert.equal(coerceToBoolean(""), false); + assert.equal(coerceToBoolean(" "), false); + assert.equal(coerceToBoolean(null), false); + assert.equal(coerceToBoolean(undefined), false); + }); + + it("reads the words true/false case-insensitively", () => { + assert.equal(coerceToBoolean("true"), true); + assert.equal(coerceToBoolean("TRUE"), true); + assert.equal(coerceToBoolean("false"), false); + assert.equal(coerceToBoolean("False"), false); + }); + + // The crux of #2387: a numeric string follows its number, so "0" is false in + // every logical function — not true in IF and false in AND. + it("follows the number in a numeric string", () => { + assert.equal(coerceToBoolean("0"), false); + assert.equal(coerceToBoolean("0.0"), false); + assert.equal(coerceToBoolean("5"), true); + assert.equal(coerceToBoolean("-3"), true); + }); + + it("treats other non-empty text as true", () => { + assert.equal(coerceToBoolean("hello"), true); + assert.equal(coerceToBoolean("no"), true); + }); +}); + +describe("IF and AND/OR/NOT agree on the same value", () => { + const evalFormula = (formula: string): unknown => + cellAt(new SpreadsheetEngine().calculate({ name: "S", data: [[{ v: formula }]] } satisfies SheetData).data, 0, 0); + + // Each value should send IF down the false branch exactly when AND/OR/NOT read + // it as false. Previously IF("0") returned 1 while AND("0") returned false. + for (const [literal, truthy] of [ + ['"0"', false], + ['"false"', false], + ['""', false], + ["0", false], + ['"5"', true], + ['"hello"', true], + ["1", true], + ] as const) { + it(`agrees that ${literal} is ${truthy ? "true" : "false"}`, () => { + assert.equal(evalFormula(`=IF(${literal}, 1, 2)`), truthy ? 1 : 2, "IF branch"); + assert.equal(evalFormula(`=AND(${literal})`), truthy, "AND"); + assert.equal(evalFormula(`=OR(${literal})`), truthy, "OR"); + assert.equal(evalFormula(`=NOT(${literal})`), !truthy, "NOT"); + }); + } +}); diff --git a/tests/engine/test_lookupBounds.ts b/tests/engine/test_lookupBounds.ts new file mode 100644 index 0000000..14d0d0c --- /dev/null +++ b/tests/engine/test_lookupBounds.ts @@ -0,0 +1,135 @@ +// VLOOKUP's col_index_num and HLOOKUP's row_index_num were used unchecked, so an +// index past the table addressed a cell OUTSIDE the range and returned whatever +// lived there — usually a silent 0 where Excel reports #REF! (#2360). INDEX +// already had this guard; these two did not. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { resolveTableOffset } from "../../src/engine/formulaRefs.ts"; +import { SpreadsheetEngine, type SheetData } from "../../src/engine/index.ts"; +import { cellAt } from "./cellAccess.ts"; + +// A1:B2 = [a, 1] / [b, 2]; the formula sits in C1. +const table = (formula: string): unknown => { + const sheet: SheetData = { + name: "S", + data: [ + [{ v: "a" }, { v: 1 }, { v: formula }], + [{ v: "b" }, { v: 2 }], + ], + }; + return cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 2); +}; + +describe("resolveTableOffset", () => { + it("maps a 1-based position to a 0-based offset", () => { + assert.equal(resolveTableOffset(1, 2), 0); + assert.equal(resolveTableOffset(2, 2), 1); + }); + + it("rejects a position past the table", () => { + assert.equal(resolveTableOffset(3, 2), null); + assert.equal(resolveTableOffset(9, 2), null); + }); + + it("rejects zero and negative positions", () => { + assert.equal(resolveTableOffset(0, 2), null); + assert.equal(resolveTableOffset(-1, 2), null); + }); + + // INDEX reads a `0` position as "the whole line" and collapses it to the only + // cell when the line is one long. A lookup index has no such meaning — its + // columns are numbered from 1 — so `0` is out of range here too (Codex review). + it("rejects zero even for a single-line table, unlike INDEX", () => { + assert.equal(resolveTableOffset(0, 1), null); + }); + + it("rejects a non-finite position", () => { + assert.equal(resolveTableOffset(NaN, 2), null); + }); + + it("truncates a fractional position toward zero, as Excel does", () => { + assert.equal(resolveTableOffset(2.9, 2), 1); + }); +}); + +describe("VLOOKUP column bounds", () => { + it("is #REF! when the column index is past the table", () => { + assert.equal(table('=VLOOKUP("a",A1:B2,9,FALSE)'), "#REF!"); + }); + + it("is #REF! for a zero or negative column index", () => { + assert.equal(table('=VLOOKUP("a",A1:B2,0,FALSE)'), "#REF!"); + assert.equal(table('=VLOOKUP("a",A1:B2,-1,FALSE)'), "#REF!"); + }); + + it("still returns the value for an in-range column", () => { + assert.equal(table('=VLOOKUP("b",A1:B2,1,FALSE)'), "b", "column 1 is the key column"); + assert.equal(table('=VLOOKUP("b",A1:B2,2,FALSE)'), 2); + }); + + it("still reports #N/A when the key is not found", () => { + assert.equal(table('=VLOOKUP("zz",A1:B2,2,FALSE)'), "#N/A", "a missing key is not a #REF!"); + }); +}); + +// Excel treats an out-of-range index as an argument error, evaluated before the +// key is searched for. Validating it after the match let a missing key mask it as +// #N/A, so a typo'd index looked like "value not in the table" (Codex review). +describe("an out-of-range index outranks a missing key", () => { + it("is #REF! for VLOOKUP with a missing key and an index past the table", () => { + assert.equal(table('=VLOOKUP("zz",A1:B2,9,FALSE)'), "#REF!"); + }); + + it("is #REF! for VLOOKUP with a missing key and a zero or negative index", () => { + assert.equal(table('=VLOOKUP("zz",A1:B2,0,FALSE)'), "#REF!"); + assert.equal(table('=VLOOKUP("zz",A1:B2,-1,FALSE)'), "#REF!"); + }); + + it("is #REF! for HLOOKUP with a missing key and an index past the table", () => { + assert.equal(table('=HLOOKUP("zz",A1:B2,9,FALSE)'), "#REF!"); + }); + + // The approximate path reaches #N/A by a different route (no candidate <= the + // key) rather than by an absent exact match, so it needs its own case. + it("is #REF! on the approximate path when the key is below every candidate", () => { + assert.equal(table("=VLOOKUP(0,A1:B2,9,TRUE)"), "#REF!"); + }); +}); + +describe("single-line tables still reject index 0", () => { + // A one-column table is where INDEX's whole-line `0` would have slipped through. + const oneColumn = (formula: string): unknown => { + const sheet: SheetData = { + name: "S", + data: [[{ v: "a" }, { v: formula }], [{ v: "b" }]], + }; + return cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 1); + }; + + it("is #REF! for VLOOKUP with index 0 on a single-column table", () => { + assert.equal(oneColumn('=VLOOKUP("a",A1:A2,0,FALSE)'), "#REF!"); + }); + + it("still returns the key column for index 1", () => { + assert.equal(oneColumn('=VLOOKUP("b",A1:A2,1,FALSE)'), "b"); + }); + + it("is #REF! when the key is missing too, not #N/A", () => { + assert.equal(oneColumn('=VLOOKUP("zz",A1:A2,0,FALSE)'), "#REF!"); + }); + + it("still reports #N/A for a missing key with a valid index", () => { + assert.equal(oneColumn('=VLOOKUP("zz",A1:A2,1,FALSE)'), "#N/A"); + }); +}); + +describe("HLOOKUP row bounds", () => { + it("is #REF! when the row index is past the table", () => { + assert.equal(table('=HLOOKUP("a",A1:B2,9,FALSE)'), "#REF!"); + }); + + it("still returns the value for an in-range row", () => { + assert.equal(table('=HLOOKUP("a",A1:B2,2,FALSE)'), "b"); + }); +}); diff --git a/tests/engine/test_lookupFunctions.ts b/tests/engine/test_lookupFunctions.ts new file mode 100644 index 0000000..698be42 --- /dev/null +++ b/tests/engine/test_lookupFunctions.ts @@ -0,0 +1,181 @@ +// Lookup functions driven through the whole engine. The cross-sheet VLOOKUP case +// is the #2390 regression: a sheet-qualified table array (`Data!A1:B3`) used to +// throw because one of VLOOKUP's two range parses ran a sheet-unaware regex and +// rejected the prefix before the sheet-aware parse could run. Now a single +// `parseRangeBounds` handles both (#2396). + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { SpreadsheetEngine, type SheetData } from "../../src/engine/index.ts"; +import { cellAt } from "./cellAccess.ts"; + +/** Calculate `formula` in cell A-after-the-data of a single sheet built from `rows`. */ +const evalInSheet = (rows: (string | number)[][], formula: string): unknown => { + const data = rows.map((row) => row.map((value) => ({ v: value }))); + data.push([{ v: formula }]); + const result = new SpreadsheetEngine().calculate({ name: "S", data }); + return cellAt(result.data, data.length - 1, 0); +}; + +describe("VLOOKUP — same sheet", () => { + const table: (string | number)[][] = [ + ["Alice", 10], + ["Bob", 20], + ["Carol", 30], + ]; + + it("returns the result-column value for an exact match", () => { + assert.equal(evalInSheet(table, '=VLOOKUP("Carol", A1:B3, 2, FALSE)'), 30); + }); + + it("returns #N/A when the value is absent", () => { + assert.equal(evalInSheet(table, '=VLOOKUP("Zoe", A1:B3, 2, FALSE)'), "#N/A"); + }); +}); + +describe("HLOOKUP — same sheet", () => { + it("looks across the top row and returns the row below", () => { + const table: (string | number)[][] = [ + ["a", "b", "c"], + [1, 2, 3], + ]; + assert.equal(evalInSheet(table, '=HLOOKUP("b", A1:C2, 2, FALSE)'), 2); + }); +}); + +describe("VLOOKUP — cross-sheet table array (#2390: no longer throws)", () => { + it("resolves a sheet-qualified table array", () => { + const data: SheetData = { + name: "Data", + data: [ + [{ v: "Alice" }, { v: 10 }], + [{ v: "Bob" }, { v: 20 }], + [{ v: "Carol" }, { v: 30 }], + ], + }; + const main: SheetData = { name: "Main", data: [[{ v: '=VLOOKUP("Bob", Data!A1:B3, 2, FALSE)' }]] }; + const [mainResult] = new SpreadsheetEngine().calculateWorkbook([main, data]); + assert.ok(mainResult); + assert.equal(cellAt(mainResult.data, 0, 0), 20); + }); +}); + +// Approximate match (#2360): the 4th argument TRUE (and omitted) must do an +// approximate match — the largest first-column/row value <= the lookup key in a +// sorted range — not fall back to exact. The literal TRUE reaches the handler as +// the string "TRUE", which the old accept-only-`true|1|"1"` check missed, so +// `VLOOKUP(4, …, TRUE)` silently returned #N/A. +describe("VLOOKUP — approximate match with TRUE (#2360)", () => { + const sorted: (string | number)[][] = [ + [1, "a"], + [3, "b"], + [5, "c"], + ]; + + it("returns the largest value <= the lookup key", () => { + assert.equal(evalInSheet(sorted, "=VLOOKUP(4, A1:B3, 2, TRUE)"), "b"); // 3 is the largest <= 4 + assert.equal(evalInSheet(sorted, "=VLOOKUP(2, A1:B3, 2, TRUE)"), "a"); // 1 is the largest <= 2 + assert.equal(evalInSheet(sorted, "=VLOOKUP(9, A1:B3, 2, TRUE)"), "c"); // past the end -> last row + }); + + it("matches an exact key on the approximate path too", () => { + assert.equal(evalInSheet(sorted, "=VLOOKUP(3, A1:B3, 2, TRUE)"), "b"); + }); + + it("treats a lowercase true the same as TRUE", () => { + assert.equal(evalInSheet(sorted, "=VLOOKUP(4, A1:B3, 2, true)"), "b"); + }); + + it("approximates when the 4th argument is omitted (default TRUE)", () => { + assert.equal(evalInSheet(sorted, "=VLOOKUP(4, A1:B3, 2)"), "b"); + }); + + it("returns #N/A when the key is below the smallest value", () => { + assert.equal(evalInSheet(sorted, "=VLOOKUP(0, A1:B3, 2, TRUE)"), "#N/A"); + }); + + it("keeps the exact FALSE path unchanged", () => { + assert.equal(evalInSheet(sorted, "=VLOOKUP(3, A1:B3, 2, FALSE)"), "b"); + assert.equal(evalInSheet(sorted, "=VLOOKUP(4, A1:B3, 2, FALSE)"), "#N/A"); + }); +}); + +describe("HLOOKUP — approximate match with TRUE (#2360)", () => { + const sorted: (string | number)[][] = [ + [10, 20, 30], + ["x", "y", "z"], + ]; + + it("returns the row-2 value under the largest column <= the lookup key", () => { + assert.equal(evalInSheet(sorted, "=HLOOKUP(25, A1:C2, 2, TRUE)"), "y"); // 20 is the largest <= 25 + assert.equal(evalInSheet(sorted, "=HLOOKUP(30, A1:C2, 2, TRUE)"), "z"); + }); + + it("returns #N/A when the key is below the smallest value", () => { + assert.equal(evalInSheet(sorted, "=HLOOKUP(5, A1:C2, 2, TRUE)"), "#N/A"); + }); +}); + +describe("INDEX — bounds (#2390)", () => { + const grid: (string | number)[][] = [ + [10, 11], + [20, 21], + [30, 31], + ]; + + it("returns the addressed cell for an in-range position", () => { + assert.equal(evalInSheet(grid, "=INDEX(A1:B3, 2, 2)"), 21); // B2 + assert.equal(evalInSheet(grid, "=INDEX(A1:A3, 3)"), 30); // A3 + }); + + it("returns #REF! when the row is past the range (was reading A5)", () => { + assert.equal(evalInSheet(grid, "=INDEX(A1:A3, 5)"), "#REF!"); + }); + + it("returns #REF! for row 0 on a multi-row range (was reading A1 above the range)", () => { + assert.equal(evalInSheet(grid, "=INDEX(A2:B3, 0, 1)"), "#REF!"); + }); +}); + +// MATCH and XLOOKUP read their ranges through the NUMERIC-ONLY reader, which +// drops every text cell. #2358 moved SUMIF/AVERAGEIF onto the raw reader for +// exactly this reason — `calculator.ts` still explains it — and these two were +// missed, so they carry both halves of that failure: text keys are invisible, +// and a lookup/return pair filtered independently falls out of row alignment +// and answers with a DIFFERENT row's value, silently. +describe("MATCH / XLOOKUP over text (#2765)", () => { + const fruit: (string | number)[][] = [ + ["apple", 1], + ["banana", 2], + ["cherry", 3], + ]; + + it("MATCH finds a text key", () => { + assert.equal(evalInSheet(fruit, '=MATCH("banana", A1:A3, 0)'), 2); + }); + + it("XLOOKUP returns the paired value for a text key", () => { + assert.equal(evalInSheet(fruit, '=XLOOKUP("banana", A1:A3, B1:B3)'), 2); + }); + + // The dangerous one: no error, just a wrong number. `A2` is text, so the + // lookup column loses that row while the return column keeps all three — + // the match at index 1 then reads B2 instead of B3. + it("XLOOKUP stays row-aligned when the lookup column holds text", () => { + const mixed: (string | number)[][] = [ + [1, 100], + ["x", 200], + [3, 300], + ]; + assert.equal(evalInSheet(mixed, "=XLOOKUP(3, A1:A3, B1:B3)"), 300); + }); + + it("MATCH keeps its 1-based index when earlier rows hold text", () => { + const mixed: (string | number)[][] = [ + ["x", 0], + [7, 0], + [9, 0], + ]; + assert.equal(evalInSheet(mixed, "=MATCH(9, A1:A3, 0)"), 3); + }); +}); diff --git a/tests/engine/test_lookupMath.ts b/tests/engine/test_lookupMath.ts new file mode 100644 index 0000000..ad063b3 --- /dev/null +++ b/tests/engine/test_lookupMath.ts @@ -0,0 +1,43 @@ +// isApproximateMatch reads VLOOKUP/HLOOKUP's range_lookup argument (#2360). The +// literal TRUE arrives as the STRING "TRUE" (the evaluator leaves bare words +// unquoted), which the old accept-only-`true|1|"1"` check missed and so fell +// back to exact match. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { isApproximateMatch } from "../../src/engine/functions/lookup-math.ts"; + +describe("isApproximateMatch", () => { + it("treats a real boolean as itself", () => { + assert.equal(isApproximateMatch(true), true); + assert.equal(isApproximateMatch(false), false); + }); + + // The fix: the bare word TRUE evaluates to the string "TRUE", not a boolean. + it("reads the string forms of TRUE/FALSE, case-insensitively", () => { + assert.equal(isApproximateMatch("TRUE"), true); + assert.equal(isApproximateMatch("true"), true); + assert.equal(isApproximateMatch("True"), true); + assert.equal(isApproximateMatch("FALSE"), false); + assert.equal(isApproximateMatch("false"), false); + }); + + it("reads numeric logicals: 0 exact, non-zero approximate", () => { + assert.equal(isApproximateMatch(1), true); + assert.equal(isApproximateMatch(0), false); + assert.equal(isApproximateMatch(2), true); + assert.equal(isApproximateMatch("1"), true); + assert.equal(isApproximateMatch("0"), false); + }); + + it("treats blank or stray text as exact (FALSE), matching Excel coercion", () => { + assert.equal(isApproximateMatch(""), false); + assert.equal(isApproximateMatch(" "), false); + assert.equal(isApproximateMatch("yes"), false); + }); + + it("ignores surrounding whitespace on the string forms", () => { + assert.equal(isApproximateMatch(" TRUE "), true); + assert.equal(isApproximateMatch(" 0 "), false); + }); +}); diff --git a/tests/engine/test_mathematicalFunctions.ts b/tests/engine/test_mathematicalFunctions.ts new file mode 100644 index 0000000..957c758 --- /dev/null +++ b/tests/engine/test_mathematicalFunctions.ts @@ -0,0 +1,169 @@ +// Domain and boundary rules for the math functions. The bugs here returned a +// plausible NUMBER (FLOOR(-2.5,2) = -4, ROUND(-2.5,0) = -2, MOD(-3,2) = -1) or a +// silent NaN/∞ instead of an Excel error (#2389). The rounding direction, the +// modulo sign and the domain guards are checked directly on the pure helpers, +// with a few end-to-end checks that the handlers surface the error values. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { + roundTo, + roundUpTo, + roundDownTo, + floorToSignificance, + ceilingToSignificance, + modulo, + power, + safeLog, + safeLog10, + safeSqrt, + logWithBase, +} from "../../src/engine/math-ops.ts"; +import { SpreadsheetEngine, type SheetData } from "../../src/engine/index.ts"; +import { DIV_ZERO_ERROR, NUM_ERROR } from "../../src/engine/spreadsheet-errors.ts"; +import { cellAt } from "./cellAccess.ts"; + +const closeTo = (actual: number, expected: number, eps = 1e-9): boolean => Math.abs(actual - expected) <= eps; + +describe("roundTo / roundUpTo / roundDownTo — direction", () => { + it("rounds half away from zero, not toward +infinity", () => { + assert.equal(roundTo(-2.5, 0), -3); + assert.equal(roundTo(2.5, 0), 3); + assert.ok(closeTo(roundTo(0.125, 2), 0.13)); + }); + + it("rounds up away from zero", () => { + assert.ok(closeTo(roundUpTo(-3.14159, 2), -3.15)); + assert.ok(closeTo(roundUpTo(3.14159, 2), 3.15)); + }); + + it("rounds down toward zero", () => { + assert.ok(closeTo(roundDownTo(-3.14159, 2), -3.14)); + assert.ok(closeTo(roundDownTo(3.19, 1), 3.1)); + }); +}); + +describe("floorToSignificance / ceilingToSignificance — sign domain", () => { + it("is a #NUM! error value when number and significance disagree in sign", () => { + assert.equal(floorToSignificance(-2.5, 2), NUM_ERROR); + assert.equal(ceilingToSignificance(2.5, -2), NUM_ERROR); + }); + + it("rounds to the multiple when the signs match", () => { + assert.equal(floorToSignificance(2.5, 2), 2); + assert.equal(floorToSignificance(-2.5, -2), -2); + assert.equal(ceilingToSignificance(2.5, 2), 4); + assert.equal(ceilingToSignificance(-2.5, -2), -4); + }); + + it("returns 0 for a zero value", () => { + assert.equal(floorToSignificance(0, 2), 0); + assert.equal(ceilingToSignificance(0, 2), 0); + }); + + // Excel is deliberately asymmetric here: FLOOR(x, 0) is #DIV/0! while + // CEILING(x, 0) is 0. Both used to answer 0, so FLOOR swallowed a divide-by- + // zero (#2360). Keep the pair pinned so neither is "made consistent" later. + it("is #DIV/0! for FLOOR with a zero significance, but 0 for CEILING", () => { + assert.equal(floorToSignificance(3, 0), DIV_ZERO_ERROR); + assert.equal(ceilingToSignificance(3, 0), 0); + }); + + // The zero check has to win over the sign check, or a negative number with a + // zero significance would report the wrong error (#NUM! instead of #DIV/0!). + it("reports #DIV/0! rather than #NUM! for a negative number over a zero significance", () => { + assert.equal(floorToSignificance(-3, 0), DIV_ZERO_ERROR); + }); +}); + +describe("modulo — divisor sign and division by zero", () => { + it("takes the sign of the divisor", () => { + assert.equal(modulo(-3, 2), 1); + assert.equal(modulo(3, -2), -1); + assert.equal(modulo(-3, -2), -1); + assert.equal(modulo(5, 3), 2); + }); + + it("is #DIV/0! when the divisor is zero", () => { + assert.equal(modulo(5, 0), DIV_ZERO_ERROR); + }); +}); + +describe("power — negative base domain", () => { + it("is #NUM! for a negative base with a non-integer exponent", () => { + assert.equal(power(-8, 1 / 3), NUM_ERROR); + assert.equal(power(-2, 0.5), NUM_ERROR); + }); + + it("computes when the exponent is an integer or the base is non-negative", () => { + assert.equal(power(-2, 3), -8); + assert.equal(power(2, 10), 1024); + assert.ok(closeTo(power(9, 0.5) as number, 3)); + }); +}); + +describe("safeSqrt / safeLog / safeLog10 — domain", () => { + it("is #NUM! outside the domain", () => { + assert.equal(safeSqrt(-1), NUM_ERROR); + assert.equal(safeLog(0), NUM_ERROR); + assert.equal(safeLog(-1), NUM_ERROR); + assert.equal(safeLog10(0), NUM_ERROR); + }); + + it("computes inside the domain", () => { + assert.equal(safeSqrt(4), 2); + assert.ok(closeTo(safeLog(Math.E) as number, 1)); + assert.equal(safeLog10(1000), 3); + }); +}); + +describe("logWithBase — number and base domain", () => { + it("computes a valid base-N log", () => { + assert.ok(closeTo(logWithBase(8, 2) as number, 3)); + assert.ok(closeTo(logWithBase(100, 10) as number, 2)); + }); + + it("is #NUM! for a non-positive number", () => { + assert.equal(logWithBase(0, 10), NUM_ERROR); + assert.equal(logWithBase(-1, 10), NUM_ERROR); + }); + + it("is #NUM! for a base that is non-positive or exactly 1", () => { + assert.equal(logWithBase(8, 1), NUM_ERROR); + assert.equal(logWithBase(8, -2), NUM_ERROR); + assert.equal(logWithBase(8, 0), NUM_ERROR); + }); +}); + +describe("the handlers surface the errors end-to-end", () => { + const evalFormula = (formula: string): unknown => + cellAt(new SpreadsheetEngine().calculate({ name: "S", data: [[{ v: formula }]] } satisfies SheetData).data, 0, 0); + + it("displays the Excel error codes through the engine", () => { + assert.equal(evalFormula("=FLOOR(-2.5, 2)"), "#NUM!"); + assert.equal(evalFormula("=FLOOR(3, 0)"), "#DIV/0!"); + assert.equal(evalFormula("=CEILING(3, 0)"), 0); + assert.equal(evalFormula("=SQRT(-1)"), "#NUM!"); + assert.equal(evalFormula("=MOD(5, 0)"), "#DIV/0!"); + assert.equal(evalFormula("=ROUND(-2.5, 0)"), -3); + assert.equal(evalFormula("=MOD(-3, 2)"), 1); + assert.equal(evalFormula("=LOG(8, 1)"), "#NUM!"); + assert.equal(evalFormula("=LOG(8, -2)"), "#NUM!"); + assert.equal(evalFormula("=LOG(8, 2)"), 3); + }); + + // Domain misses are error VALUES (not NaN/∞), so IFERROR must still catch + // them or nested formulas would surface the raw error (#2389 review). + it("lets IFERROR catch the domain errors", () => { + assert.equal(evalFormula("=IFERROR(SQRT(-1), 42)"), 42); + assert.equal(evalFormula("=IFERROR(MOD(5, 0), -1)"), -1); + assert.equal(evalFormula("=IFERROR(SQRT(4), 42)"), 2, "a non-error passes through"); + }); + + // Text that only looks like an error is real text, not an error value, so + // IFERROR returns it rather than the fallback. + it("does not treat quoted error-looking text as an error", () => { + assert.equal(evalFormula('=IFERROR("#NUM!", 42)'), "#NUM!"); + assert.equal(evalFormula('=IFERROR("hello", 42)'), "hello"); + }); +}); diff --git a/tests/engine/test_midValue.ts b/tests/engine/test_midValue.ts new file mode 100644 index 0000000..3efa3f1 --- /dev/null +++ b/tests/engine/test_midValue.ts @@ -0,0 +1,115 @@ +// MID's bounds and VALUE's parsing. Both returned a plausible answer instead of +// an error: `substring` SWAPS reversed bounds, so a negative MID count read +// backwards and produced earlier characters, and `parseFloat` stops at the first +// unreadable character, so VALUE("12abc") came back 12 (#2360). + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { takeMid, parseValueText } from "../../src/engine/functions/text.ts"; +import { VALUE_ERROR } from "../../src/engine/spreadsheet-errors.ts"; +import { SpreadsheetEngine, type SheetData } from "../../src/engine/index.ts"; +import { cellAt } from "./cellAccess.ts"; + +const evaluate = (formula: string, cell: string | number = "Hello"): unknown => { + const sheet: SheetData = { name: "S", data: [[{ v: cell }, { v: formula }]] }; + return cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 1); +}; + +describe("takeMid — bounds", () => { + it("errors on a negative count instead of reading backwards", () => { + assert.equal(takeMid("Hello", 3, -1), VALUE_ERROR); + }); + + it("errors on a non-finite count", () => { + assert.equal(takeMid("Hello", 3, NaN), VALUE_ERROR); + assert.equal(takeMid("Hello", 3, Infinity), VALUE_ERROR); + }); + + // Excel's MID is 1-based; 0 and negatives are not positions. + it("errors on a start position below 1", () => { + assert.equal(takeMid("Hello", 0, 2), VALUE_ERROR); + assert.equal(takeMid("Hello", -1, 2), VALUE_ERROR); + }); + + it("takes the requested characters from a 1-based start", () => { + assert.equal(takeMid("Hello", 1, 2), "He"); + assert.equal(takeMid("Hello", 2, 3), "ell"); + }); + + it("stops at the end of the text when the count overruns", () => { + assert.equal(takeMid("Hello", 4, 99), "lo"); + }); + + it("returns an empty string for a zero count", () => { + assert.equal(takeMid("Hello", 2, 0), ""); + }); + + it("truncates a fractional count and start toward zero", () => { + assert.equal(takeMid("Hello", 2.9, 2.9), "el"); + }); +}); + +describe("parseValueText — the whole string must be a number", () => { + it("errors on trailing text rather than salvaging the prefix", () => { + assert.equal(parseValueText("12abc"), VALUE_ERROR); + assert.equal(parseValueText("3.5kg"), VALUE_ERROR); + }); + + it("errors on an empty or blank string", () => { + assert.equal(parseValueText(""), VALUE_ERROR); + assert.equal(parseValueText(" "), VALUE_ERROR); + }); + + it("reads a plain number, tolerating surrounding whitespace", () => { + assert.equal(parseValueText("42"), 42); + assert.equal(parseValueText(" 7 "), 7); + assert.equal(parseValueText("-3.5"), -3.5); + }); + + it("strips currency symbols and thousands separators", () => { + assert.equal(parseValueText("$1,234.5"), 1234.5); + }); + + it("reads a trailing percent as a fraction", () => { + assert.equal(parseValueText("50%"), 0.5); + assert.equal(parseValueText("12abc%"), VALUE_ERROR, "still a whole-string match"); + }); + + // `Number` accepts JS-only spellings a spreadsheet never should. + it("rejects JS-only numeric syntaxes", () => { + assert.equal(parseValueText("0x10"), VALUE_ERROR, "hex"); + assert.equal(parseValueText("0X10"), VALUE_ERROR, "hex, upper case"); + assert.equal(parseValueText("0b10"), VALUE_ERROR, "binary"); + assert.equal(parseValueText("0o17"), VALUE_ERROR, "octal"); + assert.equal(parseValueText("Infinity"), VALUE_ERROR); + assert.equal(parseValueText("-Infinity"), VALUE_ERROR); + assert.equal(parseValueText("1_000"), VALUE_ERROR, "numeric separator"); + }); + + // The decimal pattern matches these, so only the finiteness check rejects + // them — without it the guard would be dead code and could be dropped unseen. + it("rejects an exponent that overflows to infinity", () => { + assert.equal(parseValueText("1e999"), VALUE_ERROR); + assert.equal(parseValueText("-1e999"), VALUE_ERROR); + assert.equal(parseValueText("1e999%"), VALUE_ERROR, "also through the percent path"); + }); + + it("still reads decimal and scientific notation", () => { + assert.equal(parseValueText("1e3"), 1000); + assert.equal(parseValueText("-2.5E-2"), -0.025); + assert.equal(parseValueText(".5"), 0.5); + assert.equal(parseValueText("+7"), 7); + }); +}); + +describe("through the engine", () => { + it("surfaces the MID and VALUE errors in the cell", () => { + assert.equal(evaluate("=MID(A1,3,-1)"), "#VALUE!"); + assert.equal(evaluate('=VALUE("12abc")'), "#VALUE!"); + }); + + it("keeps the working cases working", () => { + assert.equal(evaluate("=MID(A1,2,3)"), "ell"); + assert.equal(evaluate('=VALUE("42")'), 42); + }); +}); diff --git a/tests/engine/test_multiRangeAggregates.ts b/tests/engine/test_multiRangeAggregates.ts new file mode 100644 index 0000000..748948b --- /dev/null +++ b/tests/engine/test_multiRangeAggregates.ts @@ -0,0 +1,134 @@ +// Aggregates over more than one argument. Excel takes up to 255 (`SUM(A1:A2, +// B1:B2)`, `SUM(A1:A2, 10)`), but eight of these functions were registered with +// `maxArgs: 1` and read only `args[0]`, so the second range made the whole +// formula fail — a loud `#ERROR!` on ordinary spreadsheet usage (#2360). + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { SpreadsheetEngine, type SheetData } from "../../src/engine/index.ts"; +import { cellAt } from "./cellAccess.ts"; + +// A = 1,2 ; B = 3,4 ; the formula sits in C1. +const evaluate = (formula: string): unknown => { + const sheet: SheetData = { + name: "S", + data: [ + [{ v: 1 }, { v: 3 }, { v: formula }], + [{ v: 2 }, { v: 4 }], + ], + }; + return cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 2); +}; + +describe("aggregates accept several ranges", () => { + it("sums two ranges", () => { + assert.equal(evaluate("=SUM(A1:A2,B1:B2)"), 10); + }); + + it("averages across two ranges", () => { + assert.equal(evaluate("=AVERAGE(A1:A2,B1:B2)"), 2.5); + }); + + it("counts numbers across two ranges", () => { + assert.equal(evaluate("=COUNT(A1:A2,B1:B2)"), 4); + }); + + it("counts non-empty cells across two ranges", () => { + assert.equal(evaluate("=COUNTA(A1:A2,B1:B2)"), 4); + }); + + it("takes the median across two ranges", () => { + assert.equal(evaluate("=MEDIAN(A1:A2,B1:B2)"), 2.5); + }); +}); + +describe("aggregates mix ranges with plain values", () => { + it("sums a range plus a literal", () => { + assert.equal(evaluate("=SUM(A1:A2,10)"), 13); + }); + + it("sums a range plus a single cell reference", () => { + assert.equal(evaluate("=SUM(A1:A2,B1)"), 6); + }); +}); + +describe("a single cell reference is read as a range, not a scalar", () => { + // The scalar path coerces a blank or text cell to 0, so COUNT(A999) counted an + // empty cell as a value once multi-argument collection was introduced (Codex + // review). A bare cell ref goes through the range path instead. + const countSheet = (formula: string): unknown => { + const sheet: SheetData = { + name: "S", + data: [[{ v: 5 }, { v: formula }], [{ v: "txt" }]], + }; + return cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 1); + }; + + it("does not count an out-of-bounds cell", () => { + assert.equal(countSheet("=COUNT(A999)"), 0); + assert.equal(countSheet("=COUNTA(A999)"), 0); + }); + + it("does not count a text cell as a number", () => { + assert.equal(countSheet("=COUNT(A2)"), 0); + assert.equal(countSheet("=COUNTA(A2)"), 1, "COUNTA does count text"); + }); + + it("counts a single numeric cell", () => { + assert.equal(countSheet("=COUNT(A1)"), 1); + }); +}); + +describe("COUNT counts only the arguments that hold a number", () => { + // Multi-argument collection reads a non-reference argument as a scalar, and the + // lenient `toNumber` turns anything unreadable into 0 — so every scalar looked + // like a value and COUNT("text") answered 1 (Codex review). Excel and the + // pre-change engine both answer 0. + it("does not count a text literal", () => { + assert.equal(evaluate('=COUNT("text")'), 0); + }); + + it("counts the numbers and skips the text when both are given", () => { + assert.equal(evaluate('=COUNT(1,"text")'), 1); + assert.equal(evaluate('=COUNT(A1:A2,"text")'), 2); + }); + + it("counts a number typed directly, quoted, or computed", () => { + assert.equal(evaluate("=COUNT(1)"), 1); + assert.equal(evaluate('=COUNT("1")'), 1, "Excel counts a quoted number typed as an argument"); + assert.equal(evaluate("=COUNT(1+1)"), 1); + }); + + // Excel counts a logical typed directly as an argument; this engine has PINNED + // booleans as non-numbers throughout (`toNumber(true)` is 0, see #2391), so + // COUNT follows the engine rather than splitting the difference. + it("does not count a logical literal", () => { + assert.equal(evaluate("=COUNT(TRUE)"), 0); + assert.equal(evaluate("=COUNT(FALSE)"), 0); + }); + + // Excel reads "12abc" as text and answers 0. The engine's shared numeric parser + // reads the leading number everywhere (SUM(1,"12abc") is 13), so COUNT stays + // consistent with the engine instead. + it("counts a leading-number string, as the rest of the engine reads it", () => { + assert.equal(evaluate('=COUNT("12abc")'), 1); + }); + + it("leaves the lenient aggregates coercing as before", () => { + assert.equal(evaluate('=SUM(1,"text")'), 1); + assert.equal(evaluate('=AVERAGE(1,"text")'), 0.5); + }); + + it("still lets COUNTA count a text literal", () => { + assert.equal(evaluate('=COUNTA(1,"text")'), 2); + assert.equal(evaluate('=COUNTA("")'), 0); + }); +}); + +describe("the single-range behaviour is unchanged", () => { + it("sums, averages and counts one range as before", () => { + assert.equal(evaluate("=SUM(A1:A2)"), 3); + assert.equal(evaluate("=AVERAGE(A1:A2)"), 1.5); + assert.equal(evaluate("=COUNT(A1:A2)"), 2); + }); +}); diff --git a/tests/engine/test_normalizeData.ts b/tests/engine/test_normalizeData.ts new file mode 100644 index 0000000..1e4ec04 --- /dev/null +++ b/tests/engine/test_normalizeData.ts @@ -0,0 +1,59 @@ +// Coercing whatever shape a model emitted into a 2D cell grid. It runs before +// every calculation, and a wrong reshape silently maps every A1-style reference +// onto a different cell — the sheet computes, just against the wrong layout. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { normalizeData } from "../../src/engine/calculator.ts"; + +describe("normalizeData — already valid", () => { + it("returns a 2D array unchanged", () => { + const grid = [[{ v: 1 }, { v: 2 }], [{ v: 3 }]]; + assert.equal(normalizeData(grid), grid, "same reference, not a copy"); + }); + + it("keeps an empty row structure", () => { + const grid = [[]]; + assert.equal(normalizeData(grid), grid); + }); +}); + +describe("normalizeData — nothing to normalise", () => { + it("returns empty for null, undefined and non-arrays", () => { + assert.deepEqual(normalizeData(null), []); + assert.deepEqual(normalizeData(undefined), []); + assert.deepEqual(normalizeData("A1"), []); + assert.deepEqual(normalizeData(42), []); + assert.deepEqual(normalizeData({ v: 1 }), []); + }); + + it("returns empty for an empty array", () => { + assert.deepEqual(normalizeData([]), []); + }); + + // A flat array of primitives is not a recognised shape — pairing them would + // invent structure, so it returns empty rather than guess. + it("returns empty for a flat array of primitives", () => { + assert.deepEqual(normalizeData([1, 2, 3]), []); + assert.deepEqual(normalizeData(["a", "b"]), []); + }); +}); + +describe("normalizeData — flat cell array to 2D", () => { + // A model that emits a flat list of cell objects is reshaped into rows of two. + it("pairs a flat cell array into two-column rows", () => { + assert.deepEqual(normalizeData([{ v: 1 }, { v: 2 }, { v: 3 }, { v: 4 }]), [ + [{ v: 1 }, { v: 2 }], + [{ v: 3 }, { v: 4 }], + ]); + }); + + // An odd length leaves a one-cell final row rather than dropping or padding. + it("leaves a lone final cell in its own row when the count is odd", () => { + assert.deepEqual(normalizeData([{ v: 1 }, { v: 2 }, { v: 3 }]), [[{ v: 1 }, { v: 2 }], [{ v: 3 }]]); + }); + + it("reshapes a single cell into one row", () => { + assert.deepEqual(normalizeData([{ v: 1 }]), [[{ v: 1 }]]); + }); +}); diff --git a/tests/engine/test_npvArgs.ts b/tests/engine/test_npvArgs.ts new file mode 100644 index 0000000..0fdf971 --- /dev/null +++ b/tests/engine/test_npvArgs.ts @@ -0,0 +1,38 @@ +// NPV period assignment through the handler. The pre-refactor handler used +// `period = argIndex + rangePosition`, so a scalar argument after a multi-cell +// range landed on the range's period rather than continuing the sequence (a +// latent off-by bug). The refactor normalizes this to strictly sequential +// periods — the Excel semantics — so a mixed `NPV(rate, range, scalar)` now +// discounts every flow by its position in the flattened series (#2394 / #2442). +// This pins the intended (sequential) behavior at the handler level. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { SpreadsheetEngine, type SheetData } from "../../src/engine/index.ts"; +import { cellAt } from "./cellAccess.ts"; + +const closeTo = (actual: number, expected: number, eps = 0.01): boolean => Math.abs(actual - expected) <= eps; + +describe("NPV — sequential periods across mixed range and scalar arguments", () => { + it("discounts a scalar after a range by the next period, not the range's", () => { + // A1:A3 = 100, 200, 300 (periods 1..3); B1 = 400 must be period 4. + // 100/1.1 + 200/1.1^2 + 300/1.1^3 + 400/1.1^4 = 754.7967… + // The old arg-index math discounted B1 at period 2 (= 812.17), which is wrong. + const sheet: SheetData = { + name: "S", + data: [[{ v: 100 }, { v: 400 }, { v: "=NPV(0.1, A1:A3, B1)" }], [{ v: 200 }], [{ v: 300 }]], + }; + const result = cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 2) as number; + assert.ok(closeTo(result, 754.7967), `NPV mixed args = ${result}, expected ~754.80 (sequential)`); + }); + + it("matches a single range read as consecutive periods", () => { + // 100/1.1 + 200/1.1^2 + 300/1.1^3 = 481.5928… + const sheet: SheetData = { + name: "S", + data: [[{ v: 100 }, { v: "=NPV(0.1, A1:A3)" }], [{ v: 200 }], [{ v: 300 }]], + }; + const result = cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 1) as number; + assert.ok(closeTo(result, 481.5928), `NPV single range = ${result}, expected ~481.59`); + }); +}); diff --git a/tests/engine/test_npvFunction.ts b/tests/engine/test_npvFunction.ts new file mode 100644 index 0000000..c1f878a --- /dev/null +++ b/tests/engine/test_npvFunction.ts @@ -0,0 +1,32 @@ +// NPV through the whole engine. The #2390 bug used each value's ARGUMENT index +// as its discount period, so a scalar after a range was discounted too little: +// NPV(0.1, A1:A3, 500) put 500 at period 2 instead of period 4. The period must +// count flattened values, not arguments. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { SpreadsheetEngine } from "../../src/engine/index.ts"; +import { cellAt } from "./cellAccess.ts"; + +const closeTo = (actual: unknown, expected: number, eps = 1e-6): boolean => typeof actual === "number" && Math.abs(actual - expected) <= eps; + +describe("NPV — range followed by a scalar (#2390)", () => { + it("discounts the trailing scalar at period 4, not period 2", () => { + const sheet = { + name: "S", + data: [[{ v: 100 }], [{ v: 200 }], [{ v: 300 }], [{ v: "=NPV(0.1, A1:A3, 500)" }]], + }; + const result = new SpreadsheetEngine().calculate(sheet); + const expected = 100 / 1.1 + 200 / 1.1 ** 2 + 300 / 1.1 ** 3 + 500 / 1.1 ** 4; + const actual = cellAt(result.data, 3, 0); + assert.ok(closeTo(actual, expected), `NPV ≈ ${expected}, got ${String(actual)}`); + }); + + it("matches the all-scalar form when there is no range", () => { + const sheet = { name: "S", data: [[{ v: "=NPV(0.1, 100, 200, 300, 500)" }]] }; + const result = new SpreadsheetEngine().calculate(sheet); + const expected = 100 / 1.1 + 200 / 1.1 ** 2 + 300 / 1.1 ** 3 + 500 / 1.1 ** 4; + const actual = cellAt(result.data, 0, 0); + assert.ok(closeTo(actual, expected), `NPV ≈ ${expected}, got ${String(actual)}`); + }); +}); diff --git a/tests/engine/test_numericCoercion.ts b/tests/engine/test_numericCoercion.ts new file mode 100644 index 0000000..7808122 --- /dev/null +++ b/tests/engine/test_numericCoercion.ts @@ -0,0 +1,152 @@ +// Two numeric reads share one string parser (#2391). `toNumber` stays lenient for +// range aggregation (unreadable → 0, booleans → 0 — PINNED, since changing it +// moves every SUM / AVERAGE / COUNTIF at once). `toScalarNumber` is the strict +// scalar read that ABS / SIGN now use: booleans are Excel's 1/0 and non-numeric +// text is #VALUE! instead of a silent 0. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { holdsNumber, parseNumericString, toScalarNumber } from "../../src/engine/numericCoercion.ts"; +import { DIV_ZERO_ERROR, VALUE_ERROR } from "../../src/engine/spreadsheet-errors.ts"; +import { toNumber } from "../../src/engine/registry.ts"; + +describe("parseNumericString — reads the formats the engine has always read", () => { + it("reads a percentage as its decimal value", () => { + assert.equal(parseNumericString("5%"), 0.05); + assert.equal(parseNumericString("100%"), 1); + }); + + it("reads currency and thousands-separated numbers", () => { + assert.equal(parseNumericString("$1,000"), 1000); + assert.equal(parseNumericString("$1,000.50"), 1000.5); + assert.equal(parseNumericString("1,234,567"), 1234567); + }); + + it("reads a plain numeric string, tolerating whitespace and exponents", () => { + assert.equal(parseNumericString("42"), 42); + assert.equal(parseNumericString(" 42 "), 42); + assert.equal(parseNumericString("-3.5"), -3.5); + assert.equal(parseNumericString("1e3"), 1000); + }); + + it("takes the leading number from a partly-numeric string", () => { + assert.equal(parseNumericString("12abc"), 12); + assert.equal(parseNumericString("3.5kg"), 3.5); + }); + + // The distinction that lets the two coercions differ: null, not 0, when there + // is no number at all. toNumber maps that to 0; toScalarNumber to #VALUE!. + it("returns null when nothing parses", () => { + assert.equal(parseNumericString("abc"), null); + assert.equal(parseNumericString(""), null); + assert.equal(parseNumericString(" "), null); + assert.equal(parseNumericString("N/A"), null); + assert.equal(parseNumericString("abc12"), null); + }); +}); + +describe("toNumber — PINNED lenient behaviour (#2391 does not change this)", () => { + it("returns a number unchanged", () => { + assert.equal(toNumber(42), 42); + assert.equal(toNumber(0), 0); + assert.equal(toNumber(-7.5), -7.5); + }); + + it("maps unreadable text to 0", () => { + assert.equal(toNumber("hello"), 0); + assert.equal(toNumber(""), 0); + assert.equal(toNumber("N/A"), 0); + }); + + // The high-blast-radius case the issue asked to pin: booleans read as 0 here, + // NOT Excel's 1/0, because SUM / AVERAGE / COUNTIF all lean on this. + it("maps booleans to 0, not Excel's 1 and 0", () => { + assert.equal(toNumber(true), 0); + assert.equal(toNumber(false), 0); + }); + + it("still reads formatted strings", () => { + assert.equal(toNumber("5%"), 0.05); + assert.equal(toNumber("$1,000"), 1000); + assert.equal(toNumber("12abc"), 12); + }); +}); + +describe("toScalarNumber — strict scalar read for ABS / SIGN (#2391)", () => { + it("returns a number unchanged", () => { + assert.equal(toScalarNumber(42), 42); + assert.equal(toScalarNumber(-7.5), -7.5); + assert.equal(toScalarNumber(0), 0); + }); + + // The boolean fix: TRUE=1, FALSE=0 (Excel), where toNumber gives 0 for both. + it("reads booleans as Excel's 1 and 0", () => { + assert.equal(toScalarNumber(true), 1); + assert.equal(toScalarNumber(false), 0); + }); + + it("parses numeric and formatted text", () => { + assert.equal(toScalarNumber("5"), 5); + assert.equal(toScalarNumber("$1,000"), 1000); + assert.equal(toScalarNumber("50%"), 0.5); + }); + + // The text fix: genuinely non-numeric text is an error, not a silent 0. + it("returns #VALUE! for non-numeric text and empty strings", () => { + assert.equal(toScalarNumber("abc"), VALUE_ERROR); + assert.equal(toScalarNumber(""), VALUE_ERROR); + assert.equal(toScalarNumber(" "), VALUE_ERROR); + }); + + // Deliberate leniency, pinned: a partly-numeric string keeps its leading number + // (matching the rest of the engine) rather than erroring as strict Excel would. + it("still takes the leading number from partly-numeric text", () => { + assert.equal(toScalarNumber("12abc"), 12); + }); +}); + +// The question `toNumber` cannot answer: it maps text, booleans and a genuine 0 +// to the same 0, so COUNT could not tell "no number here" from "the number zero" +// and counted COUNT("text") as a value (Codex review on #2360). +describe("holdsNumber — is there a number in this value at all", () => { + it("is true for numbers, including zero and negatives", () => { + assert.equal(holdsNumber(42), true); + assert.equal(holdsNumber(0), true); + assert.equal(holdsNumber(-7.5), true); + }); + + it("is true for text the engine reads as a number", () => { + assert.equal(holdsNumber("1"), true); + assert.equal(holdsNumber(" 42 "), true); + assert.equal(holdsNumber("5%"), true); + assert.equal(holdsNumber("$1,000"), true); + }); + + it("is false for text holding no number", () => { + assert.equal(holdsNumber("text"), false); + assert.equal(holdsNumber(""), false); + assert.equal(holdsNumber(" "), false); + assert.equal(holdsNumber("N/A"), false); + assert.equal(holdsNumber("abc12"), false); + }); + + // Same PINNED stance as toNumber: a boolean is not a number in this engine. + it("is false for booleans", () => { + assert.equal(holdsNumber(true), false); + assert.equal(holdsNumber(false), false); + }); + + it("is false for a formula error value", () => { + assert.equal(holdsNumber(DIV_ZERO_ERROR), false); + }); + + it("is false for NaN, which is a number that is no number", () => { + assert.equal(holdsNumber(NaN), false); + }); + + // Inherited from parseNumericString, pinned deliberately: the engine reads the + // leading number everywhere, so SUM and COUNT agree that "12abc" has one. + it("is true for a leading-number string, unlike Excel", () => { + assert.equal(holdsNumber("12abc"), true); + }); +}); diff --git a/tests/engine/test_numericFunctions2391.ts b/tests/engine/test_numericFunctions2391.ts new file mode 100644 index 0000000..28bbc51 --- /dev/null +++ b/tests/engine/test_numericFunctions2391.ts @@ -0,0 +1,72 @@ +// The four #2391 scenarios end-to-end through the engine, plus pins that the +// out-of-scope lenient aggregation paths (SUM / AVERAGE over text) are unchanged +// (those blanks-as-0 cases belong to #2383, not this PR). + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { SpreadsheetEngine, type SheetData } from "../../src/engine/index.ts"; +import { cellAt } from "./cellAccess.ts"; + +// Column A holds `colA` (one value per row); `formula` goes in a row below it so a +// range like A1:A3 never overlaps the formula cell. +const evalWithColumnA = (formula: string, colA: (string | number)[] = []): unknown => { + const data: { v: string | number }[][] = colA.map((value) => [{ v: value }]); + data.push([{ v: formula }]); + const result = new SpreadsheetEngine().calculate({ name: "S", data } as SheetData); + return cellAt(result.data, data.length - 1, 0); +}; + +describe("ABS / SIGN — scalar coercion (#2391)", () => { + it("reads a logical argument as 1 / 0 (was 0)", () => { + assert.equal(evalWithColumnA("=ABS(TRUE())"), 1); + assert.equal(evalWithColumnA("=ABS(FALSE())"), 0); + assert.equal(evalWithColumnA("=SIGN(TRUE())"), 1); + }); + + it("returns #VALUE! for non-numeric text (was 0)", () => { + assert.equal(evalWithColumnA('=ABS("abc")'), "#VALUE!"); + assert.equal(evalWithColumnA('=SIGN("abc")'), "#VALUE!"); + }); + + it("errors when the argument cell holds text", () => { + assert.equal(evalWithColumnA("=ABS(A1)", ["abc"]), "#VALUE!"); + }); + + it("still works for ordinary numbers", () => { + assert.equal(evalWithColumnA("=ABS(-5)"), 5); + assert.equal(evalWithColumnA("=SIGN(-5)"), -1); + assert.equal(evalWithColumnA("=ABS(A1)", [-42]), 42); + }); +}); + +describe("MODE — no repeat is #N/A (#2391)", () => { + it("returns #N/A when every value is distinct (was the first value)", () => { + assert.equal(evalWithColumnA("=MODE(A1:A3)", [1, 2, 3]), "#N/A"); + }); + + it("still returns the most frequent value when one repeats", () => { + assert.equal(evalWithColumnA("=MODE(A1:A4)", [1, 2, 2, 3]), 2); + }); +}); + +describe("AVERAGEIF — no match is #DIV/0! (#2391)", () => { + it("returns #DIV/0! when nothing matches (was 0)", () => { + assert.equal(evalWithColumnA('=AVERAGEIF(A1:A3,">100")', [1, 2, 3]), "#DIV/0!"); + }); + + it("still averages the matching cells", () => { + assert.equal(evalWithColumnA('=AVERAGEIF(A1:A3,">1")', [1, 2, 3]), 2.5); + }); +}); + +describe("out-of-scope lenient paths stay unchanged (#2383, not this PR)", () => { + // A text cell inside a SUM range leaves the total unchanged (30, not #VALUE!) — + // the aggregation path keeps its lenient reading. This PR must not touch it. + it("SUM leaves a text cell out of the total", () => { + assert.equal(evalWithColumnA("=SUM(A1:A3)", [10, "abc", 20]), 30); + }); + + it("AVERAGE over numeric cells is unaffected", () => { + assert.equal(evalWithColumnA("=AVERAGE(A1:A3)", [10, 20, 30]), 20); + }); +}); diff --git a/tests/engine/test_parseFunctionArgs.ts b/tests/engine/test_parseFunctionArgs.ts new file mode 100644 index 0000000..93136dc --- /dev/null +++ b/tests/engine/test_parseFunctionArgs.ts @@ -0,0 +1,79 @@ +// Splitting a function's argument string into arguments. Exported but never +// tested, and it decides where every multi-argument function's arguments begin +// and end — a wrong split feeds a formula the wrong operands with no error, just +// a wrong result. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { parseFunctionArgs } from "../../src/engine/evaluator.ts"; + +describe("parseFunctionArgs — plain splitting", () => { + it("splits comma-separated arguments and trims each", () => { + assert.deepEqual(parseFunctionArgs("A1, B2, C3"), ["A1", "B2", "C3"]); + assert.deepEqual(parseFunctionArgs("1,2,3"), ["1", "2", "3"]); + }); + + it("returns a single argument when there is no comma", () => { + assert.deepEqual(parseFunctionArgs("A1"), ["A1"]); + assert.deepEqual(parseFunctionArgs("A1:A10"), ["A1:A10"]); + }); + + it("returns nothing for an empty string", () => { + assert.deepEqual(parseFunctionArgs(""), []); + assert.deepEqual(parseFunctionArgs(" "), []); + }); +}); + +describe("parseFunctionArgs — nesting", () => { + // A comma inside a nested call belongs to that call, not the outer one: + // SUM(A1, MAX(B1, C1)) is two arguments, not three. + it("keeps a comma inside a nested function with its call", () => { + assert.deepEqual(parseFunctionArgs("A1, MAX(B1, C1)"), ["A1", "MAX(B1, C1)"]); + }); + + it("handles multiple and deeply nested calls", () => { + assert.deepEqual(parseFunctionArgs("SUM(A1,A2), COUNT(B1,B2,B3)"), ["SUM(A1,A2)", "COUNT(B1,B2,B3)"]); + assert.deepEqual(parseFunctionArgs("ROUND(SUM(A1,A2)/COUNT(A1,A2), 2)"), ["ROUND(SUM(A1,A2)/COUNT(A1,A2), 2)"]); + }); + + it("splits at the top level around a nested call", () => { + assert.deepEqual(parseFunctionArgs("IF(A1>0, 1, 0), B1"), ["IF(A1>0, 1, 0)", "B1"]); + }); +}); + +describe("parseFunctionArgs — strings", () => { + // A comma or a parenthesis inside a quoted string is text, not structure. + it("keeps a comma inside a quoted string", () => { + assert.deepEqual(parseFunctionArgs('"a, b", C1'), ['"a, b"', "C1"]); + assert.deepEqual(parseFunctionArgs("'x, y', 1"), ["'x, y'", "1"]); + }); + + it("keeps a parenthesis inside a quoted string from disturbing the depth", () => { + assert.deepEqual(parseFunctionArgs('"f(x)", A1'), ['"f(x)"', "A1"]); + assert.deepEqual(parseFunctionArgs('SUM(A1), "not )a close"'), ["SUM(A1)", '"not )a close"']); + }); + + it("preserves the quotes on a quoted argument", () => { + assert.deepEqual(parseFunctionArgs('"hello"'), ['"hello"']); + }); + + // A quote is only a boundary when it is not backslash-escaped, so an escaped + // quote inside a string does not close it early. + it("does not treat a backslash-escaped quote as a boundary", () => { + assert.deepEqual(parseFunctionArgs('"say \\"hi\\", ok", B1'), ['"say \\"hi\\", ok"', "B1"]); + }); +}); + +describe("parseFunctionArgs — edge behaviour worth pinning", () => { + // Documented current behaviour, not an endorsement: a trailing empty argument + // is dropped, so IF(A1>0,"yes",) reads as TWO arguments, not three (#2359). A + // leading or interior empty argument is kept. + it("drops a trailing empty argument", () => { + assert.deepEqual(parseFunctionArgs('A1, "yes",'), ["A1", '"yes"']); + }); + + it("keeps a leading or interior empty argument", () => { + assert.deepEqual(parseFunctionArgs(",B1"), ["", "B1"]); + assert.deepEqual(parseFunctionArgs("A1,,C1"), ["A1", "", "C1"]); + }); +}); diff --git a/tests/engine/test_parseRangeBounds.ts b/tests/engine/test_parseRangeBounds.ts new file mode 100644 index 0000000..b7f60d2 --- /dev/null +++ b/tests/engine/test_parseRangeBounds.ts @@ -0,0 +1,94 @@ +// `parseRangeBounds` is the single range parser the lookup functions now share +// (#2396). It carries the sheet-prefix split that one of the four former copies +// lacked — the copy that made cross-sheet VLOOKUP throw (#2390). Columns come +// back 0-based (A=0), rows stay 1-based (A1 notation). + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { parseRangeBounds, resolveIndexTarget, type RangeBounds } from "../../src/engine/formulaRefs.ts"; + +describe("parseRangeBounds — plain ranges", () => { + it("parses A1:B10 (cols 0-based, rows 1-based)", () => { + assert.deepEqual(parseRangeBounds("A1:B10"), { sheetPrefix: "", startCol: 0, startRow: 1, endCol: 1, endRow: 10 }); + }); + + it("parses an offset range A2:C10", () => { + assert.deepEqual(parseRangeBounds("A2:C10"), { sheetPrefix: "", startCol: 0, startRow: 2, endCol: 2, endRow: 10 }); + }); + + // Column boundaries: A is Excel column 1 → index 0, Z is 26 → 25, AA is 27 → 26. + it("parses the A / Z / AA column boundaries", () => { + assert.equal(parseRangeBounds("A1:A1")?.startCol, 0); + assert.equal(parseRangeBounds("Z1:Z1")?.startCol, 25); + assert.equal(parseRangeBounds("AA1:AB2")?.startCol, 26); + assert.equal(parseRangeBounds("AA1:AB2")?.endCol, 27); + }); + + it("keeps a single-cell-wide/tall range (start == end)", () => { + assert.deepEqual(parseRangeBounds("B3:B3"), { sheetPrefix: "", startCol: 1, startRow: 3, endCol: 1, endRow: 3 }); + }); +}); + +describe("parseRangeBounds — sheet-qualified ranges (the #2390 case)", () => { + it("splits an unquoted sheet prefix", () => { + assert.deepEqual(parseRangeBounds("Sheet1!A2:C10"), { sheetPrefix: "Sheet1!", startCol: 0, startRow: 2, endCol: 2, endRow: 10 }); + }); + + it("splits a quoted sheet name containing a space", () => { + assert.deepEqual(parseRangeBounds("'My Sheet'!A1:B2"), { sheetPrefix: "'My Sheet'!", startCol: 0, startRow: 1, endCol: 1, endRow: 2 }); + }); +}); + +describe("parseRangeBounds — non-ranges return null", () => { + it("rejects a single cell (no colon)", () => { + assert.equal(parseRangeBounds("A1"), null); + assert.equal(parseRangeBounds("Sheet1!A1"), null); + }); + + it("rejects malformed input", () => { + assert.equal(parseRangeBounds("not-a-range"), null); + assert.equal(parseRangeBounds(""), null); + }); + + // Deliberate limitation, pinned so it is not "fixed" by accident: the parser + // matches uppercase `[A-Z]` with no `$`, exactly as the four former copies did. + // The engine's own cell reader (calculator.getCellValue) is likewise + // uppercase-only, so accepting these here would not make them resolve. + it("rejects lowercase and $-absolute ranges (matches prior lookup behaviour)", () => { + assert.equal(parseRangeBounds("a1:b2"), null); + assert.equal(parseRangeBounds("$A$1:$B$2"), null); + }); +}); + +describe("resolveIndexTarget — INDEX bounds (#2390)", () => { + // A2:B5 → cols 0..1, rows 2..5 (4 rows × 2 cols). + const bounds: RangeBounds = { sheetPrefix: "", startCol: 0, startRow: 2, endCol: 1, endRow: 5 }; + + it("resolves an in-range 1-based position to an absolute cell", () => { + assert.deepEqual(resolveIndexTarget(bounds, 1, 1), { colIndex: 0, row: 2 }); // A2 + assert.deepEqual(resolveIndexTarget(bounds, 4, 2), { colIndex: 1, row: 5 }); // B5 + }); + + it("returns null (→ #REF!) when the row is past the range", () => { + const single: RangeBounds = { sheetPrefix: "", startCol: 0, startRow: 1, endCol: 0, endRow: 3 }; + assert.equal(resolveIndexTarget(single, 5, 1), null); // INDEX(A1:A3,5) + }); + + it("returns null for a row/col below 1", () => { + assert.equal(resolveIndexTarget(bounds, -1, 1), null); + assert.equal(resolveIndexTarget(bounds, 1, 3), null); // col past the 2-wide range + }); + + it("treats Excel's 0 (whole line) as out of range for a multi-line dimension", () => { + assert.equal(resolveIndexTarget(bounds, 0, 1), null); // INDEX(A2:B5,0,1) — must NOT read A1 + }); + + it("collapses 0 to the only line when that dimension is a single cell", () => { + const oneRow: RangeBounds = { sheetPrefix: "", startCol: 0, startRow: 3, endCol: 2, endRow: 3 }; + assert.deepEqual(resolveIndexTarget(oneRow, 0, 2), { colIndex: 1, row: 3 }); // whole (single) row, col 2 → B3 + }); + + it("truncates a fractional index toward zero, like Excel", () => { + assert.deepEqual(resolveIndexTarget(bounds, 2.9, 1), { colIndex: 0, row: 3 }); // 2.9 → row 2 → A3 + }); +}); diff --git a/tests/engine/test_parser.ts b/tests/engine/test_parser.ts new file mode 100644 index 0000000..098ee90 --- /dev/null +++ b/tests/engine/test_parser.ts @@ -0,0 +1,93 @@ +// A1-notation column conversion. Every other reference-handling path in the +// engine is built on these two, and both fail silently: `columnToIndex` does +// arithmetic on char codes with no validation, so a bad input returns a +// plausible-looking number instead of throwing, and the caller reads the wrong +// column. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { columnToIndex, indexToColumn } from "../../src/engine/parser.ts"; + +describe("columnToIndex", () => { + it("maps the single-letter columns", () => { + assert.equal(columnToIndex("A"), 0); + assert.equal(columnToIndex("B"), 1); + assert.equal(columnToIndex("Z"), 25); + }); + + // The bijective-base-26 boundary: AA follows Z, not "BA" or index 26+1. + it("maps the two-letter columns across the Z→AA boundary", () => { + assert.equal(columnToIndex("AA"), 26); + assert.equal(columnToIndex("AB"), 27); + assert.equal(columnToIndex("AZ"), 51); + assert.equal(columnToIndex("BA"), 52); + assert.equal(columnToIndex("ZZ"), 701); + }); + + it("maps three-letter columns", () => { + assert.equal(columnToIndex("AAA"), 702); + // XFD is Excel's last column (16384 columns, 0-based → 16383). + assert.equal(columnToIndex("XFD"), 16383); + }); + + // Documented current behaviour, NOT an endorsement: the function does no + // validation, so these return numbers rather than failing. Callers all + // pre-match `[A-Z]+`, which is what keeps the garbage out today — pinned so + // that if a caller's regex is ever relaxed, the consequence is visible here. + it("returns a wrong-but-plausible index for lowercase input (no validation)", () => { + // 'a' is 97; 97 - 64 - 1 = 32, i.e. column AG. + assert.equal(columnToIndex("a"), 32); + assert.equal(columnToIndex("z"), 57); + }); + + it("returns -1 for the empty string", () => { + assert.equal(columnToIndex(""), -1); + }); +}); + +describe("indexToColumn", () => { + it("maps the single-letter columns", () => { + assert.equal(indexToColumn(0), "A"); + assert.equal(indexToColumn(1), "B"); + assert.equal(indexToColumn(25), "Z"); + }); + + it("maps the two-letter columns across the Z→AA boundary", () => { + assert.equal(indexToColumn(26), "AA"); + assert.equal(indexToColumn(27), "AB"); + assert.equal(indexToColumn(51), "AZ"); + assert.equal(indexToColumn(52), "BA"); + assert.equal(indexToColumn(701), "ZZ"); + }); + + it("maps three-letter columns", () => { + assert.equal(indexToColumn(702), "AAA"); + assert.equal(indexToColumn(16383), "XFD"); + }); + + it("returns an empty string for a negative index", () => { + assert.equal(indexToColumn(-1), ""); + }); +}); + +describe("columnToIndex / indexToColumn round-trip", () => { + // The pair is used in both directions on the same value (range expansion + // walks indices, then renders refs back), so an asymmetry anywhere in the + // range silently shifts a whole range by one column. + it("round-trips every index across the single/double/triple letter boundaries", () => { + const boundaries = [0, 1, 24, 25, 26, 27, 50, 51, 52, 700, 701, 702, 703, 16382, 16383]; + for (const index of boundaries) { + assert.equal(columnToIndex(indexToColumn(index)), index, `round-trip failed at ${index}`); + } + }); + + it("round-trips a contiguous span with no gaps or repeats", () => { + const seen = new Set(); + for (let index = 0; index <= 1000; index++) { + const col = indexToColumn(index); + assert.equal(seen.has(col), false, `duplicate column label ${col} at index ${index}`); + seen.add(col); + assert.equal(columnToIndex(col), index); + } + }); +}); diff --git a/tests/engine/test_rangeFunctions.ts b/tests/engine/test_rangeFunctions.ts new file mode 100644 index 0000000..ecceaa9 --- /dev/null +++ b/tests/engine/test_rangeFunctions.ts @@ -0,0 +1,45 @@ +// Range-consuming functions through the whole engine. `expandRangeOrCell` is +// unit-tested on its own; this drives the shapes that used to return 0 with no +// error — the failure that is invisible because 0 is a plausible answer (#2356). + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { SpreadsheetEngine, type SheetData } from "../../src/engine/index.ts"; +import { cellAt } from "./cellAccess.ts"; + +/** A single column of numbers with `formula` beside the first cell. */ +const column = (values: number[], formula: string): SheetData => ({ + name: "S", + data: values.map((value, index) => (index === 0 ? [{ v: value }, { v: formula }] : [{ v: value }])), +}); + +const result = (values: number[], formula: string): unknown => cellAt(new SpreadsheetEngine().calculate(column(values, formula)).data, 0, 1); + +describe("range references that used to return 0", () => { + it("sums an absolute range", () => { + assert.equal(result([10, 20, 30], "=SUM($A$1:$A$3)"), 60); + }); + + it("sums a lowercase range", () => { + assert.equal(result([10, 20, 30], "=sum(a1:a3)"), 60); + }); + + it("sums a single-cell argument", () => { + assert.equal(result([42], "=SUM(A1)"), 42); + }); + + // MAX/MIN took a different path already, so they worked where SUM did not. + // The two must agree now that both go through the same expansion. + it("makes SUM and MAX agree on a single cell", () => { + assert.equal(result([42], "=SUM(A1)"), result([42], "=MAX(A1)")); + }); +}); + +describe("range functions over an absolute range", () => { + it("averages, counts and finds extremes", () => { + assert.equal(result([10, 20, 30], "=AVERAGE($A$1:$A$3)"), 20); + assert.equal(result([10, 20, 30], "=COUNT($A$1:$A$3)"), 3); + assert.equal(result([10, 20, 30], "=MAX($A$1:$A$3)"), 30); + assert.equal(result([10, 20, 30], "=MIN($A$1:$A$3)"), 10); + }); +}); diff --git a/tests/engine/test_registry.ts b/tests/engine/test_registry.ts new file mode 100644 index 0000000..98e7c0d --- /dev/null +++ b/tests/engine/test_registry.ts @@ -0,0 +1,220 @@ +// The two helpers every conditional and arithmetic function leans on. +// +// Neither can fail loudly: `toNumber` returns 0 for anything it cannot read, +// and `parseCriteria` returns a predicate that answers false. So a mistake here +// does not surface as an error — a SUM comes out smaller than it should, or a +// COUNTIF reports zero matches, and both look like ordinary answers. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { parseCriteria, toNumber, toString } from "../../src/engine/registry.ts"; +import type { CellValue } from "../../src/engine/types.ts"; + +const num = (value: CellValue) => toNumber(value); + +describe("toNumber — numbers pass through", () => { + it("returns a number unchanged, including 0 and negatives", () => { + assert.equal(num(42), 42); + assert.equal(num(0), 0); + assert.equal(num(-7.5), -7.5); + }); +}); + +describe("toNumber — formatted strings", () => { + it("reads a percentage as its decimal value", () => { + assert.equal(num("5%"), 0.05); + assert.equal(num("100%"), 1); + // Dividing by 100 is a binary-float operation, so a value with enough + // decimals lands a rounding step away from the exact literal. + assert.ok(Math.abs(num("0.4167%") - 0.004167) < 1e-12); + }); + + it("reads a currency string, stripping the symbol and separators", () => { + assert.equal(num("$1000"), 1000); + assert.equal(num("$1,000"), 1000); + assert.equal(num("$1,000.50"), 1000.5); + }); + + it("reads a comma-separated number", () => { + assert.equal(num("1,000"), 1000); + assert.equal(num("1,234,567"), 1234567); + }); + + it("reads a plain numeric string, tolerating surrounding whitespace", () => { + assert.equal(num("42"), 42); + assert.equal(num(" 42 "), 42); + assert.equal(num("-3.5"), -3.5); + assert.equal(num("1e3"), 1000); + }); +}); + +describe("toNumber — the branch order is load-bearing", () => { + // The checks run `%` → `$` → `,` → plain, and each strips only its OWN + // characters. A string carrying two of them therefore falls into the first + // branch and fails to parse there, yielding 0 rather than a number. + it("returns 0 for a string mixing a percent with a currency symbol", () => { + assert.equal(num("$1,000%"), 0); + }); + + it("reads a percent string with thousands separators off by three orders of magnitude", () => { + // The `%` branch strips only "%", so the comma survives and `parseFloat` + // stops there: "1,000" reads as 1, then /100 gives 0.01. The value a user + // means by "1,000%" is 10. Pinned as current behaviour, not as correct. + assert.equal(num("1,000%"), 0.01); + }); + + it("handles a currency string with a percent-free comma correctly", () => { + assert.equal(num("$2,500.25"), 2500.25); + }); +}); + +describe("toNumber — everything unreadable becomes 0", () => { + // This is the silent path: a text cell inside a SUM range contributes 0 + // instead of raising, so the total is quietly short. + it("returns 0 for non-numeric text", () => { + assert.equal(num("hello"), 0); + assert.equal(num(""), 0); + assert.equal(num(" "), 0); + assert.equal(num("N/A"), 0); + }); + + // `CellValue` is `number | string | boolean`, so an empty cell is never + // null here — `getRawValue` in the calculator maps blanks to 0 before this + // is reached. Booleans, though, are in the type and become 0 rather than + // 1/0 as Excel would have them. + it("returns 0 for booleans, not Excel's 1 and 0", () => { + assert.equal(num(true), 0); + assert.equal(num(false), 0); + }); + + // `parseFloat` stops at the first character it cannot read rather than + // rejecting the string, so a partly-numeric cell contributes its prefix. + it("takes the leading number from a partly-numeric string", () => { + assert.equal(num("12abc"), 12); + assert.equal(num("3.5kg"), 3.5); + }); + + it("returns 0 when the number does not come first", () => { + assert.equal(num("abc12"), 0); + }); +}); + +describe("toString", () => { + it("stringifies every shape `CellValue` allows", () => { + assert.equal(toString(42), "42"); + assert.equal(toString("text"), "text"); + assert.equal(toString(true), "true"); + assert.equal(toString(false), "false"); + assert.equal(toString(""), ""); + assert.equal(toString(0), "0"); + }); +}); + +describe("parseCriteria — comparison operators", () => { + it("compares greater-than and greater-or-equal at the boundary", () => { + const greater = parseCriteria(">5"); + assert.equal(greater(5), false); + assert.equal(greater(6), true); + + const gte = parseCriteria(">=5"); + assert.equal(gte(4), false); + assert.equal(gte(5), true, "the boundary value must count for >="); + assert.equal(gte(6), true); + }); + + it("compares less-than and less-or-equal at the boundary", () => { + const less = parseCriteria("<5"); + assert.equal(less(5), false); + assert.equal(less(4), true); + + const lte = parseCriteria("<=5"); + assert.equal(lte(6), false); + assert.equal(lte(5), true, "the boundary value must count for <="); + }); + + it("accepts both spellings of equality and inequality", () => { + for (const criteria of ["=5", "==5"]) { + assert.equal(parseCriteria(criteria)(5), true, `${criteria} should match 5`); + assert.equal(parseCriteria(criteria)(6), false); + } + for (const criteria of ["!=5", "<>5"]) { + assert.equal(parseCriteria(criteria)(5), false, `${criteria} should not match 5`); + assert.equal(parseCriteria(criteria)(6), true); + } + }); + + it("strips surrounding quotes before reading the operator", () => { + assert.equal(parseCriteria('">5"')(6), true); + assert.equal(parseCriteria("'>5'")(6), true); + }); + + it("tolerates whitespace around the criteria", () => { + assert.equal(parseCriteria(" >5 ")(6), true); + }); +}); + +describe("parseCriteria — the ways it silently matches nothing", () => { + // An unrecognised operator lands in the `default` arm, which answers false + // for every value. A COUNTIF written this way reports 0 matches and reads + // like a real answer. + it("matches nothing for an operator written backwards", () => { + const backwards = parseCriteria("=>5"); + assert.equal(backwards(5), false); + assert.equal(backwards(6), false); + assert.equal(backwards(4), false); + }); + + // A non-numeric comparand makes every numeric comparison false, since + // `NaN > x` and `NaN < x` are both false. + it("matches nothing when a comparison operator is given non-numeric text", () => { + const gtText = parseCriteria(">abc"); + assert.equal(gtText(0), false); + assert.equal(gtText(1000), false); + assert.equal(gtText("abc"), false); + }); + + // `*` and `?` are Excel wildcards; `~` escapes them back to literals. + it("treats a wildcard as a pattern, and ~ escapes it", () => { + const wildcard = parseCriteria("app*"); + assert.equal(wildcard("apple"), true); + assert.equal(wildcard("app"), true, "* may match nothing"); + assert.equal(wildcard("axe"), false); + assert.equal(parseCriteria("app~*")("app*"), true, "escaped, so only the literal matches"); + assert.equal(parseCriteria("app~*")("apple"), false); + }); +}); + +describe("parseCriteria — exact match", () => { + it("matches a plain string exactly", () => { + const apple = parseCriteria("apple"); + assert.equal(apple("apple"), true); + assert.equal(apple("Apple"), true, "matching is case-insensitive, as in Excel"); + assert.equal(apple("apples"), false); + }); + + it("matches a number written either as text or as a number", () => { + const five = parseCriteria("5"); + assert.equal(five(5), true); + assert.equal(five("5"), true); + assert.equal(five("5.0"), true, "numeric equality catches a different spelling"); + assert.equal(five(6), false); + }); + + // `toNumber` turns unreadable values into 0, so a criteria of "0" matches + // every text cell in the range. Pinned because it inflates a COUNTIF + // without any sign that something went wrong. + it("matches unreadable text when the criteria is 0", () => { + const zero = parseCriteria("0"); + assert.equal(zero(0), true); + assert.equal(zero("hello"), true, "toNumber('hello') is 0, so this counts"); + }); + + it("strips quotes around an exact-match criteria too", () => { + assert.equal(parseCriteria('"apple"')("apple"), true); + }); + + it("matches an empty criteria against an empty string", () => { + assert.equal(parseCriteria("")(""), true); + assert.equal(parseCriteria("")("x"), false); + }); +}); diff --git a/tests/engine/test_requiredArg.ts b/tests/engine/test_requiredArg.ts new file mode 100644 index 0000000..1cdce48 --- /dev/null +++ b/tests/engine/test_requiredArg.ts @@ -0,0 +1,102 @@ +// #2736: adopting `noUncheckedIndexedAccess` turned every `args[N]` in a +// function handler into `string | undefined`. `requiredArg` is the single reader +// that resolves it — and the reason it can be a *reader* rather than a default +// is that the evaluator validates the registry's `minArgs` before any handler +// runs. These tests pin both halves of that contract. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { functionRegistry, requiredArg, tooFewArgumentsError, type FunctionContext } from "../../src/engine/registry.ts"; +import "../../src/engine/functions/index.ts"; +import { substituteCellRefs, findCellRefs } from "../../src/engine/evaluator.ts"; + +const stubContext = (functionName: string): FunctionContext => ({ + functionName, + getCellValue: () => 1, + getRangeValues: () => [1, 2, 3], + getRangeValuesRaw: () => [1, 2, 3], + evaluateFormula: () => 1, +}); + +/** The wording the evaluator uses for an arity violation, which `requiredArg` + * must reproduce rather than invent a second message for. */ +const TOO_FEW_MESSAGE = /^[A-Z]+ requires at least \d+ arguments?$/; + +describe("requiredArg", () => { + it("returns the argument at the index", () => { + assert.equal(requiredArg(stubContext("SUM"), ["A1", "B1"], 1), "B1"); + }); + + it("throws the evaluator's own arity wording, naming the function and the count it needed", () => { + assert.throws(() => requiredArg(stubContext("ROUND"), ["A1"], 1), { message: "ROUND requires at least 2 arguments" }); + }); + + it("uses the singular for a one-argument minimum, like the evaluator does", () => { + assert.equal(tooFewArgumentsError("UPPER", 1).message, "UPPER requires at least 1 argument"); + assert.throws(() => requiredArg(stubContext("UPPER"), [], 0), { message: "UPPER requires at least 1 argument" }); + }); + + it("never substitutes a default — an absent argument is an error, not a 0 or an empty string", () => { + assert.throws(() => requiredArg(stubContext("MID"), ["A1", "1"], 2), { message: TOO_FEW_MESSAGE }); + }); +}); + +describe("every registered function's minArgs covers what its handler reads", () => { + // The audit behind #2736's spreadsheet pass, kept as a permanent guard: a + // registration whose minArgs is LOWER than the highest index its handler reads + // unguarded is a crash path reachable from a user formula, because the + // evaluator's arity gate would let that call through. Calling each handler + // with exactly minArgs arguments makes `requiredArg` the detector. + functionRegistry.getAllFunctions().forEach((definition) => { + it(`${definition.name} reads no argument beyond its declared minimum`, () => { + const minArgs = definition.minArgs ?? 0; + const args = Array.from({ length: minArgs }, () => "A1"); + try { + definition.handler(args, stubContext(definition.name)); + } catch (error) { + // A handler may fail for unrelated reasons (a stub range is not a real + // table); only the arity wording means minArgs under-declares. + const message = error instanceof Error ? error.message : String(error); + assert.doesNotMatch(message, TOO_FEW_MESSAGE, `${definition.name} read past its declared minArgs=${minArgs}`); + } + }); + }); +}); + +describe("substituteCellRefs", () => { + // #2357: a global string replace rewrote every occurrence of the SHORTER + // reference first, so A10 became "0" and the cell showed a + // plausible wrong number. Substituting back to front is what prevents it. + it("substitutes back to front, so a longer reference is not broken by a shorter prefix", () => { + const expr = "A1+A10"; + const result = substituteCellRefs(expr, findCellRefs(expr), (ref) => (ref === "A1" ? "5" : "7")); + assert.equal(result, "5+7"); + }); + + it("leaves the caller's span list untouched", () => { + const spans = findCellRefs("A1+B2"); + substituteCellRefs("A1+B2", spans, () => "0"); + assert.deepEqual( + spans.map((span) => span.ref), + ["A1", "B2"], + ); + }); + + it("returns the expression unchanged when there is nothing to substitute", () => { + assert.equal( + substituteCellRefs("1+2", [], () => "9"), + "1+2", + ); + }); + + it("propagates a throw from the renderer, so an errored reference still poisons the expression", () => { + const expr = "A1+1"; + assert.throws( + () => + substituteCellRefs(expr, findCellRefs(expr), () => { + throw new Error("#DIV/0!"); + }), + { message: "#DIV/0!" }, + ); + }); +}); diff --git a/tests/engine/test_responseDecoder.ts b/tests/engine/test_responseDecoder.ts new file mode 100644 index 0000000..4d46122 --- /dev/null +++ b/tests/engine/test_responseDecoder.ts @@ -0,0 +1,87 @@ +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { decodeSpreadsheetResponse } from "../../src/engine/responseDecoder.js"; + +describe("decodeSpreadsheetResponse", () => { + it("ok when kind is text with valid JSON array content", () => { + const sheets = [{ name: "Sheet1", data: [] }]; + const result = decodeSpreadsheetResponse({ + kind: "text", + content: JSON.stringify(sheets), + }); + assert.equal(result.kind, "ok"); + if (result.kind !== "ok") return; + assert.deepEqual(result.sheets, sheets); + }); + + it("ok when kind is missing (legacy response)", () => { + const sheets = [{ name: "A", data: [[{ v: 1 }]] }]; + const result = decodeSpreadsheetResponse({ + content: JSON.stringify(sheets), + }); + assert.equal(result.kind, "ok"); + }); + + it("error when kind is too-large", () => { + const result = decodeSpreadsheetResponse({ + kind: "too-large", + message: "File exceeds size limit", + }); + assert.equal(result.kind, "error"); + if (result.kind !== "error") return; + assert.equal(result.message, "File exceeds size limit"); + }); + + it("error when kind is binary", () => { + const result = decodeSpreadsheetResponse({ kind: "binary" }); + assert.equal(result.kind, "error"); + if (result.kind !== "error") return; + assert.match(result.message, /binary/); + }); + + it("error when content is missing", () => { + const result = decodeSpreadsheetResponse({ kind: "text" }); + assert.equal(result.kind, "error"); + if (result.kind !== "error") return; + assert.match(result.message, /no content/i); + }); + + it("error when content is not a string", () => { + // unsafely cast to exercise the runtime branch + const result = decodeSpreadsheetResponse({ + kind: "text", + content: 123 as unknown as string, + }); + assert.equal(result.kind, "error"); + }); + + it("error when JSON is malformed", () => { + const result = decodeSpreadsheetResponse({ + kind: "text", + content: "{not valid json", + }); + assert.equal(result.kind, "error"); + if (result.kind !== "error") return; + assert.match(result.message, /malformed/i); + }); + + it("error when content is valid JSON but not an array", () => { + const result = decodeSpreadsheetResponse({ + kind: "text", + content: '{"name": "Sheet1"}', + }); + assert.equal(result.kind, "error"); + if (result.kind !== "error") return; + assert.match(result.message, /not an array/i); + }); + + it("empty array is ok (new spreadsheet)", () => { + const result = decodeSpreadsheetResponse({ + kind: "text", + content: "[]", + }); + assert.equal(result.kind, "ok"); + if (result.kind !== "ok") return; + assert.deepEqual(result.sheets, []); + }); +}); diff --git a/tests/engine/test_serialFromParts.ts b/tests/engine/test_serialFromParts.ts new file mode 100644 index 0000000..db5764d --- /dev/null +++ b/tests/engine/test_serialFromParts.ts @@ -0,0 +1,48 @@ +// serialFromParts folds the validate -> Date.UTC -> dateToSerial tail that every +// dated branch of parseDate repeated (#2482). parseDate now delegates to it, so +// pinning the rule directly here is the only place a wrong month offset or a +// dropped validity check is caught independently of parseDate itself. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { serialFromParts } from "../../src/engine/date-parser.ts"; + +describe("serialFromParts — valid triples map to the Excel serial", () => { + it("March 4, 2025 is serial 45720", () => { + assert.equal(serialFromParts(2025, 3, 4), 45720); + }); + + it("Jan 1, 1900 is serial 2 (Excel's Dec-30-1899 base with the 1900 leap-year quirk)", () => { + assert.equal(serialFromParts(1900, 1, 1), 2); + }); + + it("accepts a real leap day (Feb 29, 2024)", () => { + assert.equal(serialFromParts(2024, 2, 29), 45351); + }); + + it("accepts the upper year boundary (Dec 31, 2100)", () => { + assert.equal(serialFromParts(2100, 12, 31), 73415); + }); +}); + +describe("serialFromParts — invalid triples are null", () => { + it("rejects Feb 29 on a non-leap year", () => { + assert.equal(serialFromParts(2025, 2, 29), null); + }); + + it("rejects a day past the month length (Feb 30)", () => { + assert.equal(serialFromParts(2025, 2, 30), null); + }); + + it("rejects a month above 12", () => { + assert.equal(serialFromParts(2025, 13, 1), null); + }); + + it("rejects a year below the 1900 floor", () => { + assert.equal(serialFromParts(1800, 1, 1), null); + }); + + it("rejects a zero day", () => { + assert.equal(serialFromParts(2025, 1, 0), null); + }); +}); diff --git a/tests/engine/test_spreadsheetErrors.ts b/tests/engine/test_spreadsheetErrors.ts new file mode 100644 index 0000000..1a71958 --- /dev/null +++ b/tests/engine/test_spreadsheetErrors.ts @@ -0,0 +1,71 @@ +// The error taxonomy: which CODES the engine knows, and which values count as +// an error RESULT. Since #2451 an error is its own value, so `isErrorResult` +// deliberately rejects a look-alike string — that is what lets IFERROR tell +// SQRT(-1) apart from CONCAT("#N","UM!"). The provenance behaviour itself is +// covered in test_errorValue.ts. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { + isSpreadsheetError, + isErrorResult, + errorCodeOf, + spreadsheetError, + SPREADSHEET_ERRORS, +} from "../../src/engine/spreadsheet-errors.ts"; + +describe("isSpreadsheetError", () => { + it("recognises the Excel error strings", () => { + assert.equal(isSpreadsheetError("#NUM!"), true); + assert.equal(isSpreadsheetError("#DIV/0!"), true); + assert.equal(isSpreadsheetError("#N/A"), true); + }); + + it("recognises every code in the taxonomy, including the engine's own #ERROR!", () => { + assert.deepEqual( + SPREADSHEET_ERRORS.filter((code) => !isSpreadsheetError(code)), + [], + ); + }); + + it("rejects ordinary text and non-strings", () => { + assert.equal(isSpreadsheetError("hello"), false); + assert.equal(isSpreadsheetError("#NOPE!"), false); + assert.equal(isSpreadsheetError(0), false); + assert.equal(isSpreadsheetError(null), false); + }); +}); + +describe("errorCodeOf", () => { + it("reads the code off an error value and off a string that spells one", () => { + assert.equal(errorCodeOf(spreadsheetError("#REF!")), "#REF!"); + assert.equal(errorCodeOf("#REF!"), "#REF!"); + }); + + it("is null for anything else", () => { + assert.equal(errorCodeOf("#OOPS!"), null); + assert.equal(errorCodeOf(7), null); + assert.equal(errorCodeOf(undefined), null); + }); +}); + +describe("isErrorResult", () => { + it("treats error values, NaN/∞ and missing values as errors", () => { + assert.equal(isErrorResult(spreadsheetError("#DIV/0!")), true); + assert.equal(isErrorResult(NaN), true); + assert.equal(isErrorResult(Infinity), true); + assert.equal(isErrorResult(null), true); + assert.equal(isErrorResult(undefined), true); + }); + + it("passes ordinary values through", () => { + assert.equal(isErrorResult(0), false); + assert.equal(isErrorResult(42), false); + assert.equal(isErrorResult("text"), false); + }); + + it("does NOT treat a string that merely spells an error as one (#2451)", () => { + assert.equal(isErrorResult("#DIV/0!"), false); + assert.equal(isErrorResult("#NUM!"), false); + }); +}); diff --git a/tests/engine/test_statisticalFunctions.ts b/tests/engine/test_statisticalFunctions.ts new file mode 100644 index 0000000..c5f8a98 --- /dev/null +++ b/tests/engine/test_statisticalFunctions.ts @@ -0,0 +1,101 @@ +// STDEV / VAR through the whole engine (#2360). Excel's STDEV / VAR are the +// SAMPLE estimators (divide by n-1); the engine used to divide by n, which is +// the POPULATION estimator (Excel's STDEVP / VARP) — a silent wrong answer. +// A single value has no n-1 to divide by, so Excel reports #DIV/0!. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { SpreadsheetEngine } from "../../src/engine/index.ts"; +import { cellAt } from "./cellAccess.ts"; + +/** Calculate `formula` in the cell just below a single column of `values`. */ +const evalOverColumn = (values: (string | number)[], formula: string): unknown => { + const data: { v: string | number }[][] = values.map((value) => [{ v: value }]); + data.push([{ v: formula }]); + const result = new SpreadsheetEngine().calculate({ name: "S", data }); + return cellAt(result.data, data.length - 1, 0); +}; + +// {2,4,4,4,5,5,7,9}: mean 5, Σ(x-μ)² = 32. +// Sample: 32 / (8-1) = 4.5714… → stdev 2.1380… +// Population: 32 / 8 = 4 → stdev 2.0 (the old wrong answer). +const SAMPLE_VALUES = [2, 4, 4, 4, 5, 5, 7, 9]; + +describe("STDEV — sample estimator (#2360)", () => { + it("divides by n-1, not n", () => { + const result = evalOverColumn(SAMPLE_VALUES, "=STDEV(A1:A8)"); + assert.equal(typeof result, "number"); + assert.ok(Math.abs((result as number) - 2.138089935) < 1e-6, `expected ~2.1381 sample stdev, got ${result}`); + }); + + it("returns #DIV/0! for a single value (no n-1 to divide by)", () => { + assert.equal(evalOverColumn([42], "=STDEV(A1:A1)"), "#DIV/0!"); + }); +}); + +describe("VAR — sample estimator (#2360)", () => { + it("divides by n-1, not n", () => { + const result = evalOverColumn(SAMPLE_VALUES, "=VAR(A1:A8)"); + assert.equal(typeof result, "number"); + assert.ok(Math.abs((result as number) - 32 / 7) < 1e-9, `expected 32/7 sample variance, got ${result}`); + }); + + it("returns #DIV/0! for a single value", () => { + assert.equal(evalOverColumn([42], "=VAR(A1:A1)"), "#DIV/0!"); + }); +}); + +// A range holding no numbers is not a range of zeros. AVERAGE and MEDIAN used +// to answer 0 for it, which reads like a genuine result (#2360). MAX / MIN / +// SUM / COUNT are NOT part of this: Excel really does answer 0 there. + +const BLANKS = ["", "", ""]; +const TEXTS = ["apple", "banana", "cherry"]; + +describe("AVERAGE — no numbers to average is #DIV/0! (#2360)", () => { + it("returns #DIV/0! for an all-blank range", () => { + assert.equal(evalOverColumn(BLANKS, "=AVERAGE(A1:A3)"), "#DIV/0!"); + }); + + it("returns #DIV/0! for a text-only range (Excel ignores text)", () => { + assert.equal(evalOverColumn(TEXTS, "=AVERAGE(A1:A3)"), "#DIV/0!"); + }); + + it("still averages when at least one number is present", () => { + assert.equal(evalOverColumn(["", 10, 20], "=AVERAGE(A1:A3)"), 15); + }); + + it("is an error VALUE, so IFERROR catches it", () => { + assert.equal(evalOverColumn(BLANKS, "=IFERROR(AVERAGE(A1:A3), 99)"), 99); + }); +}); + +describe("MEDIAN — no numbers has no middle, so #NUM! (#2360)", () => { + it("returns #NUM! for an all-blank range", () => { + assert.equal(evalOverColumn(BLANKS, "=MEDIAN(A1:A3)"), "#NUM!"); + }); + + it("returns #NUM! for a text-only range", () => { + assert.equal(evalOverColumn(TEXTS, "=MEDIAN(A1:A3)"), "#NUM!"); + }); + + it("still takes the median when numbers are present", () => { + assert.equal(evalOverColumn([3, 1, 2], "=MEDIAN(A1:A3)"), 2); + assert.equal(evalOverColumn(["", 1, 3], "=MEDIAN(A1:A3)"), 2, "blanks are ignored, not averaged in as 0"); + }); + + it("is an error VALUE, so IFERROR catches it", () => { + assert.equal(evalOverColumn(BLANKS, "=IFERROR(MEDIAN(A1:A3), 99)"), 99); + }); +}); + +// Excel's own boundary for these four is 0, so the engine's 0 is correct and +// must NOT be "fixed" into an error. +describe("MAX / MIN / SUM / COUNT over an empty range stay 0 (Excel agrees)", () => { + it("answers 0 rather than an error", () => { + assert.equal(evalOverColumn(BLANKS, "=MAX(A1:A3)"), 0); + assert.equal(evalOverColumn(BLANKS, "=MIN(A1:A3)"), 0); + assert.equal(evalOverColumn(BLANKS, "=SUM(A1:A3)"), 0); + assert.equal(evalOverColumn(BLANKS, "=COUNT(A1:A3)"), 0); + }); +}); diff --git a/tests/engine/test_statisticalMath.ts b/tests/engine/test_statisticalMath.ts new file mode 100644 index 0000000..0d3fbc6 --- /dev/null +++ b/tests/engine/test_statisticalMath.ts @@ -0,0 +1,130 @@ +// MODE's rule in isolation (#2391): the most frequent value, or #N/A when nothing +// repeats. The old handler returned the FIRST value for an all-distinct set, a +// silent wrong answer that reads like a real mode. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { + computeAverage, + computeMedian, + computeMode, + sampleVariance, + sampleStdev, +} from "../../src/engine/functions/statistical-math.ts"; +import { NA_ERROR, DIV_ZERO_ERROR, NUM_ERROR } from "../../src/engine/spreadsheet-errors.ts"; + +// AVERAGE and MEDIAN of nothing used to be 0 — a number a reader cannot tell +// from a real average of zeros (#2360). Excel reports the absence, and with +// DIFFERENT codes: AVERAGE divides by a count of zero (#DIV/0!) while MEDIAN +// has no division at all, just no middle element (#NUM!). + +describe("computeAverage", () => { + it("is the arithmetic mean", () => { + assert.equal(computeAverage([10, 20, 30]), 20); + assert.equal(computeAverage([1, 2]), 1.5); + }); + + it("averages a single value to itself", () => { + assert.equal(computeAverage([7]), 7); + }); + + it("keeps stored zeros in the denominator", () => { + assert.equal(computeAverage([0, 0, 3]), 1); + }); + + it("returns #DIV/0! for an empty set", () => { + assert.equal(computeAverage([]), DIV_ZERO_ERROR); + }); +}); + +describe("computeMedian", () => { + it("returns the middle value of an odd-sized set", () => { + assert.equal(computeMedian([3, 1, 2]), 2); + }); + + it("averages the two middle values of an even-sized set", () => { + assert.equal(computeMedian([4, 1, 3, 2]), 2.5); + }); + + it("returns the value itself for a single-value set", () => { + assert.equal(computeMedian([7]), 7); + }); + + it("sorts numerically, not lexicographically", () => { + assert.equal(computeMedian([10, 9, 100]), 10); + }); + + it("does not reorder the caller's array", () => { + const values = [3, 1, 2]; + computeMedian(values); + assert.deepEqual(values, [3, 1, 2]); + }); + + it("returns #NUM! for an empty set", () => { + assert.equal(computeMedian([]), NUM_ERROR); + }); +}); + +describe("computeMode", () => { + it("returns the single most frequent value", () => { + assert.equal(computeMode([1, 2, 2, 3]), 2); + assert.equal(computeMode([5, 5, 5, 1, 2]), 5); + }); + + // The fix: no value repeats, so the mode is undefined → #N/A, not values[0]. + it("returns #N/A when every value is distinct", () => { + assert.equal(computeMode([1, 2, 3]), NA_ERROR); + assert.equal(computeMode([9]), NA_ERROR); + }); + + it("returns #N/A for an empty set", () => { + assert.equal(computeMode([]), NA_ERROR); + }); + + // On a tie the earliest-appearing value wins (Map preserves insertion order). + it("breaks a frequency tie toward the first-appearing value", () => { + assert.equal(computeMode([2, 2, 1, 1]), 2); + assert.equal(computeMode([1, 1, 2, 2]), 1); + }); + + it("counts repeated negatives and zeros", () => { + assert.equal(computeMode([0, 0, 1]), 0); + assert.equal(computeMode([-3, -3, 4]), -3); + }); +}); + +// STDEV / VAR are Excel's SAMPLE estimators: divide by n-1, not n. The old +// handler divided by n (the POPULATION estimator, Excel's STDEVP / VARP), a +// silent understatement of the spread (#2360). + +describe("sampleVariance", () => { + // {2,4,4,4,5,5,7,9}: mean 5, Σ(x-μ)² = 32. Sample 32/7 vs population 32/8 = 4. + it("divides the summed squared deviations by n-1", () => { + assert.equal(sampleVariance([2, 4, 4, 4, 5, 5, 7, 9]), 32 / 7); + }); + + it("gives 0.5 for two values one apart, not the population 0.25", () => { + assert.equal(sampleVariance([1, 2]), 0.5); + }); + + it("returns #DIV/0! for a single value (no n-1)", () => { + assert.equal(sampleVariance([5]), DIV_ZERO_ERROR); + }); + + it("returns #DIV/0! for an empty set", () => { + assert.equal(sampleVariance([]), DIV_ZERO_ERROR); + }); +}); + +describe("sampleStdev", () => { + it("is the square root of the sample variance", () => { + const result = sampleStdev([2, 4, 4, 4, 5, 5, 7, 9]); + assert.equal(typeof result, "number"); + assert.ok(Math.abs((result as number) - Math.sqrt(32 / 7)) < 1e-12); + }); + + it("propagates #DIV/0! from the fewer-than-two boundary", () => { + assert.equal(sampleStdev([5]), DIV_ZERO_ERROR); + assert.equal(sampleStdev([]), DIV_ZERO_ERROR); + }); +}); diff --git a/tests/engine/test_textFormat.ts b/tests/engine/test_textFormat.ts new file mode 100644 index 0000000..6221961 --- /dev/null +++ b/tests/engine/test_textFormat.ts @@ -0,0 +1,254 @@ +// Excel number-format codes as read by the TEXT function. +// +// TEXT used to ignore the pattern's shape and hard-code its own: `#,##0` +// grouping was dropped entirely, and the `$` / `%` branches always produced two +// decimals — so `TEXT(1234.5,"$#,##0.00")` returned "$1234.50", `TEXT(0.5,"0%")` +// returned "50.00%" and `TEXT(5,"$0")` returned "$5.00". Every one of those is a +// plausible-looking string, which is why they survived (#2360). +// +// The pattern interpreter is tested directly; the end-to-end path through +// SpreadsheetEngine confirms the TEXT handler wires it up. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { formatWithPattern, parseNumberPattern } from "../../src/engine/textFormat.ts"; +import { groupThousands } from "../../src/engine/formatter.ts"; +import { SpreadsheetEngine } from "../../src/engine/index.ts"; +import { cellAt } from "./cellAccess.ts"; + +const engine = new SpreadsheetEngine(); +const evalFormula = (formula: string): unknown => cellAt(engine.calculate(engine.createSheet("S", [[`=${formula}`]])).data, 0, 0); + +describe("formatWithPattern — the cases reported in #2360", () => { + it("groups digits for a currency pattern", () => { + assert.equal(formatWithPattern(1234.5, "$#,##0.00"), "$1,234.50"); + }); + + it("takes the percent decimals from the pattern instead of forcing two", () => { + assert.equal(formatWithPattern(0.5, "0%"), "50%"); + }); + + it("takes the currency decimals from the pattern instead of forcing two", () => { + assert.equal(formatWithPattern(5, "$0"), "$5"); + }); +}); + +describe("formatWithPattern — grouping boundaries", () => { + it("leaves a three-digit number ungrouped", () => { + assert.equal(formatWithPattern(999, "#,##0"), "999"); + }); + + it("groups from four digits up", () => { + assert.equal(formatWithPattern(1000, "#,##0"), "1,000"); + }); + + it("groups every third digit of a large number", () => { + assert.equal(formatWithPattern(1234567, "#,##0"), "1,234,567"); + }); + + it("groups the carried digit when rounding crosses a boundary", () => { + assert.equal(formatWithPattern(999.5, "#,##0"), "1,000"); + }); + + it("omits grouping when the pattern has no comma", () => { + assert.equal(formatWithPattern(1234567, "0"), "1234567"); + }); +}); + +describe("formatWithPattern — negatives and zero", () => { + it("puts the sign in front of the whole rendering, currency included", () => { + assert.equal(formatWithPattern(-1234.5, "$#,##0.00"), "-$1,234.50"); + }); + + it("keeps the sign on a grouped plain number", () => { + assert.equal(formatWithPattern(-1234.5, "#,##0.00"), "-1,234.50"); + }); + + it("renders zero with the pattern's decimals", () => { + assert.equal(formatWithPattern(0, "$#,##0.00"), "$0.00"); + }); + + // The sign follows the ROUNDED digits: a value that rounds away to zero must + // not render as "-0.00", which reads as a negative amount that isn't there. + it("drops the sign when rounding leaves nothing but zeros", () => { + assert.equal(formatWithPattern(-0.001, "0.00"), "0.00"); + }); + + it("keeps the sign when rounding leaves a non-zero digit", () => { + assert.equal(formatWithPattern(-0.006, "0.00"), "-0.01"); + }); +}); + +describe("formatWithPattern — decimals", () => { + it("rounds to the pattern's fixed decimals", () => { + assert.equal(formatWithPattern(1234.5678, "0.00"), "1234.57"); + }); + + it("pads to the pattern's fixed decimals", () => { + assert.equal(formatWithPattern(2, "0.000"), "2.000"); + }); + + it("rounds to a whole number when the pattern has no decimal point", () => { + assert.equal(formatWithPattern(1234.5, "0"), "1235"); + }); + + it("drops an optional trailing decimal written as #", () => { + assert.equal(formatWithPattern(0.5, "0.0#"), "0.5"); + }); + + it("keeps an optional decimal that is not a trailing zero", () => { + assert.equal(formatWithPattern(0.25, "0.0#"), "0.25"); + }); + + it("drops every decimal when all of them are optional", () => { + assert.equal(formatWithPattern(2, "0.##"), "2"); + }); + + it("pads the whole part to the pattern's leading zeros", () => { + assert.equal(formatWithPattern(5, "000"), "005"); + }); +}); + +describe("formatWithPattern — percent", () => { + it("scales by 100 and keeps the pattern's two decimals", () => { + assert.equal(formatWithPattern(0.1234, "0.00%"), "12.34%"); + }); + + it("scales a value above one", () => { + assert.equal(formatWithPattern(1, "0%"), "100%"); + }); + + it("groups a large percentage", () => { + assert.equal(formatWithPattern(12.3456, "#,##0.0%"), "1,234.6%"); + }); + + it("keeps the sign of a negative percentage", () => { + assert.equal(formatWithPattern(-0.5, "0%"), "-50%"); + }); +}); + +describe("formatWithPattern — literals", () => { + it("keeps a leading literal", () => { + assert.equal(formatWithPattern(1234.5, "USD #,##0.00"), "USD 1,234.50"); + }); + + it("keeps a trailing literal", () => { + assert.equal(formatWithPattern(1234.5, "#,##0.0 kg"), "1,234.5 kg"); + }); +}); + +describe("formatWithPattern — formats it deliberately does not render", () => { + it("declines a pattern with no digit placeholder", () => { + assert.equal(formatWithPattern(1234.5, "MM/DD/YYYY"), null); + }); + + it("declines an empty pattern", () => { + assert.equal(formatWithPattern(1234.5, ""), null); + }); + + it("declines scientific notation", () => { + assert.equal(formatWithPattern(1234.5, "0.00E+00"), null); + }); + + it("declines a multi-section pattern, whose negative/zero sections it cannot honour", () => { + assert.equal(formatWithPattern(-1, "0.00;(0.00)"), null); + }); + + it("declines a quoted literal", () => { + assert.equal(formatWithPattern(1, '0" units"'), null); + }); + + it("declines a fraction pattern", () => { + assert.equal(formatWithPattern(1.5, "# ?/?"), null); + }); + + // A comma AFTER the last placeholder means "scale by a thousand" in Excel; + // rendering it as ordinary grouping would be off by 1000. + it("declines the thousands-scaling trailing comma", () => { + assert.equal(formatWithPattern(1234567, "#,##0,"), null); + }); + + it("declines a non-finite value", () => { + assert.equal(formatWithPattern(NaN, "0.00"), null); + assert.equal(formatWithPattern(Infinity, "0.00"), null); + }); +}); + +describe("parseNumberPattern", () => { + it("splits a currency pattern into literal, grouping and decimals", () => { + assert.deepEqual(parseNumberPattern("$#,##0.00"), { + prefix: "$", + suffix: "", + useGrouping: true, + integerMinDigits: 1, + minDecimals: 2, + maxDecimals: 2, + isPercent: false, + }); + }); + + it("marks a trailing percent sign and counts optional decimals separately", () => { + assert.deepEqual(parseNumberPattern("0.0#%"), { + prefix: "", + suffix: "%", + useGrouping: false, + integerMinDigits: 1, + minDecimals: 1, + maxDecimals: 2, + isPercent: true, + }); + }); + + it("returns null for a code it cannot render", () => { + assert.equal(parseNumberPattern("MMM D, YYYY"), null); + }); +}); + +describe("groupThousands", () => { + it("returns short runs unchanged", () => { + assert.equal(groupThousands(""), ""); + assert.equal(groupThousands("7"), "7"); + assert.equal(groupThousands("999"), "999"); + }); + + it("separates every third digit from the right", () => { + assert.equal(groupThousands("1000"), "1,000"); + assert.equal(groupThousands("1234567"), "1,234,567"); + assert.equal(groupThousands("100000000"), "100,000,000"); + }); +}); + +describe("TEXT — end to end through SpreadsheetEngine", () => { + it("groups digits for a currency pattern", () => { + assert.equal(evalFormula('TEXT(1234.5,"$#,##0.00")'), "$1,234.50"); + }); + + it("renders a bare percent pattern without inventing decimals", () => { + assert.equal(evalFormula('TEXT(0.5,"0%")'), "50%"); + }); + + it("renders a bare currency pattern without inventing decimals", () => { + assert.equal(evalFormula('TEXT(5,"$0")'), "$5"); + }); + + it("groups a large plain number", () => { + assert.equal(evalFormula('TEXT(1234567,"#,##0")'), "1,234,567"); + }); + + it("keeps the sign of a negative value", () => { + assert.equal(evalFormula('TEXT(-1234.5,"#,##0.00")'), "-1,234.50"); + }); + + it("formats a cell reference the same way", () => { + const sheet = engine.createSheet("S", [[{ v: 1234.5 }, { v: '=TEXT(A1,"$#,##0.00")' }]]); + assert.equal(cellAt(engine.calculate(sheet).data, 0, 1), "$1,234.50"); + }); + + it("returns the value's own text for a format code it does not render", () => { + assert.equal(evalFormula('TEXT(1234.5,"MM/DD/YYYY")'), "1234.5"); + }); + + it("passes non-numeric input through unchanged", () => { + assert.equal(evalFormula('TEXT("abc","0.00")'), "abc"); + }); +}); diff --git a/tests/engine/test_textFunctions.ts b/tests/engine/test_textFunctions.ts new file mode 100644 index 0000000..599abc5 --- /dev/null +++ b/tests/engine/test_textFunctions.ts @@ -0,0 +1,190 @@ +// Edge-case behaviour of the SUBSTITUTE / RIGHT / LEFT / PROPER text functions. +// +// Each fix targets a case that fails silently: SUBSTITUTE with an empty +// old_text inserted the replacement between every character, a non-positive +// instance was ignored instead of erroring, RIGHT/LEFT turned a negative count +// into an empty string, and PROPER only broke words on spaces — all returning a +// plausible-looking string rather than the Excel result or a #VALUE! error. +// +// The pure helpers are tested directly; the end-to-end path through +// SpreadsheetEngine confirms the handlers wire them up. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { substituteText, takeLeft, takeRight, toProperCase } from "../../src/engine/functions/text.ts"; +import { SpreadsheetEngine } from "../../src/engine/index.ts"; +import { VALUE_ERROR } from "../../src/engine/spreadsheet-errors.ts"; +import { cellAt } from "./cellAccess.ts"; + +// The engine's display pass renders an error VALUE back to its code, so the +// end-to-end assertions compare against the string a cell shows. +const VALUE_ERROR_TEXT = VALUE_ERROR.code; + +const engine = new SpreadsheetEngine(); +const evalFormula = (formula: string): unknown => cellAt(engine.calculate(engine.createSheet("S", [[`=${formula}`]])).data, 0, 0); + +describe("substituteText — empty old_text", () => { + it("returns the text unchanged instead of inserting between characters", () => { + assert.equal(substituteText("abc", "", "-"), "abc"); + }); + + it("stays unchanged even when an instance is supplied", () => { + assert.equal(substituteText("abc", "", "-", 1), "abc"); + }); + + it("stays unchanged for an empty text as well", () => { + assert.equal(substituteText("", "", "-"), ""); + }); +}); + +describe("substituteText — instance validation", () => { + it("errors when the instance is zero", () => { + assert.equal(substituteText("aa", "a", "b", 0), VALUE_ERROR); + }); + + it("errors when the instance is negative", () => { + assert.equal(substituteText("aa", "a", "b", -1), VALUE_ERROR); + }); + + it("errors when the instance is not a finite number", () => { + assert.equal(substituteText("aa", "a", "b", NaN), VALUE_ERROR); + }); + + it("truncates a fractional instance toward zero, so <1 errors", () => { + assert.equal(substituteText("aa", "a", "b", 0.9), VALUE_ERROR); + assert.equal(substituteText("aaa", "a", "b", 1.9), "baa", "1.9 truncates to the 1st occurrence"); + }); +}); + +describe("substituteText — replacing occurrences", () => { + it("replaces every occurrence when no instance is given", () => { + assert.equal(substituteText("Hello World", "World", "Earth"), "Hello Earth"); + assert.equal(substituteText("a-b-c", "-", "+"), "a+b+c"); + }); + + it("replaces only the requested 1-based occurrence", () => { + assert.equal(substituteText("a-b-c", "-", "+", 1), "a+b-c"); + assert.equal(substituteText("a-b-c", "-", "+", 2), "a-b+c"); + }); + + it("returns the text unchanged when the instance exceeds the occurrence count", () => { + assert.equal(substituteText("a-b-c", "-", "+", 3), "a-b-c"); + }); + + it("keeps matches non-overlapping like the replace-all path", () => { + // "aa" occurs once in "aaa" when scanned non-overlapping, so a 2nd instance + // does not exist and the text is returned unchanged. + assert.equal(substituteText("aaa", "aa", "b", 2), "aaa"); + assert.equal(substituteText("aaa", "aa", "b", 1), "ba"); + }); +}); + +describe("takeRight — negative count is an error", () => { + it("errors on a negative count instead of returning an empty string", () => { + assert.equal(takeRight("Hello", -1), VALUE_ERROR); + assert.equal(takeRight("Hello", -0.5), VALUE_ERROR); + }); + + it("returns an empty string for a zero count", () => { + assert.equal(takeRight("Hello", 0), ""); + }); + + it("returns the whole string when the count exceeds its length", () => { + assert.equal(takeRight("Hello", 10), "Hello"); + }); + + it("returns the rightmost characters for a normal count", () => { + assert.equal(takeRight("Hello", 2), "lo"); + }); + + it("truncates a fractional count toward zero", () => { + assert.equal(takeRight("Hello", 2.5), "lo"); + }); + + it("errors on a non-finite count", () => { + assert.equal(takeRight("Hello", NaN), VALUE_ERROR); + assert.equal(takeRight("Hello", Infinity), VALUE_ERROR); + }); + + it("returns an empty string from an empty text", () => { + assert.equal(takeRight("", 3), ""); + }); +}); + +describe("takeLeft — negative count is an error", () => { + it("errors on a negative count instead of returning an empty string", () => { + assert.equal(takeLeft("Hello", -1), VALUE_ERROR); + }); + + it("returns an empty string for a zero count", () => { + assert.equal(takeLeft("Hello", 0), ""); + }); + + it("returns the whole string when the count exceeds its length", () => { + assert.equal(takeLeft("Hello", 10), "Hello"); + }); + + it("returns the leftmost characters for a normal count", () => { + assert.equal(takeLeft("Hello", 2), "He"); + }); + + it("truncates a fractional count and errors on a non-finite one", () => { + assert.equal(takeLeft("Hello", 2.9), "He"); + assert.equal(takeLeft("Hello", NaN), VALUE_ERROR); + }); +}); + +describe("toProperCase — word boundaries include punctuation", () => { + it("capitalises after an apostrophe and a hyphen, not only spaces", () => { + assert.equal(toProperCase("o'neil-jr"), "O'Neil-Jr"); + }); + + it("capitalises the first letter of each space-separated word", () => { + assert.equal(toProperCase("hello world"), "Hello World"); + }); + + it("lowercases the remaining letters of an all-caps word", () => { + assert.equal(toProperCase("HELLO"), "Hello"); + }); + + it("treats a digit as a non-letter boundary", () => { + assert.equal(toProperCase("abc2def"), "Abc2Def"); + }); + + it("returns an empty string for empty input", () => { + assert.equal(toProperCase(""), ""); + }); + + // A decomposed accented letter is a base letter + a combining mark; the mark + // must not read as a word boundary and capitalize the next letter (#2388). + it("keeps a decomposed accented letter as one word", () => { + assert.equal(toProperCase("e\u0301clair"), "E\u0301clair", "NFD: e + combining acute"); + assert.equal(toProperCase("\u00e9clair"), "\u00c9clair", "NFC composed form"); + }); + + it("leaves leading punctuation in place and capitalises the first letter", () => { + assert.equal(toProperCase("'hello"), "'Hello"); + }); +}); + +describe("SpreadsheetEngine — text edge cases end to end", () => { + it("evaluates SUBSTITUTE with empty old_text and bad instance", () => { + assert.equal(evalFormula('SUBSTITUTE("abc","","-")'), "abc"); + assert.equal(evalFormula('SUBSTITUTE("aa","a","b",0)'), VALUE_ERROR_TEXT); + }); + + it("evaluates RIGHT / LEFT with a negative count as an error", () => { + assert.equal(evalFormula('RIGHT("Hello",-1)'), VALUE_ERROR_TEXT); + assert.equal(evalFormula('LEFT("Hello",-1)'), VALUE_ERROR_TEXT); + }); + + it("errors on a non-numeric count and truncates a fractional one", () => { + assert.equal(evalFormula('LEFT("Hello","x")'), VALUE_ERROR_TEXT); + assert.equal(evalFormula('RIGHT("Hello","x")'), VALUE_ERROR_TEXT); + assert.equal(evalFormula('RIGHT("Hello",2.5)'), "lo"); + }); + + it("evaluates PROPER across punctuation boundaries", () => { + assert.equal(evalFormula('PROPER("o\'neil-jr")'), "O'Neil-Jr"); + }); +}); diff --git a/tests/engine/test_translateFormula.ts b/tests/engine/test_translateFormula.ts new file mode 100644 index 0000000..531a794 --- /dev/null +++ b/tests/engine/test_translateFormula.ts @@ -0,0 +1,200 @@ +// Excel-formula → JS-expression translation. These are the pure string +// transforms the evaluator runs on an already-substituted expression before +// handing it to `new Function`. A mistranslated operator can still produce a +// plausible wrong NUMBER (`=2^3^2` = 512, not 64 — the associativity gap tracked +// by the sibling issues). What #2359 fixed: a formula that reaches `new Function` +// as invalid JS (`=5<>6`) no longer comes back as its raw text — it surfaces as +// an #ERROR!. This suite pins both the remaining known-wrong number cases and the +// post-#2359 error surfacing. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { + caretToPow, + replaceConcatOperator, + rewriteComparisonEq, + isSafeArithmetic, + isSafeComparison, + SpreadsheetEngine, + type SheetData, +} from "../../src/engine/index.ts"; +import { cellAt } from "./cellAccess.ts"; + +describe("caretToPow", () => { + it("rewrites a single caret to the JS exponentiation operator", () => { + assert.equal(caretToPow("2^3"), "2**3"); + }); + + it("rewrites every caret in the expression", () => { + assert.equal(caretToPow("2^3+4^5"), "2**3+4**5"); + assert.equal(caretToPow("A^B^C"), "A**B**C"); + }); + + // Excel's `^` is LEFT-associative (`2^3^2` = `(2^3)^2` = 64); JS `**` is + // RIGHT-associative (`2**3**2` = `2**(3**2)` = 512). This transform is a plain + // substitution and does NOT bridge that difference — a chained caret keeps + // producing the JS answer. Pinned as a known limitation (#2359). + it("does not correct the left-vs-right associativity of chained carets", () => { + assert.equal(caretToPow("2^3^2"), "2**3**2"); + assert.equal(new Function("return (2**3**2)")(), 512, "JS is right-associative, so 512 not 64"); + }); + + it("leaves an expression with no caret unchanged", () => { + assert.equal(caretToPow("2+3*4"), "2+3*4"); + assert.equal(caretToPow(""), ""); + }); +}); + +describe("replaceConcatOperator", () => { + it("rewrites a concatenation ampersand to +", () => { + assert.equal(replaceConcatOperator('"a"&"b"'), '"a"+"b"'); + assert.equal(replaceConcatOperator("5&6"), "5+6"); + }); + + it("rewrites every ampersand outside string literals", () => { + assert.equal(replaceConcatOperator('"a"&"b"&"c"'), '"a"+"b"+"c"'); + }); + + // An ampersand INSIDE a literal is text, not an operator — flipping it would + // corrupt the string content. + it("preserves an ampersand inside a double-quoted literal", () => { + assert.equal(replaceConcatOperator('"a&b"&"c"'), '"a&b"+"c"'); + }); + + it("preserves an ampersand inside a single-quoted literal", () => { + assert.equal(replaceConcatOperator("'a&b'&'c'"), "'a&b'+'c'"); + }); + + // A backslash-escaped quote does not end the literal, so an ampersand after it + // is still inside the string and must be preserved. + it("honours a backslash-escaped quote when tracking literal boundaries", () => { + assert.equal(replaceConcatOperator('"a\\"&b"&"c"'), '"a\\"&b"+"c"'); + }); + + it("leaves an expression with no ampersand unchanged", () => { + assert.equal(replaceConcatOperator("2+3"), "2+3"); + assert.equal(replaceConcatOperator('"plain"'), '"plain"'); + assert.equal(replaceConcatOperator(""), ""); + }); + + it("rewrites a leading or trailing ampersand outside a literal", () => { + assert.equal(replaceConcatOperator('&"x"'), '+"x"'); + assert.equal(replaceConcatOperator('"x"&'), '"x"+'); + }); +}); + +describe("rewriteComparisonEq", () => { + it("rewrites a single equality to the JS == operator", () => { + assert.equal(rewriteComparisonEq("5=5"), "5==5"); + assert.equal(rewriteComparisonEq("5=6"), "5==6"); + }); + + it("does not disturb <=, >= or != which are already valid JS", () => { + assert.equal(rewriteComparisonEq("5<=6"), "5<=6"); + assert.equal(rewriteComparisonEq("5>=6"), "5>=6"); + assert.equal(rewriteComparisonEq("5!=6"), "5!=6"); + }); + + // The regex matches the SECOND `=` of a `==` (its left neighbour, the first + // `=`, is not one of `<>!`), so an already-doubled `==` becomes `===`. Excel + // never emits `==`, so this only bites a hand-typed oddity — pinned as a known + // quirk (#2359) rather than relied upon. + it("turns an already-doubled == into === (known quirk)", () => { + assert.equal(rewriteComparisonEq("5==6"), "5===6"); + }); + + // The match consumes both flanking characters, so replacements cannot overlap: + // in `5=6=7` only the first `=` is rewritten. Pinned as a known limitation + // (#2359) — the non-overlapping replacement is a real bug to be fixed later. + it("rewrites only the first of two adjacent equalities (non-overlapping)", () => { + assert.equal(rewriteComparisonEq("5=6=7"), "5==6=7"); + }); + + // A `=` with nothing on one side has no flanking character to match, so it is + // left as a lone `=`. Pinned known limitation (#2359). + it("does not rewrite an = at the start or end of the expression", () => { + assert.equal(rewriteComparisonEq("=5"), "=5"); + assert.equal(rewriteComparisonEq("5="), "5="); + }); + + it("leaves an expression with no equals unchanged", () => { + assert.equal(rewriteComparisonEq("5<6"), "5<6"); + assert.equal(rewriteComparisonEq(""), ""); + }); +}); + +describe("isSafeArithmetic", () => { + it("accepts digits, arithmetic operators, parentheses, dot and space", () => { + assert.equal(isSafeArithmetic("2+3*4"), true); + assert.equal(isSafeArithmetic("(2 + 3) / 4.5"), true); + assert.equal(isSafeArithmetic("2**3"), true, "** survives the caret rewrite and must pass"); + }); + + it("rejects letters, quotes, ampersands and comparison characters", () => { + assert.equal(isSafeArithmetic("A1+2"), false); + assert.equal(isSafeArithmetic('"a"+"b"'), false); + assert.equal(isSafeArithmetic("5&6"), false); + assert.equal(isSafeArithmetic("5<6"), false); + assert.equal(isSafeArithmetic("5=6"), false); + }); + + // The allowlist requires at least one character, so the empty string is not + // "safe" — there is nothing to evaluate. + it("rejects the empty string", () => { + assert.equal(isSafeArithmetic(""), false); + }); +}); + +describe("isSafeComparison", () => { + it("accepts the arithmetic set plus < > ! =", () => { + assert.equal(isSafeComparison("5<=6"), true); + assert.equal(isSafeComparison("5<>6"), true, "the raw Excel <> passes the gate even though JS rejects it"); + assert.equal(isSafeComparison("(2**3) >= 6"), true); + }); + + it("rejects letters, quotes and ampersands", () => { + assert.equal(isSafeComparison("A1<6"), false); + assert.equal(isSafeComparison('"a"="b"'), false); + assert.equal(isSafeComparison("5&6"), false); + }); + + it("rejects the empty string", () => { + assert.equal(isSafeComparison(""), false); + }); +}); + +// End-to-end characterization: the pure functions compose inside the evaluator +// to the exact behaviour observed before the extraction. Includes the +// intentionally-wrong cases so the eventual #2359 fix shows up as a diff here. +describe("translation through the engine (characterization)", () => { + const calc = (formula: string): unknown => cellAt(new SpreadsheetEngine().calculate({ name: "S", data: [[{ v: formula }]] } satisfies SheetData).data, 0, 0); + + it("exponentiates (and keeps the JS-associativity quirk)", () => { + assert.equal(calc("=2^3"), 8); + assert.equal(calc("=2^3^2"), 512); + }); + + it("evaluates equality and the ordering comparisons", () => { + assert.equal(calc("=5=5"), true); + assert.equal(calc("=5=6"), false); + assert.equal(calc("=5<=6"), true); + assert.equal(calc("=5>=6"), false); + }); + + // `<>` is Excel's not-equal; it is not translated, so it reaches `new Function` + // as invalid JS and throws. Post-#2359 that throw surfaces as #ERROR! rather + // than the raw formula text (translating `<>` itself is a sibling issue). + it("surfaces the untranslated <> operator as #ERROR!, not raw text", () => { + assert.equal(calc("=5<>6"), "#ERROR!"); + }); + + it("concatenates strings and mixed operands", () => { + assert.equal(calc('="a"&"b"'), "ab"); + assert.equal(calc('="x"&5'), "x5"); + }); + + it("evaluates plain arithmetic", () => { + assert.equal(calc("=2*3+1"), 7); + assert.equal(calc("=(2+3)*4"), 20); + }); +});