From 52a32fbb292c8dcc13f6995ab2e5ebe98b0546de Mon Sep 17 00:00:00 2001 From: isamu Date: Thu, 24 Sep 2026 17:43:07 +0900 Subject: [PATCH] feat(engine): adopt the spreadsheet engine developed in MulmoClaude The engine here and the one in MulmoClaude's src/plugins/spreadsheet started as one copy. The copy kept going: its formula handling was split into small units with tests of their own, and several Excel behaviours were corrected along the way. This brings that engine back, with its tests. What arrives with it: - src/engine is replaced wholesale. The reference handling moved from parser.ts into formulaRefs.ts, the aggregates read through cellBuilder / condition / numericCoercion, and the financial, statistical, lookup and math helpers are separate units. - src/engine/guards.ts holds the three runtime helpers the engine needs (isRecord, isObj, errorMessage). They come from MulmoClaude's leaf package; copied rather than depended on, because a gui-chat plugin must not take a dependency on a host's packages. - tests/engine gains the engine's own node:test suite, run by a new `test:engine` script and folded into `test:all`. Two expected values in the financial fixture were wrong and now match Excel: the first interest payment had the sign of money coming in rather than going out, and the first principal payment was not PMT - IPMT. The reason is recorded in tests/engine/fixtures/README.md so a future change cannot quietly flip them back. Two test-side repairs, neither of which changes the engine: - parser.test.ts kept exercising parseCellRef / parseRangeRef / cellRefToA1, which this engine does not have; expandRangeOrCell covers that ground and has its own tests. Column conversion stays, since the Vue view still uses it. - run-evaluator-tests.ts built a context whose range readers answered nothing for a single cell, so MAX(B1, A1:A3, 5) silently dropped B1. The real context resolves every reference through expandRangeOrCell; the fake now does too. typecheck, lint, build, and every test script pass. Co-Authored-By: Claude Opus 5 (1M context) --- package.json | 3 +- src/engine/calculator.ts | 520 ++++++-------- src/engine/cellBuilder.ts | 81 +++ src/engine/cellEmpty.ts | 26 + src/engine/cellFormatting.ts | 59 ++ src/engine/coerce-boolean.ts | 26 + src/engine/condition.ts | 182 +++++ src/engine/date-locale.ts | 47 ++ src/engine/date-parser.ts | 196 +++-- src/engine/date-utils.ts | 41 +- src/engine/datedif.ts | 93 +++ src/engine/engine.ts | 20 +- src/engine/evaluator.ts | 673 ++++++++++-------- src/engine/financial-math.ts | 114 +++ src/engine/formatter.ts | 154 ++-- src/engine/formulaError.ts | 67 ++ src/engine/formulaRefs.ts | 202 ++++++ src/engine/functions/date.ts | 152 +--- src/engine/functions/financial.ts | 341 ++------- src/engine/functions/logical.ts | 196 ++--- src/engine/functions/lookup-math.ts | 28 + src/engine/functions/lookup.ts | 386 +++------- src/engine/functions/mathematical.ts | 126 ++-- src/engine/functions/statistical-math.ts | 91 +++ src/engine/functions/statistical.ts | 273 +++---- src/engine/functions/text.ts | 325 ++++----- src/engine/guards.ts | 32 + src/engine/index.ts | 11 + src/engine/jsonCellLocator.ts | 111 +++ src/engine/math-ops.ts | 98 +++ src/engine/numericCoercion.ts | 61 ++ src/engine/parser.ts | 104 --- src/engine/registry.ts | 129 ++-- src/engine/responseDecoder.ts | 68 ++ src/engine/spreadsheet-errors.ts | 82 +++ src/engine/textFormat.ts | 108 +++ src/engine/translateFormula.ts | 69 ++ src/engine/types.ts | 23 +- tests/engine/cellAccess.ts | 18 + tests/engine/fixtures/README.md | 9 + .../expected/financial-functions.json | 4 +- tests/engine/parser.test.ts | 169 +---- tests/engine/run-evaluator-tests.ts | 9 +- tests/engine/test_argCountValidation.ts | 128 ++++ tests/engine/test_calculateDateLocale.ts | 98 +++ tests/engine/test_cellBuilder.ts | 130 ++++ tests/engine/test_cellEmpty.ts | 123 ++++ tests/engine/test_cellFormatting.ts | 73 ++ tests/engine/test_cellRefSubstitution.ts | 179 +++++ tests/engine/test_concatSafety.ts | 109 +++ tests/engine/test_condition.ts | 292 ++++++++ tests/engine/test_conditionalAggregates.ts | 47 ++ tests/engine/test_criteria.ts | 90 +++ tests/engine/test_crossSheetReference.ts | 105 +++ tests/engine/test_dateLocale.ts | 95 +++ tests/engine/test_dateParser.ts | 211 ++++++ tests/engine/test_dateUtils.ts | 123 ++++ tests/engine/test_datedif.ts | 133 ++++ tests/engine/test_errorReporting.ts | 101 +++ tests/engine/test_errorValue.ts | 180 +++++ tests/engine/test_expandRangeOrCell.ts | 83 +++ tests/engine/test_financialMath.ts | 188 +++++ .../engine/test_financialPeriodicHandlers.ts | 34 + tests/engine/test_formatter.ts | 157 ++++ tests/engine/test_formulaError.ts | 94 +++ tests/engine/test_formulaRefs.ts | 339 +++++++++ tests/engine/test_ifBranchEvaluation.ts | 61 ++ tests/engine/test_ifsInjection.ts | 221 ++++++ tests/engine/test_jsonCellLocator.ts | 118 +++ tests/engine/test_locateSubstring.ts | 52 ++ tests/engine/test_logicalFunctions.ts | 76 ++ tests/engine/test_lookupBounds.ts | 135 ++++ tests/engine/test_lookupFunctions.ts | 181 +++++ tests/engine/test_lookupMath.ts | 43 ++ tests/engine/test_mathematicalFunctions.ts | 169 +++++ tests/engine/test_midValue.ts | 115 +++ tests/engine/test_multiRangeAggregates.ts | 134 ++++ tests/engine/test_normalizeData.ts | 59 ++ tests/engine/test_npvArgs.ts | 38 + tests/engine/test_npvFunction.ts | 32 + tests/engine/test_numericCoercion.ts | 152 ++++ tests/engine/test_numericFunctions2391.ts | 72 ++ tests/engine/test_parseFunctionArgs.ts | 79 ++ tests/engine/test_parseRangeBounds.ts | 94 +++ tests/engine/test_parser.ts | 93 +++ tests/engine/test_rangeFunctions.ts | 45 ++ tests/engine/test_registry.ts | 220 ++++++ tests/engine/test_requiredArg.ts | 102 +++ tests/engine/test_responseDecoder.ts | 87 +++ tests/engine/test_serialFromParts.ts | 48 ++ tests/engine/test_spreadsheetErrors.ts | 71 ++ tests/engine/test_statisticalFunctions.ts | 101 +++ tests/engine/test_statisticalMath.ts | 130 ++++ tests/engine/test_textFormat.ts | 254 +++++++ tests/engine/test_textFunctions.ts | 190 +++++ tests/engine/test_translateFormula.ts | 200 ++++++ 96 files changed, 9642 insertions(+), 2369 deletions(-) create mode 100644 src/engine/cellBuilder.ts create mode 100644 src/engine/cellEmpty.ts create mode 100644 src/engine/cellFormatting.ts create mode 100644 src/engine/coerce-boolean.ts create mode 100644 src/engine/condition.ts create mode 100644 src/engine/date-locale.ts create mode 100644 src/engine/datedif.ts create mode 100644 src/engine/financial-math.ts create mode 100644 src/engine/formulaError.ts create mode 100644 src/engine/formulaRefs.ts create mode 100644 src/engine/functions/lookup-math.ts create mode 100644 src/engine/functions/statistical-math.ts create mode 100644 src/engine/guards.ts create mode 100644 src/engine/jsonCellLocator.ts create mode 100644 src/engine/math-ops.ts create mode 100644 src/engine/numericCoercion.ts create mode 100644 src/engine/responseDecoder.ts create mode 100644 src/engine/spreadsheet-errors.ts create mode 100644 src/engine/textFormat.ts create mode 100644 src/engine/translateFormula.ts create mode 100644 tests/engine/cellAccess.ts create mode 100644 tests/engine/test_argCountValidation.ts create mode 100644 tests/engine/test_calculateDateLocale.ts create mode 100644 tests/engine/test_cellBuilder.ts create mode 100644 tests/engine/test_cellEmpty.ts create mode 100644 tests/engine/test_cellFormatting.ts create mode 100644 tests/engine/test_cellRefSubstitution.ts create mode 100644 tests/engine/test_concatSafety.ts create mode 100644 tests/engine/test_condition.ts create mode 100644 tests/engine/test_conditionalAggregates.ts create mode 100644 tests/engine/test_criteria.ts create mode 100644 tests/engine/test_crossSheetReference.ts create mode 100644 tests/engine/test_dateLocale.ts create mode 100644 tests/engine/test_dateParser.ts create mode 100644 tests/engine/test_dateUtils.ts create mode 100644 tests/engine/test_datedif.ts create mode 100644 tests/engine/test_errorReporting.ts create mode 100644 tests/engine/test_errorValue.ts create mode 100644 tests/engine/test_expandRangeOrCell.ts create mode 100644 tests/engine/test_financialMath.ts create mode 100644 tests/engine/test_financialPeriodicHandlers.ts create mode 100644 tests/engine/test_formatter.ts create mode 100644 tests/engine/test_formulaError.ts create mode 100644 tests/engine/test_formulaRefs.ts create mode 100644 tests/engine/test_ifBranchEvaluation.ts create mode 100644 tests/engine/test_ifsInjection.ts create mode 100644 tests/engine/test_jsonCellLocator.ts create mode 100644 tests/engine/test_locateSubstring.ts create mode 100644 tests/engine/test_logicalFunctions.ts create mode 100644 tests/engine/test_lookupBounds.ts create mode 100644 tests/engine/test_lookupFunctions.ts create mode 100644 tests/engine/test_lookupMath.ts create mode 100644 tests/engine/test_mathematicalFunctions.ts create mode 100644 tests/engine/test_midValue.ts create mode 100644 tests/engine/test_multiRangeAggregates.ts create mode 100644 tests/engine/test_normalizeData.ts create mode 100644 tests/engine/test_npvArgs.ts create mode 100644 tests/engine/test_npvFunction.ts create mode 100644 tests/engine/test_numericCoercion.ts create mode 100644 tests/engine/test_numericFunctions2391.ts create mode 100644 tests/engine/test_parseFunctionArgs.ts create mode 100644 tests/engine/test_parseRangeBounds.ts create mode 100644 tests/engine/test_parser.ts create mode 100644 tests/engine/test_rangeFunctions.ts create mode 100644 tests/engine/test_registry.ts create mode 100644 tests/engine/test_requiredArg.ts create mode 100644 tests/engine/test_responseDecoder.ts create mode 100644 tests/engine/test_serialFromParts.ts create mode 100644 tests/engine/test_spreadsheetErrors.ts create mode 100644 tests/engine/test_statisticalFunctions.ts create mode 100644 tests/engine/test_statisticalMath.ts create mode 100644 tests/engine/test_textFormat.ts create mode 100644 tests/engine/test_textFunctions.ts create mode 100644 tests/engine/test_translateFormula.ts diff --git a/package.json b/package.json index fb0fdb7..9094c84 100644 --- a/package.json +++ b/package.json @@ -33,12 +33,13 @@ "typecheck": "vue-tsc --noEmit", "lint": "eslint src demo", "test": "vitest run", + "test:engine": "tsx --test \"tests/engine/test_*.ts\"", "test:watch": "vitest", "test:fixtures": "tsx tests/engine/run-all-fixtures.ts", "test:calculator": "tsx tests/engine/run-calculator-tests.ts", "test:evaluator": "tsx tests/engine/run-evaluator-tests.ts", "test:functions": "tsx tests/engine/test-functions.ts", - "test:all": "vitest run && tsx tests/engine/run-all-fixtures.ts" + "test:all": "vitest run && yarn run test:engine && tsx tests/engine/run-all-fixtures.ts" }, "peerDependencies": { "gui-chat-protocol": "^2.0.0", diff --git a/src/engine/calculator.ts b/src/engine/calculator.ts index 1721bb3..0464594 100644 --- a/src/engine/calculator.ts +++ b/src/engine/calculator.ts @@ -4,18 +4,35 @@ * Core calculation engine with circular reference detection and cross-sheet support */ -import { formatNumber } from "./formatter"; -import { columnToIndex } from "./parser"; +import { formatCellForDisplay } from "./cellFormatting"; import { evaluateFormula as evaluateFormulaFn } from "./evaluator"; +import { expandRangeOrCell, parseSingleCellRef } from "./formulaRefs"; import { parseDate, getDefaultDateFormat } from "./date-parser"; -import type { - SheetData, - CellValue, - CalculatedSheet, - CalculationError, - FormulaInfo, - SpreadsheetCell, -} from "./types"; +import type { SheetData, CellValue, CalculatedSheet, CalculationError, FormulaInfo, SpreadsheetCell, CalculateOptions } from "./types"; +import { isObj } from "./guards"; +import { isEmptyCell } from "./cellEmpty"; +import { errorMessage } from "./guards"; +import { classifyThrownError, invalidRefError } from "./formulaError"; +import { isSpreadsheetErrorValue, spreadsheetError } from "./spreadsheet-errors"; + +// The grid a reference should read from, plus the sheet-name-stripped ref and +// whether it points at the sheet currently being calculated (which decides +// whether recursive formula evaluation is allowed for the cell it lands on). +interface ResolvedSheetRef { + sheetData: (SpreadsheetCell | CellValue)[][]; + ref: string; + isCurrentSheet: boolean; +} + +// Where a cell sits in the grid CURRENTLY being calculated, carrying the row it +// belongs to so a formula's result is written back without re-indexing. Only a +// same-sheet cell has one: a cross-sheet read is resolved by that sheet's own +// calculateSheet pass, so it has no position here to recurse from. +interface CellPosition { + cells: any[]; + row: number; + col: number; +} /** * Normalize malformed data structures @@ -24,7 +41,7 @@ import type { * @param data - Potentially malformed sheet data * @returns Normalized 2D array */ -function normalizeData(data: any): SpreadsheetCell[][] { +export function normalizeData(data: any): SpreadsheetCell[][] { // Handle null/undefined if (!data) { return []; @@ -48,7 +65,7 @@ function normalizeData(data: any): SpreadsheetCell[][] { // If data is a flat array of cell objects, convert to 2D by pairing cells // Pattern: [cell1, cell2, cell3, cell4] -> [[cell1, cell2], [cell3, cell4]] // This handles the case where models output flat arrays instead of rows - if (typeof data[0] === "object" && data[0] !== null) { + if (isObj(data[0])) { const rows: SpreadsheetCell[][] = []; for (let i = 0; i < data.length; i += 2) { const row = [data[i]]; @@ -71,11 +88,11 @@ function normalizeData(data: any): SpreadsheetCell[][] { * @param data - Raw sheet data * @returns Processed data with dates converted to serial numbers */ -function preprocessDates(data: SpreadsheetCell[][]): SpreadsheetCell[][] { +function preprocessDates(data: SpreadsheetCell[][], preferDDMMYYYY: boolean): SpreadsheetCell[][] { return data.map((row) => row.map((cell) => { // Skip if not a cell object or if it has a formula - if (!cell || typeof cell !== "object" || !("v" in cell)) { + if (!isObj(cell) || !("v" in cell)) { return cell; } @@ -83,13 +100,13 @@ function preprocessDates(data: SpreadsheetCell[][]): SpreadsheetCell[][] { // Only parse strings that aren't formulas if (typeof value === "string" && !value.startsWith("=")) { - const dateSerial = parseDate(value); + const dateSerial = parseDate(value, preferDDMMYYYY); if (dateSerial !== null) { // It's a date! Convert to serial number return { v: dateSerial, - f: cell.f || getDefaultDateFormat(value), // Use existing format or detect from input + f: cell.f || getDefaultDateFormat(value, preferDDMMYYYY), // Use existing format or detect from input }; } } @@ -100,6 +117,12 @@ function preprocessDates(data: SpreadsheetCell[][]): SpreadsheetCell[][] { ); } +// `skipFormatting` is internal, not a public knob: a cross-sheet reference +// computes its target sheet only to READ values, so the display-formatting pass +// is skipped there — a date must stay a serial, not become "03/04/2025" that a +// downstream parseFloat reads as 3 (issue #2332). +type SheetCalculateOptions = CalculateOptions & { skipFormatting?: boolean }; + /** * Calculate formulas in a single sheet * @@ -107,20 +130,19 @@ function preprocessDates(data: SpreadsheetCell[][]): SpreadsheetCell[][] { * @param allSheets - All sheets for cross-sheet references * @returns Calculated sheet with formulas evaluated */ -export function calculateSheet( - sheet: SheetData, - allSheets?: SheetData[], -): CalculatedSheet { +export function calculateSheet(sheet: SheetData, allSheets?: SheetData[], options: SheetCalculateOptions = {}): CalculatedSheet { + const preferDDMMYYYY = options.preferDDMMYYYY ?? false; + const skipFormatting = options.skipFormatting ?? false; // Normalize malformed data structures first const normalizedData = normalizeData(sheet.data); // Pre-process dates before calculation - const processedData = preprocessDates(normalizedData); + const processedData = preprocessDates(normalizedData, preferDDMMYYYY); // Also preprocess all sheets if provided const processedAllSheets = allSheets?.map((s) => ({ ...s, - data: preprocessDates(normalizeData(s.data)), + data: preprocessDates(normalizeData(s.data), preferDDMMYYYY), })); const data = processedData; @@ -138,12 +160,50 @@ export function calculateSheet( // Track cells being calculated to detect circular references const calculating = new Set(); + // Cells whose result is already stored in `calculated`, of ANY type. A number + // check alone missed string/error results, so a cell referenced before the + // top loop reached it was re-evaluated (and, once formulas can throw, would + // re-emit its error). Membership here means "read the cached value, do not + // re-run". + const evaluated = new Set(); + + // Evaluate one formula cell, guarding circular references, caching the result, + // and turning a thrown failure into a typed errors[] entry plus the Excel + // error value in the cell — never a swallowed bare string/number (#2359). + const resolveFormulaCell = (formulaText: string, { cells, row, col }: CellPosition): CellValue => { + const cellKey = `${row},${col}`; + if (calculating.has(cellKey)) { + errors.push({ cell: { row, col }, formula: formulaText, error: "Circular reference detected", type: "circular" }); + return 0; + } + if (evaluated.has(cellKey)) return cells[col]; + calculating.add(cellKey); + try { + const result = evaluateFormula(formulaText.substring(1)); // drop leading "=" + cells[col] = result; + return result; + } catch (error) { + const { type, display } = classifyThrownError(error); + errors.push({ cell: { row, col }, formula: formulaText, error: errorMessage(error), type }); + // The error VALUE, not its text: a cell that reads this one must see a + // real error, and the display pass renders it back to `#DIV/0!`. + const errorValue = spreadsheetError(display); + cells[col] = errorValue; + return errorValue; + } finally { + calculating.delete(cellKey); + evaluated.add(cellKey); + } + }; // Helper to extract raw value from cell with recursive formula evaluation - const getRawValue = (cell: any, row?: number, col?: number): CellValue => { + const getRawValue = (cell: any, position?: CellPosition): CellValue => { // Handle null/undefined cells - treat as 0 if (cell === null || cell === undefined) return 0; + // An already-calculated cell can hold a formula error; it stays an error. + if (isSpreadsheetErrorValue(cell)) return cell; + if (typeof cell === "number") return cell; // Handle string values (for legacy or calculated cells) @@ -175,65 +235,22 @@ export function calculateSheet( } // Handle new cell format {v, f} - if (typeof cell === "object" && cell !== null && "v" in cell) { + if (isObj(cell) && "v" in cell) { const value = cell.v; // If value is a string starting with "=", it's a formula if (typeof value === "string" && value.startsWith("=")) { - // Check if we have row/col info to evaluate recursively - if (row !== undefined && col !== undefined) { - const cellKey = `${row},${col}`; - - // Check for circular reference - if (calculating.has(cellKey)) { - console.warn( - `Circular reference detected at row ${row}, col ${col}`, - ); - errors.push({ - cell: { row, col }, - formula: value, - error: "Circular reference detected", - type: "circular", - }); - return 0; - } - - // Check if already calculated (result is cached as a number) - const calculatedCell = calculated[row][col]; - if (typeof calculatedCell === "number") { - return calculatedCell; - } - - // Recursively evaluate the formula - calculating.add(cellKey); - try { - const formula = value.substring(1); // Remove "=" prefix - const result = evaluateFormula(formula); - calculating.delete(cellKey); - - // Cache the calculated result (preserve strings and numbers) - calculated[row][col] = result; - - return result; - } catch (error) { - calculating.delete(cellKey); - console.error( - `Error evaluating formula at row ${row}, col ${col}:`, - error, - ); - errors.push({ - cell: { row, col }, - formula: value, - error: error instanceof Error ? error.message : String(error), - type: "unknown", - }); - return 0; - } - } - return 0; // No position info, can't evaluate + // Only evaluatable when we know the cell's position (for recursion + + // circular tracking); otherwise treat as 0. + return position ? resolveFormulaCell(value, position) : 0; } - // Try to parse as number, but preserve strings - const num = parseFloat(value); - return isNaN(num) ? value : num; + // Try to parse as number, but preserve original type on failure + if (typeof value === "number") return value; + if (typeof value === "boolean") return value; + if (typeof value === "string") { + const num = parseFloat(value); + return isNaN(num) ? value : num; + } + return String(value); } // Try to parse cell as number, but preserve strings @@ -241,137 +258,87 @@ export function calculateSheet( return isNaN(num) ? cell : num; }; + // Resolve a possibly cross-sheet reference to the grid it reads from, plus the + // sheet-name-stripped ref. A same-sheet ref returns the current `calculated` + // grid; a cross-sheet ref computes and caches its target sheet. null means the + // named sheet does not exist — the caller picks the terminal action (#REF! for + // a single cell, [] for a range). The two-stage cache seed (raw copy published + // BEFORE recursing, real result after) is the cross-sheet infinite-loop guard. + const resolveSheetData = (fullRef: string): ResolvedSheetRef | null => { + const sheetMatch = fullRef.match(/^(?:'([^']+)'|([^!]+))!(.+)$/); + if (!sheetMatch) return { sheetData: calculated, ref: fullRef, isCurrentSheet: true }; + + // Exactly one of the two name branches participates; the reference part + // always does. A shortfall would mean the pattern and this read disagree, so + // it lands on the caller's "sheet not found" path rather than a guess. + const [, quotedName, plainName, innerRef] = sheetMatch; + const targetSheetName = quotedName ?? plainName; + if (targetSheetName === undefined || innerRef === undefined) return null; + + // Check cache first to prevent infinite loops + const cached = sheetsCache.get(targetSheetName); + if (cached) return { sheetData: cached, ref: innerRef, isCurrentSheet: false }; + + const targetSheet = processedAllSheets?.find((s) => s.name === targetSheetName); + if (!targetSheet || !targetSheet.data) return null; + + // Seed the cache with a raw copy BEFORE recursing so a cyclic back-reference + // finds this sheet mid-flight, then overwrite it with the calculated result. + // Resolve cross-sheet values RAW (skip display formatting) so a date cell + // reads as its serial, not the presentation string "03/04/2025". + const targetCalculated = targetSheet.data.map((row) => [...row]); + sheetsCache.set(targetSheetName, targetCalculated); + const targetResult = calculateSheet(targetSheet, processedAllSheets, { preferDDMMYYYY, skipFormatting: true }); + sheetsCache.set(targetSheetName, targetResult.data); + return { sheetData: targetResult.data, ref: innerRef, isCurrentSheet: false }; + }; + // Helper to get cell value by reference (e.g., "B2", "$B$2", or "'Sheet1'!B2") const getCellValue = (ref: string): CellValue => { - let sheetData: any[][] = calculated; - let cellRef = ref; - let isCurrentSheet = true; - - // Check for cross-sheet reference (e.g., 'Sheet Name'!B2 or Sheet1!B2) - const sheetMatch = ref.match(/^(?:'([^']+)'|([^!]+))!(.+)$/); - if (sheetMatch) { - const targetSheetName = sheetMatch[1] || sheetMatch[2]; // Quoted or unquoted sheet name - cellRef = sheetMatch[3]; // Cell reference part - isCurrentSheet = false; - - // Check cache first to prevent infinite loops - if (sheetsCache.has(targetSheetName)) { - sheetData = sheetsCache.get(targetSheetName)!; - } else { - // Find the sheet in all sheets - const targetSheet = processedAllSheets?.find( - (s) => s.name === targetSheetName, - ); - if (targetSheet && targetSheet.data) { - // Calculate formulas for the target sheet with cache - const targetCalculated = targetSheet.data.map((row) => [...row]); - sheetsCache.set(targetSheetName, targetCalculated); - - // Recursively calculate the target sheet - const targetResult = calculateSheet(targetSheet, processedAllSheets); - sheetsCache.set(targetSheetName, targetResult.data); - sheetData = targetResult.data as any[][]; - } else { - return 0; // Sheet not found - } - } - } - - // Remove $ symbols for absolute references - const cleanRef = cellRef.replace(/\$/g, ""); - const match = cleanRef.match(/^([A-Z]+)(\d+)$/); - if (!match) return 0; + const resolved = resolveSheetData(ref); + if (!resolved) throw invalidRefError(ref); // Sheet not found → #REF! + const { sheetData, ref: cellRef, isCurrentSheet } = resolved; - const col = columnToIndex(match[1]); // A=0, B=1, ..., Z=25, AA=26, etc. - const row = parseInt(match[2]) - 1; // 1-indexed to 0-indexed + // `$` symbols and the A1 shape are parsed by the shared single-cell reader. + const coord = parseSingleCellRef(cellRef); + if (!coord) return 0; - if ( - row < 0 || - row >= sheetData.length || - col < 0 || - col >= sheetData[row].length - ) { - return 0; - } + const { row, col } = coord; + const gridRow = row >= 0 ? sheetData[row] : undefined; + if (!gridRow || col < 0 || col >= gridRow.length) return 0; - const cell = sheetData[row][col]; - // Pass row/col only if this is the current sheet (for recursive evaluation) - return getRawValue( - cell, - isCurrentSheet ? row : undefined, - isCurrentSheet ? col : undefined, - ); + // Pass the position only if this is the current sheet (for recursive evaluation) + return getRawValue(gridRow[col], isCurrentSheet ? { cells: gridRow, row, col } : undefined); }; - const collectRangeValues = ( - range: string, - options: { numericOnly: boolean }, - ): CellValue[] => { - let sheetData: any[][] = calculated; - let rangeRef = range; - let isCurrentSheet = true; - - // Check for cross-sheet reference - const sheetMatch = range.match(/^(?:'([^']+)'|([^!]+))!(.+)$/); - if (sheetMatch) { - const targetSheetName = sheetMatch[1] || sheetMatch[2]; - rangeRef = sheetMatch[3]; - isCurrentSheet = false; - - // Check cache first - if (sheetsCache.has(targetSheetName)) { - sheetData = sheetsCache.get(targetSheetName)!; - } else { - // Find and calculate the target sheet - const targetSheet = processedAllSheets?.find( - (s) => s.name === targetSheetName, - ); - if (targetSheet && targetSheet.data) { - const targetCalculated = targetSheet.data.map((row) => [...row]); - sheetsCache.set(targetSheetName, targetCalculated); - - // Recursively calculate the target sheet - const targetResult = calculateSheet(targetSheet, processedAllSheets); - sheetsCache.set(targetSheetName, targetResult.data); - sheetData = targetResult.data as any[][]; - } else { - return []; - } - } - } - - const match = rangeRef.match(/^([A-Z]+)(\d+):([A-Z]+)(\d+)$/); - if (!match) return []; + const collectRangeValues = (range: string, options: { numericOnly: boolean }): CellValue[] => { + const resolved = resolveSheetData(range); + if (!resolved) return []; // Sheet not found → empty range + const { sheetData, ref: rangeRef, isCurrentSheet } = resolved; - const startCol = columnToIndex(match[1]); - const startRow = parseInt(match[2]) - 1; - const endCol = columnToIndex(match[3]); - const endRow = parseInt(match[4]) - 1; + const coords = expandRangeOrCell(rangeRef); + if (!coords) return []; const values: CellValue[] = []; - for (let row = startRow; row <= endRow; row++) { - for (let col = startCol; col <= endCol; col++) { - if ( - row >= 0 && - row < sheetData.length && - col >= 0 && - col < sheetData[row].length - ) { - const cell = sheetData[row][col]; - // Pass row/col only if current sheet (for recursive evaluation) - const rawValue = getRawValue( - cell, - isCurrentSheet ? row : undefined, - isCurrentSheet ? col : undefined, - ); - - if (options.numericOnly) { - if (!isNaN(rawValue as number)) { - values.push(rawValue); - } - } else { + for (const { row, col } of coords) { + const gridRow = row >= 0 ? sheetData[row] : undefined; + if (gridRow && col >= 0 && col < gridRow.length) { + const cell = gridRow[col]; + // Pass the position only if current sheet (for recursive evaluation) + const rawValue = getRawValue(cell, isCurrentSheet ? { cells: gridRow, row, col } : undefined); + + if (options.numericOnly) { + // A blank cell is not a value. Dropping it from the NUMERIC list keeps + // SUM unchanged (a blank read as 0) while stopping it from inflating + // AVERAGE's denominator and COUNT's tally (#2358). The raw list keeps + // every cell so SUMIF/AVERAGEIF's criteria and value ranges stay + // row-aligned; dropping there would shift indexes and aggregate the + // wrong rows (Codex review). + if (!isEmptyCell(cell) && !isNaN(rawValue as number)) { values.push(rawValue); } + } else { + values.push(rawValue); } } } @@ -379,12 +346,10 @@ export function calculateSheet( }; // Helper to get numeric-only range values (legacy behavior) - const getRangeValues = (range: string): CellValue[] => - collectRangeValues(range, { numericOnly: true }); + const getRangeValues = (range: string): CellValue[] => collectRangeValues(range, { numericOnly: true }); // Helper to get raw range values including text - const getRangeValuesRaw = (range: string): CellValue[] => - collectRangeValues(range, { numericOnly: false }); + const getRangeValuesRaw = (range: string): CellValue[] => collectRangeValues(range, { numericOnly: false }); // Evaluate a formula with context const evaluateFormula = (formula: string): CellValue => { @@ -393,111 +358,62 @@ export function calculateSheet( getRangeValues, getRangeValuesRaw, evaluateFormula, + preferDDMMYYYY, }); }; - // Process all cells and calculate formulas - for (let rowIdx = 0; rowIdx < data.length; rowIdx++) { - for (let colIdx = 0; colIdx < data[rowIdx].length; colIdx++) { - const originalCell = data[rowIdx][colIdx]; - const calculatedCell = calculated[rowIdx][colIdx]; - - // Skip if cell was already calculated recursively - if ( - typeof calculatedCell === "number" && - originalCell && - typeof originalCell === "object" && - "f" in originalCell - ) { - // Cell was already evaluated - keep it as number for now - // Formatting will be applied at the end - continue; - } + // Compute one cell into its calculated row. A cell not in {v, f} format is + // left as-is (it is already a plain value). + const calculateCell = (originalCell: SpreadsheetCell, position: CellPosition): void => { + if (!isObj(originalCell) || !("v" in originalCell)) return; + const value = originalCell.v; - // Handle cell format {v, f} - if ( - originalCell && - typeof originalCell === "object" && - "v" in originalCell - ) { - const value = originalCell.v; - - // Check if value is a formula (string starting with "=") - if (typeof value === "string" && value.startsWith("=")) { - // Remove the "=" prefix and evaluate the formula - const formula = value.substring(1); - - // Track formula info - formulas.push({ - cell: { row: rowIdx, col: colIdx }, - formula: value, - dependencies: [], // TODO: Extract dependencies from formula - result: 0, // Will be updated below - }); - - const result = evaluateFormula(formula); - - // Update formula result - formulas[formulas.length - 1].result = result; - - // Store result as-is (formatting will be applied at the end) - calculated[rowIdx][colIdx] = result; - } else { - // Regular value cell (not a formula) - // Convert to plain value (important for range evaluation) - calculated[rowIdx][colIdx] = value; - } - } - // If cell is not in {v, f} format, leave it as-is (already a plain value) + // A plain value is copied through, so range evaluation reads it. + if (typeof value !== "string" || !value.startsWith("=")) { + position.cells[position.col] = value; + return; } - } - // Final formatting pass: apply formatting to all cells with format codes - for (let rowIdx = 0; rowIdx < data.length; rowIdx++) { - for (let colIdx = 0; colIdx < data[rowIdx].length; colIdx++) { - const originalCell = data[rowIdx][colIdx]; - const calculatedValue = calculated[rowIdx][colIdx]; - - if ( - originalCell && - typeof originalCell === "object" && - "v" in originalCell - ) { - const isFormula = - typeof originalCell.v === "string" && originalCell.v.startsWith("="); - - // Apply formatting if cell has a format code and calculated value is a number - if ( - "f" in originalCell && - originalCell.f && - typeof calculatedValue === "number" - ) { - calculated[rowIdx][colIdx] = formatNumber( - calculatedValue, - originalCell.f, - ); - } - // Auto-format date serial numbers from formulas without explicit format - else if ( - isFormula && - typeof calculatedValue === "number" && - calculatedValue >= 36000 && - calculatedValue <= 63499 && - Number.isInteger(calculatedValue) && - (!("f" in originalCell) || !originalCell.f) - ) { - // Check if this looks like a date serial number - // 36000 = Jul 1998, 63499 = Dec 2073 - // Must be integer (dates without time component) - // Avoids formatting calculated averages/sums as dates - // Apply default date format - calculated[rowIdx][colIdx] = formatNumber( - calculatedValue, - "MM/DD/YYYY", - ); - } - } - } + const info: FormulaInfo = { + cell: { row: position.row, col: position.col }, + formula: value, + dependencies: [], // TODO: Extract dependencies from formula + result: 0, // Will be updated below + }; + formulas.push(info); + // Route through the protected path so a thrown failure is classified into + // errors[] instead of escaping this walk, and a cell already resolved via + // another formula's recursion is read from cache (#2359). + info.result = resolveFormulaCell(value, position); + // Store result as-is (formatting will be applied at the end) + position.cells[position.col] = info.result; + }; + + // Walk every cell of the grid. `calculated` is built as a row-for-row copy of + // `data` and never resized, so a missing row cannot happen; skipping one keeps + // the walk total rather than asserting the invariant at each cell. + const forEachCell = (visit: (originalCell: SpreadsheetCell, position: CellPosition) => void): void => { + data.forEach((dataRow, row) => { + const cells = calculated[row]; + if (!cells) return; + dataRow.forEach((originalCell, col) => visit(originalCell, { cells, row, col })); + }); + }; + + // Process all cells and calculate formulas. A cell already evaluated through + // another formula's recursion keeps its number; formatting comes at the end. + forEachCell((originalCell, position) => { + const alreadyCalculated = typeof position.cells[position.col] === "number" && isObj(originalCell) && "f" in originalCell; + if (!alreadyCalculated) calculateCell(originalCell, position); + }); + + // Final display-formatting pass: turn raw serials into presentation strings. + // Skipped when this sheet is computed only to resolve a cross-sheet reference, + // so the referencing cell reads the underlying value, not a display string. + if (!skipFormatting) { + forEachCell((originalCell, { cells, col }) => { + cells[col] = formatCellForDisplay(originalCell, cells[col], preferDDMMYYYY); + }); } return { @@ -514,6 +430,6 @@ export function calculateSheet( * @param sheets - Array of sheets to calculate * @returns Array of calculated sheets */ -export function calculateWorkbook(sheets: SheetData[]): CalculatedSheet[] { - return sheets.map((sheet) => calculateSheet(sheet, sheets)); +export function calculateWorkbook(sheets: SheetData[], options: CalculateOptions = {}): CalculatedSheet[] { + return sheets.map((sheet) => calculateSheet(sheet, sheets, options)); } diff --git a/src/engine/cellBuilder.ts b/src/engine/cellBuilder.ts new file mode 100644 index 0000000..5eaf209 --- /dev/null +++ b/src/engine/cellBuilder.ts @@ -0,0 +1,81 @@ +/** + * Build a SpreadsheetCell from the raw input captured by the mini + * editor (type + value / formula / format). Extracted from + * `saveMiniEditor` in `src/plugins/spreadsheet/View.vue` where it + * was inlined as ~30 lines of nested if/else that pushed the + * surrounding function over the cognitive-complexity threshold. + * + * Pure — no refs, no DOM, no side effects. Given the same inputs + * it always returns the same SpreadsheetCell. Tested in + * `test/plugins/spreadsheet/engine/test_cellBuilder.ts`. + */ + +import type { SpreadsheetCell } from "./types.js"; + +/** Inputs to the cell builder. Mirrors the mini editor refs in the + * View but as plain values so unit tests don't need a Vue runtime. */ +export interface MiniEditorInput { + /** "string" → value is stored as-is as a string. + * Anything else → the `formula` field is parsed (formula / number / raw string). */ + type: string; + /** Used when type === "string". Coerced to string. */ + value: unknown; + /** Used when type !== "string". Trimmed before classification. */ + formula?: string; + /** Optional format code (e.g. "$#,##0.00"). */ + format?: string; +} + +// Anchored at the start of the input (after optional unary +/-) so we +// only treat expressions that clearly begin with a function call as +// formulas. Unanchored would match "abc FOO(" inside ordinary text. +const FORMULA_FUNCTION_CALL = /^[-+]?\s*[A-Z]+\s*\(/i; + +// `A1 + B2` style — cell reference next to an arithmetic operator. +const FORMULA_CELL_OP = /[A-Z]+\d+\s*[+\-*/^]/; + +// `6/100`, `5 * 2` — arithmetic between two literal numbers. +const FORMULA_NUMERIC_OP = /\d+\s*[+\-*/^]\s*\d+/; + +// Strict numeric literal. `parseFloat` accepts trailing junk +// ("42abc" → 42) which silently corrupts user input; this anchor +// ensures the ENTIRE trimmed string is a number. +const STRICT_NUMBER = /^[-+]?(?:\d+\.?\d*|\.\d+)(?:[eE][-+]?\d+)?$/; + +/** + * Best-effort formula detection. The rules are conservative enough + * that plain text like "hello world" stays as text, but any input + * with arithmetic operators or function calls is treated as a + * formula and gets the "=" prefix the engine expects. + */ +export function looksLikeFormula(input: string): boolean { + return FORMULA_FUNCTION_CALL.test(input) || FORMULA_CELL_OP.test(input) || FORMULA_NUMERIC_OP.test(input); +} + +/** + * Parse the raw (non-string-type) editor input into a cell value. + * Priority: formula > number > raw string > empty string. + */ +export function parseNonStringInput(raw: string): number | string { + const input = raw.trim(); + if (input === "") return ""; + if (looksLikeFormula(input)) return `=${input}`; + return STRICT_NUMBER.test(input) ? Number(input) : input; +} + +/** + * Build the full SpreadsheetCell from a mini editor input record. + * String type short-circuits to `{ v: String(value) }`. + * Everything else goes through parseNonStringInput for formula / + * number / text classification, then optionally attaches `f`. + */ +export function buildCellFromInput(input: MiniEditorInput): SpreadsheetCell { + if (input.type === "string") { + return { v: String(input.value) }; + } + const cell: SpreadsheetCell = { v: parseNonStringInput(input.formula ?? "") }; + if (input.format && input.format.length > 0) { + cell.f = input.format; + } + return cell; +} diff --git a/src/engine/cellEmpty.ts b/src/engine/cellEmpty.ts new file mode 100644 index 0000000..55c2398 --- /dev/null +++ b/src/engine/cellEmpty.ts @@ -0,0 +1,26 @@ +/** + * Telling a genuinely empty cell apart from one that holds the number 0. + * + * The calculator reads a blank cell as 0 for arithmetic (`=A1+1` on a blank A1 + * is 1, as in Excel). But an aggregate must not: `AVERAGE` divides by the count + * of real values, and `COUNT` counts numbers — a blank that reads as 0 inflates + * the denominator and the count. So range collection needs to skip the blanks, + * which means distinguishing them from a stored 0, which this does. + */ + +import { isObj } from "./guards"; + +/** True when a cell holds no value at all — absent, null, or an empty/whitespace + * string, in either the bare or the `{ v }` form. A cell containing the number + * 0, `false`, or any non-empty text is NOT empty. */ +export function isEmptyCell(cell: unknown): boolean { + if (cell === null || cell === undefined) return true; + if (typeof cell === "string") return cell.trim() === ""; + if (isObj(cell)) { + if (!("v" in cell)) return true; + const value = (cell as { v: unknown }).v; + if (value === null || value === undefined) return true; + return typeof value === "string" && value.trim() === ""; + } + return false; +} diff --git a/src/engine/cellFormatting.ts b/src/engine/cellFormatting.ts new file mode 100644 index 0000000..aea68ad --- /dev/null +++ b/src/engine/cellFormatting.ts @@ -0,0 +1,59 @@ +/** + * Cell display formatting + * + * Turns a cell's raw calculated value into its display value (currency, + * percentage, date, ...). Pure — no engine state — so cross-sheet reference + * resolution can deliberately SKIP it and keep raw serial numbers, while the + * final output pass applies it for presentation. + */ + +import { formatNumber } from "./formatter"; +import { isRecord } from "./guards"; +import { isSpreadsheetErrorValue } from "./spreadsheet-errors"; +import type { CellValue, SpreadsheetCell, StoredCellValue } from "./types"; + +// Integer serials the engine is willing to auto-format as dates without an +// explicit format code: ~Jul 1998 (36000) through ~Dec 2073 (63499). Narrow on +// purpose so ordinary sums/averages are not mistaken for dates. +const DATE_SERIAL_MIN = 36000; +const DATE_SERIAL_MAX = 63499; + +const isSpreadsheetCell = (value: unknown): value is SpreadsheetCell => isRecord(value) && "v" in value; + +/** An integer within the date-serial window — a calculated number the engine + * should display as a date when the cell carries no explicit format. */ +export const isLikelyDateSerial = (value: CellValue): boolean => + typeof value === "number" && Number.isInteger(value) && value >= DATE_SERIAL_MIN && value <= DATE_SERIAL_MAX; + +/** + * Resolve the display value of one cell from its original definition and its + * calculated value. + * + * - A formula error renders as its code, so the cell still reads `#NUM!`. + * - Explicit format code wins (currency, percentage, date, ...). + * - A formula that produced a date serial auto-formats as a date. + * - Everything else (text, plain numbers, empty) passes through unchanged. + * + * The result is always a STORED value: this is the boundary where a computed + * error becomes the text a cell shows and a workbook serializes. + */ +export const formatCellForDisplay = (originalCell: unknown, calculatedValue: CellValue, preferDDMMYYYY: boolean): StoredCellValue => { + if (isSpreadsheetErrorValue(calculatedValue)) { + return calculatedValue.code; + } + if (!isSpreadsheetCell(originalCell) || typeof calculatedValue !== "number") { + return calculatedValue; + } + + const explicitFormat = typeof originalCell.f === "string" ? originalCell.f : ""; + if (explicitFormat) { + return formatNumber(calculatedValue, explicitFormat); + } + + const isFormula = typeof originalCell.v === "string" && originalCell.v.startsWith("="); + if (isFormula && isLikelyDateSerial(calculatedValue)) { + return formatNumber(calculatedValue, preferDDMMYYYY ? "DD/MM/YYYY" : "MM/DD/YYYY"); + } + + return calculatedValue; +}; diff --git a/src/engine/coerce-boolean.ts b/src/engine/coerce-boolean.ts new file mode 100644 index 0000000..5b56c78 --- /dev/null +++ b/src/engine/coerce-boolean.ts @@ -0,0 +1,26 @@ +import type { CellValue } from "./types"; +import { isSpreadsheetErrorValue } from "./spreadsheet-errors"; + +/** Excel-style truthiness, shared by IF and AND/OR/NOT so the same value cannot + * read as true in one function and false in another. A number is false only + * when 0; blank and empty text are false; the words `true`/`false` are their + * logical values (case-insensitively); a numeric string follows its number + * (`"0"` → false); any other non-empty text is true. */ +export function coerceToBoolean(value: CellValue | null | undefined): boolean { + if (typeof value === "boolean") return value; + if (value === null || value === undefined) return false; + if (typeof value === "number") return value !== 0; + // Pinned: an error reads as non-empty text, i.e. true — the same answer the + // error strings gave before they became values. + if (isSpreadsheetErrorValue(value)) return true; + + const text = value.trim(); + if (text === "") return false; + + const lowered = text.toLowerCase(); + if (lowered === "true") return true; + if (lowered === "false") return false; + + const asNumber = Number(text); + return Number.isNaN(asNumber) ? true : asNumber !== 0; +} diff --git a/src/engine/condition.ts b/src/engine/condition.ts new file mode 100644 index 0000000..5f571e3 --- /dev/null +++ b/src/engine/condition.ts @@ -0,0 +1,182 @@ +/** + * Evaluating a spreadsheet condition without running it as code. + * + * A condition is one comparison, or a bare value tested for truthiness. That is + * the whole grammar — small enough to read directly, which is the point: the + * previous implementation handed the substituted text to `eval`, so a cell + * containing `globalThis.x = 1` executed when any IFS referenced it. + */ + +import type { CellValue } from "./types"; +import { isSpreadsheetErrorValue } from "./spreadsheet-errors"; + +export type ComparisonOperator = ">=" | "<=" | "<>" | "!=" | "==" | "=" | ">" | "<"; + +// Longest first: `>=` must win over `>`, and `<>` / `<=` over `<`. +const OPERATORS: readonly ComparisonOperator[] = [">=", "<=", "<>", "!=", "==", "=", ">", "<"]; + +export interface Comparison { + left: string; + operator: ComparisonOperator; + right: string; +} + +/** Remove parentheses that wrap the WHOLE expression, repeatedly. `(1>0)` is + * the same condition as `1>0`, but `(A)=(B)` is not `A)=(B` — the leading `(` + * closes before the end, so it wraps only its own operand and must stay. + * Unbalanced input is left untouched rather than guessed at. */ +export function stripOuterParens(condition: string): string { + let text = condition.trim(); + while (text.startsWith("(") && text.endsWith(")")) { + let depth = 0; + let quote: string | null = null; + let wrapsAll = true; + for (let index = 0; index < text.length; index++) { + const char = text[index]; + if (quote !== null) { + if (char === "\\") + index++; // skip the escaped character + else if (char === quote) quote = null; + continue; + } + if (char === '"' || char === "'") { + quote = char; + continue; + } + if (char === "(") depth++; + else if (char === ")") { + depth--; + // Back to zero before the end means this `(` closed early. + if (depth === 0 && index < text.length - 1) { + wrapsAll = false; + break; + } + if (depth < 0) return text; // unbalanced + } + } + if (!wrapsAll || depth !== 0) return text; + text = text.slice(1, -1).trim(); + } + return text; +} + +/** Split a condition into its two sides, or null when it holds no comparison. + * Only the FIRST top-level operator counts — `a>b>c` is not a chain here, and + * treating it as one is what let `1=1=1` reach a JS parser before. + * + * "Top-level" means outside quotes: a cell holding `a>b` substitutes into the + * condition as `"a>b"`, and splitting on that `>` would compare two fragments + * of one string literal. */ +export function splitComparison(condition: string): Comparison | null { + const text = stripOuterParens(condition); + let quote: string | null = null; + for (let index = 0; index < text.length; index++) { + const char = text[index]; + if (quote !== null) { + if (char === "\\") + index++; // skip the escaped character + else if (char === quote) quote = null; + continue; + } + if (char === '"' || char === "'") { + quote = char; + continue; + } + for (const operator of OPERATORS) { + if (!text.startsWith(operator, index)) continue; + return { left: text.slice(0, index).trim(), operator, right: text.slice(index + operator.length).trim() }; + } + } + return null; +} + +/** Strip one matching pair of surrounding quotes, and undo the `\"` / `\\` + * escaping that `renderConditionOperand` applies when it quotes a cell value. */ +function unquote(text: string): { value: string; quoted: boolean } { + const isQuoted = text.length >= 2 && ((text.startsWith('"') && text.endsWith('"')) || (text.startsWith("'") && text.endsWith("'"))); + if (!isQuoted) return { value: text, quoted: false }; + const inner = text.slice(1, -1).replace(/\\(["'\\])/g, "$1"); + return { value: inner, quoted: true }; +} + +/** Read an operand as the value it denotes: a quoted string stays text, a + * numeric literal becomes a number, `TRUE`/`FALSE` become booleans, and + * anything else stays the text it already is. */ +export function readOperand(raw: string): CellValue { + const { value, quoted } = unquote(raw.trim()); + if (quoted) return value; + if (value === "") return ""; + const upper = value.toUpperCase(); + if (upper === "TRUE") return true; + if (upper === "FALSE") return false; + // `Number` rather than `parseFloat`: it rejects trailing garbage, so "12abc" + // stays text instead of becoming 12. + const numeric = Number(value); + return Number.isNaN(numeric) ? value : numeric; +} + +function compareValues(left: CellValue, right: CellValue): number | null { + if (typeof left === "number" && typeof right === "number") return left - right; + if (typeof left === "boolean" || typeof right === "boolean") return null; + return String(left).localeCompare(String(right)); +} + +function applyOperator(operator: ComparisonOperator, left: CellValue, right: CellValue): boolean { + // Equality does not need an ordering, so it works for every type pair — + // including the boolean combinations `compareValues` refuses to order. + if (operator === "=" || operator === "==") return left === right; + if (operator === "<>" || operator === "!=") return left !== right; + const ordering = compareValues(left, right); + if (ordering === null) return false; + if (operator === ">") return ordering > 0; + if (operator === ">=") return ordering >= 0; + if (operator === "<") return ordering < 0; + return ordering <= 0; +} + +/** Whether a resolved condition value counts as satisfied. Mirrors the + * spreadsheet convention rather than JavaScript's: 0 and an empty string are + * false, every other value is true. */ +function valueIsTruthy(value: CellValue): boolean { + if (typeof value === "boolean") return value; + if (typeof value === "number") return value !== 0; + return value !== ""; +} + +/** True when a bare (non-comparison) condition counts as satisfied. */ +export function isTruthyCondition(raw: string): boolean { + return valueIsTruthy(readOperand(stripOuterParens(raw))); +} + +/** Evaluate a condition — one comparison, or a value tested for truthiness. + * Never executes its input. */ +export function evaluateCondition(condition: string): boolean { + const comparison = splitComparison(condition); + if (!comparison) return isTruthyCondition(condition); + return applyOperator(comparison.operator, readOperand(comparison.left), readOperand(comparison.right)); +} + +/** Like `evaluateCondition`, but each operand is resolved by `evaluate` — so a + * caller holding the engine can compute arithmetic and sub-expressions + * (`5+1>10`) instead of reading each side as a bare string. It still never runs + * the condition as code: it only splits on the top-level comparison and + * compares the two resolved values. */ +export function evaluateConditionValues(condition: string, evaluate: (operand: string) => CellValue): boolean { + const comparison = splitComparison(condition); + if (!comparison) return valueIsTruthy(evaluate(stripOuterParens(condition))); + return applyOperator(comparison.operator, evaluate(comparison.left), evaluate(comparison.right)); +} + +/** Render a cell's value as an operand for a condition string. A string is + * quoted, with its own quotes and backslashes escaped, so its contents cannot + * be re-parsed as operators — `evaluateCondition` then unquotes it back to the + * original text. A missing value becomes an empty string; numbers and booleans + * render as themselves. */ +export function renderConditionOperand(value: CellValue | null | undefined): string { + if (value === null || value === undefined) return '""'; + // A formula error renders as its quoted code, so a condition compares it as + // the text a cell shows rather than as a bare `#NUM!` token. + const text = isSpreadsheetErrorValue(value) ? value.code : value; + if (typeof text === "string") return `"${text.replace(/\\/g, "\\\\").replace(/"/g, '\\"')}"`; + return text.toString(); +} diff --git a/src/engine/date-locale.ts b/src/engine/date-locale.ts new file mode 100644 index 0000000..850bfa1 --- /dev/null +++ b/src/engine/date-locale.ts @@ -0,0 +1,47 @@ +/** + * Which order an ambiguous slash date is written in, per locale. + * + * Only dates whose two leading numbers are BOTH 12 or under are ambiguous — + * `13/04/2025` can only be day-first, and `04/13/2025` can only be month-first, + * so those decide themselves. This is purely about `03/04/2025`. + */ + +// Day-first locales among the ones the app ships. The rest are month-first +// here: `en` because the app cannot see the region (`en-GB` is folded to `en` +// before it reaches any plugin), and ja / zh / ko because their conventional +// order is year-month-day, which puts the month before the day in a two-part +// date just as US order does. +const DAY_FIRST_LANGUAGES = new Set(["es", "pt", "fr", "de", "it", "nl", "ru", "pl", "tr", "id", "vi", "th"]); + +// English regions that write month-first. Everywhere else that speaks English +// writes day-first. +const MONTH_FIRST_EN_REGIONS = new Set(["us", "ca", "ph"]); + +/** The region subtag, or "" when the tag carries none. Scanning rather than + * taking position 1: a script subtag (`en-Latn-US`) sits between the language + * and the region, and reading it as the region flips month-first English + * locales to day-first. A region is two letters or three digits; a + * single-character subtag starts an extension, so nothing past it is one. */ +function regionOf(subtags: readonly string[]): string { + for (const subtag of subtags) { + if (subtag.length === 1) break; + if (/^[a-z]{2}$/.test(subtag) || /^[0-9]{3}$/.test(subtag)) return subtag; + } + return ""; +} + +/** True when an ambiguous `A/B/YYYY` should read as day-first. Accepts a full + * BCP 47 tag or a bare language subtag; the region is used when present so a + * caller that can supply `en-GB` gets the right answer even though the app's + * own locale resolution discards it. */ +export function prefersDayFirst(locale: string | undefined | null): boolean { + if (!locale) return false; + const [language = "", ...rest] = locale.toLowerCase().split(/[-_]/); + // English splits on region rather than language. A bare `en` carries no + // region to split on, so it keeps the US default. + if (language === "en") { + const region = regionOf(rest); + return region !== "" && !MONTH_FIRST_EN_REGIONS.has(region); + } + return DAY_FIRST_LANGUAGES.has(language); +} diff --git a/src/engine/date-parser.ts b/src/engine/date-parser.ts index 2f6a2f9..1cd1c58 100644 --- a/src/engine/date-parser.ts +++ b/src/engine/date-parser.ts @@ -4,11 +4,7 @@ * Parse various date string formats into Excel serial numbers. */ -import { - dateToSerial, - MONTH_NAMES_SHORT, - MONTH_NAMES_FULL, -} from "./date-utils"; +import { dateToSerial, MONTH_NAMES_SHORT, MONTH_NAMES_FULL } from "./date-utils"; /** * Check if a string looks like a date @@ -45,6 +41,37 @@ export function isDateLike(str: string): boolean { return datePatterns.some((pattern) => pattern.test(str.trim())); } +/** + * Build the Excel serial number for a year/month/day triple, or null when the + * triple is not a valid calendar date. Every dated branch of `parseDate` ends in + * this same validate → Date.UTC → dateToSerial step. + */ +export function serialFromParts(year: number, month: number, day: number): number | null { + if (!isValidDate(year, month, day)) return null; + return dateToSerial(new Date(Date.UTC(year, month - 1, day))); +} + +/** + * The three capture groups of a date pattern, as a tuple the caller can + * destructure. `null` when the text does not match — or when a group did not + * participate, which every pattern here makes impossible, so it lands on the + * same "not a date" answer rather than reading an absent group as text. + */ +function matchDateParts(text: string, pattern: RegExp): [string, string, string] | null { + const [, first, second, third] = text.match(pattern) ?? []; + if (first === undefined || second === undefined || third === undefined) return null; + return [first, second, third]; +} + +const SLASH_DATE_PATTERN = /^(\d{1,2})\/(\d{1,2})\/(\d{2,4})$/; +const MAX_MONTH = 12; + +/** Whether `A/B/YYYY` reads day-first. A number no month can hold decides on its + * own; an ambiguous pair follows the reading preference. Shared with + * `getDefaultDateFormat` so a slash date renders in the order it was READ. */ +const readsDayFirst = (first: number, second: number, preferDDMMYYYY: boolean): boolean => + first > MAX_MONTH || (second <= MAX_MONTH && first <= MAX_MONTH && preferDDMMYYYY); + /** * Parse a month name to month number (1-12) */ @@ -52,15 +79,11 @@ function parseMonthName(monthStr: string): number | null { const month = monthStr.toLowerCase(); // Try short names - const shortIndex = MONTH_NAMES_SHORT.findIndex( - (m) => m.toLowerCase() === month, - ); + const shortIndex = MONTH_NAMES_SHORT.findIndex((m) => m.toLowerCase() === month); if (shortIndex !== -1) return shortIndex + 1; // Try full names - const fullIndex = MONTH_NAMES_FULL.findIndex( - (m) => m.toLowerCase() === month, - ); + const fullIndex = MONTH_NAMES_FULL.findIndex((m) => m.toLowerCase() === month); if (fullIndex !== -1) return fullIndex + 1; return null; @@ -80,108 +103,69 @@ function parseMonthName(monthStr: string): number | null { * @param preferDDMMYYYY - Prefer DD/MM/YYYY over MM/DD/YYYY for ambiguous dates (default: false) * @returns Serial number or null if not a valid date */ -export function parseDate( - dateStr: string, - preferDDMMYYYY: boolean = false, -): number | null { +export function parseDate(dateStr: string, preferDDMMYYYY: boolean = false): number | null { if (!isDateLike(dateStr)) return null; const trimmed = dateStr.trim(); // Try YYYY-MM-DD or YYYY/MM/DD (ISO format) - const isoMatch = trimmed.match(/^(\d{4})[-/](\d{1,2})[-/](\d{1,2})$/); - if (isoMatch) { - const year = parseInt(isoMatch[1]); - const month = parseInt(isoMatch[2]); - const day = parseInt(isoMatch[3]); - - if (isValidDate(year, month, day)) { - const date = new Date(Date.UTC(year, month - 1, day)); - return dateToSerial(date); - } - return null; + const isoParts = matchDateParts(trimmed, /^(\d{4})[-/](\d{1,2})[-/](\d{1,2})$/); + if (isoParts) { + const [year, month, day] = isoParts; + return serialFromParts(parseInt(year), parseInt(month), parseInt(day)); } // Try DD-MMM-YYYY or D-MMM-YYYY - const dmmyMatch = trimmed.match(/^(\d{1,2})-([A-Za-z]{3})-(\d{2,4})$/); - if (dmmyMatch) { - const day = parseInt(dmmyMatch[1]); - const monthName = dmmyMatch[2]; - let year = parseInt(dmmyMatch[3]); - - // Handle 2-digit years - if (year < 100) { - year = year < 30 ? 2000 + year : 1900 + year; - } - - const month = parseMonthName(monthName); - if (month && isValidDate(year, month, day)) { - const date = new Date(Date.UTC(year, month - 1, day)); - return dateToSerial(date); - } - return null; + const dmmyParts = matchDateParts(trimmed, /^(\d{1,2})-([A-Za-z]{3})-(\d{2,4})$/); + if (dmmyParts) { + const [day, monthName, year] = dmmyParts; + return serialFromNamedMonth(expandTwoDigitYear(parseInt(year)), monthName, parseInt(day)); } // Try MMM D, YYYY or MMMM D, YYYY - const mmmMatch = trimmed.match(/^([A-Za-z]{3,9})\s+(\d{1,2}),?\s+(\d{4})$/); - if (mmmMatch) { - const monthName = mmmMatch[1]; - const day = parseInt(mmmMatch[2]); - const year = parseInt(mmmMatch[3]); - - const month = parseMonthName(monthName); - if (month && isValidDate(year, month, day)) { - const date = new Date(Date.UTC(year, month - 1, day)); - return dateToSerial(date); - } - return null; + const mmmParts = matchDateParts(trimmed, /^([A-Za-z]{3,9})\s+(\d{1,2}),?\s+(\d{4})$/); + if (mmmParts) { + const [monthName, day, year] = mmmParts; + return serialFromNamedMonth(parseInt(year), monthName, parseInt(day)); } // Try D MMM YYYY - const dMmmMatch = trimmed.match(/^(\d{1,2})\s+([A-Za-z]{3,9})\s+(\d{4})$/); - if (dMmmMatch) { - const day = parseInt(dMmmMatch[1]); - const monthName = dMmmMatch[2]; - const year = parseInt(dMmmMatch[3]); - - const month = parseMonthName(monthName); - if (month && isValidDate(year, month, day)) { - const date = new Date(Date.UTC(year, month - 1, day)); - return dateToSerial(date); - } - return null; + const dMmmParts = matchDateParts(trimmed, /^(\d{1,2})\s+([A-Za-z]{3,9})\s+(\d{4})$/); + if (dMmmParts) { + const [day, monthName, year] = dMmmParts; + return serialFromNamedMonth(parseInt(year), monthName, parseInt(day)); } // Try MM/DD/YYYY or DD/MM/YYYY - const slashMatch = trimmed.match(/^(\d{1,2})\/(\d{1,2})\/(\d{2,4})$/); - if (slashMatch) { - const first = parseInt(slashMatch[1]); - const second = parseInt(slashMatch[2]); - let year = parseInt(slashMatch[3]); - - // Handle 2-digit years - if (year < 100) { - year = year < 30 ? 2000 + year : 1900 + year; - } - - // Determine if it's MM/DD/YYYY or DD/MM/YYYY - // If first > 12, it must be DD/MM; if second > 12, it must be MM/DD - // Otherwise use preference (default to MM/DD for US format) - const isDayFirst = - first > 12 || (second <= 12 && first <= 12 && preferDDMMYYYY); - const month = isDayFirst ? second : first; - const day = isDayFirst ? first : second; - - if (isValidDate(year, month, day)) { - const date = new Date(Date.UTC(year, month - 1, day)); - return dateToSerial(date); - } - return null; + const slashParts = matchDateParts(trimmed, SLASH_DATE_PATTERN); + if (slashParts) { + const [first, second, year] = slashParts; + // If first > 12, it must be DD/MM; if second > 12, it must be MM/DD. + // Otherwise use preference (default to MM/DD for US format). + const dayFirst = readsDayFirst(parseInt(first), parseInt(second), preferDDMMYYYY); + const month = dayFirst ? parseInt(second) : parseInt(first); + const day = dayFirst ? parseInt(first) : parseInt(second); + return serialFromParts(expandTwoDigitYear(parseInt(year)), month, day); } return null; } +/** A two-digit year reads as this century up to 29, the previous one after. */ +const TWO_DIGIT_YEAR_LIMIT = 100; +const CENTURY_PIVOT = 30; + +function expandTwoDigitYear(year: number): number { + if (year >= TWO_DIGIT_YEAR_LIMIT) return year; + return year < CENTURY_PIVOT ? 2000 + year : 1900 + year; +} + +function serialFromNamedMonth(year: number, monthName: string, day: number): number | null { + const month = parseMonthName(monthName); + if (month === null) return null; + return serialFromParts(year, month, day); +} + /** * Validate that a date is valid */ @@ -195,11 +179,7 @@ function isValidDate(year: number, month: number, day: number): boolean { const date = new Date(Date.UTC(year, month - 1, day)); // If the date rolls over to the next month, it's invalid - return ( - date.getUTCFullYear() === year && - date.getUTCMonth() === month - 1 && - date.getUTCDate() === day - ); + return date.getUTCFullYear() === year && date.getUTCMonth() === month - 1 && date.getUTCDate() === day; } /** @@ -208,7 +188,7 @@ function isValidDate(year: number, month: number, day: number): boolean { * @param originalStr - Original date string * @returns Appropriate format code */ -export function getDefaultDateFormat(originalStr: string): string { +export function getDefaultDateFormat(originalStr: string, preferDDMMYYYY: boolean = false): string { const trimmed = originalStr.trim(); // YYYY-MM-DD → use same format @@ -216,6 +196,14 @@ export function getDefaultDateFormat(originalStr: string): string { return "YYYY-MM-DD"; } + // YYYY/MM/DD parses as ISO, so it must keep a year-first label. Without this + // branch it fell through to the slash default and re-rendered as MM/DD or + // DD/MM — the same digits in a different order, which reads as a different + // date (Codex review). + if (/^\d{4}\/\d{1,2}\/\d{1,2}$/.test(trimmed)) { + return "YYYY/MM/DD"; + } + // DD-MMM-YYYY → use same format if (/^\d{1,2}-[A-Za-z]{3}-\d{2,4}$/.test(trimmed)) { return "DD-MMM-YYYY"; @@ -231,6 +219,16 @@ export function getDefaultDateFormat(originalStr: string): string { return "MMMM D, YYYY"; } - // Default to MM/DD/YYYY for slash-separated dates - return "MM/DD/YYYY"; + // A slash date must render in the order it was READ, or the cell shows the + // user's own input with its two halves swapped. `parseDate` takes the first + // number as the day whenever it cannot be a month, whatever the preference + // says, so that case is decided here the same way. + const slashParts = matchDateParts(trimmed, SLASH_DATE_PATTERN); + if (slashParts) { + const [first, second] = slashParts; + return readsDayFirst(parseInt(first), parseInt(second), preferDDMMYYYY) ? "DD/MM/YYYY" : "MM/DD/YYYY"; + } + + // Anything unrecognised keeps the reading order's default. + return preferDDMMYYYY ? "DD/MM/YYYY" : "MM/DD/YYYY"; } diff --git a/src/engine/date-utils.ts b/src/engine/date-utils.ts index e75ff7d..e5f37f1 100644 --- a/src/engine/date-utils.ts +++ b/src/engine/date-utils.ts @@ -41,25 +41,16 @@ export const serialToDate = (serial: number): Date => { return date; }; +/** A name list that always has a first entry, so a formatter can fall back to it + * when a date is unreadable and its month/weekday index comes out NaN. */ +export type NameList = readonly [string, ...string[]]; + /** * Month names for formatting */ -export const MONTH_NAMES_SHORT = [ - "Jan", - "Feb", - "Mar", - "Apr", - "May", - "Jun", - "Jul", - "Aug", - "Sep", - "Oct", - "Nov", - "Dec", -]; +export const MONTH_NAMES_SHORT: NameList = ["Jan", "Feb", "Mar", "Apr", "May", "Jun", "Jul", "Aug", "Sep", "Oct", "Nov", "Dec"]; -export const MONTH_NAMES_FULL = [ +export const MONTH_NAMES_FULL: NameList = [ "January", "February", "March", @@ -77,22 +68,6 @@ export const MONTH_NAMES_FULL = [ /** * Day names for formatting */ -export const DAY_NAMES_SHORT = [ - "Sun", - "Mon", - "Tue", - "Wed", - "Thu", - "Fri", - "Sat", -]; +export const DAY_NAMES_SHORT: NameList = ["Sun", "Mon", "Tue", "Wed", "Thu", "Fri", "Sat"]; -export const DAY_NAMES_FULL = [ - "Sunday", - "Monday", - "Tuesday", - "Wednesday", - "Thursday", - "Friday", - "Saturday", -]; +export const DAY_NAMES_FULL: NameList = ["Sunday", "Monday", "Tuesday", "Wednesday", "Thursday", "Friday", "Saturday"]; diff --git a/src/engine/datedif.ts b/src/engine/datedif.ts new file mode 100644 index 0000000..8e5ca53 --- /dev/null +++ b/src/engine/datedif.ts @@ -0,0 +1,93 @@ +/** + * DATEDIF — complete elapsed time between two dates, in a chosen unit. + * + * Pure: two Excel serials and a unit in, a number (or a formula error value) + * out. The unit branches each have their own boundary handling (month-end + * borrowing, year wraparound), which is exactly what makes them worth testing + * apart from the handler that reads the arguments. + */ + +import { serialToDate } from "./date-utils"; +import { NUM_ERROR, type SpreadsheetError } from "./spreadsheet-errors"; + +const MS_PER_DAY = 24 * 60 * 60 * 1000; +const MONTHS_PER_YEAR = 12; + +/** `date` advanced by `months`, clamping the day to the target month's length so + * adding a month to Jan 30 lands on the last day of a shorter month instead of + * overflowing into the next one. */ +function addMonthsClamped(date: Date, months: number): Date { + const monthIndex = date.getUTCMonth() + months; + const year = date.getUTCFullYear() + Math.floor(monthIndex / MONTHS_PER_YEAR); + const month = ((monthIndex % MONTHS_PER_YEAR) + MONTHS_PER_YEAR) % MONTHS_PER_YEAR; + const lastDayOfMonth = new Date(Date.UTC(year, month + 1, 0)).getUTCDate(); + const day = Math.min(date.getUTCDate(), lastDayOfMonth); + return new Date(Date.UTC(year, month, day)); +} + +/** Complete `unit`s between two dates, or `#NUM!` when start is after end or + * the unit is not one of Y / M / D / MD / YM / YD. `unit` is matched + * case-insensitively. */ +export function computeDatedif(startSerial: number, endSerial: number, unit: string): number | SpreadsheetError { + if (startSerial > endSerial) return NUM_ERROR; + + const startDate = serialToDate(startSerial); + const endDate = serialToDate(endSerial); + + const yearDiff = endDate.getUTCFullYear() - startDate.getUTCFullYear(); + const monthDiff = endDate.getUTCMonth() - startDate.getUTCMonth(); + const dayDiff = endDate.getUTCDate() - startDate.getUTCDate(); + + switch (unit.toUpperCase()) { + case "Y": { + // Complete years: back off one if the end has not yet reached the + // start's month-and-day within its year. + const years = yearDiff; + return monthDiff < 0 || (monthDiff === 0 && dayDiff < 0) ? years - 1 : years; + } + + case "M": { + // Complete months, backing off one when the day-of-month has not been reached. + const months = yearDiff * MONTHS_PER_YEAR + monthDiff; + return dayDiff < 0 ? months - 1 : months; + } + + case "D": + return Math.floor(endSerial - startSerial); + + case "MD": { + // Days left after the complete months "M" counts. Anchoring on start plus + // those months keeps the result non-negative and self-consistent (start + + // M months + MD days == end). Subtracting the calendar month before `end` + // instead goes negative when the start day outruns that month's length — + // Jan 30 → Mar 1 borrowed Feb's 28 days and returned -1. + const completeMonths = yearDiff * MONTHS_PER_YEAR + monthDiff - (dayDiff < 0 ? 1 : 0); + const anchor = addMonthsClamped(startDate, completeMonths); + // Compare whole days only. `end` may carry a time-of-day (a datetime + // serial); anchor is already UTC midnight, so strip end's time too or the + // remainder would swing with the clock (Codex review). + const endMidnight = Date.UTC(endDate.getUTCFullYear(), endDate.getUTCMonth(), endDate.getUTCDate()); + return Math.round((endMidnight - anchor.getTime()) / MS_PER_DAY); + } + + case "YM": { + // Month difference, ignoring years — wraps into 0..11. + const ym = dayDiff < 0 ? monthDiff - 1 : monthDiff; + return ym < 0 ? ym + MONTHS_PER_YEAR : ym; + } + + case "YD": { + // Day difference, ignoring years: move the start into the end's year, + // stepping back a year if that would put it after the end. + const startInEndYear = new Date(startDate); + startInEndYear.setUTCFullYear(endDate.getUTCFullYear()); + if (startInEndYear.getTime() - endDate.getTime() > 0) { + startInEndYear.setUTCFullYear(endDate.getUTCFullYear() - 1); + } + return Math.floor((endDate.getTime() - startInEndYear.getTime()) / MS_PER_DAY); + } + + default: + return NUM_ERROR; + } +} diff --git a/src/engine/engine.ts b/src/engine/engine.ts index 65fd0a9..516fffb 100644 --- a/src/engine/engine.ts +++ b/src/engine/engine.ts @@ -5,12 +5,8 @@ */ import { calculateSheet, calculateWorkbook } from "./calculator"; -import type { - SheetData, - CalculatedSheet, - EngineOptions, - SpreadsheetCell, -} from "./types"; +import type { SheetData, CalculatedSheet, EngineOptions, SpreadsheetCell } from "./types"; +import { isObj } from "./guards"; /** * SpreadsheetEngine - Main calculation engine class @@ -49,6 +45,7 @@ export class SpreadsheetEngine { maxIterations: options.maxIterations ?? 100, enableCrossSheetRefs: options.enableCrossSheetRefs ?? true, strictMode: options.strictMode ?? false, + preferDDMMYYYY: options.preferDDMMYYYY ?? false, }; } @@ -77,7 +74,7 @@ export class SpreadsheetEngine { * ``` */ calculate(sheet: SheetData, allSheets?: SheetData[]): CalculatedSheet { - return calculateSheet(sheet, allSheets); + return calculateSheet(sheet, allSheets, { preferDDMMYYYY: this.options.preferDDMMYYYY }); } /** @@ -99,7 +96,7 @@ export class SpreadsheetEngine { * ``` */ calculateWorkbook(sheets: SheetData[]): CalculatedSheet[] { - return calculateWorkbook(sheets); + return calculateWorkbook(sheets, { preferDDMMYYYY: this.options.preferDDMMYYYY }); } /** @@ -146,15 +143,12 @@ export class SpreadsheetEngine { * ]); * ``` */ - createSheet( - name: string, - data: Array>, - ): SheetData { + createSheet(name: string, data: Array>): SheetData { return { name, data: data.map((row) => row.map((cell) => { - if (typeof cell === "object" && cell !== null && "v" in cell) { + if (isObj(cell) && "v" in cell) { return cell as SpreadsheetCell; } return { v: cell }; diff --git a/src/engine/evaluator.ts b/src/engine/evaluator.ts index a555727..922db3b 100644 --- a/src/engine/evaluator.ts +++ b/src/engine/evaluator.ts @@ -4,9 +4,12 @@ * Evaluates spreadsheet formulas including functions, cell references, and arithmetic */ -import { functionRegistry } from "./registry"; +import { functionRegistry, tooFewArgumentsError } from "./registry"; import type { CellValue } from "./types"; import { parseDate } from "./date-parser"; +import { caretToPow, replaceConcatOperator, rewriteComparisonEq, isSafeArithmetic, isSafeComparison } from "./translateFormula"; +import { divZeroError, unknownError, nameError, propagatedError } from "./formulaError"; +import { errorCodeOf, isSpreadsheetErrorValue } from "./spreadsheet-errors"; /** * Evaluation context for formulas @@ -14,8 +17,155 @@ import { parseDate } from "./date-parser"; export interface EvaluatorContext { getCellValue: (ref: string) => CellValue; getRangeValues: (range: string) => CellValue[]; - getRangeValuesRaw?: (range: string) => CellValue[]; + getRangeValuesRaw?: ((range: string) => CellValue[]) | undefined; evaluateFormula: (formula: string) => CellValue; + /** Reading order for an ambiguous slash date; see engine/date-locale.ts. */ + preferDDMMYYYY?: boolean | undefined; +} + +/** Render a cell's value as the text that stands in for it inside an + * expression. Strings are quoted so they cannot be read as identifiers or + * operators, and their own quotes and backslashes are escaped so the literal + * cannot be closed early. A missing value becomes 0, matching how blanks are + * treated everywhere else in the engine. */ +export function renderOperand(value: CellValue | null | undefined): string { + if (value === null || value === undefined) return "0"; + const text = isSpreadsheetErrorValue(value) ? value.code : value; + if (typeof text === "string") return `"${text.replace(/\\/g, "\\\\").replace(/"/g, '\\"')}"`; + return text.toString(); +} + +/** Replace every quoted string literal with an empty pair of quotes, honouring + * `\` escapes so an escaped quote does not end the literal early. Used to + * validate the STRUCTURE of a concat/arithmetic expression without letting the + * arbitrary CONTENT of a string decide whether the whole thing looks safe. */ +export function maskStringLiterals(expr: string): string { + let out = ""; + let quote: string | null = null; + for (let index = 0; index < expr.length; index++) { + const char = expr[index]; + if (quote !== null) { + if (char === "\\") { + index++; // skip the escaped character — it is part of the literal + continue; + } + if (char === quote) { + quote = null; + out += char; + } + continue; + } + if (char === '"' || char === "'") { + quote = char; + out += char; + continue; + } + out += char; + } + return out; +} + +/** Whether a `&`-to-`+` concatenation expression is safe to evaluate. Checks the + * structure with string CONTENT masked out — a string literal may hold any + * character (a `!`, a `\`), and validating those against a character allowlist + * rejected valid formulas like `=A1&"!"` and every escaped operand. Once the + * literals are masked, only the joining structure remains to validate. */ +export function isSafeConcatExpression(expr: string): boolean { + // A boolean cell renders as a bare `true` / `false` here (renderOperand) — + // the only non-literal identifier the substitution can emit. Drop those words + // before validating so a boolean operand (`=A1&"!"` with A1 = true) still + // evaluates, while any other identifier keeps the structure from matching and + // never reaches `new Function`. + const structure = maskStringLiterals(expr).replace(/\b(?:true|false)\b/g, ""); + return /^[\d+\-*/(). "']*$/.test(structure); +} + +/** Index just past the string literal that opens at `start` (a quote char), + * honouring `\` escapes so an escaped quote does not close it early. Returns + * `expr.length` when the literal is never closed. */ +export function endOfStringLiteral(expr: string, start: number): number { + const quote = expr[start]; + let index = start + 1; + while (index < expr.length) { + if (expr[index] === "\\") { + index += 2; // skip the escaped character — it is part of the literal + continue; + } + if (expr[index] === quote) return index + 1; + index++; + } + return expr.length; +} + +/** A `'Sheet Name'!A1` reference beginning at `start` (a `'`), or null when the + * quotes open a plain string literal rather than a sheet-qualified reference. */ +function matchQuotedSheetRef(expr: string, start: number): string | null { + const endQuote = expr.indexOf("'", start + 1); + if (endQuote === -1 || expr[endQuote + 1] !== "!") return null; + const cellPart = expr.substring(endQuote + 2).match(/^(\$?[A-Z]+\$?\d+)/); + if (!cellPart) return null; + return expr.substring(start, endQuote + 2 + cellPart[0].length); +} + +/** A `Sheet!A1` or bare `A1` / `$A$1` reference beginning at `start`, or null. */ +function matchUnquotedRef(expr: string, start: number): string | null { + const rest = expr.substring(start); + const sheetMatch = rest.match(/^([A-Z][A-Z0-9]*)!/i); + if (sheetMatch) { + const cellPart = rest.substring(sheetMatch[0].length).match(/^(\$?[A-Z]+\$?\d+)/); + if (cellPart) return sheetMatch[0] + cellPart[0]; + } + const cellMatch = rest.match(/^(\$?[A-Z]+\$?\d+)/); + return cellMatch ? cellMatch[0] : null; +} + +/** Every cell reference in an expression, as `{ref, start}` spans in source + * order. Quoted string literals are skipped whole: a `"A1"` in the text is a + * constant, not a reference, and substituting it would turn `="A1"&"!"` into + * A1's value. A `'` opens either a `'Sheet'!A1` reference or a string literal — + * only the former is a reference; the latter is skipped like a `"` literal. */ +export interface CellRefSpan { + ref: string; + start: number; +} + +export function findCellRefs(expr: string): CellRefSpan[] { + const cellRefs: CellRefSpan[] = []; + let i = 0; + while (i < expr.length) { + const char = expr[i]; + if (char === '"') { + i = endOfStringLiteral(expr, i); + continue; + } + if (char === "'") { + const sheetRef = matchQuotedSheetRef(expr, i); + if (sheetRef) { + cellRefs.push({ ref: sheetRef, start: i }); + i += sheetRef.length; + } else { + i = endOfStringLiteral(expr, i); + } + continue; + } + const ref = matchUnquotedRef(expr, i); + if (ref) { + cellRefs.push({ ref, start: i }); + i += ref.length; + continue; + } + i++; + } + return cellRefs; +} + +/** Replace every reference span in `expr` with the text `render` gives for it, + * walking BACK TO FRONT so the spans ahead of each edit stay valid. A global + * string replace rewrote every occurrence of the shorter reference first, so + * `=A1+A10` had its `A10` broken into `0` and produced a plausible + * wrong number (#2357). */ +export function substituteCellRefs(expr: string, cellRefs: CellRefSpan[], render: (ref: string) => string): string { + return [...cellRefs].reverse().reduce((text, { ref, start }) => text.slice(0, start) + render(ref) + text.slice(start + ref.length), expr); } /** @@ -71,6 +221,27 @@ export function parseFunctionArgs(argsStr: string): string[] { return args; } +/** Run `new Function` on an expression the callers have already gated with a + * character allowlist. A JS parse/eval failure becomes #ERROR! — a genuinely + * broken formula that used to be swallowed into a bare string (#2359). A + * non-finite NUMBER result is only reachable by dividing by zero (comparisons + * yield booleans, concatenation yields strings), so it becomes #DIV/0!. */ +function evalValidatedExpression(jsExpr: string): CellValue { + let result: unknown; + try { + // eslint-disable -- sonarjs/code-eval + result = new Function(`return (${jsExpr})`)(); + } catch { + throw unknownError(); + } + if (typeof result === "number") { + if (!Number.isFinite(result)) throw divZeroError(); + return result; + } + if (typeof result === "string" || typeof result === "boolean") return result; + throw unknownError(); +} + /** * Evaluate a formula string * @@ -80,333 +251,239 @@ export function parseFunctionArgs(argsStr: string): string[] { * - Arithmetic: 2+3, A1*B1, (A1+B1)/2 * - Nested expressions: ROUND(SUM(A1:A10)/COUNT(A1:A10), 2) * + * A genuine failure THROWS (a typed FormulaError or any handler error) rather + * than returning the raw formula text; the calculator classifies it into + * errors[] and shows the Excel error value in the cell (#2359). + * * @param formula - Formula string (without leading =) * @param context - Evaluation context with cell/range accessors * @returns Evaluated result (number or string) */ -export function evaluateFormula( - formula: string, - context: EvaluatorContext, -): CellValue { - try { - // Handle string literals - remove surrounding quotes - // But NOT string concatenations (which contain & operators) - const trimmed = formula.trim(); - if ( - ((trimmed.startsWith('"') && trimmed.endsWith('"')) || - (trimmed.startsWith("'") && trimmed.endsWith("'"))) && - !trimmed.includes("&") // Exclude string concatenations - ) { - const stringValue = trimmed.slice(1, -1); // Remove first and last character (quotes) - - // Auto-parse date strings to serial numbers for compatibility with date arithmetic - // This allows formulas like =HLOOKUP("6/1/2024", ...) to work with parsed date cells - const dateSerial = parseDate(stringValue); - if (dateSerial !== null) { - return dateSerial; - } - - return stringValue; +export function evaluateFormula(formula: string, context: EvaluatorContext): CellValue { + // Handle string literals - remove surrounding quotes + // But NOT string concatenations (which contain & operators) + const trimmed = formula.trim(); + if ( + ((trimmed.startsWith('"') && trimmed.endsWith('"')) || (trimmed.startsWith("'") && trimmed.endsWith("'"))) && + !trimmed.includes("&") // Exclude string concatenations + ) { + const stringValue = trimmed.slice(1, -1); // Remove first and last character (quotes) + + // Auto-parse date strings to serial numbers for compatibility with date arithmetic + // This allows formulas like =HLOOKUP("6/1/2024", ...) to work with parsed date cells + const dateSerial = parseDate(stringValue, context.preferDDMMYYYY); + if (dateSerial !== null) { + return dateSerial; } - // Check if it's a SIMPLE function call (not a complex expression) - // We need to ensure the formula is JUST a function, not "FUNC(...) + something" - const funcMatch = formula.match(/^([A-Z]+)\((.*)\)$/i); - if (funcMatch) { - const [, funcName, argsStr] = funcMatch; - - // Check that the closing paren is actually the end of the function - // by counting parentheses in argsStr - let parenDepth = 0; - let isValidFunction = true; - for (const char of argsStr) { - if (char === "(") parenDepth++; - else if (char === ")") { - parenDepth--; - if (parenDepth < 0) { - // More closing parens than opening - this means we matched too much - isValidFunction = false; - break; - } - } - } - - // Normalize function name to uppercase for registry lookup - const normalizedFuncName = funcName.toUpperCase(); - const func = functionRegistry.get(normalizedFuncName); - - if (func && isValidFunction) { - const args = parseFunctionArgs(argsStr); + return stringValue; + } - // Validate argument count - if (func.minArgs !== undefined && args.length < func.minArgs) { - throw new Error( - `${normalizedFuncName} requires at least ${func.minArgs} argument${func.minArgs !== 1 ? "s" : ""}`, - ); - } - if (func.maxArgs !== undefined && args.length > func.maxArgs) { - throw new Error( - `${normalizedFuncName} accepts at most ${func.maxArgs} argument${func.maxArgs !== 1 ? "s" : ""}`, - ); + // Check if it's a SIMPLE function call (not a complex expression) + // We need to ensure the formula is JUST a function, not "FUNC(...) + something" + const [, funcName, argsStr] = formula.match(/^([A-Z]+)\((.*)\)$/i) ?? []; + if (funcName !== undefined && argsStr !== undefined) { + // Check that the closing paren is actually the end of the function + // by counting parentheses in argsStr + let parenDepth = 0; + let isValidFunction = true; + for (const char of argsStr) { + if (char === "(") parenDepth++; + else if (char === ")") { + parenDepth--; + if (parenDepth < 0) { + // More closing parens than opening - this means we matched too much + isValidFunction = false; + break; } - - // Execute function with context - return func.handler(args, { - getCellValue: context.getCellValue, - getRangeValues: context.getRangeValues, - getRangeValuesRaw: context.getRangeValuesRaw, - evaluateFormula: context.evaluateFormula, - }); } } - // Handle simple arithmetic expressions with cell references - // First, replace any function calls within the expression - let expr = formula; - - // Find and evaluate function calls (e.g., TODAY(), SUM(A1:A10), LOWER(A1), etc.) - // Use a simpler approach: find function names followed by parentheses - // and manually parse the matching closing parenthesis - let searchIndex = 0; - const maxIterations = 100; // Prevent infinite loops - let iterations = 0; - - while (searchIndex < expr.length && iterations < maxIterations) { - iterations++; - const funcNameMatch = expr.substring(searchIndex).match(/^([A-Z]+)\(/i); - if (!funcNameMatch) { - // No more functions found, move to next character - searchIndex++; - if (searchIndex >= expr.length) break; - continue; - } + // Normalize function name to uppercase for registry lookup + const normalizedFuncName = funcName.toUpperCase(); + const func = functionRegistry.get(normalizedFuncName); - const funcStartIndex = searchIndex; - const funcName = funcNameMatch[1]; - const argsStartIndex = searchIndex + funcName.length + 1; - - // Find matching closing parenthesis - let depth = 1; - let argsEndIndex = argsStartIndex; - let inString = false; - let stringChar = ""; - - while (argsEndIndex < expr.length && depth > 0) { - const char = expr[argsEndIndex]; - const prevChar = argsEndIndex > 0 ? expr[argsEndIndex - 1] : ""; - - // Track string boundaries - if ((char === '"' || char === "'") && prevChar !== "\\") { - if (!inString) { - inString = true; - stringChar = char; - } else if (char === stringChar) { - inString = false; - stringChar = ""; - } - } + if (func && isValidFunction) { + const args = parseFunctionArgs(argsStr); - // Only count parens outside of strings - if (!inString) { - if (char === "(") depth++; - else if (char === ")") depth--; - } - argsEndIndex++; + // Validate argument count + if (func.minArgs !== undefined && args.length < func.minArgs) { + throw tooFewArgumentsError(normalizedFuncName, func.minArgs); } - - if (depth === 0) { - const fullMatch = expr.substring(funcStartIndex, argsEndIndex); - const result = context.evaluateFormula(fullMatch); - // For string results, wrap in quotes; for numbers, wrap in parentheses - const replacement = - typeof result === "string" ? `"${result}"` : `(${result})`; - expr = - expr.substring(0, funcStartIndex) + - replacement + - expr.substring(argsEndIndex); - // Continue from after the replacement - searchIndex = funcStartIndex + replacement.length; - } else { - searchIndex++; + if (func.maxArgs !== undefined && args.length > func.maxArgs) { + throw new Error(`${normalizedFuncName} accepts at most ${func.maxArgs} argument${func.maxArgs !== 1 ? "s" : ""}`); } + + // Execute function with context + return func.handler(args, { + functionName: normalizedFuncName, + getCellValue: context.getCellValue, + getRangeValues: context.getRangeValues, + getRangeValuesRaw: context.getRangeValuesRaw, + evaluateFormula: context.evaluateFormula, + }); } - // Then replace cell references with their values - // Match cell references manually to avoid complex regex - const cellRefs: string[] = []; - let i = 0; - while (i < expr.length) { - // Check for cross-sheet reference (quoted or unquoted) - let ref = ""; - if (expr[i] === "'") { - // Quoted sheet name - const endQuote = expr.indexOf("'", i + 1); - if (endQuote !== -1 && expr[endQuote + 1] === "!") { - const cellPart = expr - .substring(endQuote + 2) - .match(/^(\$?[A-Z]+\$?\d+)/); - if (cellPart) { - ref = expr.substring(i, endQuote + 2 + cellPart[0].length); - cellRefs.push(ref); - i += ref.length; - continue; - } - } - } else { - // Unquoted sheet name or simple cell ref - const sheetMatch = expr.substring(i).match(/^([A-Z][A-Z0-9]*)!/i); - if (sheetMatch) { - const cellPart = expr - .substring(i + sheetMatch[0].length) - .match(/^(\$?[A-Z]+\$?\d+)/); - if (cellPart) { - ref = sheetMatch[0] + cellPart[0]; - cellRefs.push(ref); - i += ref.length; - continue; - } - } - // Simple cell reference - const cellMatch = expr.substring(i).match(/^(\$?[A-Z]+\$?\d+)/); - if (cellMatch) { - ref = cellMatch[0]; - cellRefs.push(ref); - i += ref.length; - continue; - } - } - i++; + // A balanced `NAME(...)` spanning the whole formula whose NAME is not a + // registered function is an unrecognized name — Excel's #NAME?. Surfacing + // it here also stops the arithmetic path below from recursing on it forever + // and leaving the literal text in the cell (#2359). + if (isValidFunction && !func) { + throw nameError(normalizedFuncName); } + } - if (cellRefs.length > 0) { - for (const ref of cellRefs) { - const value = context.getCellValue(ref); - // Escape special regex characters - const escapedRef = ref.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"); - // Wrap string values in quotes for proper evaluation - // Handle null/undefined values by treating them as 0 - let replacement: string; - if (value === null || value === undefined) { - replacement = "0"; - } else if (typeof value === "string") { - replacement = `"${value}"`; - } else { - replacement = value.toString(); - } - expr = expr.replace(new RegExp(escapedRef, "g"), replacement); - } + // Handle simple arithmetic expressions with cell references + // First, replace any function calls within the expression + let expr = formula; + + // Find and evaluate function calls (e.g., TODAY(), SUM(A1:A10), LOWER(A1), etc.) + // Use a simpler approach: find function names followed by parentheses + // and manually parse the matching closing parenthesis + let searchIndex = 0; + const maxIterations = 100; // Prevent infinite loops + let iterations = 0; + + while (searchIndex < expr.length && iterations < maxIterations) { + iterations++; + const [, funcName] = expr.substring(searchIndex).match(/^([A-Z]+)\(/i) ?? []; + if (funcName === undefined) { + // No more functions found, move to next character + searchIndex++; + if (searchIndex >= expr.length) break; + continue; } - // Parse date strings in arithmetic expressions (e.g., "06/01/2025" → serial number) - // This allows formulas like =B3-"06/01/2025" to work correctly - expr = expr.replace(/"([^"]+)"/g, (match, dateStr) => { - const dateSerial = parseDate(dateStr); - if (dateSerial !== null) { - return dateSerial.toString(); - } - return match; // Keep original if not a date - }); - - // Replace ^ with ** for exponentiation - expr = expr.replace(/\^/g, "**"); - - // Check if this is a string concatenation expression (contains & and quoted strings) - const hasStringConcat = expr.includes("&"); - const hasQuotedStrings = /["']/.test(expr); - - // If it contains string concatenation, handle it specially - if (hasStringConcat && hasQuotedStrings) { - try { - // Convert & to + for JavaScript string concatenation - // We need to be careful to only replace & that are not inside strings - let inString = false; - let stringChar = ""; - let result = ""; - - for (let index = 0; index < expr.length; index++) { - const char = expr[index]; - const prevChar = index > 0 ? expr[index - 1] : ""; - - // Handle string boundaries - if ((char === '"' || char === "'") && prevChar !== "\\") { - if (!inString) { - inString = true; - stringChar = char; - } else if (char === stringChar) { - inString = false; - stringChar = ""; - } - } - - // Replace & with + when not in a string - if (char === "&" && !inString) { - result += "+"; - } else { - result += char; - } - } + const funcStartIndex = searchIndex; + const argsStartIndex = searchIndex + funcName.length + 1; + + // Find matching closing parenthesis + let depth = 1; + let argsEndIndex = argsStartIndex; + let inString = false; + let stringChar = ""; - // Validate the expression contains only safe characters - // Allow: numbers, letters, strings (with quotes), operators, parentheses, whitespace, @, ., comma - if (/^[a-zA-Z0-9+\-*/(). "'@,]+$/.test(result)) { - // eslint-disable -- sonarjs/code-eval - const evalResult = new Function(`return (${result})`)(); - return evalResult; + while (argsEndIndex < expr.length && depth > 0) { + const char = expr[argsEndIndex]; + const prevChar = argsEndIndex > 0 ? expr[argsEndIndex - 1] : ""; + + // Track string boundaries + if ((char === '"' || char === "'") && prevChar !== "\\") { + if (!inString) { + inString = true; + stringChar = char; + } else if (char === stringChar) { + inString = false; + stringChar = ""; } - } catch (error) { - console.error( - `Failed to evaluate string concatenation: ${expr}`, - error, - ); - return formula; } - } - // Safely evaluate comparison expressions (e.g., 5=6, (5)>(6)) - // Allow numbers, comparison operators (=, !=, <, >, <=, >=), parentheses, whitespace - if (/^[\d+\-*/(). <>!=]+$/.test(expr)) { - try { - // Replace = with == for JavaScript comparison (but not <= or >=) - const jsExpr = expr.replace(/([^<>!])=([^=])/g, "$1==$2"); - - // Use Function constructor which is safer than eval - // eslint-disable -- sonarjs/code-eval - const result = new Function(`return (${jsExpr})`)(); - return result; - } catch { - return formula; + // Only count parens outside of strings + if (!inString) { + if (char === "(") depth++; + else if (char === ")") depth--; } + argsEndIndex++; } - // Safely evaluate arithmetic expressions using Function constructor instead of eval - // Allow numbers, operators, parentheses, whitespace, and decimal points - if (/^[\d+\-*/(). ]+$/.test(expr)) { - try { - // Use Function constructor which is safer than eval because: - // 1. The expression is strictly validated (only numbers and math operators) - // 2. No access to local scope variables - // 3. No this binding issues - // This is safe because we validate the expression first - // eslint-disable -- sonarjs/code-eval - const result = new Function(`return (${expr})`)(); - return result; - } catch { - return formula; - } + if (depth === 0) { + const fullMatch = expr.substring(funcStartIndex, argsEndIndex); + const result = context.evaluateFormula(fullMatch); + // A nested call that errored poisons the expression the same way an + // errored reference does — substituting it produced `"#NUM!"+1` text. + // Only the error VALUE counts: text that merely spells an error (CONCAT) + // is an ordinary operand. + if (isSpreadsheetErrorValue(result)) throw propagatedError(result.code); + // For string results, wrap in quotes; for numbers, wrap in parentheses + const replacement = typeof result === "string" ? `"${result}"` : `(${result})`; + expr = expr.substring(0, funcStartIndex) + replacement + expr.substring(argsEndIndex); + // Continue from after the replacement + searchIndex = funcStartIndex + replacement.length; + } else { + searchIndex++; } + } + + // Then replace cell references with their values. Detection skips quoted + // string literals so a `"A1"` constant is not read as a reference. + const cellRefs = findCellRefs(expr); + + // A formula that is nothing but one reference returns that cell's value + // directly. Rendering it into an expression first would mean escaping the + // text, and the escapes would survive into the result — `=A1` on a cell + // holding `say "hi"` would come back `say \"hi\"`. Surrounding whitespace + // (`= A1`, `=A1 `) is part of "nothing but one reference", so compare the + // span against the trimmed expression rather than the raw one. + const soleRef = cellRefs.length === 1 ? cellRefs[0] : undefined; + if (soleRef) { + const before = expr.slice(0, soleRef.start); + const after = expr.slice(soleRef.start + soleRef.ref.length); + if (before.trim() === "" && after.trim() === "") { + return context.getCellValue(soleRef.ref); + } + } - // If the final expression is a quoted string literal, unwrap it - const trimmedExpr = expr.trim(); - if ( - (trimmedExpr.startsWith('"') && trimmedExpr.endsWith('"')) || - (trimmedExpr.startsWith("'") && trimmedExpr.endsWith("'")) - ) { - return trimmedExpr.slice(1, -1); // Remove quotes + expr = substituteCellRefs(expr, cellRefs, (ref) => { + const value = context.getCellValue(ref); + // A referenced cell holding an error poisons the whole expression: + // rendering it would produce `"#DIV/0!"+1` garbage, so propagate the error + // instead of substituting it (#2359, Excel behaviour). A cell whose stored + // TEXT spells an error counts here too — Excel stores a typed error literal + // as an error, and arithmetic over it is never meaningful. + const referencedErrorCode = errorCodeOf(value); + if (referencedErrorCode !== null) throw propagatedError(referencedErrorCode); + return renderOperand(value); + }); + + // Parse date strings in arithmetic expressions (e.g., "06/01/2025" → serial number) + // This allows formulas like =B3-"06/01/2025" to work correctly + expr = expr.replace(/"([^"]+)"/g, (match, dateStr) => { + const dateSerial = parseDate(dateStr, context.preferDDMMYYYY); + if (dateSerial !== null) { + return dateSerial.toString(); + } + return match; // Keep original if not a date + }); + + // Replace ^ with ** for exponentiation + expr = caretToPow(expr); + + // Check if this is a string concatenation expression (contains & and quoted strings) + const hasStringConcat = expr.includes("&"); + const hasQuotedStrings = /["']/.test(expr); + + // If it contains string concatenation, handle it specially. Convert & to + + // for JS concatenation (leaving any & inside a literal untouched), then + // validate the STRUCTURE with literals masked — a literal may hold any + // character, so masking is what lets `=A1&"!"` and escaped operands through + // while still rejecting a genuinely unsafe expression. A non-safe structure + // falls through to the paths below rather than erroring. + if (hasStringConcat && hasQuotedStrings) { + const result = replaceConcatOperator(expr); + if (isSafeConcatExpression(result)) { + return evalValidatedExpression(result); } + } - return expr; // Return processed expression (with cell refs replaced, etc.) - } catch (error) { - console.error(`Failed to evaluate formula: ${formula}`, error); - return formula; + // Safely evaluate comparison expressions (e.g., 5=6, (5)>(6)). The allowlist + // (numbers, `= != < > <= >=`, parens, whitespace) gates the evaluation. It is + // a superset of the arithmetic allowlist, so a plain `1/0` is handled here — + // evalValidatedExpression turns its Infinity into #DIV/0! (#2359). + if (isSafeComparison(expr)) { + return evalValidatedExpression(rewriteComparisonEq(expr)); } + + // Safely evaluate arithmetic expressions (numbers, operators, parens, + // whitespace, decimal points). + if (isSafeArithmetic(expr)) { + return evalValidatedExpression(expr); + } + + // If the final expression is a quoted string literal, unwrap it + const trimmedExpr = expr.trim(); + if ((trimmedExpr.startsWith('"') && trimmedExpr.endsWith('"')) || (trimmedExpr.startsWith("'") && trimmedExpr.endsWith("'"))) { + return trimmedExpr.slice(1, -1); // Remove quotes + } + + return expr; // Return processed expression (with cell refs replaced, etc.) } diff --git a/src/engine/financial-math.ts b/src/engine/financial-math.ts new file mode 100644 index 0000000..d1361c0 --- /dev/null +++ b/src/engine/financial-math.ts @@ -0,0 +1,114 @@ +/** + * Annuity math for the financial functions, as pure numeric helpers. + * + * Excel's cash-flow sign convention: money received is positive, money paid out + * is negative — so a loan's `pv` is positive and its payment and interest come + * back negative. Keeping the per-period split as plain number-in / number-out + * functions lets it be tested against known Excel values without routing string + * arguments back through the formula evaluator. + */ + +import { NUM_ERROR, type SpreadsheetError } from "./spreadsheet-errors"; + +/** Future value after `nper` periods. `type` is 0 (end of period) or 1 (begin). */ +export function computeFv(rate: number, nper: number, pmt: number, pv: number, type: number): number { + if (rate === 0) return -(pv + pmt * nper); + const growth = Math.pow(1 + rate, nper); + return -pv * growth - (pmt * (growth - 1) * (1 + rate * type)) / rate; +} + +/** Constant per-period payment amortising a present value `pv` to `fv`. */ +export function computePmt(rate: number, nper: number, pv: number, fv: number, type: number): number { + if (rate === 0) return -(fv + pv) / nper; + const growth = Math.pow(1 + rate, nper); + return (-rate * (fv + pv * growth)) / ((growth - 1) * (1 + rate * type)); +} + +/** Interest portion of the `per`-th payment. */ +export function computeIpmt(rate: number, per: number, nper: number, pv: number, fv: number, type: number): number { + const pmt = computePmt(rate, nper, pv, fv, type); + if (per === 1 && type === 1) return 0; + const priorPeriods = type === 1 ? per - 2 : per - 1; + // Interest is the rate applied to the balance outstanding at the start of the + // period. computeFv already carries the payment-negative sign, so this product + // is the interest as a (negative) cash outflow — negating it again inverted the + // sign (IPMT came back +1250 for a -1250 payment) and PPMT amplified the error. + const interest = computeFv(rate, priorPeriods, pmt, pv, type) * rate; + return type === 1 ? interest / (1 + rate) : interest; +} + +/** Principal portion of the `per`-th payment (total payment minus interest). */ +export function computePpmt(rate: number, per: number, nper: number, pv: number, fv: number, type: number): number { + return computePmt(rate, nper, pv, fv, type) - computeIpmt(rate, per, nper, pv, fv, type); +} + +/** Present value of an annuity that grows to `fv` after `nper` payments of `pmt`. */ +export function computePv(rate: number, nper: number, pmt: number, fv: number, type: number): number { + if (rate === 0) return -(fv + pmt * nper); + const growth = Math.pow(1 + rate, nper); + return (-fv - (pmt * (growth - 1) * (1 + rate * type)) / rate) / growth; +} + +/** Number of periods needed to amortise `pv` to `fv` at constant `pmt`. */ +export function computeNper(rate: number, pmt: number, pv: number, fv: number, type: number): number { + if (rate === 0) return -(fv + pv) / pmt; + const pmtWithType = pmt * (1 + rate * type); + return Math.log((pmtWithType - fv * rate) / (pmtWithType + pv * rate)) / Math.log(1 + rate); +} + +// Newton-Raphson is iterative; a valid annuity / cash-flow series converges well +// inside these bounds. A series with no real solution never does, so exhausting +// the loop means "no answer" — Excel reports #NUM! there, and returning the last +// iterate instead would hand back a divergent value or NaN as if it were a rate. +const NEWTON_MAX_ITERATIONS = 100; +const NEWTON_TOLERANCE = 1e-7; + +/** Interest rate per period solving the annuity equation, via Newton-Raphson. */ +export function computeRate(nper: number, pmt: number, pv: number, fv: number, type: number, guess: number): number | SpreadsheetError { + let rate = guess; + for (let i = 0; i < NEWTON_MAX_ITERATIONS; i++) { + if (Math.abs(rate) < NEWTON_TOLERANCE) rate = NEWTON_TOLERANCE; // avoid division by zero + const growth = Math.pow(1 + rate, nper); + const f = pv * growth + pmt * ((growth - 1) / rate) * (1 + rate * type) + fv; + const df = + nper * pv * Math.pow(1 + rate, nper - 1) + + (pmt * (1 + rate * type) * (nper * Math.pow(1 + rate, nper - 1) * rate - (growth - 1))) / (rate * rate) + + pmt * type * ((growth - 1) / rate); + const newRate = rate - f / df; + if (Math.abs(newRate - rate) < NEWTON_TOLERANCE) return Number.isFinite(newRate) ? newRate : NUM_ERROR; + rate = newRate; + } + return NUM_ERROR; +} + +/** Net present value of `cashflows`, each discounted one period further into the future. */ +export function computeNpv(rate: number, cashflows: number[]): number { + return cashflows.reduce((npv, value, index) => npv + value / Math.pow(1 + rate, index + 1), 0); +} + +/** The present value of `values` (element 0 at period 0) discounted at `rate`, + * together with its derivative — the pair one Newton-Raphson step needs. */ +function discountedSeries(values: number[], rate: number): { npv: number; dnpv: number } { + return values.reduce( + (totals, value, period) => { + const factor = Math.pow(1 + rate, period); + return { npv: totals.npv + value / factor, dnpv: totals.dnpv - (period * value) / (factor * (1 + rate)) }; + }, + { npv: 0, dnpv: 0 }, + ); +} + +/** Internal rate of return of `values` (element 0 at period 0), via Newton-Raphson. */ +export function computeIrr(values: number[], guess: number): number | SpreadsheetError { + let rate = guess; + for (let i = 0; i < NEWTON_MAX_ITERATIONS; i++) { + const { npv, dnpv } = discountedSeries(values, rate); + if (Math.abs(npv) < NEWTON_TOLERANCE) return rate; + // A flat derivative leaves nowhere to step: the series has no root here. + if (Math.abs(dnpv) < NEWTON_TOLERANCE) return NUM_ERROR; + const newRate = rate - npv / dnpv; + if (Math.abs(newRate - rate) < NEWTON_TOLERANCE) return Number.isFinite(newRate) ? newRate : NUM_ERROR; + rate = newRate; + } + return NUM_ERROR; +} diff --git a/src/engine/formatter.ts b/src/engine/formatter.ts index bd7728f..d7775a5 100644 --- a/src/engine/formatter.ts +++ b/src/engine/formatter.ts @@ -4,13 +4,7 @@ * Handles Excel-style format codes for currency, percentages, decimals, dates, etc. */ -import { - serialToDate, - MONTH_NAMES_SHORT, - MONTH_NAMES_FULL, - DAY_NAMES_SHORT, - DAY_NAMES_FULL, -} from "./date-utils"; +import { serialToDate, MONTH_NAMES_SHORT, MONTH_NAMES_FULL, DAY_NAMES_SHORT, DAY_NAMES_FULL } from "./date-utils"; /** * Check if a format code is for dates @@ -37,6 +31,12 @@ function isDateFormat(format: string): boolean { * @param format - Date format code * @returns Formatted date string */ +// Every token, longest-first within each family so `MMMM` wins over `MMM` and +// `MM`. Matched in ONE pass: a sequence of `replace` calls re-scans its own +// output, so an inserted "March" had its `M` rewritten by the later month-number +// step and "AM/PM" was destroyed before the meridiem branch could see it. +const DATE_TOKEN_RE = /YYYY|YY|MMMM|MMM|MM|M|dddd|ddd|DD|D|AM\/PM|am\/pm|HH|H|hh|h|mm|ss/g; + function formatDate(serial: number, format: string): string { const date = serialToDate(serial); @@ -48,59 +48,56 @@ function formatDate(serial: number, format: string): string { const seconds = date.getUTCSeconds(); const dayOfWeek = date.getUTCDay(); // 0-6 - let result = format; - - // Replace year tokens - result = result.replace(/YYYY/g, year.toString()); - result = result.replace(/YY/g, (year % 100).toString().padStart(2, "0")); - - // Replace month tokens (order matters - do longer patterns first) - result = result.replace( - /MMMM/g, - MONTH_NAMES_FULL[month] || MONTH_NAMES_FULL[0], - ); - result = result.replace( - /MMM/g, - MONTH_NAMES_SHORT[month] || MONTH_NAMES_SHORT[0], - ); - result = result.replace(/MM/g, (month + 1).toString().padStart(2, "0")); - result = result.replace(/M/g, (month + 1).toString()); - - // Replace day tokens - result = result.replace(/DD/g, day.toString().padStart(2, "0")); - result = result.replace(/D/g, day.toString()); - - // Replace day of week tokens - result = result.replace( - /dddd/g, - DAY_NAMES_FULL[dayOfWeek] || DAY_NAMES_FULL[0], - ); - result = result.replace( - /ddd/g, - DAY_NAMES_SHORT[dayOfWeek] || DAY_NAMES_SHORT[0], - ); - - // Replace time tokens - // Handle 12-hour format with AM/PM - if (result.includes("AM/PM") || result.includes("am/pm")) { - const isPM = hours >= 12; - const hours12 = hours % 12 || 12; // 0 becomes 12 - - result = result.replace(/h/g, hours12.toString()); - result = result.replace(/AM\/PM/g, isPM ? "PM" : "AM"); - result = result.replace(/am\/pm/g, isPM ? "pm" : "am"); - } else { - // 24-hour format - result = result.replace(/HH/g, hours.toString().padStart(2, "0")); - result = result.replace(/H/g, hours.toString()); - result = result.replace(/h/g, hours.toString()); - } + // `h` means the 12-hour clock only when the format also asks for a meridiem; + // decided from the ORIGINAL format, before any substitution. + const uses12Hour = /AM\/PM|am\/pm/.test(format); + const hours12 = hours % 12 || 12; // 0 becomes 12 + const isPM = hours >= 12; + + const replacements: Record = { + YYYY: year.toString(), + YY: (year % 100).toString().padStart(2, "0"), + MMMM: MONTH_NAMES_FULL[month] ?? MONTH_NAMES_FULL[0], + MMM: MONTH_NAMES_SHORT[month] ?? MONTH_NAMES_SHORT[0], + MM: (month + 1).toString().padStart(2, "0"), + M: (month + 1).toString(), + dddd: DAY_NAMES_FULL[dayOfWeek] ?? DAY_NAMES_FULL[0], + ddd: DAY_NAMES_SHORT[dayOfWeek] ?? DAY_NAMES_SHORT[0], + DD: day.toString().padStart(2, "0"), + D: day.toString(), + "AM/PM": isPM ? "PM" : "AM", + "am/pm": isPM ? "pm" : "am", + HH: hours.toString().padStart(2, "0"), + H: hours.toString(), + hh: (uses12Hour ? hours12 : hours).toString().padStart(2, "0"), + h: (uses12Hour ? hours12 : hours).toString(), + mm: minutes.toString().padStart(2, "0"), + ss: seconds.toString().padStart(2, "0"), + }; + + return format.replace(DATE_TOKEN_RE, (token) => replacements[token] ?? token); +} - result = result.replace(/mm/g, minutes.toString().padStart(2, "0")); - result = result.replace(/ss/g, seconds.toString().padStart(2, "0")); +const THOUSANDS_GROUP_SIZE = 3; - return result; -} +/** + * Insert thousands separators into a run of digits ("1234567" → "1,234,567"). + * Regex free on purpose: the classic `\B(?=(\d{3})+(?!\d))` lookahead + * backtracks badly on a long run of digits. + */ +export const groupThousands = (digits: string): string => + Array.from(digits).reduce((grouped, digit, index) => { + const needsSeparator = index > 0 && (digits.length - index) % THOUSANDS_GROUP_SIZE === 0; + return needsSeparator ? `${grouped},${digit}` : `${grouped}${digit}`; + }, ""); + +/** Group the integer part of a formatted number ("1234567.89" → "1,234,567.89"), + * leaving any fractional part after the decimal point untouched. */ +export const addThousandSeparators = (formatted: string): string => { + const [whole, ...fraction] = formatted.split("."); + if (whole === undefined) return formatted; + return [groupThousands(whole), ...fraction].join("."); +}; /** * Format a number according to Excel format code @@ -128,26 +125,10 @@ export function formatNumber(value: number, format: string): string { // Handle currency formats if (format.includes("$")) { const decimals = (format.match(/\.0+/) || [""])[0].length - 1; - const hasComma = format.includes(","); - - let formatted = Math.abs(value).toFixed(decimals >= 0 ? decimals : 0); - if (hasComma) { - // Add thousand separators without regex to avoid performance issues - const parts = formatted.split("."); - const integerPart = parts[0]; - let result = ""; - for (let i = integerPart.length - 1, count = 0; i >= 0; i--, count++) { - if (count > 0 && count % 3 === 0) { - result = "," + result; - } - result = integerPart[i] + result; - } - parts[0] = result; - formatted = parts.join("."); - } - formatted = "$" + formatted; - if (value < 0) formatted = "-" + formatted; - return formatted; + const magnitude = Math.abs(value).toFixed(decimals >= 0 ? decimals : 0); + const grouped = format.includes(",") ? addThousandSeparators(magnitude) : magnitude; + const withSymbol = "$" + grouped; + return value < 0 ? "-" + withSymbol : withSymbol; } // Handle percentage @@ -159,21 +140,8 @@ export function formatNumber(value: number, format: string): string { // Handle comma separator if (format.includes(",")) { const decimals = (format.match(/\.0+/) || [""])[0].length - 1; - let formatted = Math.abs(value).toFixed(decimals >= 0 ? decimals : 0); - // Add thousand separators without regex to avoid performance issues - const parts = formatted.split("."); - const integerPart = parts[0]; - let result = ""; - for (let i = integerPart.length - 1, count = 0; i >= 0; i--, count++) { - if (count > 0 && count % 3 === 0) { - result = "," + result; - } - result = integerPart[i] + result; - } - parts[0] = result; - formatted = parts.join("."); - if (value < 0) formatted = "-" + formatted; - return formatted; + const grouped = addThousandSeparators(Math.abs(value).toFixed(decimals >= 0 ? decimals : 0)); + return value < 0 ? "-" + grouped : grouped; } // Handle decimal places diff --git a/src/engine/formulaError.ts b/src/engine/formulaError.ts new file mode 100644 index 0000000..7e34282 --- /dev/null +++ b/src/engine/formulaError.ts @@ -0,0 +1,67 @@ +/** + * Typed formula-evaluation errors + * + * A failed formula must surface as a typed error with an Excel-style error value + * in the cell, never as a swallowed bare string (issue #2359). This module holds + * the error taxonomy, the throwable carrier, and the pure helpers that classify a + * thrown error and propagate an error value across cell references. No engine + * state is captured — every export is input → output only. + */ + +import type { CalculationError } from "./types"; +import type { SpreadsheetErrorCode } from "./spreadsheet-errors"; + +/** The evaluator-facing error kinds: every `CalculationError["type"]` except + * `circular`, which the calculator raises directly (not via a thrown error). */ +export type FormulaErrorType = Exclude; + +/** The error code shown for each kind, matching Excel's error literals. */ +export const FORMULA_ERROR_VALUES: Record = { + div_zero: "#DIV/0!", + invalid_ref: "#REF!", + syntax: "#NAME?", + unknown: "#ERROR!", +}; + +/** Reverse map for propagation: an error code back to the kind it represents. + * Codes without a dedicated kind (`#N/A`, `#NUM!`, …) propagate as `unknown`. */ +const ERROR_VALUE_TO_TYPE: Record = { + [FORMULA_ERROR_VALUES.div_zero]: "div_zero", + [FORMULA_ERROR_VALUES.invalid_ref]: "invalid_ref", + [FORMULA_ERROR_VALUES.syntax]: "syntax", +}; + +/** A recoverable formula failure carrying the kind and the code to show. */ +export class FormulaError extends Error { + constructor( + readonly errorType: FormulaErrorType, + readonly display: SpreadsheetErrorCode, + message?: string, + ) { + super(message ?? display); + this.name = "FormulaError"; + } +} + +export const isFormulaError = (error: unknown): error is FormulaError => error instanceof FormulaError; + +export const divZeroError = (): FormulaError => new FormulaError("div_zero", FORMULA_ERROR_VALUES.div_zero, "Division by zero"); + +export const invalidRefError = (ref: string): FormulaError => new FormulaError("invalid_ref", FORMULA_ERROR_VALUES.invalid_ref, `Invalid reference: ${ref}`); + +export const nameError = (funcName: string): FormulaError => new FormulaError("syntax", FORMULA_ERROR_VALUES.syntax, `Unknown function: ${funcName}`); + +export const unknownError = (message?: string): FormulaError => new FormulaError("unknown", FORMULA_ERROR_VALUES.unknown, message); + +/** The FormulaError that re-raises an error read from a referenced cell or a + * nested call, so `=A1+1` inherits A1's error instead of corrupting into + * `"#DIV/0!"+1`. */ +export const propagatedError = (value: SpreadsheetErrorCode): FormulaError => + new FormulaError(ERROR_VALUE_TO_TYPE[value] ?? "unknown", value, `Propagated error: ${value}`); + +/** Map any thrown value onto the typed entry the calculator records. A + * `FormulaError` keeps its own kind and display; anything else is `unknown`. */ +export const classifyThrownError = (error: unknown): { type: FormulaErrorType; display: SpreadsheetErrorCode } => { + if (isFormulaError(error)) return { type: error.errorType, display: error.display }; + return { type: "unknown", display: FORMULA_ERROR_VALUES.unknown }; +}; diff --git a/src/engine/formulaRefs.ts b/src/engine/formulaRefs.ts new file mode 100644 index 0000000..c6cb9a6 --- /dev/null +++ b/src/engine/formulaRefs.ts @@ -0,0 +1,202 @@ +/** + * Extract the set of cells that a formula references. + * + * Extracted from `src/plugins/spreadsheet/View.vue` (was the body of + * `extractCellReferences`, cognitive complexity 32). The original + * function combined regex scanning, range expansion, single-cell + * parsing, and deduplication all in one body; splitting each concern + * into a named helper brings the top-level function well under the + * sonarjs/cognitive-complexity threshold of 15 and makes the pure + * logic unit-testable in isolation (see + * `test/plugins/spreadsheet/engine/test_formulaRefs.ts`). + * + * Tracks #175. No behavioural change — the wrapper in View.vue + * still returns exactly the same `{ row, col }` list as before. + */ + +import { columnToIndex } from "./parser.js"; + +export interface CellCoord { + row: number; + col: number; +} + +// `A1:B10`, `$A$1:$B$10`, `Sheet` refs are out of scope here — the +// caller only passes the formula body, and cross-sheet ranges never +// reached the original regex anyway. Keeping the patterns identical +// to the pre-refactor code preserves behaviour exactly. +const RANGE_REGEX = /\$?[A-Z]+\$?\d+:\$?[A-Z]+\$?\d+/g; +const CELL_REGEX = /\$?[A-Z]+\$?\d+/g; + +// Excel formulas start with `=`. Strip it for uniform handling. +// Keeps any inner `=` intact (Excel does not allow them but the +// caller may pass partial text during live editing). +export function stripFormulaPrefix(formula: string): string { + return formula.startsWith("=") ? formula.slice(1) : formula; +} + +const A1_REGEX = /^([A-Z]+)(\d+)$/; + +// One `A1` token as its column index (0-based) and its row NUMBER as written +// (1-based). The single place a column letter + row digit pair is read, so the +// range and single-cell parsers below cannot drift apart. +function parseA1Token(token: string): { col: number; rowNumber: number } | null { + const [, column, row] = token.match(A1_REGEX) ?? []; + if (column === undefined || row === undefined) return null; + return { col: columnToIndex(column), rowNumber: parseInt(row, 10) }; +} + +// The two endpoints of an `A1:B3` range, or null for anything that is not +// exactly two `A1` tokens joined by one colon. +function parseRangeEndpoints(body: string): { start: { col: number; rowNumber: number }; end: { col: number; rowNumber: number } } | null { + const [startToken, endToken, ...extra] = body.split(":"); + if (startToken === undefined || endToken === undefined || extra.length > 0) return null; + const start = parseA1Token(startToken); + const end = parseA1Token(endToken); + return start && end ? { start, end } : null; +} + +// Expand a single range token (`A1:B3`, `$A$1:$C$5`) into every +// coordinate the range covers. Returns an empty array for malformed +// input so callers never have to handle exceptions; the worst case +// is "we silently ignored a weird-looking substring," which matches +// the original inline behaviour. +export function expandRange(rangeStr: string): CellCoord[] { + const endpoints = parseRangeEndpoints(rangeStr.replace(/\$/g, "")); + if (!endpoints) return []; + const { start, end } = endpoints; + const cells: CellCoord[] = []; + for (let row = start.rowNumber - 1; row <= end.rowNumber - 1; row++) { + for (let col = start.col; col <= end.col; col++) { + cells.push({ row, col }); + } + } + return cells; +} + +// Expand a range OR a single cell into coordinates, upcasing first so +// lowercase references (`a1:b2`, which spreadsheets accept) are not dropped, +// and falling back to a single cell when there is no colon. `collectRangeValues` +// in the calculator used a range-only, case-sensitive regex, so `A1`, +// `$A$1:$A$10` and `a1:a10` all silently produced no values (#2356). Ordering +// matches `expandRange`: top-to-bottom, left-to-right. +export function expandRangeOrCell(ref: string): CellCoord[] | null { + const upper = ref.trim().toUpperCase(); + if (upper.includes(":")) { + const cells = expandRange(upper); + return cells.length > 0 ? cells : null; + } + const single = parseSingleCellRef(upper); + return single ? [single] : null; +} + +// Parse a single cell ref (`A1`, `$A$1`, `AA100`) into a coord. +// Returns null for malformed input rather than throwing — keeps the +// caller's loop flat (the engine-layer `parseCellRef` throws, which +// is fine for the evaluator but wrong for a best-effort scanner). +export function parseSingleCellRef(refStr: string): CellCoord | null { + const parsed = parseA1Token(refStr.replace(/\$/g, "")); + return parsed ? { col: parsed.col, row: parsed.rowNumber - 1 } : null; +} + +// Numeric bounds of a `A2:C10` range, with any `Sheet1!` / `'My Sheet'!` +// prefix kept verbatim so callers can rebuild sheet-qualified refs. Columns +// are 0-based (via `columnToIndex`); rows stay 1-based, matching A1 notation. +export interface RangeBounds { + sheetPrefix: string; + startCol: number; + startRow: number; + endCol: number; + endRow: number; +} + +// Split an optional sheet prefix from a range, then parse the `A2:C10` body. +// The prefix is everything up to and including the last `!`, so a quoted sheet +// name that itself contains no `!` (the common case) is preserved intact. The +// lookup functions each carried their own copy of this parse; one copy ran a +// sheet-unaware regex against the whole string and threw on `Sheet1!A2:C10` +// before the sheet-aware copy could run (#2390). Returns null for anything that +// is not a two-endpoint range so callers surface one "Invalid range" message. +export function parseRangeBounds(range: string): RangeBounds | null { + const bang = range.lastIndexOf("!"); + const sheetPrefix = bang >= 0 ? range.slice(0, bang + 1) : ""; + const body = bang >= 0 ? range.slice(bang + 1) : range; + const endpoints = parseRangeEndpoints(body); + if (!endpoints) return null; + const { start, end } = endpoints; + return { sheetPrefix, startCol: start.col, startRow: start.rowNumber, endCol: end.col, endRow: end.rowNumber }; +} + +// Excel's `0` row/column index selects the entire row/column. This scalar engine +// returns a single cell, so `0` is representable only when that dimension is one +// line long; otherwise it is out of range. +const WHOLE_LINE_INDEX = 0; +const FIRST_INDEX = 1; + +// Map a 1-based INDEX position onto a 0-based offset within a `size`-long line, +// truncating toward zero as Excel does. Returns null when the position falls +// outside the line — including a `0` (whole-line) request the scalar engine +// cannot collapse to one cell. Callers turn null into #REF!. +function lineOffset(position: number, size: number): number | null { + const index = Math.trunc(position); + if (!Number.isFinite(index)) return null; + if (index === WHOLE_LINE_INDEX) return size === 1 ? 0 : null; + if (index < FIRST_INDEX || index > size) return null; + return index - FIRST_INDEX; +} + +// Resolve INDEX(range, rowNum, colNum)'s 1-based position to an absolute cell +// within the range, or null (→ #REF!) when it falls outside. The former handler +// computed the target with no bounds check, so an out-of-range index silently +// read a cell outside the range (#2390: `INDEX(A1:A3,5)` read A5, `INDEX(A2:B5, +// 0,1)` read A1). Returned column is 0-based; row is 1-based (A1 notation). +export function resolveIndexTarget(bounds: RangeBounds, rowNum: number, colNum: number): { colIndex: number; row: number } | null { + const rowOffset = lineOffset(rowNum, bounds.endRow - bounds.startRow + 1); + const colOffset = lineOffset(colNum, bounds.endCol - bounds.startCol + 1); + if (rowOffset === null || colOffset === null) return null; + return { colIndex: bounds.startCol + colOffset, row: bounds.startRow + rowOffset }; +} + +// VLOOKUP's col_index_num / HLOOKUP's row_index_num as a 0-based offset inside +// the table, or null (→ #REF!) when it points outside. Excel rejects an index +// past the table's width; without the check the handler addressed a cell beyond +// the range and returned whatever lived there — usually a silent 0 (#2360). +export function resolveTableOffset(position: number, size: number): number | null { + // Not `lineOffset`: INDEX reads a `0` position as "the whole line", which it + // can collapse to one cell when the line is one long. A lookup index has no + // such meaning — VLOOKUP's columns are numbered from 1 — so `0` is out of + // range even for a single-column table (Codex review). + const index = Math.trunc(position); + if (!Number.isFinite(index)) return null; + if (index < FIRST_INDEX || index > size) return null; + return index - FIRST_INDEX; +} + +// Top-level: scan the formula, expand any ranges, then pick up +// remaining single-cell refs, deduplicating as we go. Kept short +// (~15 lines) so the cognitive-complexity signal lands on the +// helpers if anything grows here. +export function extractCellReferences(formula: string): CellCoord[] { + const clean = stripFormulaPrefix(formula); + const refs: CellCoord[] = []; + const seen = new Set(); + const addUnique = (coord: CellCoord): void => { + const key = `${coord.row},${coord.col}`; + if (seen.has(key)) return; + seen.add(key); + refs.push(coord); + }; + + for (const range of clean.match(RANGE_REGEX) ?? []) { + for (const coord of expandRange(range)) addUnique(coord); + } + // Strip matched ranges so the cell-regex doesn't re-emit their + // endpoints as standalone refs (mirrors the original's second + // `.replace(rangeRegex, "")` pass). + const withoutRanges = clean.replace(RANGE_REGEX, ""); + for (const cellStr of withoutRanges.match(CELL_REGEX) ?? []) { + const coord = parseSingleCellRef(cellStr); + if (coord) addUnique(coord); + } + return refs; +} diff --git a/src/engine/functions/date.ts b/src/engine/functions/date.ts index 0e6e33a..311ab70 100644 --- a/src/engine/functions/date.ts +++ b/src/engine/functions/date.ts @@ -4,51 +4,30 @@ * January 1, 1900 is serial number 1. */ -import { - functionRegistry, - toNumber, - toString, - type FunctionHandler, -} from "../registry"; +import { functionRegistry, requiredArg, toNumber, toString, type FunctionHandler } from "../registry"; +import { computeDatedif } from "../datedif"; import { dateToSerial, serialToDate } from "../date-utils"; -const MS_PER_DAY = 24 * 60 * 60 * 1000; - -const nowHandler: FunctionHandler = (args) => { - if (args.length !== 0) throw new Error("NOW requires 0 arguments"); +const nowHandler: FunctionHandler = () => { // Return current date and time as serial number // We need to adjust for timezone offset because Excel dates are "local" usually // But for simplicity we'll use local time converted to serial const now = new Date(); // Create a UTC date that matches the local time components - const localAsUtc = new Date( - Date.UTC( - now.getFullYear(), - now.getMonth(), - now.getDate(), - now.getHours(), - now.getMinutes(), - now.getSeconds(), - ), - ); + const localAsUtc = new Date(Date.UTC(now.getFullYear(), now.getMonth(), now.getDate(), now.getHours(), now.getMinutes(), now.getSeconds())); return dateToSerial(localAsUtc); }; -const todayHandler: FunctionHandler = (args) => { - if (args.length !== 0) throw new Error("TODAY requires 0 arguments"); +const todayHandler: FunctionHandler = () => { const now = new Date(); - const today = new Date( - Date.UTC(now.getFullYear(), now.getMonth(), now.getDate()), - ); + const today = new Date(Date.UTC(now.getFullYear(), now.getMonth(), now.getDate())); return dateToSerial(today); }; const dateHandler: FunctionHandler = (args, context) => { - if (args.length !== 3) throw new Error("DATE requires 3 arguments"); - - const year = toNumber(context.evaluateFormula(args[0])); - const month = toNumber(context.evaluateFormula(args[1])); - const day = toNumber(context.evaluateFormula(args[2])); + const year = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + const month = toNumber(context.evaluateFormula(requiredArg(context, args, 1))); + const day = toNumber(context.evaluateFormula(requiredArg(context, args, 2))); // JS Date constructor handles overflow (e.g. month 13 becomes Jan of next year) // Month is 0-indexed in JS, 1-indexed in Excel @@ -57,11 +36,9 @@ const dateHandler: FunctionHandler = (args, context) => { }; const timeHandler: FunctionHandler = (args, context) => { - if (args.length !== 3) throw new Error("TIME requires 3 arguments"); - - const hour = toNumber(context.evaluateFormula(args[0])); - const minute = toNumber(context.evaluateFormula(args[1])); - const second = toNumber(context.evaluateFormula(args[2])); + const hour = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + const minute = toNumber(context.evaluateFormula(requiredArg(context, args, 1))); + const second = toNumber(context.evaluateFormula(requiredArg(context, args, 2))); // Time is a fraction of a day // 1 hour = 1/24 @@ -76,29 +53,25 @@ const timeHandler: FunctionHandler = (args, context) => { }; const yearHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("YEAR requires 1 argument"); - const serial = toNumber(context.evaluateFormula(args[0])); + const serial = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); const date = serialToDate(serial); return date.getUTCFullYear(); }; const monthHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("MONTH requires 1 argument"); - const serial = toNumber(context.evaluateFormula(args[0])); + const serial = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); const date = serialToDate(serial); return date.getUTCMonth() + 1; // 1-indexed }; const dayHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("DAY requires 1 argument"); - const serial = toNumber(context.evaluateFormula(args[0])); + const serial = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); const date = serialToDate(serial); return date.getUTCDate(); }; const hourHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("HOUR requires 1 argument"); - const serial = toNumber(context.evaluateFormula(args[0])); + const serial = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); // Get fractional part const timePart = serial - Math.floor(serial); const totalSeconds = Math.round(timePart * 86400); @@ -106,103 +79,25 @@ const hourHandler: FunctionHandler = (args, context) => { }; const minuteHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("MINUTE requires 1 argument"); - const serial = toNumber(context.evaluateFormula(args[0])); + const serial = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); const timePart = serial - Math.floor(serial); const totalSeconds = Math.round(timePart * 86400); return Math.floor((totalSeconds % 3600) / 60); }; const secondHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("SECOND requires 1 argument"); - const serial = toNumber(context.evaluateFormula(args[0])); + const serial = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); const timePart = serial - Math.floor(serial); const totalSeconds = Math.round(timePart * 86400); return totalSeconds % 60; }; const datedifHandler: FunctionHandler = (args, context) => { - if (args.length !== 3) throw new Error("DATEDIF requires 3 arguments"); - - const startSerial = toNumber(context.evaluateFormula(args[0])); - const endSerial = toNumber(context.evaluateFormula(args[1])); - const unit = toString(context.evaluateFormula(args[2])).toUpperCase(); - - if (startSerial > endSerial) return "#NUM!"; - - const startDate = serialToDate(startSerial); - const endDate = serialToDate(endSerial); - - const yearDiff = endDate.getUTCFullYear() - startDate.getUTCFullYear(); - const monthDiff = endDate.getUTCMonth() - startDate.getUTCMonth(); - const dayDiff = endDate.getUTCDate() - startDate.getUTCDate(); - - switch (unit) { - case "Y": { - // Complete years - let years = yearDiff; - if (monthDiff < 0 || (monthDiff === 0 && dayDiff < 0)) { - years--; - } - return years; - } - - case "M": { - // Complete months - let months = yearDiff * 12 + monthDiff; - if (dayDiff < 0) { - months--; - } - return months; - } - - case "D": - // Complete days - return Math.floor(endSerial - startSerial); - - case "MD": { - // Difference in days, ignoring months and years - // This is tricky. It's basically day of month difference, but handling wrap around - // E.g. Jan 30 to Mar 1. - // Standard implementation: - const startD = startDate.getUTCDate(); - const endD = endDate.getUTCDate(); - - if (endD >= startD) return endD - startD; - - // Need to borrow days from previous month - const prevMonthDate = new Date( - Date.UTC(endDate.getUTCFullYear(), endDate.getUTCMonth(), 0), - ); - return prevMonthDate.getUTCDate() - startD + endD; - } - - case "YM": { - // Difference in months, ignoring years - let ym = monthDiff; - if (dayDiff < 0) ym--; - if (ym < 0) ym += 12; - return ym; - } - - case "YD": { - // Difference in days, ignoring years - // Treat start date as being in the same year as end date - // If start > end (after adjusting year), move start to previous year - const startCopy = new Date(startDate); - startCopy.setUTCFullYear(endDate.getUTCFullYear()); - - const diff = (startCopy.getTime() - endDate.getTime()) / MS_PER_DAY; - if (diff > 0) { - startCopy.setUTCFullYear(endDate.getUTCFullYear() - 1); - } - - return Math.floor((endDate.getTime() - startCopy.getTime()) / MS_PER_DAY); - } + const startSerial = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + const endSerial = toNumber(context.evaluateFormula(requiredArg(context, args, 1))); + const unit = toString(context.evaluateFormula(requiredArg(context, args, 2))); - default: - return "#NUM!"; - } + return computeDatedif(startSerial, endSerial, unit); }; // Register functions @@ -311,8 +206,7 @@ functionRegistry.register({ handler: datedifHandler, minArgs: 3, maxArgs: 3, - description: - "Calculates the number of days, months, or years between two dates", + description: "Calculates the number of days, months, or years between two dates", examples: ['DATEDIF(A1, B1, "Y")', 'DATEDIF(DATE(2020,1,1), TODAY(), "D")'], category: "Date & Time", }); diff --git a/src/engine/functions/financial.ts b/src/engine/functions/financial.ts index b0ed85a..4cbc19a 100644 --- a/src/engine/functions/financial.ts +++ b/src/engine/functions/financial.ts @@ -2,7 +2,8 @@ * Financial Functions */ -import { functionRegistry, toNumber, type FunctionHandler } from "../registry"; +import { functionRegistry, requiredArg, toNumber, type FunctionContext, type FunctionHandler } from "../registry"; +import { computeFv, computePmt, computeIpmt, computePpmt, computePv, computeNper, computeRate, computeNpv, computeIrr } from "../financial-math"; /** * FV - Future Value @@ -15,25 +16,13 @@ import { functionRegistry, toNumber, type FunctionHandler } from "../registry"; * - type: 0 = end of period, 1 = beginning of period (optional, default 0) */ const fvHandler: FunctionHandler = (args, context) => { - if (args.length < 3 || args.length > 5) { - throw new Error("FV requires 3 to 5 arguments"); - } - - const rate = toNumber(context.evaluateFormula(args[0])); - const nper = toNumber(context.evaluateFormula(args[1])); - const pmt = toNumber(context.evaluateFormula(args[2])); - const pv = args.length >= 4 ? toNumber(context.evaluateFormula(args[3])) : 0; - const type = - args.length >= 5 ? toNumber(context.evaluateFormula(args[4])) : 0; - - if (rate === 0) { - return -(pv + pmt * nper); - } + const rate = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + const nper = toNumber(context.evaluateFormula(requiredArg(context, args, 1))); + const pmt = toNumber(context.evaluateFormula(requiredArg(context, args, 2))); + const pv = args.length >= 4 ? toNumber(context.evaluateFormula(requiredArg(context, args, 3))) : 0; + const type = args.length >= 5 ? toNumber(context.evaluateFormula(requiredArg(context, args, 4))) : 0; - const pvFactor = Math.pow(1 + rate, nper); - const fv = -pv * pvFactor - (pmt * (pvFactor - 1) * (1 + rate * type)) / rate; - - return fv; + return computeFv(rate, nper, pmt, pv, type); }; /** @@ -42,26 +31,13 @@ const fvHandler: FunctionHandler = (args, context) => { * PV(rate, nper, pmt, [fv], [type]) */ const pvHandler: FunctionHandler = (args, context) => { - if (args.length < 3 || args.length > 5) { - throw new Error("PV requires 3 to 5 arguments"); - } - - const rate = toNumber(context.evaluateFormula(args[0])); - const nper = toNumber(context.evaluateFormula(args[1])); - const pmt = toNumber(context.evaluateFormula(args[2])); - const fv = args.length >= 4 ? toNumber(context.evaluateFormula(args[3])) : 0; - const type = - args.length >= 5 ? toNumber(context.evaluateFormula(args[4])) : 0; - - if (rate === 0) { - return -(fv + pmt * nper); - } - - const pvFactor = Math.pow(1 + rate, nper); - const pv = - (-fv - (pmt * (pvFactor - 1) * (1 + rate * type)) / rate) / pvFactor; + const rate = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + const nper = toNumber(context.evaluateFormula(requiredArg(context, args, 1))); + const pmt = toNumber(context.evaluateFormula(requiredArg(context, args, 2))); + const fv = args.length >= 4 ? toNumber(context.evaluateFormula(requiredArg(context, args, 3))) : 0; + const type = args.length >= 5 ? toNumber(context.evaluateFormula(requiredArg(context, args, 4))) : 0; - return pv; + return computePv(rate, nper, pmt, fv, type); }; /** @@ -70,26 +46,13 @@ const pvHandler: FunctionHandler = (args, context) => { * PMT(rate, nper, pv, [fv], [type]) */ const pmtHandler: FunctionHandler = (args, context) => { - if (args.length < 3 || args.length > 5) { - throw new Error("PMT requires 3 to 5 arguments"); - } - - const rate = toNumber(context.evaluateFormula(args[0])); - const nper = toNumber(context.evaluateFormula(args[1])); - const pv = toNumber(context.evaluateFormula(args[2])); - const fv = args.length >= 4 ? toNumber(context.evaluateFormula(args[3])) : 0; - const type = - args.length >= 5 ? toNumber(context.evaluateFormula(args[4])) : 0; + const rate = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + const nper = toNumber(context.evaluateFormula(requiredArg(context, args, 1))); + const pv = toNumber(context.evaluateFormula(requiredArg(context, args, 2))); + const fv = args.length >= 4 ? toNumber(context.evaluateFormula(requiredArg(context, args, 3))) : 0; + const type = args.length >= 5 ? toNumber(context.evaluateFormula(requiredArg(context, args, 4))) : 0; - if (rate === 0) { - return -(fv + pv) / nper; - } - - const pvFactor = Math.pow(1 + rate, nper); - const pmt = - (-rate * (fv + pv * pvFactor)) / ((pvFactor - 1) * (1 + rate * type)); - - return pmt; + return computePmt(rate, nper, pv, fv, type); }; /** @@ -98,27 +61,13 @@ const pmtHandler: FunctionHandler = (args, context) => { * NPER(rate, pmt, pv, [fv], [type]) */ const nperHandler: FunctionHandler = (args, context) => { - if (args.length < 3 || args.length > 5) { - throw new Error("NPER requires 3 to 5 arguments"); - } + const rate = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + const pmt = toNumber(context.evaluateFormula(requiredArg(context, args, 1))); + const pv = toNumber(context.evaluateFormula(requiredArg(context, args, 2))); + const fv = args.length >= 4 ? toNumber(context.evaluateFormula(requiredArg(context, args, 3))) : 0; + const type = args.length >= 5 ? toNumber(context.evaluateFormula(requiredArg(context, args, 4))) : 0; - const rate = toNumber(context.evaluateFormula(args[0])); - const pmt = toNumber(context.evaluateFormula(args[1])); - const pv = toNumber(context.evaluateFormula(args[2])); - const fv = args.length >= 4 ? toNumber(context.evaluateFormula(args[3])) : 0; - const type = - args.length >= 5 ? toNumber(context.evaluateFormula(args[4])) : 0; - - if (rate === 0) { - return -(fv + pv) / pmt; - } - - const pmtWithType = pmt * (1 + rate * type); - const nper = - Math.log((pmtWithType - fv * rate) / (pmtWithType + pv * rate)) / - Math.log(1 + rate); - - return nper; + return computeNper(rate, pmt, pv, fv, type); }; /** @@ -128,161 +77,63 @@ const nperHandler: FunctionHandler = (args, context) => { * Uses Newton-Raphson method for iteration */ const rateHandler: FunctionHandler = (args, context) => { - if (args.length < 3 || args.length > 6) { - throw new Error("RATE requires 3 to 6 arguments"); - } - - const nper = toNumber(context.evaluateFormula(args[0])); - const pmt = toNumber(context.evaluateFormula(args[1])); - const pv = toNumber(context.evaluateFormula(args[2])); - const fv = args.length >= 4 ? toNumber(context.evaluateFormula(args[3])) : 0; - const type = - args.length >= 5 ? toNumber(context.evaluateFormula(args[4])) : 0; - const guess = - args.length >= 6 ? toNumber(context.evaluateFormula(args[5])) : 0.1; - - // Use Newton-Raphson method to find rate - let rate = guess; - const maxIterations = 100; - const tolerance = 1e-7; - - for (let i = 0; i < maxIterations; i++) { - if (Math.abs(rate) < tolerance) { - rate = tolerance; // Avoid division by zero - } - - const y = Math.pow(1 + rate, nper); - const f = pv * y + pmt * ((y - 1) / rate) * (1 + rate * type) + fv; - - const df = - nper * pv * Math.pow(1 + rate, nper - 1) + - (pmt * - (1 + rate * type) * - (nper * Math.pow(1 + rate, nper - 1) * rate - - (Math.pow(1 + rate, nper) - 1))) / - (rate * rate) + - pmt * type * ((Math.pow(1 + rate, nper) - 1) / rate); + const nper = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + const pmt = toNumber(context.evaluateFormula(requiredArg(context, args, 1))); + const pv = toNumber(context.evaluateFormula(requiredArg(context, args, 2))); + const fv = args.length >= 4 ? toNumber(context.evaluateFormula(requiredArg(context, args, 3))) : 0; + const type = args.length >= 5 ? toNumber(context.evaluateFormula(requiredArg(context, args, 4))) : 0; + const guess = args.length >= 6 ? toNumber(context.evaluateFormula(requiredArg(context, args, 5))) : 0.1; + + return computeRate(nper, pmt, pv, fv, type, guess); +}; - const newRate = rate - f / df; +// IPMT and PPMT take the SAME six operands — (rate, per, nper, pv, [fv], [type]) +// — and differ only in which annuity component they compute. One factory parses +// the shared shape so the two handlers can't drift in their optional-arg defaults. +type PeriodicComponent = (rate: number, per: number, nper: number, pv: number, fv: number, type: number) => number; - if (Math.abs(newRate - rate) < tolerance) { - return newRate; - } +const makePeriodicComponentHandler = + (compute: PeriodicComponent): FunctionHandler => + (args, context) => { + const rate = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + const per = toNumber(context.evaluateFormula(requiredArg(context, args, 1))); + const nper = toNumber(context.evaluateFormula(requiredArg(context, args, 2))); + const pv = toNumber(context.evaluateFormula(requiredArg(context, args, 3))); + const fv = args.length >= 5 ? toNumber(context.evaluateFormula(requiredArg(context, args, 4))) : 0; + const type = args.length >= 6 ? toNumber(context.evaluateFormula(requiredArg(context, args, 5))) : 0; - rate = newRate; - } - - return rate; -}; + return compute(rate, per, nper, pv, fv, type); + }; /** * IPMT - Interest Payment * Calculates the interest payment for a given period for an investment based on periodic, constant payments and a constant interest rate. * IPMT(rate, per, nper, pv, [fv], [type]) */ -const ipmtHandler: FunctionHandler = (args, context) => { - if (args.length < 4 || args.length > 6) { - throw new Error("IPMT requires 4 to 6 arguments"); - } - - const rate = toNumber(context.evaluateFormula(args[0])); - const per = toNumber(context.evaluateFormula(args[1])); - const type = - args.length >= 6 ? toNumber(context.evaluateFormula(args[5])) : 0; - - // Calculate payment first - const pmt = pmtHandler( - [ - args[0], - args[2], - args[3], - ...(args.length >= 5 ? [args[4]] : []), - ...(args.length >= 6 ? [args[5]] : []), - ], - context, - ); - - if (per === 1 && type === 1) { - return 0; // No interest in first period when payment is at beginning - } - - // Calculate remaining balance at previous period - const fvPrevious = fvHandler( - [ - args[0], - String(type === 1 ? per - 2 : per - 1), - String(pmt), - args[3], - ...(args.length >= 6 ? [args[5]] : []), - ], - context, - ); - - const ipmt = -fvPrevious * rate; - - return type === 1 ? ipmt / (1 + rate) : ipmt; -}; +const ipmtHandler: FunctionHandler = makePeriodicComponentHandler(computeIpmt); /** * PPMT - Principal Payment * Calculates the payment on the principal for a given period for an investment based on periodic, constant payments and a constant interest rate. * PPMT(rate, per, nper, pv, [fv], [type]) */ -const ppmtHandler: FunctionHandler = (args, context) => { - if (args.length < 4 || args.length > 6) { - throw new Error("PPMT requires 4 to 6 arguments"); - } - - // Calculate total payment - const pmt = pmtHandler( - [ - args[0], - args[2], - args[3], - ...(args.length >= 5 ? [args[4]] : []), - ...(args.length >= 6 ? [args[5]] : []), - ], - context, - ); - - // Calculate interest payment - const ipmt = ipmtHandler(args, context); - - // Principal payment = Total payment - Interest payment - return toNumber(pmt) - toNumber(ipmt); -}; +const ppmtHandler: FunctionHandler = makePeriodicComponentHandler(computePpmt); /** * NPV - Net Present Value * Calculates the net present value of an investment based on a discount rate and a series of future cash flows. * NPV(rate, value1, [value2], ...) */ -const npvHandler: FunctionHandler = (args, context) => { - if (args.length < 2) { - throw new Error("NPV requires at least 2 arguments"); - } +// Flatten NPV's value arguments into one ordered list of numeric cash flows: +// ranges expand in place (numeric cells only), scalars contribute one value. +// The order defines each flow's discount period, so ranges and the scalars +// after them must stay in argument order. +const collectCashFlows = (valueArgs: string[], context: FunctionContext): number[] => + valueArgs.flatMap((arg) => (arg.includes(":") ? context.getRangeValues(arg) : [context.evaluateFormula(arg)]).map(toNumber)); - const rate = toNumber(context.evaluateFormula(args[0])); - let npv = 0; - - // Process each cash flow - for (let i = 1; i < args.length; i++) { - // Check if argument is a range - if (args[i].includes(":")) { - const values = context.getRangeValues(args[i]); - for (let j = 0; j < values.length; j++) { - const value = toNumber(values[j]); - // Period starts from i for first range element, then continues - const period = i + j; - npv += value / Math.pow(1 + rate, period); - } - } else { - const value = toNumber(context.evaluateFormula(args[i])); - npv += value / Math.pow(1 + rate, i); - } - } - - return npv; +const npvHandler: FunctionHandler = (args, context) => { + const rate = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + return computeNpv(rate, collectCashFlows(args.slice(1), context)); }; /** @@ -292,51 +143,14 @@ const npvHandler: FunctionHandler = (args, context) => { * Uses Newton-Raphson method for iteration */ const irrHandler: FunctionHandler = (args, context) => { - if (args.length < 1 || args.length > 2) { - throw new Error("IRR requires 1 or 2 arguments"); - } - - const values = context.getRangeValues(args[0]).map(toNumber); - const guess = - args.length === 2 ? toNumber(context.evaluateFormula(args[1])) : 0.1; + const values = context.getRangeValues(requiredArg(context, args, 0)).map(toNumber); + const guess = args.length === 2 ? toNumber(context.evaluateFormula(requiredArg(context, args, 1))) : 0.1; if (values.length === 0) { throw new Error("IRR requires at least one value"); } - // Use Newton-Raphson method - let rate = guess; - const maxIterations = 100; - const tolerance = 1e-7; - - for (let i = 0; i < maxIterations; i++) { - let npv = 0; - let dnpv = 0; - - for (let j = 0; j < values.length; j++) { - const factor = Math.pow(1 + rate, j); - npv += values[j] / factor; - dnpv -= (j * values[j]) / (factor * (1 + rate)); - } - - if (Math.abs(npv) < tolerance) { - return rate; - } - - if (Math.abs(dnpv) < tolerance) { - throw new Error("IRR cannot converge"); - } - - const newRate = rate - npv / dnpv; - - if (Math.abs(newRate - rate) < tolerance) { - return newRate; - } - - rate = newRate; - } - - return rate; + return computeIrr(values, guess); }; // Register all financial functions @@ -345,8 +159,7 @@ functionRegistry.register({ handler: fvHandler, minArgs: 3, maxArgs: 5, - description: - "Calculates the future value of an investment based on periodic, constant payments and a constant interest rate", + description: "Calculates the future value of an investment based on periodic, constant payments and a constant interest rate", examples: ["FV(0.06/12, 12, -100, -1000, 1)", "FV(A1, A2, A3)"], category: "Financial", }); @@ -356,8 +169,7 @@ functionRegistry.register({ handler: pvHandler, minArgs: 3, maxArgs: 5, - description: - "Calculates the present value of an investment based on periodic, constant payments and a constant interest rate", + description: "Calculates the present value of an investment based on periodic, constant payments and a constant interest rate", examples: ["PV(0.08/12, 12*20, 500, 0, 0)", "PV(A1, A2, A3)"], category: "Financial", }); @@ -367,8 +179,7 @@ functionRegistry.register({ handler: pmtHandler, minArgs: 3, maxArgs: 5, - description: - "Calculates the payment for a loan based on constant payments and a constant interest rate", + description: "Calculates the payment for a loan based on constant payments and a constant interest rate", examples: ["PMT(0.06/12, 30*12, 250000)", "PMT(A1, A2, A3)"], category: "Financial", }); @@ -378,8 +189,7 @@ functionRegistry.register({ handler: nperHandler, minArgs: 3, maxArgs: 5, - description: - "Calculates the number of periods for an investment based on periodic, constant payments and a constant interest rate", + description: "Calculates the number of periods for an investment based on periodic, constant payments and a constant interest rate", examples: ["NPER(0.06/12, -1000, 50000, 0, 0)", "NPER(A1, A2, A3)"], category: "Financial", }); @@ -389,8 +199,7 @@ functionRegistry.register({ handler: rateHandler, minArgs: 3, maxArgs: 6, - description: - "Calculates the interest rate per period of an annuity using iteration", + description: "Calculates the interest rate per period of an annuity using iteration", examples: ["RATE(12, -100, 1000, 0, 0, 0.1)", "RATE(A1, A2, A3)"], category: "Financial", }); @@ -400,8 +209,7 @@ functionRegistry.register({ handler: ipmtHandler, minArgs: 4, maxArgs: 6, - description: - "Calculates the interest payment for a given period for an investment", + description: "Calculates the interest payment for a given period for an investment", examples: ["IPMT(0.06/12, 1, 30*12, 250000)", "IPMT(A1, A2, A3, A4)"], category: "Financial", }); @@ -411,8 +219,7 @@ functionRegistry.register({ handler: ppmtHandler, minArgs: 4, maxArgs: 6, - description: - "Calculates the payment on the principal for a given period for an investment", + description: "Calculates the payment on the principal for a given period for an investment", examples: ["PPMT(0.06/12, 1, 30*12, 250000)", "PPMT(A1, A2, A3, A4)"], category: "Financial", }); @@ -421,8 +228,7 @@ functionRegistry.register({ name: "NPV", handler: npvHandler, minArgs: 2, - description: - "Calculates the net present value of an investment based on a discount rate and a series of future cash flows", + description: "Calculates the net present value of an investment based on a discount rate and a series of future cash flows", examples: ["NPV(0.1, -10000, 3000, 4200, 6800)", "NPV(A1, B1:B10)"], category: "Financial", }); @@ -432,8 +238,7 @@ functionRegistry.register({ handler: irrHandler, minArgs: 1, maxArgs: 2, - description: - "Calculates the internal rate of return for a series of cash flows", + description: "Calculates the internal rate of return for a series of cash flows", examples: ["IRR(A1:A5, 0.1)", "IRR(B1:B10)"], category: "Financial", }); diff --git a/src/engine/functions/logical.ts b/src/engine/functions/logical.ts index aa65377..515df31 100644 --- a/src/engine/functions/logical.ts +++ b/src/engine/functions/logical.ts @@ -2,32 +2,21 @@ * Logical Functions */ - - -import { functionRegistry, type FunctionHandler } from "../registry"; +import { evaluateConditionValues, readOperand, renderConditionOperand } from "../condition"; +import { findCellRefs, substituteCellRefs } from "../evaluator"; +import { functionRegistry, requiredArg, type FunctionHandler } from "../registry"; +import { isErrorResult, isSpreadsheetErrorValue, NA_ERROR } from "../spreadsheet-errors"; +import { coerceToBoolean } from "../coerce-boolean"; +import type { CellValue } from "../types"; const ifHandler: FunctionHandler = (args, context) => { - if (args.length !== 3) throw new Error("IF requires 3 arguments"); - - const condition = args[0]; - const trueValue = args[1]; - const falseValue = args[2]; + const condition = requiredArg(context, args, 0); + const trueValue = requiredArg(context, args, 1); + const falseValue = requiredArg(context, args, 2); // Evaluate condition - use evaluateFormula to handle nested functions like MONTH() const conditionValue = context.evaluateFormula(condition); - - // Convert to boolean - let conditionResult = false; - if (typeof conditionValue === "boolean") { - conditionResult = conditionValue; - } else if (typeof conditionValue === "number") { - conditionResult = conditionValue !== 0; - } else if (typeof conditionValue === "string") { - conditionResult = - conditionValue.toLowerCase() === "true" || conditionValue !== ""; - } else { - conditionResult = !!conditionValue; - } + const conditionResult = coerceToBoolean(conditionValue); // Return the appropriate value based on condition const resultValue = conditionResult ? trueValue : falseValue; @@ -37,40 +26,17 @@ const ifHandler: FunctionHandler = (args, context) => { return resultValue.slice(1, -1); } - // If result is a nested formula, evaluate it recursively - if (/^(SUM|AVERAGE|MAX|MIN|COUNT|IF|AND|OR|NOT)\(/i.test(resultValue)) { - return context.evaluateFormula(resultValue); - } - - // Otherwise evaluate as expression - let expr = resultValue; - - const refs = resultValue.match( - /(?:'[^']+'|[^'!\s]+)![A-Z]+\d+|\$?[A-Z]+\$?\d+/g, - ); - if (refs) { - for (const ref of refs) { - const value = context.getCellValue(ref); - const escapedRef = ref.replace(/\$/g, "\\$").replace(/'/g, "\\'"); - expr = expr.replace( - new RegExp(escapedRef.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"), "g"), - String(value), - ); - } - } - - const numResult = parseFloat(expr); - return isNaN(numResult) ? expr : numResult; + // Everything else — a nested call, an arithmetic expression, a reference — is + // evaluated by the engine. Hand-rolling it here silently returned a plausible + // wrong value twice over: a hard-coded list of nine function names sent + // `ROUND(A1,1)` back as its own text, and the fallback read `A1+1` through + // `parseFloat("3+1")`, yielding 3. + return context.evaluateFormula(resultValue); }; const andHandler: FunctionHandler = (args, context) => { - if (args.length === 0) throw new Error("AND requires at least 1 argument"); - for (const arg of args) { - const value = context.evaluateFormula(arg.trim()); - // Check if value is falsy (0, false, empty string, etc.) - // Note: !value already covers false, so we check for 0 and "0" explicitly - if (!value || value === 0 || value === "0") { + if (!coerceToBoolean(context.evaluateFormula(arg.trim()))) { return false; } } @@ -78,12 +44,8 @@ const andHandler: FunctionHandler = (args, context) => { }; const orHandler: FunctionHandler = (args, context) => { - if (args.length === 0) throw new Error("OR requires at least 1 argument"); - for (const arg of args) { - const value = context.evaluateFormula(arg.trim()); - // Check if value is truthy (non-zero, non-empty) - if (value && value !== 0 && value !== "0") { + if (coerceToBoolean(context.evaluateFormula(arg.trim()))) { return true; } } @@ -91,83 +53,67 @@ const orHandler: FunctionHandler = (args, context) => { }; const notHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("NOT requires 1 argument"); - - const value = context.evaluateFormula(args[0]); - // Note: !value already covers false - return !value || value === 0 || value === "0"; + return !coerceToBoolean(context.evaluateFormula(requiredArg(context, args, 0))); }; const iferrorHandler: FunctionHandler = (args, context) => { - if (args.length !== 2) throw new Error("IFERROR requires 2 arguments"); - try { - const result = context.evaluateFormula(args[0]); - // Check if result is an error (NaN, Infinity, etc.) - if ( - result === null || - result === undefined || - (typeof result === "number" && (isNaN(result) || !isFinite(result))) - ) { - return context.evaluateFormula(args[1]); + const result = context.evaluateFormula(requiredArg(context, args, 0)); + // Catches NaN/∞ and the formula error VALUES functions return (a math domain + // miss like SQRT(-1) → #NUM!), so IFERROR(SQRT(-1), 0) is 0. Text that + // merely spells an error is not an error value, so it passes through + // whether it was written as a literal (IFERROR("#NUM!", 42)) or computed + // (IFERROR(CONCAT("#N","UM!"), 42)) — the computed case is why errors carry + // provenance at all (#2451). + if (isErrorResult(result)) { + return context.evaluateFormula(requiredArg(context, args, 1)); } return result; } catch { // If evaluation throws an error, return the fallback value - return context.evaluateFormula(args[1]); + return context.evaluateFormula(requiredArg(context, args, 1)); } }; const ifnaHandler: FunctionHandler = (args, context) => { - if (args.length !== 2) throw new Error("IFNA requires 2 arguments"); - - const result = context.evaluateFormula(args[0]); - // Check if result is N/A (could be represented as specific error value) - if (result === null || result === undefined || result === "#N/A") { - return context.evaluateFormula(args[1]); + const result = context.evaluateFormula(requiredArg(context, args, 0)); + const isNotAvailable = isSpreadsheetErrorValue(result) && result.code === NA_ERROR.code; + if (result === null || result === undefined || isNotAvailable) { + return context.evaluateFormula(requiredArg(context, args, 1)); } return result; }; const ifsHandler: FunctionHandler = (args, context) => { if (args.length < 2 || args.length % 2 !== 0) { - throw new Error( - "IFS requires an even number of arguments (condition-value pairs)", - ); + throw new Error("IFS requires an even number of arguments (condition-value pairs)"); } // Iterate through condition-value pairs for (let i = 0; i < args.length; i += 2) { - const condition = args[i]; - const value = args[i + 1]; - - // Evaluate condition - let condExpr = condition; - - const cellRefs = condition.match( - /(?:'[^']+'|[^'!\s]+)![A-Z]+\d+|\$?[A-Z]+\$?\d+/g, - ); - if (cellRefs) { - for (const ref of cellRefs) { - const cellValue = context.getCellValue(ref); - const escapedRef = ref.replace(/\$/g, "\\$").replace(/'/g, "\\'"); - condExpr = condExpr.replace( - new RegExp(escapedRef.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"), "g"), - String(cellValue), - ); - } - } - - // Evaluate the condition - let conditionResult = false; - - if (/>=|<=|>|<|==|!=/.test(condExpr)) { - conditionResult = eval(condExpr); - } else { - conditionResult = !!eval(condExpr); - } - - if (conditionResult) { + const condition = requiredArg(context, args, i); + const value = requiredArg(context, args, i + 1); + + // Substitute references by POSITION (back to front), skipping any that sit + // inside a quoted string literal: `IFS(A1="B2", …)` must compare A1 to the + // TEXT "B2", not to cell B2's value (Codex review). findCellRefs already + // skips literals and matches absolute / sheet-qualified refs, so this also + // avoids the earlier regex double-escaping. renderConditionOperand quotes a + // text cell so its own operators are not re-parsed as comparisons. + const condExpr = substituteCellRefs(condition, findCellRefs(condition), (ref) => renderConditionOperand(context.getCellValue(ref))); + + // Parsed, not executed. This used to call `eval` on `condExpr`, which is + // the substituted text — so a cell containing `globalThis.x = 1` ran as + // code whenever an IFS referenced it, and so did anything written into the + // formula itself. `readOperand` resolves the simple operands (TRUE/FALSE -> + // boolean, quoted text, numbers); an arithmetic expression it leaves as raw + // text is handed to the engine's safe evaluator so `A1+1>10` is computed. + // Only the top-level comparison is applied — the condition is never run. + const evaluateOperand = (operand: string): CellValue => { + const parsed = readOperand(operand); + return typeof parsed === "string" && parsed === operand.trim() ? context.evaluateFormula(operand) : parsed; + }; + if (evaluateConditionValues(condExpr, evaluateOperand)) { // If result is a quoted string, return without quotes if (/^["'](.*)["']$/.test(value)) { @@ -179,18 +125,12 @@ const ifsHandler: FunctionHandler = (args, context) => { } // If no conditions match, return error - return "#N/A"; + return NA_ERROR; }; -const trueHandler: FunctionHandler = (args) => { - if (args.length !== 0) throw new Error("TRUE requires 0 arguments"); - return true; -}; +const trueHandler: FunctionHandler = () => true; -const falseHandler: FunctionHandler = (args) => { - if (args.length !== 0) throw new Error("FALSE requires 0 arguments"); - return false; -}; +const falseHandler: FunctionHandler = () => false; // Register all logical functions functionRegistry.register({ @@ -236,12 +176,8 @@ functionRegistry.register({ handler: iferrorHandler, minArgs: 2, maxArgs: 2, - description: - "Returns a value if expression is an error, otherwise returns the expression", - examples: [ - "IFERROR(A1/B1, 0)", - 'IFERROR(VLOOKUP(A1, B1:C10, 2), "Not found")', - ], + description: "Returns a value if expression is an error, otherwise returns the expression", + examples: ["IFERROR(A1/B1, 0)", 'IFERROR(VLOOKUP(A1, B1:C10, 2), "Not found")'], category: "Logical", }); @@ -250,8 +186,7 @@ functionRegistry.register({ handler: ifnaHandler, minArgs: 2, maxArgs: 2, - description: - "Returns a value if expression is #N/A, otherwise returns the expression", + description: "Returns a value if expression is #N/A, otherwise returns the expression", examples: ['IFNA(A1, "N/A")', "IFNA(MATCH(A1, B1:B10), 0)"], category: "Logical", }); @@ -261,10 +196,7 @@ functionRegistry.register({ handler: ifsHandler, minArgs: 2, description: "Checks multiple conditions and returns the first true result", - examples: [ - 'IFS(A1>90, "A", A1>80, "B", A1>70, "C")', - 'IFS(B1="Yes", 1, B1="No", 0)', - ], + examples: ['IFS(A1>90, "A", A1>80, "B", A1>70, "C")', 'IFS(B1="Yes", 1, B1="No", 0)'], category: "Logical", }); diff --git a/src/engine/functions/lookup-math.ts b/src/engine/functions/lookup-math.ts new file mode 100644 index 0000000..aa12348 --- /dev/null +++ b/src/engine/functions/lookup-math.ts @@ -0,0 +1,28 @@ +/** + * Pure lookup rules, separated from the range-reading handlers so they can be + * unit-tested directly. + */ + +import type { CellValue } from "../types"; +import { isSpreadsheetErrorValue } from "../spreadsheet-errors"; + +/** + * Interpret VLOOKUP/HLOOKUP's 4th argument (range_lookup): TRUE (or omitted) = + * approximate match, FALSE = exact. + * + * The literal `TRUE` reaches the handler as the STRING `"TRUE"` — the evaluator + * leaves bare words unquoted — so an accept-only-`true | 1 | "1"` check silently + * fell back to exact match and returned `#N/A` for a valid approximate lookup + * (#2360). Read it the way Excel coerces a logical instead: a boolean as-is, a + * non-zero number as TRUE, and the words `TRUE` / `FALSE` (case-insensitive) + * and `"1"` / `"0"` as their logical value. Anything else (blank, stray text) + * is treated as FALSE, matching Excel's coercion of a blank range_lookup — and + * so is a formula error, which is neither of the two logicals. + */ +export function isApproximateMatch(rangeLookup: CellValue): boolean { + if (typeof rangeLookup === "boolean") return rangeLookup; + if (typeof rangeLookup === "number") return rangeLookup !== 0; + if (isSpreadsheetErrorValue(rangeLookup)) return false; + const normalized = rangeLookup.trim().toUpperCase(); + return normalized === "TRUE" || normalized === "1"; +} diff --git a/src/engine/functions/lookup.ts b/src/engine/functions/lookup.ts index fd3f4ec..de96924 100644 --- a/src/engine/functions/lookup.ts +++ b/src/engine/functions/lookup.ts @@ -2,33 +2,35 @@ * Lookup and Reference Functions */ -import { - functionRegistry, - toNumber, - parseCriteria, - type FunctionHandler, -} from "../registry"; +import { functionRegistry, rawRangeReader, requiredArg, toNumber, parseCriteria, type FunctionHandler, type FunctionContext } from "../registry"; +import { indexToColumn } from "../parser"; +import { parseRangeBounds, resolveIndexTarget, resolveTableOffset } from "../formulaRefs"; +import { isApproximateMatch } from "./lookup-math"; import type { CellValue } from "../types"; - -// Helper to convert Excel column letters to 0-based index (A=0, Z=25, AA=26, etc.) -const colToIndex = (col: string): number => { - let result = 0; - for (let i = 0; i < col.length; i++) { - result = result * 26 + (col.charCodeAt(i) - 64); - } - return result - 1; -}; - -// Helper to convert 0-based index to Excel column letters (0=A, 25=Z, 26=AA, etc.) -const indexToCol = (index: number): string => { - let col = ""; - let num = index + 1; - while (num > 0) { - const remainder = (num - 1) % 26; - col = String.fromCharCode(65 + remainder) + col; - num = Math.floor((num - 1) / 26); - } - return col; +import { NA_ERROR, REF_ERROR } from "../spreadsheet-errors"; + +const inclusiveRange = (start: number, end: number): number[] => Array.from({ length: Math.max(0, end - start + 1) }, (_, i) => start + i); + +// Read a vertical slice (one column, `startRow`..`endRow`) as VLOOKUP's lookup column. +const columnValues = (context: FunctionContext, sheetPrefix: string, colStr: string, startRow: number, endRow: number): CellValue[] => + inclusiveRange(startRow, endRow).map((r) => context.getCellValue(`${sheetPrefix}${colStr}${r}`)); + +// Read a horizontal slice (one row, `startCol`..`endCol`) as HLOOKUP's lookup row. +const rowValues = (context: FunctionContext, sheetPrefix: string, row: number, startCol: number, endCol: number): CellValue[] => + inclusiveRange(startCol, endCol).map((c) => context.getCellValue(`${sheetPrefix}${indexToColumn(c)}${row}`)); + +/** The index of the LAST value that satisfies `matches`, or -1. Array.findIndex + * only walks forward, and XLOOKUP's `searchMode: -1` searches from the end. */ +const findLastMatchIndex = (values: CellValue[], matches: (value: CellValue) => boolean): number => + values.reduce((found, value, index) => (matches(value) ? index : found), -1); + +/** The index of the last value in a SORTED list before `keepGoing` stops + * holding, or -1 when the very first value already fails. The list is assumed + * sorted (as Excel's approximate match requires), so the first failure ends the + * run. */ +const lastIndexWhile = (values: CellValue[], keepGoing: (value: CellValue) => boolean): number => { + const stop = values.findIndex((value) => !keepGoing(value)); + return (stop === -1 ? values.length : stop) - 1; }; // Helper to find match index @@ -45,271 +47,110 @@ const findMatchIndex = ( // Exact match if (matchType === 0) { - // Handle wildcards for strings if it's an exact match request - if ( - typeof lookupValue === "string" && - (lookupValue.includes("*") || lookupValue.includes("?")) - ) { - const criteriaFn = parseCriteria(lookupValue); - - if (searchMode === 1) { - return lookupArray.findIndex((item) => criteriaFn(item)); - } else { - for (let i = lookupArray.length - 1; i >= 0; i--) { - if (criteriaFn(lookupArray[i])) return i; - } - return -1; - } - } - - if (searchMode === 1) { - return lookupArray.findIndex((item) => item == lookupValue); // Loose equality for "10" == 10 - } else { - for (let i = lookupArray.length - 1; i >= 0; i--) { - if (lookupArray[i] == lookupValue) return i; - } - return -1; - } + // Handle wildcards for strings if it's an exact match request. + // Loose equality otherwise, for "10" == 10. + const usesWildcards = typeof lookupValue === "string" && (lookupValue.includes("*") || lookupValue.includes("?")); + const isMatch = usesWildcards ? parseCriteria(lookupValue) : (item: CellValue) => item == lookupValue; + return searchMode === 1 ? lookupArray.findIndex(isMatch) : findLastMatchIndex(lookupArray, isMatch); } // Approximate match (requires sorted array) // We'll assume the user knows what they are doing regarding sorting, as per Excel behavior - if (matchType === 1) { - // Less than or equal to - // Array must be sorted ascending - let bestIdx = -1; - for (let i = 0; i < lookupArray.length; i++) { - const item = lookupArray[i]; - if (compare(item, lookupValue) <= 0) { - bestIdx = i; - } else { - // Since it's sorted ascending, once we exceed, we can stop - break; - } - } - return bestIdx; - } + // Less than or equal to, over an array sorted ascending: once we exceed, stop. + if (matchType === 1) return lastIndexWhile(lookupArray, (item) => compare(item, lookupValue) <= 0); - if (matchType === -1) { - // Greater than or equal to - // Array must be sorted descending - let bestIdx = -1; - for (let i = 0; i < lookupArray.length; i++) { - const item = lookupArray[i]; - if (compare(item, lookupValue) >= 0) { - bestIdx = i; - } else { - break; - } - } - return bestIdx; - } + // Greater than or equal to, over an array sorted descending. + if (matchType === -1) return lastIndexWhile(lookupArray, (item) => compare(item, lookupValue) >= 0); return -1; }; const vlookupHandler: FunctionHandler = (args, context) => { - if (args.length < 3 || args.length > 4) { - throw new Error("VLOOKUP requires 3 or 4 arguments"); - } - - const lookupValue = context.evaluateFormula(args[0]); - const tableArrayRange = args[1]; - const colIndexNum = toNumber(context.evaluateFormula(args[2])); - const rangeLookup = - args.length === 4 ? context.evaluateFormula(args[3]) : true; - - // Convert rangeLookup to boolean/number logic - // TRUE/1/omitted = approximate match (default) - // FALSE/0 = exact match - const isApprox = - rangeLookup === true || rangeLookup === 1 || rangeLookup === "1"; - const matchType = isApprox ? 1 : 0; - - // Get the full table data - // We need to parse the range string to get dimensions - const match = tableArrayRange.match(/^([A-Z]+)(\d+):([A-Z]+)(\d+)$/); - if (!match) throw new Error("Invalid table array range"); - - // We need to get the first column for looking up - // And the specific column for the result - // This is a bit tricky with the current getRangeValues which flattens everything - // We need to manually reconstruct the table structure or request specific cells - - // Let's parse the range to get start/end col/row - // Note: This relies on the context.getCellValue implementation details or we need to implement - // a smarter way to get 2D data. - // For now, we will iterate row by row. - - // Parse range manually to get boundaries - // We can't easily use context.getRangeValues because it flattens 2D arrays to 1D - // So we will iterate through the rows of the first column - - // Extract sheet name if present - let sheetName = ""; - let rangePart = tableArrayRange; - if (tableArrayRange.includes("!")) { - const parts = tableArrayRange.split("!"); - sheetName = parts[0] + "!"; - rangePart = parts[1]; - } - - const rangeMatch = rangePart.match(/^([A-Z]+)(\d+):([A-Z]+)(\d+)$/); - if (!rangeMatch) throw new Error("Invalid range format"); - - const startColStr = rangeMatch[1]; - const startRow = parseInt(rangeMatch[2]); - const endRow = parseInt(rangeMatch[4]); - - const startColIdx = colToIndex(startColStr); - const resultColIdx = startColIdx + colIndexNum - 1; - const resultColStr = indexToCol(resultColIdx); - - // Build lookup array (first column) - const lookupArray: CellValue[] = []; - for (let r = startRow; r <= endRow; r++) { - const cellRef = `${sheetName}${startColStr}${r}`; - lookupArray.push(context.getCellValue(cellRef)); - } - + const lookupValue = context.evaluateFormula(requiredArg(context, args, 0)); + const bounds = parseRangeBounds(requiredArg(context, args, 1)); + if (!bounds) throw new Error("Invalid table array range"); + const colIndexNum = toNumber(context.evaluateFormula(requiredArg(context, args, 2))); + const rangeLookup = args.length === 4 ? context.evaluateFormula(requiredArg(context, args, 3)) : true; + const matchType = isApproximateMatch(rangeLookup) ? 1 : 0; + + // Excel rejects a col_index_num past the table's width; reading on regardless + // addressed a cell outside the range and usually returned a silent 0 (#2360). + // Checked before the search: an out-of-range index is an argument error, so it + // wins over the #N/A a missing key would otherwise mask it with (Codex review). + const colOffset = resolveTableOffset(colIndexNum, bounds.endCol - bounds.startCol + 1); + if (colOffset === null) return REF_ERROR; + + const startColStr = indexToColumn(bounds.startCol); + const lookupArray = columnValues(context, bounds.sheetPrefix, startColStr, bounds.startRow, bounds.endRow); const matchIdx = findMatchIndex(lookupValue, lookupArray, matchType); + if (matchIdx === -1) return NA_ERROR; - if (matchIdx === -1) return "#N/A"; - - // Get result - const resultRow = startRow + matchIdx; - const resultRef = `${sheetName}${resultColStr}${resultRow}`; - - return context.getCellValue(resultRef); + const resultColStr = indexToColumn(bounds.startCol + colOffset); + const resultRow = bounds.startRow + matchIdx; + return context.getCellValue(`${bounds.sheetPrefix}${resultColStr}${resultRow}`); }; const hlookupHandler: FunctionHandler = (args, context) => { - if (args.length < 3 || args.length > 4) { - throw new Error("HLOOKUP requires 3 or 4 arguments"); - } - - const lookupValue = context.evaluateFormula(args[0]); - const tableArrayRange = args[1]; - const rowIndexNum = toNumber(context.evaluateFormula(args[2])); - const rangeLookup = - args.length === 4 ? context.evaluateFormula(args[3]) : true; - - const isApprox = - rangeLookup === true || rangeLookup === 1 || rangeLookup === "1"; - const matchType = isApprox ? 1 : 0; - - // Extract sheet name if present - let sheetName = ""; - let rangePart = tableArrayRange; - if (tableArrayRange.includes("!")) { - const parts = tableArrayRange.split("!"); - sheetName = parts[0] + "!"; - rangePart = parts[1]; - } + const lookupValue = context.evaluateFormula(requiredArg(context, args, 0)); + const bounds = parseRangeBounds(requiredArg(context, args, 1)); + if (!bounds) throw new Error("Invalid range format"); + const rowIndexNum = toNumber(context.evaluateFormula(requiredArg(context, args, 2))); + const rangeLookup = args.length === 4 ? context.evaluateFormula(requiredArg(context, args, 3)) : true; + const matchType = isApproximateMatch(rangeLookup) ? 1 : 0; - const rangeMatch = rangePart.match(/^([A-Z]+)(\d+):([A-Z]+)(\d+)$/); - if (!rangeMatch) throw new Error("Invalid range format"); - - const startColStr = rangeMatch[1]; - const endColStr = rangeMatch[3]; - const startRow = parseInt(rangeMatch[2]); - - const startColIdx = colToIndex(startColStr); - const endColIdx = colToIndex(endColStr); - - // Build lookup array (first row) - const lookupArray: CellValue[] = []; - for (let c = startColIdx; c <= endColIdx; c++) { - const colStr = indexToCol(c); - const cellRef = `${sheetName}${colStr}${startRow}`; - lookupArray.push(context.getCellValue(cellRef)); - } + const rowOffset = resolveTableOffset(rowIndexNum, bounds.endRow - bounds.startRow + 1); + if (rowOffset === null) return REF_ERROR; + const lookupArray = rowValues(context, bounds.sheetPrefix, bounds.startRow, bounds.startCol, bounds.endCol); const matchIdx = findMatchIndex(lookupValue, lookupArray, matchType); + if (matchIdx === -1) return NA_ERROR; - if (matchIdx === -1) return "#N/A"; - - // Get result - const resultColIdx = startColIdx + matchIdx; - const resultColStr = indexToCol(resultColIdx); - const resultRow = startRow + rowIndexNum - 1; - const resultRef = `${sheetName}${resultColStr}${resultRow}`; - - return context.getCellValue(resultRef); + const resultColStr = indexToColumn(bounds.startCol + matchIdx); + const resultRow = bounds.startRow + rowOffset; + return context.getCellValue(`${bounds.sheetPrefix}${resultColStr}${resultRow}`); }; const matchHandler: FunctionHandler = (args, context) => { - if (args.length < 2 || args.length > 3) { - throw new Error("MATCH requires 2 or 3 arguments"); - } + const lookupValue = context.evaluateFormula(requiredArg(context, args, 0)); + const lookupArrayRange = requiredArg(context, args, 1); + const matchType = args.length === 3 ? toNumber(context.evaluateFormula(requiredArg(context, args, 2))) : 1; - const lookupValue = context.evaluateFormula(args[0]); - const lookupArrayRange = args[1]; - const matchType = - args.length === 3 ? toNumber(context.evaluateFormula(args[2])) : 1; - - const lookupArray = context.getRangeValues(lookupArrayRange); + // Raw, not numeric-only: MATCH answers with a POSITION, so a dropped text + // cell renumbers everything after it and returns a different row (#2765). + const lookupArray = rawRangeReader(context)(lookupArrayRange); const index = findMatchIndex(lookupValue, lookupArray, matchType); - return index === -1 ? "#N/A" : index + 1; // 1-based index + return index === -1 ? NA_ERROR : index + 1; // 1-based index }; const indexHandler: FunctionHandler = (args, context) => { - if (args.length < 2 || args.length > 4) { - throw new Error("INDEX requires 2 to 4 arguments"); - } - - const arrayRange = args[0]; - const rowNum = toNumber(context.evaluateFormula(args[1])); - const colNum = - args.length >= 3 ? toNumber(context.evaluateFormula(args[2])) : 1; // Default to 1 if omitted (for 1D arrays) - - // Parse range to find the specific cell - let sheetName = ""; - let rangePart = arrayRange; - if (arrayRange.includes("!")) { - const parts = arrayRange.split("!"); - sheetName = parts[0] + "!"; - rangePart = parts[1]; - } - - const rangeMatch = rangePart.match(/^([A-Z]+)(\d+):([A-Z]+)(\d+)$/); - if (!rangeMatch) throw new Error("Invalid range format"); - - const startColStr = rangeMatch[1]; - const startRow = parseInt(rangeMatch[2]); - - const startColIdx = colToIndex(startColStr); - - // Calculate target cell - // rowNum and colNum are 1-based relative to the range - const targetRow = startRow + rowNum - 1; - const targetColIdx = startColIdx + colNum - 1; - const targetColStr = indexToCol(targetColIdx); - - const ref = `${sheetName}${targetColStr}${targetRow}`; - return context.getCellValue(ref); + const bounds = parseRangeBounds(requiredArg(context, args, 0)); + if (!bounds) throw new Error("Invalid range format"); + const rowNum = toNumber(context.evaluateFormula(requiredArg(context, args, 1))); + const colNum = args.length >= 3 ? toNumber(context.evaluateFormula(requiredArg(context, args, 2))) : 1; // Default to 1 if omitted (for 1D arrays) + + const target = resolveIndexTarget(bounds, rowNum, colNum); + if (!target) return REF_ERROR; + return context.getCellValue(`${bounds.sheetPrefix}${indexToColumn(target.colIndex)}${target.row}`); }; const xlookupHandler: FunctionHandler = (args, context) => { - if (args.length < 3 || args.length > 6) { - throw new Error("XLOOKUP requires 3 to 6 arguments"); - } - - const lookupValue = context.evaluateFormula(args[0]); - const lookupArrayRange = args[1]; - const returnArrayRange = args[2]; - const ifNotFound = - args.length >= 4 ? context.evaluateFormula(args[3]) : "#N/A"; - const matchMode = - args.length >= 5 ? toNumber(context.evaluateFormula(args[4])) : 0; - const searchMode = - args.length >= 6 ? toNumber(context.evaluateFormula(args[5])) : 1; - - const lookupArray = context.getRangeValues(lookupArrayRange); - const returnArray = context.getRangeValues(returnArrayRange); + const lookupValue = context.evaluateFormula(requiredArg(context, args, 0)); + const lookupArrayRange = requiredArg(context, args, 1); + const returnArrayRange = requiredArg(context, args, 2); + const ifNotFound = args.length >= 4 ? context.evaluateFormula(requiredArg(context, args, 3)) : NA_ERROR; + const matchMode = args.length >= 5 ? toNumber(context.evaluateFormula(requiredArg(context, args, 4))) : 0; + const searchMode = args.length >= 6 ? toNumber(context.evaluateFormula(requiredArg(context, args, 5))) : 1; + + // Both raw, and for a second reason beyond MATCH's: the numeric-only reader + // filters these two INDEPENDENTLY, so a text cell in one range shifts it + // against the other and the match index reads a different row's value — a + // wrong number with no error, which is the worst outcome a formula can have. + const readRange = rawRangeReader(context); + const lookupArray = readRange(lookupArrayRange); + const returnArray = readRange(returnArrayRange); // XLOOKUP match modes: // 0 = Exact match (default) @@ -358,11 +199,10 @@ const xlookupHandler: FunctionHandler = (args, context) => { if (matchIdx === -1) return ifNotFound; - if (matchIdx >= 0 && matchIdx < returnArray.length) { - return returnArray[matchIdx]; - } - - return "#N/A"; + // A match past the end of the return range has nothing to return — the two + // ranges are independent arguments and need not be the same length. + const found = returnArray[matchIdx]; + return found === undefined ? NA_ERROR : found; }; // Register functions @@ -371,8 +211,7 @@ functionRegistry.register({ handler: vlookupHandler, minArgs: 3, maxArgs: 4, - description: - "Looks for a value in the leftmost column of a table, and then returns a value in the same row from a column you specify", + description: "Looks for a value in the leftmost column of a table, and then returns a value in the same row from a column you specify", examples: ["VLOOKUP(105, A2:C10, 2)", 'VLOOKUP("Smith", A2:E10, 5, FALSE)'], category: "Lookup & Reference", }); @@ -382,8 +221,7 @@ functionRegistry.register({ handler: hlookupHandler, minArgs: 3, maxArgs: 4, - description: - "Looks for a value in the top row of a table, and then returns a value in the same column from a row you specify", + description: "Looks for a value in the top row of a table, and then returns a value in the same column from a row you specify", examples: ['HLOOKUP("Axles", A1:C10, 2, TRUE)'], category: "Lookup & Reference", }); @@ -393,8 +231,7 @@ functionRegistry.register({ handler: matchHandler, minArgs: 2, maxArgs: 3, - description: - "Returns the relative position of an item in an array that matches a specified value", + description: "Returns the relative position of an item in an array that matches a specified value", examples: ["MATCH(25, A1:A10, 0)", 'MATCH("b", A1:A5, 0)'], category: "Lookup & Reference", }); @@ -404,8 +241,7 @@ functionRegistry.register({ handler: indexHandler, minArgs: 2, maxArgs: 4, - description: - "Returns the value of an element in a table or an array, selected by the row and column number indexes", + description: "Returns the value of an element in a table or an array, selected by the row and column number indexes", examples: ["INDEX(A1:B5, 2, 2)", "INDEX(A1:A10, 5)"], category: "Lookup & Reference", }); @@ -415,11 +251,7 @@ functionRegistry.register({ handler: xlookupHandler, minArgs: 3, maxArgs: 6, - description: - "Searches a range or an array, and returns an item corresponding to the first match it finds", - examples: [ - "XLOOKUP(A1, B1:B10, C1:C10)", - 'XLOOKUP("USA", Countries, Populations)', - ], + description: "Searches a range or an array, and returns an item corresponding to the first match it finds", + examples: ["XLOOKUP(A1, B1:B10, C1:C10)", 'XLOOKUP("USA", Countries, Populations)'], category: "Lookup & Reference", }); diff --git a/src/engine/functions/mathematical.ts b/src/engine/functions/mathematical.ts index 278b439..f444f3a 100644 --- a/src/engine/functions/mathematical.ts +++ b/src/engine/functions/mathematical.ts @@ -2,129 +2,113 @@ * Mathematical Functions */ -import { functionRegistry, toNumber, type FunctionHandler } from "../registry"; +import { functionRegistry, requiredArg, toNumber, type FunctionHandler } from "../registry"; +import { + roundTo, + roundUpTo, + roundDownTo, + floorToSignificance, + ceilingToSignificance, + modulo, + power, + safeLog, + safeLog10, + safeSqrt, + logWithBase, +} from "../math-ops"; +import { toScalarNumber } from "../numericCoercion"; +import { isSpreadsheetErrorValue } from "../spreadsheet-errors"; const roundHandler: FunctionHandler = (args, context) => { - if (args.length !== 2) throw new Error("ROUND requires 2 arguments"); - const number = toNumber(context.evaluateFormula(args[0])); - const digits = toNumber(context.evaluateFormula(args[1])); - const multiplier = Math.pow(10, digits); - return Math.round(number * multiplier) / multiplier; + const number = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + const digits = toNumber(context.evaluateFormula(requiredArg(context, args, 1))); + return roundTo(number, digits); }; const roundupHandler: FunctionHandler = (args, context) => { - if (args.length !== 2) throw new Error("ROUNDUP requires 2 arguments"); - const number = toNumber(context.evaluateFormula(args[0])); - const digits = toNumber(context.evaluateFormula(args[1])); - const multiplier = Math.pow(10, digits); - return Math.ceil(number * multiplier) / multiplier; + const number = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + const digits = toNumber(context.evaluateFormula(requiredArg(context, args, 1))); + return roundUpTo(number, digits); }; const rounddownHandler: FunctionHandler = (args, context) => { - if (args.length !== 2) throw new Error("ROUNDDOWN requires 2 arguments"); - const number = toNumber(context.evaluateFormula(args[0])); - const digits = toNumber(context.evaluateFormula(args[1])); - const multiplier = Math.pow(10, digits); - return Math.floor(number * multiplier) / multiplier; + const number = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + const digits = toNumber(context.evaluateFormula(requiredArg(context, args, 1))); + return roundDownTo(number, digits); }; const floorHandler: FunctionHandler = (args, context) => { - if (args.length !== 2) throw new Error("FLOOR requires 2 arguments"); - const number = toNumber(context.evaluateFormula(args[0])); - const significance = toNumber(context.evaluateFormula(args[1])); - if (significance === 0) return 0; - return Math.floor(number / significance) * significance; + const number = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + const significance = toNumber(context.evaluateFormula(requiredArg(context, args, 1))); + return floorToSignificance(number, significance); }; const ceilingHandler: FunctionHandler = (args, context) => { - if (args.length !== 2) throw new Error("CEILING requires 2 arguments"); - const number = toNumber(context.evaluateFormula(args[0])); - const significance = toNumber(context.evaluateFormula(args[1])); - if (significance === 0) return 0; - return Math.ceil(number / significance) * significance; + const number = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + const significance = toNumber(context.evaluateFormula(requiredArg(context, args, 1))); + return ceilingToSignificance(number, significance); }; const absHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("ABS requires 1 argument"); - const number = toNumber(context.evaluateFormula(args[0])); - return Math.abs(number); + const number = toScalarNumber(context.evaluateFormula(requiredArg(context, args, 0))); + return isSpreadsheetErrorValue(number) ? number : Math.abs(number); }; const powerHandler: FunctionHandler = (args, context) => { - if (args.length !== 2) throw new Error("POWER requires 2 arguments"); - const base = toNumber(context.evaluateFormula(args[0])); - const exponent = toNumber(context.evaluateFormula(args[1])); - return Math.pow(base, exponent); + const base = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + const exponent = toNumber(context.evaluateFormula(requiredArg(context, args, 1))); + return power(base, exponent); }; const sqrtHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("SQRT requires 1 argument"); - const number = toNumber(context.evaluateFormula(args[0])); - return Math.sqrt(number); + const number = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + return safeSqrt(number); }; const modHandler: FunctionHandler = (args, context) => { - if (args.length !== 2) throw new Error("MOD requires 2 arguments"); - const number = toNumber(context.evaluateFormula(args[0])); - const divisor = toNumber(context.evaluateFormula(args[1])); - if (divisor === 0) return 0; - return number % divisor; + const number = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + const divisor = toNumber(context.evaluateFormula(requiredArg(context, args, 1))); + return modulo(number, divisor); }; const intHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("INT requires 1 argument"); - const number = toNumber(context.evaluateFormula(args[0])); + const number = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); return Math.floor(number); }; const truncHandler: FunctionHandler = (args, context) => { - if (args.length < 1 || args.length > 2) { - throw new Error("TRUNC requires 1 or 2 arguments"); - } - const number = toNumber(context.evaluateFormula(args[0])); - const digits = - args.length === 2 ? toNumber(context.evaluateFormula(args[1])) : 0; + const number = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + const digits = args.length === 2 ? toNumber(context.evaluateFormula(requiredArg(context, args, 1))) : 0; const multiplier = Math.pow(10, digits); return Math.trunc(number * multiplier) / multiplier; }; const signHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("SIGN requires 1 argument"); - const number = toNumber(context.evaluateFormula(args[0])); - return Math.sign(number); + const number = toScalarNumber(context.evaluateFormula(requiredArg(context, args, 0))); + return isSpreadsheetErrorValue(number) ? number : Math.sign(number); }; -const piHandler: FunctionHandler = (args) => { - if (args.length !== 0) throw new Error("PI requires 0 arguments"); - return Math.PI; -}; +const piHandler: FunctionHandler = () => Math.PI; const expHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("EXP requires 1 argument"); - const number = toNumber(context.evaluateFormula(args[0])); + const number = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); return Math.exp(number); }; const lnHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("LN requires 1 argument"); - const number = toNumber(context.evaluateFormula(args[0])); - return Math.log(number); + const number = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + return safeLog(number); }; const logHandler: FunctionHandler = (args, context) => { - if (args.length < 1 || args.length > 2) { - throw new Error("LOG requires 1 or 2 arguments"); - } - const number = toNumber(context.evaluateFormula(args[0])); - const base = - args.length === 2 ? toNumber(context.evaluateFormula(args[1])) : 10; - return Math.log(number) / Math.log(base); + const number = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + const base = args.length === 2 ? toNumber(context.evaluateFormula(requiredArg(context, args, 1))) : 10; + return logWithBase(number, base); }; const log10Handler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("LOG10 requires 1 argument"); - const number = toNumber(context.evaluateFormula(args[0])); - return Math.log10(number); + const number = toNumber(context.evaluateFormula(requiredArg(context, args, 0))); + return safeLog10(number); }; // Register all mathematical functions diff --git a/src/engine/functions/statistical-math.ts b/src/engine/functions/statistical-math.ts new file mode 100644 index 0000000..cd0ad94 --- /dev/null +++ b/src/engine/functions/statistical-math.ts @@ -0,0 +1,91 @@ +/** + * Pure statistical rules, separated from the range-reading handlers so they can + * be unit-tested directly. + */ + +import { DIV_ZERO_ERROR, NA_ERROR, NUM_ERROR, type SpreadsheetError } from "../spreadsheet-errors"; + +const arithmeticMean = (values: number[]): number => values.reduce((sum, value) => sum + value, 0) / values.length; + +/** + * The arithmetic mean, or `#DIV/0!` when there is nothing to average. + * + * Excel divides by the count of NUMBERS, so a range of blanks or text leaves a + * zero denominator and reports the division rather than a 0 that reads like a + * genuine average of zeros. + */ +export function computeAverage(values: number[]): number | SpreadsheetError { + if (values.length === 0) return DIV_ZERO_ERROR; + return arithmeticMean(values); +} + +/** + * The middle value (the mean of the middle two when the count is even), or + * `#NUM!` when there is no value to sit in the middle. + * + * Excel's code here is `#NUM!`, not AVERAGE's `#DIV/0!` — nothing is divided by + * zero, the median of an empty set simply does not exist. Sorts a copy so the + * caller's array keeps its order. + */ +export function computeMedian(values: number[]): number | SpreadsheetError { + if (values.length === 0) return NUM_ERROR; + return arithmeticMean(middleValues([...values].sort((a, b) => a - b))); +} + +/** The one or two values sitting in the middle of a sorted list — two when the + * count is even, one when it is odd. Sliced rather than indexed, so the middle + * of an empty list is simply nothing. */ +function middleValues(sorted: number[]): number[] { + const half = Math.floor(sorted.length / 2); + return sorted.slice(sorted.length % 2 === 0 ? half - 1 : half, half + 1); +} + +/** + * The most frequently occurring value, or `#N/A` when no value repeats. + * + * Excel's MODE is undefined for an all-distinct set, so returning the first + * element (as a naive "highest frequency wins" loop does when every count is 1) + * is a silent wrong answer. Ties resolve to the value that appears first, which + * Map insertion order preserves. + */ +export function computeMode(values: number[]): number | SpreadsheetError { + const frequency = new Map(); + for (const value of values) { + frequency.set(value, (frequency.get(value) ?? 0) + 1); + } + + let topFrequency = 0; + let mode: number | SpreadsheetError = NA_ERROR; + for (const [value, count] of frequency.entries()) { + if (count > topFrequency) { + topFrequency = count; + mode = value; + } + } + + const REPEAT_THRESHOLD = 2; + return topFrequency >= REPEAT_THRESHOLD ? mode : NA_ERROR; +} + +/** Sample size below which a sample variance/stdev is undefined. */ +const MIN_SAMPLE_SIZE = 2; + +/** + * Sample variance (Excel VAR): the mean squared deviation divided by `n - 1`, + * not `n`. Dividing by `n` is the POPULATION variance (Excel's VARP); using it + * for VAR understates the spread. Fewer than two values leave no `n - 1` to + * divide by, so Excel reports `#DIV/0!` rather than a silent 0. + */ +export function sampleVariance(values: number[]): number | SpreadsheetError { + if (values.length < MIN_SAMPLE_SIZE) return DIV_ZERO_ERROR; + const mean = arithmeticMean(values); + const sumSquaredDiffs = values.reduce((sum, value) => sum + (value - mean) ** 2, 0); + return sumSquaredDiffs / (values.length - 1); +} + +/** Sample standard deviation (Excel STDEV): the square root of the sample + * variance, and `#DIV/0!` on the same fewer-than-two-values boundary. */ +export function sampleStdev(values: number[]): number | SpreadsheetError { + const variance = sampleVariance(values); + return typeof variance === "number" ? Math.sqrt(variance) : variance; +} diff --git a/src/engine/functions/statistical.ts b/src/engine/functions/statistical.ts index d4a8f01..a6995a2 100644 --- a/src/engine/functions/statistical.ts +++ b/src/engine/functions/statistical.ts @@ -4,230 +4,163 @@ import { functionRegistry, + rawRangeReader, + requiredArg, toNumber, parseCriteria, type FunctionContext, type FunctionHandler, + type RangeGetter, } from "../registry"; +import { computeAverage, computeMedian, computeMode, sampleStdev, sampleVariance } from "./statistical-math"; +import { DIV_ZERO_ERROR } from "../spreadsheet-errors"; +import { holdsNumber } from "../numericCoercion"; +import type { CellValue } from "../types"; -const isLetter = (char: string): boolean => /[A-Z]/i.test(char); +// Excel accepts up to 255 arguments for its aggregate functions. +const MAX_AGGREGATE_ARGS = 255; -const isCellReference = (segment: string): boolean => { - if (!segment) return false; - let index = 0; - if (segment[index] === "$") index++; - const colStart = index; - while (index < segment.length && isLetter(segment[index])) { - index++; - } - if (index === colStart) return false; // Require at least one column letter - if (segment[index] === "$") index++; - if (index >= segment.length) return false; // Require row digits - for (; index < segment.length; index++) { - const char = segment[index]; - if (char < "0" || char > "9") { - return false; - } - } - return true; -}; +// `A1`, `$A$1`, `AA100`: column letters then row digits, each half optionally +// prefixed by `$`. +const CELL_REFERENCE_PATTERN = /^\$?[A-Z]+\$?\d+$/i; + +const isCellReference = (segment: string): boolean => CELL_REFERENCE_PATTERN.test(segment); + +// Everything after the last `!` — the reference without its sheet name, or the +// whole string when it carries none. +const withoutSheetPrefix = (value: string): string => value.slice(value.lastIndexOf("!") + 1); const isRangeReference = (value: string): boolean => { if (!value) return false; - const rangePart = value.includes("!") ? value.split("!").slice(-1)[0] : value; - const [start, end] = rangePart.split(":"); + const [start, end] = withoutSheetPrefix(value).split(":"); if (!start || !end) return false; return isCellReference(start) && isCellReference(end); }; -const collectNumericValues = ( - args: string[], - context: FunctionContext, -): number[] => { - const values: number[] = []; +// A bare cell reference (`A1`, `Sheet1!B2`) is read through the RANGE path, not +// evaluated as a scalar: the scalar path coerces a blank or text cell to 0, so +// COUNT(A999) counted an empty cell as a value. The range path yields nothing +// for a cell that holds nothing, which is what the count functions need. +const isReference = (arg: string): boolean => isRangeReference(arg) || isCellReference(withoutSheetPrefix(arg)); + +/** One value an argument contributed, tagged by where it came from. A range cell + * was already filtered by the range getter; a scalar is whatever the argument + * evaluated to and may hold no number at all. */ +interface ArgumentValue { + value: CellValue; + isScalar: boolean; +} + +const collectArgumentValues = (args: string[], context: FunctionContext, readRange: RangeGetter): ArgumentValue[] => { + const collected: ArgumentValue[] = []; for (const rawArg of args) { const arg = rawArg?.trim(); if (!arg) continue; - if (isRangeReference(arg)) { - const rangeValues = context.getRangeValues(arg).map(toNumber); - values.push(...rangeValues); + if (isReference(arg)) { + readRange(arg).forEach((value) => collected.push({ value, isScalar: false })); } else { - const evaluated = context.evaluateFormula(arg); - values.push(toNumber(evaluated)); + collected.push({ value: context.evaluateFormula(arg), isScalar: true }); } } - return values; + return collected; }; +const collectNumericValues = (args: string[], context: FunctionContext): number[] => + collectArgumentValues(args, context, context.getRangeValues).map(({ value }) => toNumber(value)); + +// Same walk as `collectNumericValues`, but keeping each cell as it is: COUNTA +// counts non-empty cells, so text must survive the trip. +const collectRawValues = (args: string[], context: FunctionContext): CellValue[] => + collectArgumentValues(args, context, rawRangeReader(context)).map(({ value }) => value); + const sumHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("SUM requires 1 argument"); - const values = context.getRangeValues(args[0]); - return values.reduce((sum: number, val) => sum + toNumber(val), 0); + const values = collectNumericValues(args, context); + return values.reduce((sum: number, value) => sum + value, 0); }; -const averageHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("AVERAGE requires 1 argument"); - const values = context.getRangeValues(args[0]); - if (values.length === 0) return 0; - const sum = values.reduce((acc: number, val) => acc + toNumber(val), 0); - return sum / values.length; -}; +// Multi-argument collection (#2360) feeding the empty-range error rule (#2501). +const averageHandler: FunctionHandler = (args, context) => computeAverage(collectNumericValues(args, context)); const maxHandler: FunctionHandler = (args, context) => { - if (args.length === 0) { - throw new Error("MAX requires at least 1 argument"); - } const values = collectNumericValues(args, context); return values.length > 0 ? Math.max(...values) : 0; }; const minHandler: FunctionHandler = (args, context) => { - if (args.length === 0) { - throw new Error("MIN requires at least 1 argument"); - } const values = collectNumericValues(args, context); return values.length > 0 ? Math.min(...values) : 0; }; -const countHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("COUNT requires 1 argument"); - const values = context.getRangeValues(args[0]); - return values.length; -}; - -const medianHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("MEDIAN requires 1 argument"); - const values = context - .getRangeValues(args[0]) - .map(toNumber) - .sort((a, b) => a - b); - - if (values.length === 0) return 0; - const mid = Math.floor(values.length / 2); - return values.length % 2 === 0 - ? (values[mid - 1] + values[mid]) / 2 - : values[mid]; -}; +// COUNT counts NUMBERS, so it cannot go through the lenient numeric collection +// the other aggregates share: `toNumber("text")` is 0, which made COUNT("text") +// answer 1 where Excel answers 0 (Codex review). A range cell reached the list +// only by being numeric already; a scalar has to be asked. +const countsAsNumber = ({ value, isScalar }: ArgumentValue): boolean => !isScalar || holdsNumber(value); -const modeHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("MODE requires 1 argument"); - const values = context.getRangeValues(args[0]).map(toNumber); +const countHandler: FunctionHandler = (args, context) => collectArgumentValues(args, context, context.getRangeValues).filter(countsAsNumber).length; - if (values.length === 0) return 0; +const medianHandler: FunctionHandler = (args, context) => computeMedian(collectNumericValues(args, context)); - // Count frequency of each value - const frequency = new Map(); - for (const val of values) { - frequency.set(val, (frequency.get(val) || 0) + 1); - } - - // Find the value with highest frequency - let maxFreq = 0; - let mode = values[0]; - for (const [val, freq] of frequency.entries()) { - if (freq > maxFreq) { - maxFreq = freq; - mode = val; - } - } - - return mode; +const modeHandler: FunctionHandler = (args, context) => { + return computeMode(collectNumericValues(args, context)); }; const stdevHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("STDEV requires 1 argument"); - const values = context.getRangeValues(args[0]).map(toNumber); - - if (values.length === 0) return 0; - - const mean = values.reduce((sum, val) => sum + val, 0) / values.length; - const squaredDiffs = values.map((val) => Math.pow(val - mean, 2)); - const variance = - squaredDiffs.reduce((sum, val) => sum + val, 0) / values.length; - return Math.sqrt(variance); + return sampleStdev(collectNumericValues(args, context)); }; const varHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("VAR requires 1 argument"); - const values = context.getRangeValues(args[0]).map(toNumber); - - if (values.length === 0) return 0; - - const mean = values.reduce((sum, val) => sum + val, 0) / values.length; - const squaredDiffs = values.map((val) => Math.pow(val - mean, 2)); - return squaredDiffs.reduce((sum, val) => sum + val, 0) / values.length; + return sampleVariance(collectNumericValues(args, context)); }; const countaHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("COUNTA requires 1 argument"); - const values = - context.getRangeValuesRaw?.(args[0]) ?? context.getRangeValues(args[0]); + const values = collectRawValues(args, context); // Count non-empty cells - return values.filter((v) => v !== null && v !== undefined && v !== "").length; + return values.filter((value) => value !== null && value !== undefined && value !== "").length; }; const countifHandler: FunctionHandler = (args, context) => { - if (args.length !== 2) throw new Error("COUNTIF requires 2 arguments"); - const values = - context.getRangeValuesRaw?.(args[0]) ?? context.getRangeValues(args[0]); - const criteria = args[1].trim(); - const compareFn = parseCriteria(criteria); - return values.filter(compareFn).length; + const values = rawRangeReader(context)(requiredArg(context, args, 0)); + return values.filter(parseCriteria(requiredArg(context, args, 1).trim())).length; }; -const sumifHandler: FunctionHandler = (args, context) => { - if (args.length < 2 || args.length > 3) { - throw new Error("SUMIF requires 2 or 3 arguments"); - } - - const criteriaRange = - context.getRangeValuesRaw?.(args[0]) ?? context.getRangeValues(args[0]); - const criteria = args[1].trim(); - const sumRange = - args.length === 3 - ? context.getRangeValues(args[2]) - : context.getRangeValues(args[0]); +/** The criteria range, the matcher and the value range SUMIF and AVERAGEIF both + * read. Both value ranges are RAW, not numeric-only: dropping blanks would + * shift the value range out of alignment with the (raw) criteria range, so a + * blank would pull a later row's number into an earlier match (#2358). */ +const readConditionalRanges = (args: string[], context: FunctionContext) => { + const criteriaRef = requiredArg(context, args, 0); + const readRaw = rawRangeReader(context); + const valueRef = args.length === 3 ? requiredArg(context, args, 2) : criteriaRef; + return { + criteriaRange: readRaw(criteriaRef), + valueRange: readRaw(valueRef), + matches: parseCriteria(requiredArg(context, args, 1).trim()), + }; +}; - const compareFn = parseCriteria(criteria); +/** Sum and count the values whose row in `criteriaRange` matches. The two + * ranges stay row-aligned, so a matched row with no value contributes 0. */ +const aggregateMatchedRows = (criteriaRange: CellValue[], valueRange: CellValue[], matches: (value: CellValue) => boolean) => + criteriaRange.reduce( + (totals, criteriaValue, index) => (matches(criteriaValue) ? { sum: totals.sum + toNumber(valueRange[index] ?? 0), count: totals.count + 1 } : totals), + { sum: 0, count: 0 }, + ); - let sum = 0; - for (let i = 0; i < criteriaRange.length; i++) { - if (compareFn(criteriaRange[i])) { - sum += toNumber(sumRange[i] ?? 0); - } - } - - return sum; +const sumifHandler: FunctionHandler = (args, context) => { + const { criteriaRange, valueRange, matches } = readConditionalRanges(args, context); + return aggregateMatchedRows(criteriaRange, valueRange, matches).sum; }; const averageifHandler: FunctionHandler = (args, context) => { - if (args.length < 2 || args.length > 3) { - throw new Error("AVERAGEIF requires 2 or 3 arguments"); - } - - const criteriaRange = - context.getRangeValuesRaw?.(args[0]) ?? context.getRangeValues(args[0]); - const criteria = args[1].trim(); - const avgRange = - args.length === 3 - ? context.getRangeValues(args[2]) - : context.getRangeValues(args[0]); - - const compareFn = parseCriteria(criteria); - - let sum = 0; - let count = 0; - for (let i = 0; i < criteriaRange.length; i++) { - if (compareFn(criteriaRange[i])) { - sum += toNumber(avgRange[i] ?? 0); - count++; - } - } - - return count > 0 ? sum / count : 0; + const { criteriaRange, valueRange, matches } = readConditionalRanges(args, context); + const { sum, count } = aggregateMatchedRows(criteriaRange, valueRange, matches); + // Excel returns #DIV/0! when no cell matches (the average of nothing is + // undefined), rather than a silent 0. + return count > 0 ? sum / count : DIV_ZERO_ERROR; }; // Register all statistical functions @@ -235,7 +168,7 @@ functionRegistry.register({ name: "SUM", handler: sumHandler, minArgs: 1, - maxArgs: 1, + maxArgs: MAX_AGGREGATE_ARGS, description: "Returns the sum of all numbers in a range", examples: ["SUM(A1:A10)", "SUM(B2:B20)"], category: "Statistical", @@ -245,7 +178,7 @@ functionRegistry.register({ name: "AVERAGE", handler: averageHandler, minArgs: 1, - maxArgs: 1, + maxArgs: MAX_AGGREGATE_ARGS, description: "Returns the average (arithmetic mean) of numbers in a range", examples: ["AVERAGE(A1:A10)", "AVERAGE(B2:B20)"], category: "Statistical", @@ -273,7 +206,7 @@ functionRegistry.register({ name: "COUNT", handler: countHandler, minArgs: 1, - maxArgs: 1, + maxArgs: MAX_AGGREGATE_ARGS, description: "Counts the number of cells in a range", examples: ["COUNT(A1:A10)", "COUNT(B2:B20)"], category: "Statistical", @@ -283,7 +216,7 @@ functionRegistry.register({ name: "MEDIAN", handler: medianHandler, minArgs: 1, - maxArgs: 1, + maxArgs: MAX_AGGREGATE_ARGS, description: "Returns the median (middle) value in a range", examples: ["MEDIAN(A1:A10)", "MEDIAN(B2:B20)"], category: "Statistical", @@ -293,7 +226,7 @@ functionRegistry.register({ name: "MODE", handler: modeHandler, minArgs: 1, - maxArgs: 1, + maxArgs: MAX_AGGREGATE_ARGS, description: "Returns the most frequently occurring value in a range", examples: ["MODE(A1:A10)", "MODE(B2:B20)"], category: "Statistical", @@ -303,7 +236,7 @@ functionRegistry.register({ name: "STDEV", handler: stdevHandler, minArgs: 1, - maxArgs: 1, + maxArgs: MAX_AGGREGATE_ARGS, description: "Returns the standard deviation of numbers in a range", examples: ["STDEV(A1:A10)", "STDEV(B2:B20)"], category: "Statistical", @@ -313,7 +246,7 @@ functionRegistry.register({ name: "VAR", handler: varHandler, minArgs: 1, - maxArgs: 1, + maxArgs: MAX_AGGREGATE_ARGS, description: "Returns the variance of numbers in a range", examples: ["VAR(A1:A10)", "VAR(B2:B20)"], category: "Statistical", @@ -323,7 +256,7 @@ functionRegistry.register({ name: "COUNTA", handler: countaHandler, minArgs: 1, - maxArgs: 1, + maxArgs: MAX_AGGREGATE_ARGS, description: "Counts the number of non-empty cells in a range", examples: ["COUNTA(A1:A10)", "COUNTA(B2:B20)"], category: "Statistical", diff --git a/src/engine/functions/text.ts b/src/engine/functions/text.ts index 9a67cce..85b85d4 100644 --- a/src/engine/functions/text.ts +++ b/src/engine/functions/text.ts @@ -2,12 +2,67 @@ * Text Functions */ -import { functionRegistry, toString, type FunctionHandler } from "../registry"; +import { functionRegistry, requiredArg, toString, type FunctionHandler } from "../registry"; +import { VALUE_ERROR, isSpreadsheetErrorValue, type SpreadsheetError } from "../spreadsheet-errors"; +import { formatWithPattern } from "../textFormat"; + +// A letter or a combining mark. A decomposed accented letter (e + U+0301) is two +// code points; counting the mark as part of the word stops PROPER from treating +// the base letter that follows it as a new word (éclair → Éclair, not ÉClair). +const isWordCharacter = (char: string): boolean => /[\p{L}\p{M}]/u.test(char); + +// Excel PROPER capitalises a letter at the start of the text or after any +// non-letter (space, punctuation, digit) and lowercases the rest — so word +// boundaries include "'" and "-", which a space-only split misses. +export const toProperCase = (text: string): string => { + const chars = Array.from(text); + const cased = chars.map((char, index) => { + const previous = chars[index - 1]; + return previous === undefined || !isWordCharacter(previous) ? char.toUpperCase() : char.toLowerCase(); + }); + return cased.join(""); +}; -const concatenateHandler: FunctionHandler = (args, context) => { - if (args.length === 0) - throw new Error("CONCATENATE requires at least 1 argument"); +// Excel LEFT/RIGHT reject a negative count with #VALUE!; 0 and over-length +// counts keep substring's clamping. +// Excel truncates a fractional count toward zero and rejects a non-finite or +// negative one with #VALUE! (LEFT/RIGHT with "x" or -1). Normalising once keeps +// LEFT and RIGHT consistent instead of each feeding a raw Number() to substring. +const normalizeCharCount = (count: number): number | SpreadsheetError => { + // Test the sign before truncating: Math.trunc(-0.5) is -0, which is not < 0. + if (!Number.isFinite(count) || count < 0) return VALUE_ERROR; + return Math.trunc(count); +}; + +export const takeLeft = (text: string, count: number): string | SpreadsheetError => { + const chars = normalizeCharCount(count); + return isSpreadsheetErrorValue(chars) ? chars : text.substring(0, chars); +}; + +export const takeRight = (text: string, count: number): string | SpreadsheetError => { + const chars = normalizeCharCount(count); + return isSpreadsheetErrorValue(chars) ? chars : text.substring(text.length - chars); +}; + +// Replace the nth (1-based) occurrence; split/join keeps matches non-overlapping, +// matching the replace-all path. +const replaceNthOccurrence = (text: string, oldText: string, newText: string, nth: number): string => { + const parts = text.split(oldText); + if (nth > parts.length - 1) return text; + return parts.slice(0, nth).join(oldText) + newText + parts.slice(nth).join(oldText); +}; + +// Excel SUBSTITUTE: empty old_text returns the text unchanged (never inserts +// between characters); a supplied instance ≤ 0 or non-finite is a #VALUE! error. +export const substituteText = (text: string, oldText: string, newText: string, instance?: number): string | SpreadsheetError => { + if (oldText === "") return text; + if (instance === undefined) return text.split(oldText).join(newText); + const nth = Math.trunc(instance); + if (!Number.isFinite(nth) || nth <= 0) return VALUE_ERROR; + return replaceNthOccurrence(text, oldText, newText, nth); +}; +const concatenateHandler: FunctionHandler = (args, context) => { return args .map((arg) => { const value = context.evaluateFormula(arg.trim()); @@ -19,217 +74,156 @@ const concatenateHandler: FunctionHandler = (args, context) => { const concatHandler: FunctionHandler = concatenateHandler; // Alias const leftHandler: FunctionHandler = (args, context) => { - if (args.length < 1 || args.length > 2) { - throw new Error("LEFT requires 1 or 2 arguments"); - } - - const text = toString(context.evaluateFormula(args[0])); - const numChars = - args.length === 2 ? Number(context.evaluateFormula(args[1])) : 1; + const text = toString(context.evaluateFormula(requiredArg(context, args, 0))); + const numChars = args.length === 2 ? Number(context.evaluateFormula(requiredArg(context, args, 1))) : 1; - return text.substring(0, numChars); + return takeLeft(text, numChars); }; const rightHandler: FunctionHandler = (args, context) => { - if (args.length < 1 || args.length > 2) { - throw new Error("RIGHT requires 1 or 2 arguments"); - } + const text = toString(context.evaluateFormula(requiredArg(context, args, 0))); + const numChars = args.length === 2 ? Number(context.evaluateFormula(requiredArg(context, args, 1))) : 1; - const text = toString(context.evaluateFormula(args[0])); - const numChars = - args.length === 2 ? Number(context.evaluateFormula(args[1])) : 1; + return takeRight(text, numChars); +}; - return text.substring(text.length - numChars); +/** Excel MID: 1-based start, count of characters. `substring` SWAPS its bounds + * when they are reversed, so a negative count read backwards from `start` and + * returned earlier characters — `MID("Hello",3,-1)` gave "e" instead of an + * error. Both arguments are validated here instead. */ +export const takeMid = (text: string, start: number, count: number): string | SpreadsheetError => { + const chars = normalizeCharCount(count); + if (isSpreadsheetErrorValue(chars)) return chars; + if (!Number.isFinite(start) || start < 1) return VALUE_ERROR; + const from = Math.trunc(start) - 1; + return text.substring(from, from + chars); }; const midHandler: FunctionHandler = (args, context) => { - if (args.length !== 3) throw new Error("MID requires 3 arguments"); + const text = toString(context.evaluateFormula(requiredArg(context, args, 0))); + const start = Number(context.evaluateFormula(requiredArg(context, args, 1))); + const numChars = Number(context.evaluateFormula(requiredArg(context, args, 2))); - const text = toString(context.evaluateFormula(args[0])); - const start = Number(context.evaluateFormula(args[1])) - 1; // 1-indexed to 0-indexed - const numChars = Number(context.evaluateFormula(args[2])); - - return text.substring(start, start + numChars); + return takeMid(text, start, numChars); }; const lenHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("LEN requires 1 argument"); - - const text = toString(context.evaluateFormula(args[0])); + const text = toString(context.evaluateFormula(requiredArg(context, args, 0))); return text.length; }; const upperHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("UPPER requires 1 argument"); - - const text = toString(context.evaluateFormula(args[0])); + const text = toString(context.evaluateFormula(requiredArg(context, args, 0))); return text.toUpperCase(); }; const lowerHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("LOWER requires 1 argument"); - - const text = toString(context.evaluateFormula(args[0])); + const text = toString(context.evaluateFormula(requiredArg(context, args, 0))); return text.toLowerCase(); }; const properHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("PROPER requires 1 argument"); - - const text = toString(context.evaluateFormula(args[0])); - return text - .toLowerCase() - .split(" ") - .map((word) => - word.length > 0 ? word[0].toUpperCase() + word.slice(1) : "", - ) - .join(" "); + const text = toString(context.evaluateFormula(requiredArg(context, args, 0))); + return toProperCase(text); }; const trimHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("TRIM requires 1 argument"); - - const text = toString(context.evaluateFormula(args[0])); + const text = toString(context.evaluateFormula(requiredArg(context, args, 0))); // Trim leading/trailing spaces and replace multiple spaces with single space return text.trim().replace(/\s+/g, " "); }; const substituteHandler: FunctionHandler = (args, context) => { - if (args.length < 3 || args.length > 4) { - throw new Error("SUBSTITUTE requires 3 or 4 arguments"); - } + const text = toString(context.evaluateFormula(requiredArg(context, args, 0))); + const oldText = toString(context.evaluateFormula(requiredArg(context, args, 1))); + const newText = toString(context.evaluateFormula(requiredArg(context, args, 2))); + const instance = args.length === 4 ? Number(context.evaluateFormula(requiredArg(context, args, 3))) : undefined; - const text = toString(context.evaluateFormula(args[0])); - const oldText = toString(context.evaluateFormula(args[1])); - const newText = toString(context.evaluateFormula(args[2])); - - if (args.length === 4) { - // Replace specific instance - const instance = Number(context.evaluateFormula(args[3])); - let count = 0; - let index = 0; - - while (index < text.length) { - const pos = text.indexOf(oldText, index); - if (pos === -1) break; - - count++; - if (count === instance) { - return ( - text.substring(0, pos) + - newText + - text.substring(pos + oldText.length) - ); - } - index = pos + 1; - } - return text; // Instance not found - } else { - // Replace all instances - return text.split(oldText).join(newText); - } + return substituteText(text, oldText, newText, instance); }; const replaceHandler: FunctionHandler = (args, context) => { - if (args.length !== 4) throw new Error("REPLACE requires 4 arguments"); - - const oldText = toString(context.evaluateFormula(args[0])); - const startPos = Number(context.evaluateFormula(args[1])) - 1; // 1-indexed to 0-indexed - const numChars = Number(context.evaluateFormula(args[2])); - const newText = toString(context.evaluateFormula(args[3])); - - return ( - oldText.substring(0, startPos) + - newText + - oldText.substring(startPos + numChars) - ); + const oldText = toString(context.evaluateFormula(requiredArg(context, args, 0))); + const startPos = Number(context.evaluateFormula(requiredArg(context, args, 1))) - 1; // 1-indexed to 0-indexed + const numChars = Number(context.evaluateFormula(requiredArg(context, args, 2))); + const newText = toString(context.evaluateFormula(requiredArg(context, args, 3))); + + return oldText.substring(0, startPos) + newText + oldText.substring(startPos + numChars); }; -const findHandler: FunctionHandler = (args, context) => { - if (args.length < 2 || args.length > 3) { - throw new Error("FIND requires 2 or 3 arguments"); - } +// FIND and SEARCH share everything but case sensitivity: same 0-based start, +// same #VALUE! on no match, same 1-based result. SEARCH folds case first. +export const locateSubstring = (find: string, within: string, start: number, options: { caseInsensitive: boolean }): number | SpreadsheetError => { + const needle = options.caseInsensitive ? find.toLowerCase() : find; + const haystack = options.caseInsensitive ? within.toLowerCase() : within; + const index = haystack.indexOf(needle, start); + return index === -1 ? VALUE_ERROR : index + 1; // 1-indexed position +}; - const findText = toString(context.evaluateFormula(args[0])); - const withinText = toString(context.evaluateFormula(args[1])); - const startPos = - args.length === 3 ? Number(context.evaluateFormula(args[2])) - 1 : 0; +const findHandler: FunctionHandler = (args, context) => { + const findText = toString(context.evaluateFormula(requiredArg(context, args, 0))); + const withinText = toString(context.evaluateFormula(requiredArg(context, args, 1))); + const startPos = args.length === 3 ? Number(context.evaluateFormula(requiredArg(context, args, 2))) - 1 : 0; - const index = withinText.indexOf(findText, startPos); - return index === -1 ? "#VALUE!" : index + 1; // Return 1-indexed position + return locateSubstring(findText, withinText, startPos, { caseInsensitive: false }); }; const searchHandler: FunctionHandler = (args, context) => { - if (args.length < 2 || args.length > 3) { - throw new Error("SEARCH requires 2 or 3 arguments"); - } - - const findText = toString(context.evaluateFormula(args[0])); - const withinText = toString(context.evaluateFormula(args[1])); - const startPos = - args.length === 3 ? Number(context.evaluateFormula(args[2])) - 1 : 0; - - // SEARCH is case-insensitive - const lowerFind = findText.toLowerCase(); - const lowerWithin = withinText.toLowerCase(); + const findText = toString(context.evaluateFormula(requiredArg(context, args, 0))); + const withinText = toString(context.evaluateFormula(requiredArg(context, args, 1))); + const startPos = args.length === 3 ? Number(context.evaluateFormula(requiredArg(context, args, 2))) - 1 : 0; - const index = lowerWithin.indexOf(lowerFind, startPos); - return index === -1 ? "#VALUE!" : index + 1; // Return 1-indexed position + return locateSubstring(findText, withinText, startPos, { caseInsensitive: true }); }; -const textHandler: FunctionHandler = (args, context) => { - if (args.length !== 2) throw new Error("TEXT requires 2 arguments"); - - const value = context.evaluateFormula(args[0]); - const format = toString(context.evaluateFormula(args[1])).replace( - // eslint-disable -- sonarjs/anchor-precedence - /^["']|["']$/g, - "", - ); - - // Simple format code handling - if (typeof value === "number") { - // Handle common format codes - if (format.includes("$")) { - const decimals = (format.match(/\.0+/) || [""])[0].length - 1; - return "$" + value.toFixed(decimals >= 0 ? decimals : 2); - } - if (format.includes("%")) { - const decimals = (format.match(/\.0+/) || [""])[0].length - 1; - return (value * 100).toFixed(decimals >= 0 ? decimals : 2) + "%"; - } - if (format.includes("0")) { - const decimals = (format.match(/\.0+/) || [""])[0].length - 1; - return value.toFixed(decimals >= 0 ? decimals : 0); - } - } +// A format code written as a literal still carries its quotes when it reaches +// the handler. +const stripSurroundingQuotes = (text: string): string => text.replace(/^["']/, "").replace(/["']$/, ""); - return toString(value); +const textHandler: FunctionHandler = (args, context) => { + const value = context.evaluateFormula(requiredArg(context, args, 0)); + const format = stripSurroundingQuotes(toString(context.evaluateFormula(requiredArg(context, args, 1)))); + if (typeof value !== "number") return toString(value); + return formatWithPattern(value, format) ?? toString(value); }; -const valueHandler: FunctionHandler = (args, context) => { - if (args.length !== 1) throw new Error("VALUE requires 1 argument"); - - const text = toString(context.evaluateFormula(args[0])); - - // Remove currency symbols and commas - const cleaned = text.replace(/[$,]/g, "").trim(); +/** Read a numeric string in full, or null. `Number` rather than `parseFloat`: + * `parseFloat` stops at the first character it cannot read, so `VALUE("12abc")` + * came back 12 where Excel reports #VALUE!. An empty string is not a number + * here even though `Number("")` is 0. */ +// Decimal or scientific notation only. `Number` alone also accepts JS-only +// spellings a spreadsheet never should — `0x10` → 16, `0b10` → 2, `Infinity` — +// so the shape is checked before converting. +const DECIMAL_NUMBER = /^[+-]?(?:\d+(?:\.\d*)?|\.\d+)(?:[eE][+-]?\d+)?$/; + +const wholeNumberOrNull = (text: string): number | null => { + if (!DECIMAL_NUMBER.test(text)) return null; + const parsed = Number(text); + // The pattern admits an exponent that overflows to Infinity (`1e999`), which + // is no more a spreadsheet number than the literal spelling is. + return Number.isFinite(parsed) ? parsed : null; +}; - // Handle percentages - if (cleaned.includes("%")) { - const num = parseFloat(cleaned.replace("%", "")); - return isNaN(num) ? "#VALUE!" : num / 100; +/** Excel VALUE: the WHOLE string must be a number once its currency symbols and + * thousands separators are stripped; trailing text is an error, not a prefix to + * salvage. */ +export const parseValueText = (raw: string): number | SpreadsheetError => { + const cleaned = raw.replace(/[$,]/g, "").trim(); + if (cleaned.endsWith("%")) { + const percent = wholeNumberOrNull(cleaned.slice(0, -1).trim()); + return percent === null ? VALUE_ERROR : percent / 100; } + const parsed = wholeNumberOrNull(cleaned); + return parsed === null ? VALUE_ERROR : parsed; +}; - const num = parseFloat(cleaned); - return isNaN(num) ? "#VALUE!" : num; +const valueHandler: FunctionHandler = (args, context) => { + return parseValueText(toString(context.evaluateFormula(requiredArg(context, args, 0)))); }; const exactHandler: FunctionHandler = (args, context) => { - if (args.length !== 2) throw new Error("EXACT requires 2 arguments"); - - const text1 = toString(context.evaluateFormula(args[0])); - const text2 = toString(context.evaluateFormula(args[1])); + const text1 = toString(context.evaluateFormula(requiredArg(context, args, 0))); + const text2 = toString(context.evaluateFormula(requiredArg(context, args, 1))); return text1 === text2; }; @@ -248,8 +242,7 @@ functionRegistry.register({ name: "CONCAT", handler: concatHandler, minArgs: 1, - description: - "Joins several text strings into one string (same as CONCATENATE)", + description: "Joins several text strings into one string (same as CONCATENATE)", examples: ['CONCAT("Hello", " ", "World")', "CONCAT(A1, B1)"], category: "Text", }); @@ -340,10 +333,7 @@ functionRegistry.register({ minArgs: 3, maxArgs: 4, description: "Replaces old text with new text in a string", - examples: [ - 'SUBSTITUTE("Hello World", "World", "Earth")', - 'SUBSTITUTE(A1, "old", "new", 1)', - ], + examples: ['SUBSTITUTE("Hello World", "World", "Earth")', 'SUBSTITUTE(A1, "old", "new", 1)'], category: "Text", }); @@ -353,10 +343,7 @@ functionRegistry.register({ minArgs: 4, maxArgs: 4, description: "Replaces part of a text string with a different text string", - examples: [ - 'REPLACE("Hello World", 7, 5, "Earth")', - 'REPLACE(A1, 1, 3, "New")', - ], + examples: ['REPLACE("Hello World", 7, 5, "Earth")', 'REPLACE(A1, 1, 3, "New")'], category: "Text", }); diff --git a/src/engine/guards.ts b/src/engine/guards.ts new file mode 100644 index 0000000..6610250 --- /dev/null +++ b/src/engine/guards.ts @@ -0,0 +1,32 @@ +// The three runtime helpers the engine needs, kept here so the engine stays what +// its own header claims: framework-agnostic, with nothing above it to import. +// +// They arrive with the engine that was developed in MulmoClaude, where they came +// from that repo's own leaf package. Copied rather than depended on: a +// gui-chat plugin must not take a dependency on a host's packages. + +/** Narrow `unknown` to a plain object (not null, not array). */ +export function isRecord(value: unknown): value is Record { + return typeof value === "object" && value !== null && !Array.isArray(value); +} + +/** Narrow `unknown` to any object (not null, arrays allowed). + * Use `isRecord` when you need to access string keys. */ +export function isObj(value: unknown): value is object { + return typeof value === "object" && value !== null; +} + +const hasStringProp = (value: unknown, key: K): value is Record & Record => + isRecord(value) && typeof value[key] === "string"; + +/** The message to show for a thrown value. A non-Error object with a non-empty + * string `details` (the gRPC convention) or `message` surfaces that field — + * `details` wins — instead of the `[object Object]` a bare `String(err)` + * would print. */ +export function errorMessage(err: unknown, fallback?: string): string { + if (err instanceof Error) return err.message; + if (hasStringProp(err, "details") && err.details) return err.details; + if (hasStringProp(err, "message") && err.message) return err.message; + if (fallback !== undefined) return fallback; + return String(err); +} diff --git a/src/engine/index.ts b/src/engine/index.ts index 50e157b..84f55da 100644 --- a/src/engine/index.ts +++ b/src/engine/index.ts @@ -9,9 +9,20 @@ export * from "./types"; // Export utilities export * from "./parser"; +export * from "./condition"; +export * from "./date-locale"; +export * from "./cellEmpty"; +export * from "./datedif"; export * from "./formatter"; +export * from "./translateFormula"; +export * from "./formulaError"; +export * from "./spreadsheet-errors"; export * from "./evaluator"; export * from "./calculator"; +export * from "./formulaRefs"; +export * from "./cellBuilder"; +export * from "./responseDecoder"; +export * from "./jsonCellLocator"; // Export function registry export * from "./registry"; diff --git a/src/engine/jsonCellLocator.ts b/src/engine/jsonCellLocator.ts new file mode 100644 index 0000000..b996fb3 --- /dev/null +++ b/src/engine/jsonCellLocator.ts @@ -0,0 +1,111 @@ +/** + * Locate the start offset of a specific cell value inside a + * pretty-printed JSON spreadsheet document. Extracted from + * `handleTableClick` in `src/plugins/spreadsheet/View.vue` where the + * inline scanner pushed cognitive complexity to 163. + * + * The scanner walks the raw editor text character-by-character, + * tracking string boundaries, bracket depth, and object depth to + * find the n-th cell of the m-th row within a named sheet's `data` + * array. We deliberately do NOT parse the JSON — we need the + * character offset inside the user's text buffer (which may not be + * valid JSON mid-edit), so a positional scan is required. + * + * Pure — no refs, no DOM. Returns -1 if the cell can't be located. + * Tested in `test/plugins/spreadsheet/engine/test_jsonCellLocator.ts`. + */ + +// Advance `pos` through `text` until we reach the `rowIndex`-th +// opening `[` after `startPos` (counting from -1 so the first `[` +// encountered is index 0). Returns the position just after that +// opening bracket, or -1 if we ran off the end. +function findRowOpenBracket(text: string, startPos: number, rowIndex: number): number { + let currentRow = -1; + let inString = false; + for (let i = startPos; i < text.length; i++) { + const c = text[i]; + const prevChar = i > 0 ? text[i - 1] : ""; + // Track string literal boundaries so that a `[` inside a cell + // value like `"has [bracket]"` doesn't get mistaken for a row + // opener and throw off the row offset. + if (c === '"' && prevChar !== "\\") { + inString = !inString; + continue; + } + if (!inString && c === "[") { + currentRow++; + if (currentRow === rowIndex) return i + 1; + } + } + return -1; +} + +// Starting just inside the row's `[`, scan for the start offset of +// the `colIndex`-th cell. Tracks string/object/bracket state so +// commas inside cell objects don't miscounted as cell separators. +// Returns -1 if the row ends before we reach colIndex. +function findCellStartWithinRow(text: string, rowStart: number, colIndex: number): number { + let currentCol = 0; + let inString = false; + let inObject = 0; + let bracketDepth = 1; // we already stepped past one `[` + + for (let i = rowStart; i < text.length; i++) { + const c = text[i]; + const prevChar = i > 0 ? text[i - 1] : ""; + + if (c === '"' && prevChar !== "\\") { + inString = !inString; + } + + if (!inString) { + if (c === "[") bracketDepth++; + if (c === "]") { + bracketDepth--; + if (bracketDepth === 0) return -1; // row ended before colIndex + } + if (c === "{") inObject++; + if (c === "}") inObject--; + if (c === "," && inObject === 0 && bracketDepth === 1) { + currentCol++; + } + } + + // Once currentCol matches, skip any structural whitespace / + // opening bracket / comma and return the first content char. + if (currentCol === colIndex) { + if (c !== " " && c !== "\n" && c !== "\t" && c !== "[" && c !== ",") { + return i; + } + } + } + return -1; +} + +/** + * Given the pretty-printed JSON editor text, a sheet name, and + * (rowIndex, colIndex), return the character offset of that cell's + * JSON token. The returned offset can be fed into + * `textarea.setSelectionRange(offset, offset + cellJsonLength)` to + * highlight the cell. + * + * Returns -1 if the sheet isn't found or the (row, col) is out of + * range. Never throws. + */ +export function findCellJsonPosition(editorText: string, sheetName: string, rowIndex: number, colIndex: number): number { + // JSON.stringify escapes embedded quotes/backslashes so the marker + // matches the way the sheet name actually appears in editorText. + const sheetStartMarker = `"name": ${JSON.stringify(sheetName)}`; + const dataStartMarker = `"data": [`; + + const sheetPos = editorText.indexOf(sheetStartMarker); + if (sheetPos === -1) return -1; + + const dataPos = editorText.indexOf(dataStartMarker, sheetPos); + if (dataPos === -1) return -1; + + const rowStart = findRowOpenBracket(editorText, dataPos + dataStartMarker.length, rowIndex); + if (rowStart === -1) return -1; + + return findCellStartWithinRow(editorText, rowStart, colIndex); +} diff --git a/src/engine/math-ops.ts b/src/engine/math-ops.ts new file mode 100644 index 0000000..44e4667 --- /dev/null +++ b/src/engine/math-ops.ts @@ -0,0 +1,98 @@ +/** + * Rounding, modulo and significance math for the mathematical functions. + * + * Pure number-in / (number | formula-error-value)-out helpers. The rounding + * direction and the domain rules are exactly where the engine diverged from + * Excel (rounding toward +∞ instead of away from zero, modulo taking the + * dividend's sign, negative bases and domains slipping through as NaN/∞), so + * they are worth testing apart from the argument-reading handlers. + */ + +import { DIV_ZERO_ERROR, NUM_ERROR, isSpreadsheetErrorValue, type SpreadsheetError } from "./spreadsheet-errors"; + +const DECIMAL_BASE = 10; + +const scale = (digits: number): number => Math.pow(DECIMAL_BASE, digits); + +/** Excel ROUND: half away from zero (JS `Math.round` breaks half toward +∞, so + * `ROUND(-2.5, 0)` came back -2 instead of -3). */ +export function roundTo(value: number, digits: number): number { + const factor = scale(digits); + return (Math.sign(value) * Math.round(Math.abs(value) * factor)) / factor; +} + +/** Excel ROUNDUP: away from zero. */ +export function roundUpTo(value: number, digits: number): number { + const factor = scale(digits); + return (Math.sign(value) * Math.ceil(Math.abs(value) * factor)) / factor; +} + +/** Excel ROUNDDOWN: toward zero. */ +export function roundDownTo(value: number, digits: number): number { + const factor = scale(digits); + return (Math.sign(value) * Math.floor(Math.abs(value) * factor)) / factor; +} + +/** Whether `value` and `significance` point the same way; opposite signs are the + * #NUM! case for FLOOR / CEILING. A zero value is always in range. */ +const sameSign = (value: number, significance: number): boolean => value === 0 || Math.sign(value) === Math.sign(significance); + +/** Excel FLOOR: nearest multiple of `significance` toward zero; opposite signs + * are #NUM!, and a zero significance is #DIV/0! — the division by the + * significance is what FLOOR reports, so it wins over the sign check. */ +export function floorToSignificance(value: number, significance: number): number | SpreadsheetError { + if (significance === 0) return DIV_ZERO_ERROR; + if (!sameSign(value, significance)) return NUM_ERROR; + return Math.floor(value / significance) * significance; +} + +/** Excel CEILING: nearest multiple of `significance` away from zero; opposite + * signs are #NUM!, and a zero significance is 0 — deliberately NOT FLOOR's + * #DIV/0!, an asymmetry Excel keeps and this engine has to match. */ +export function ceilingToSignificance(value: number, significance: number): number | SpreadsheetError { + if (significance === 0) return 0; + if (!sameSign(value, significance)) return NUM_ERROR; + return Math.ceil(value / significance) * significance; +} + +/** Excel MOD: result takes the divisor's sign (`MOD(-3, 2) === 1`); dividing by + * zero is #DIV/0!, not a silent 0. */ +export function modulo(value: number, divisor: number): number | SpreadsheetError { + if (divisor === 0) return DIV_ZERO_ERROR; + return value - divisor * Math.floor(value / divisor); +} + +/** Excel POWER: a negative base with a non-integer exponent has no real root, so + * it is #NUM! rather than JS's NaN. */ +export function power(base: number, exponent: number): number | SpreadsheetError { + if (base < 0 && !Number.isInteger(exponent)) return NUM_ERROR; + return Math.pow(base, exponent); +} + +/** Natural log guarded to its domain: `LN(x)` for `x <= 0` is #NUM!, not + * -∞ / NaN. Reused by LN and LOG. */ +export function safeLog(value: number): number | SpreadsheetError { + if (value <= 0) return NUM_ERROR; + return Math.log(value); +} + +/** Excel LOG(value, base): the base must also be positive and not 1, or the + * result is #NUM! rather than the ∞ / NaN a bare division would give. */ +export function logWithBase(value: number, base: number): number | SpreadsheetError { + const lnValue = safeLog(value); + if (isSpreadsheetErrorValue(lnValue)) return lnValue; + if (base <= 0 || base === 1) return NUM_ERROR; + return lnValue / Math.log(base); +} + +/** Base-10 log guarded to its domain (`LOG10(x)` for `x <= 0` is #NUM!). */ +export function safeLog10(value: number): number | SpreadsheetError { + if (value <= 0) return NUM_ERROR; + return Math.log10(value); +} + +/** Square root guarded to its domain (`SQRT(x)` for `x < 0` is #NUM!, not NaN). */ +export function safeSqrt(value: number): number | SpreadsheetError { + if (value < 0) return NUM_ERROR; + return Math.sqrt(value); +} diff --git a/src/engine/numericCoercion.ts b/src/engine/numericCoercion.ts new file mode 100644 index 0000000..2cd3b3b --- /dev/null +++ b/src/engine/numericCoercion.ts @@ -0,0 +1,61 @@ +/** + * Numeric coercion helpers for the spreadsheet engine. + * + * Two intentionally different reads share one string parser (`parseNumericString`): + * - `registry.toNumber` — lenient, for range aggregation (SUM / AVERAGE / …): + * anything unreadable becomes 0. PINNED (booleans are 0, not Excel's 1/0). + * - `toScalarNumber` — strict, for single-value math functions (ABS / SIGN): + * booleans are 1/0 and non-numeric text is `#VALUE!`, matching Excel. + * + * `holdsNumber` asks the question neither read can answer once it has coerced: + * was there a number here at all, or is this 0 the reading of something else. + */ + +import type { CellValue } from "./types"; +import { VALUE_ERROR, type SpreadsheetError } from "./spreadsheet-errors"; + +const numberOrNull = (num: number): number | null => (isNaN(num) ? null : num); + +/** + * Parse a spreadsheet string as a number, or `null` when it holds no number. + * + * Mirrors the engine's long-standing lenient read: a percentage, a currency + * amount, or a thousands-separated value, else a bare `parseFloat` (which reads a + * leading number out of "12abc"). The branch order is load-bearing — each strips + * only its own characters, so a string mixing "%" and "$" fails in the first. + */ +export function parseNumericString(value: string): number | null { + if (value.includes("%")) { + const num = numberOrNull(parseFloat(value.replace("%", "").trim())); + return num === null ? null : num / 100; + } + if (value.includes("$")) return numberOrNull(parseFloat(value.replace(/[$,]/g, "").trim())); + if (value.includes(",")) return numberOrNull(parseFloat(value.replace(/,/g, "").trim())); + return numberOrNull(parseFloat(value)); +} + +/** + * Whether a value holds a number at all, under the same reading `toNumber` uses. + * + * `toNumber` answers 0 for text, for a boolean and for a genuine 0 alike, so a + * caller that must tell "no number here" from "the number zero" — COUNT — cannot + * ask it. Booleans are not numbers here, matching `toNumber`'s PINNED behaviour. + */ +export function holdsNumber(value: CellValue): boolean { + if (typeof value === "number") return !isNaN(value); + if (typeof value !== "string") return false; + return parseNumericString(value) !== null; +} + +/** + * Strict scalar coercion for single-value math functions (ABS, SIGN). Follows + * Excel's scalar rules: a boolean is 1 / 0 and genuinely non-numeric text is + * `#VALUE!` rather than a silent 0. A partly-numeric string still yields its + * leading number, matching the engine's other numeric reads. + */ +export function toScalarNumber(value: CellValue): number | SpreadsheetError { + if (typeof value === "number") return value; + if (typeof value === "boolean") return value ? 1 : 0; + if (typeof value !== "string") return value; + return parseNumericString(value) ?? VALUE_ERROR; +} diff --git a/src/engine/parser.ts b/src/engine/parser.ts index 504eef1..d040eb7 100644 --- a/src/engine/parser.ts +++ b/src/engine/parser.ts @@ -4,7 +4,6 @@ * Handles Excel A1 notation parsing and conversion */ -import type { CellRef, RangeRef } from "./types"; /** * Convert Excel column letters to 0-based index @@ -38,106 +37,3 @@ export function indexToColumn(index: number): string { } return col; } - -/** - * Parse a cell reference to its components - * Supports: A1, $A$1, $A1, A$1, Sheet1!A1, 'My Sheet'!A1 - * - * @param ref - Cell reference string - * @returns Parsed cell reference object - */ -export function parseCellRef(ref: string): CellRef { - let cellRef = ref; - let sheetName: string | undefined; - - // Check for cross-sheet reference (e.g., 'Sheet Name'!B2 or Sheet1!B2) - const sheetMatch = ref.match(/^(?:'([^']+)'|([^!]+))!(.+)$/); - if (sheetMatch) { - sheetName = sheetMatch[1] || sheetMatch[2]; // Quoted or unquoted sheet name - cellRef = sheetMatch[3]; // Cell reference part - } - - // Parse absolute references ($A$1) - const absoluteRow = cellRef.includes("$") && cellRef.match(/\$\d+/); - const absoluteCol = cellRef.includes("$") && cellRef.match(/\$[A-Z]+/); - - // Remove $ symbols - const cleanRef = cellRef.replace(/\$/g, ""); - const match = cleanRef.match(/^([A-Z]+)(\d+)$/); - - if (!match) { - throw new Error(`Invalid cell reference: ${ref}`); - } - - const col = columnToIndex(match[1]); - const row = parseInt(match[2]) - 1; // 1-indexed to 0-indexed - - const result: CellRef = { row, col }; - - if (sheetName) { - result.sheet = sheetName; - } - - if (absoluteRow || absoluteCol) { - result.absolute = { - row: !!absoluteRow, - col: !!absoluteCol, - }; - } - - return result; -} - -/** - * Parse a range reference to its components - * Supports: A1:B10, $A$1:$B$10, Sheet1!A1:B10 - * - * @param range - Range reference string - * @returns Parsed range reference object - */ -export function parseRangeRef(range: string): RangeRef { - // Use non-greedy match to improve performance - const colonIndex = range.lastIndexOf(":"); - if (colonIndex === -1) { - throw new Error(`Invalid range reference: ${range}`); - } - - const start = parseCellRef(range.substring(0, colonIndex)); - const end = parseCellRef(range.substring(colonIndex + 1)); - - return { start, end }; -} - -/** - * Convert a cell reference object back to A1 notation - * - * @param ref - Cell reference object - * @returns A1 notation string - */ -export function cellRefToA1(ref: CellRef): string { - const col = indexToColumn(ref.col); - const row = ref.row + 1; // 0-based to 1-based - - let result = ""; - - if (ref.absolute?.col) { - result += "$"; - } - result += col; - - if (ref.absolute?.row) { - result += "$"; - } - result += row; - - if (ref.sheet) { - // Quote sheet name if it contains spaces - if (ref.sheet.includes(" ")) { - result = `'${ref.sheet}'!${result}`; - } else { - result = `${ref.sheet}!${result}`; - } - } - - return result; -} diff --git a/src/engine/registry.ts b/src/engine/registry.ts index f332ea4..6242674 100644 --- a/src/engine/registry.ts +++ b/src/engine/registry.ts @@ -6,22 +6,54 @@ */ import type { CellValue } from "./types"; +import { parseNumericString } from "./numericCoercion"; export type { CellValue }; export type CellGetter = (ref: string) => CellValue; export type RangeGetter = (range: string) => CellValue[]; export type RawRangeGetter = (range: string) => CellValue[]; export interface FunctionContext { + /** The registry name the formula invoked, so a handler can report a failure + * the way the evaluator names it. */ + functionName: string; getCellValue: CellGetter; getRangeValues: RangeGetter; - getRangeValuesRaw?: RawRangeGetter; + getRangeValuesRaw?: RawRangeGetter | undefined; evaluateFormula: (formula: string) => CellValue; } -export type FunctionHandler = ( - args: string[], - context: FunctionContext, -) => CellValue; +/** The too-few-arguments error, raised by the evaluator from the registry's + * `minArgs` before any handler runs — and by `requiredArg` for the argument a + * handler actually reads, so both report one wording. */ +export const tooFewArgumentsError = (funcName: string, minArgs: number): Error => + new Error(`${funcName} requires at least ${minArgs} argument${minArgs !== 1 ? "s" : ""}`); + +/** The argument at `index`, which the caller has already established is there — + * the registry's `minArgs` for a mandatory argument, an `args.length` branch for + * an optional one. An absent one means that guarantee is wrong, so it raises the + * arity error the guarantee should have raised. Never a default value: a + * substituted 0 or "" computes a plausible wrong answer, which is worse than + * the error the formula deserves. */ +export const requiredArg = (context: FunctionContext, args: string[], index: number): string => { + const arg = args[index]; + if (arg === undefined) throw tooFewArgumentsError(context.functionName, index + 1); + return arg; +}; + +/** The range reader that keeps every cell, for functions whose answer depends on + * a range's POSITIONS rather than its numbers. + * + * `getRangeValues` drops non-numeric cells, which silently renumbers the rows + * underneath: a criteria/value pair read that way falls out of alignment and + * aggregates a different row, and an index returned from it points at the wrong + * one. SUMIF/AVERAGEIF were moved here in #2358; MATCH/XLOOKUP in #2765. + * + * Falls back to the numeric reader because `getRangeValuesRaw` is optional on + * `FunctionContext` — a host that predates it keeps its old behaviour rather + * than throwing. */ +export const rawRangeReader = (context: FunctionContext): RangeGetter => context.getRangeValuesRaw ?? context.getRangeValues; + +export type FunctionHandler = (args: string[], context: FunctionContext) => CellValue; export interface FunctionDefinition { name: string; @@ -70,35 +102,13 @@ class FunctionRegistry { export const functionRegistry = new FunctionRegistry(); /** - * Helper function to convert a value to a number + * Lenient numeric coercion for range aggregation: anything unreadable is 0. + * PINNED behaviour (booleans are 0, not Excel's 1/0) — the string parsing lives + * in numericCoercion.parseNumericString, shared with the strict scalar read. */ export function toNumber(value: CellValue): number { if (typeof value === "number") return value; - - // Handle percentage strings like "5%" or "0.4167%" - if (typeof value === "string" && value.includes("%")) { - const numericPart = value.replace("%", "").trim(); - const num = parseFloat(numericPart); - return isNaN(num) ? 0 : num / 100; - } - - // Handle currency strings like "$1,000" or "$1,000.00" - if (typeof value === "string" && value.includes("$")) { - const numericPart = value.replace(/[$,]/g, "").trim(); - const num = parseFloat(numericPart); - return isNaN(num) ? 0 : num; - } - - // Handle comma-separated numbers like "1,000" - if (typeof value === "string" && value.includes(",")) { - const numericPart = value.replace(/,/g, "").trim(); - const num = parseFloat(numericPart); - return isNaN(num) ? 0 : num; - } - - // Handle regular numeric strings - const num = parseFloat(String(value)); - return isNaN(num) ? 0 : num; + return parseNumericString(String(value)) ?? 0; } /** @@ -108,6 +118,34 @@ export function toString(value: CellValue): string { return String(value); } +const REGEXP_METACHARACTERS = /[.*+?^${}()|[\]\\]/; + +const escapeRegExpChar = (char: string): string => (REGEXP_METACHARACTERS.test(char) ? `\\${char}` : char); + +// One criteria token: `~x` (an escaped character) or any single character. The +// escape branch needs a following character, so a TRAILING `~` falls through to +// the single-character branch and stands for itself. +const CRITERIA_TOKEN = /~([\s\S])|[\s\S]/g; + +const criteriaRegexSource = (pattern: string): string => + pattern.replace(CRITERIA_TOKEN, (token, escaped: string | undefined) => { + if (escaped !== undefined) return escapeRegExpChar(escaped); + if (token === "*") return ".*"; + if (token === "?") return "."; + return escapeRegExpChar(token); + }); + +/** + * Match text the way a spreadsheet criteria does: case-insensitively, with `*` + * standing for any run of characters, `?` for exactly one, and `~` escaping the + * next character. A plain `String(v) === criteria` missed both — `COUNTIF(range, + * "yes")` skipped a cell holding `Yes`, and `"A*"` was compared literally. + */ +function textMatcher(pattern: string): (text: string) => boolean { + const regex = new RegExp(`^${criteriaRegexSource(pattern)}$`, "iu"); + return (text) => regex.test(text); +} + /** * Helper to parse criteria for conditional functions like COUNTIF, SUMIF * Returns a comparison function that tests if a value matches the criteria @@ -118,11 +156,13 @@ export function parseCriteria(criteria: string): (value: CellValue) => boolean { // Check for comparison operators // eslint-disable -- sonarjs/slow-regex - const opMatch = trimmedCriteria.match(/^([><=!]+)(.+)$/); - if (opMatch) { - const [, op, value] = opMatch; + const [, op, value] = trimmedCriteria.match(/^([><=!]+)(.+)$/) ?? []; + if (op !== undefined && value !== undefined) { const numValue = parseFloat(value); + // `=` / `<>` compare like the bare criteria does — case-insensitive text + // with wildcards, or the number. `<>` is exactly its negation. + const equals = matchesTextOrNumber(value); switch (op) { case ">": return (v) => toNumber(v) > numValue; @@ -134,20 +174,23 @@ export function parseCriteria(criteria: string): (value: CellValue) => boolean { return (v) => toNumber(v) <= numValue; case "=": case "==": - return (v) => String(v) === value || toNumber(v) === numValue; + return equals; case "!=": case "<>": - return (v) => String(v) !== value && toNumber(v) !== numValue; + return (v) => !equals(v); default: return () => false; } } - // Exact match (string or number) - return (v) => { - const strMatch = String(v) === trimmedCriteria; - const numCriteria = parseFloat(trimmedCriteria); - const numMatch = !isNaN(numCriteria) && toNumber(v) === numCriteria; - return strMatch || numMatch; - }; + return matchesTextOrNumber(trimmedCriteria); +} + +/** A value matches when its text matches the criteria (case-insensitively, with + * wildcards) or, for a numeric criteria, when its number is equal. */ +function matchesTextOrNumber(criteria: string): (value: CellValue) => boolean { + const matchesText = textMatcher(criteria); + const numCriteria = parseFloat(criteria); + const hasNumber = !isNaN(numCriteria); + return (value) => matchesText(String(value)) || (hasNumber && toNumber(value) === numCriteria); } diff --git a/src/engine/responseDecoder.ts b/src/engine/responseDecoder.ts new file mode 100644 index 0000000..1cae216 --- /dev/null +++ b/src/engine/responseDecoder.ts @@ -0,0 +1,68 @@ +/** + * Decode the `/api/files/content` response for a spreadsheet file + * into an "ok with sheets" / "error with message" discriminated + * union. Extracted from `fetchSheets` in + * `src/plugins/spreadsheet/View.vue` where the decision tree was + * inlined as several nested try/catch + if branches. + * + * Pure — no fetch, no refs. Takes the parsed JSON body and returns + * a result the caller can pattern-match on. Tested in + * `test/plugins/spreadsheet/engine/test_responseDecoder.ts`. + */ + +import type { SheetData } from "./types.js"; +import { errorMessage } from "./guards"; + +/** Shape of the `/api/files/content` response we care about. The + * server returns more fields (kind, size, modifiedMs, …) but this + * decoder only depends on the three that drive branching. */ +export interface FilesContentResponseLike { + kind?: string; + content?: string; + message?: string; +} + +export type DecodeResult = { kind: "ok"; sheets: SheetData[] } | { kind: "error"; message: string }; + +/** + * Turn a parsed `/files/content` body into an OK/error decision: + * + * - `kind` present and not "text" → error with the server's message + * (e.g. "too-large", "binary"). Spreadsheets only live in text + * JSON files. + * - Missing or non-string `content` → error. + * - `content` is not valid JSON → error. + * - Parsed `content` is not an array → error (server should never + * return a non-array but the guard protects downstream render). + * - Otherwise → ok with the sheets array. + */ +export function decodeSpreadsheetResponse(body: FilesContentResponseLike): DecodeResult { + if (body.kind && body.kind !== "text") { + return { + kind: "error", + message: body.message ?? `Cannot load spreadsheet: ${body.kind}`, + }; + } + if (typeof body.content !== "string") { + return { kind: "error", message: "Spreadsheet file has no content" }; + } + let parsed: unknown; + try { + parsed = JSON.parse(body.content); + } catch (err) { + return { + kind: "error", + message: `Spreadsheet JSON is malformed: ${errorMessage(err, "parse error")}`, + }; + } + if (!Array.isArray(parsed)) { + return { + kind: "error", + message: "Spreadsheet content is not an array of sheets", + }; + } + // Array.isArray narrows to unknown[]; we trust the server contract + // and type the local explicitly rather than using an inline `as`. + const sheets: SheetData[] = parsed; + return { kind: "ok", sheets }; +} diff --git a/src/engine/spreadsheet-errors.ts b/src/engine/spreadsheet-errors.ts new file mode 100644 index 0000000..8e2040d --- /dev/null +++ b/src/engine/spreadsheet-errors.ts @@ -0,0 +1,82 @@ +/** + * Formula errors as a distinct VALUE type. + * + * A `#NUM!` produced by `SQRT(-1)` and the text `"#NUM!"` produced by + * `CONCAT("#N","UM!")` used to be the same string, so IFERROR could not tell a + * real error from text that merely spells one (#2451). An error is now its own + * value carrying the code, which is what the error-aware functions check; the + * display pass renders it back to `#NUM!` so cells look unchanged. + */ + +export const SPREADSHEET_ERRORS = ["#NULL!", "#DIV/0!", "#VALUE!", "#REF!", "#NAME?", "#NUM!", "#N/A", "#ERROR!"] as const; + +export type SpreadsheetErrorCode = (typeof SPREADSHEET_ERRORS)[number]; + +/** A formula error, distinct from any string. `toString` yields the code so the + * value renders as `#NUM!` wherever the engine coerces a cell value to text. */ +export class SpreadsheetError { + constructor(readonly code: SpreadsheetErrorCode) {} + + toString(): string { + return this.code; + } + + toJSON(): string { + return this.code; + } +} + +export const NULL_ERROR = new SpreadsheetError("#NULL!"); +export const DIV_ZERO_ERROR = new SpreadsheetError("#DIV/0!"); +export const VALUE_ERROR = new SpreadsheetError("#VALUE!"); +export const REF_ERROR = new SpreadsheetError("#REF!"); +export const NAME_ERROR = new SpreadsheetError("#NAME?"); +export const NUM_ERROR = new SpreadsheetError("#NUM!"); +export const NA_ERROR = new SpreadsheetError("#N/A"); +export const UNKNOWN_ERROR = new SpreadsheetError("#ERROR!"); + +// One instance per code: two errors of the same kind compare equal, which is how +// error strings behaved and what the condition/comparison paths still rely on. +const ERROR_VALUES: Record = { + "#NULL!": NULL_ERROR, + "#DIV/0!": DIV_ZERO_ERROR, + "#VALUE!": VALUE_ERROR, + "#REF!": REF_ERROR, + "#NAME?": NAME_ERROR, + "#NUM!": NUM_ERROR, + "#N/A": NA_ERROR, + "#ERROR!": UNKNOWN_ERROR, +}; + +/** The error value for a code. */ +export const spreadsheetError = (code: SpreadsheetErrorCode): SpreadsheetError => ERROR_VALUES[code]; + +const errorSet: ReadonlySet = new Set(SPREADSHEET_ERRORS); + +/** Whether a value is one of the error CODES as a plain string. Text that spells + * an error is not an error value — see `isSpreadsheetErrorValue`. */ +export function isSpreadsheetError(value: unknown): value is SpreadsheetErrorCode { + return typeof value === "string" && errorSet.has(value); +} + +/** Whether a value IS a formula error, as opposed to text that spells one. */ +export function isSpreadsheetErrorValue(value: unknown): value is SpreadsheetError { + return value instanceof SpreadsheetError; +} + +/** The code behind a value: an error value's own code, or the code a plain + * string spells. Used where a literal `#REF!` written into a cell must still + * poison the arithmetic that reads it, provenance aside. */ +export function errorCodeOf(value: unknown): SpreadsheetErrorCode | null { + if (isSpreadsheetErrorValue(value)) return value.code; + return isSpreadsheetError(value) ? value : null; +} + +/** Whether an evaluated result should be treated as an error: a formula error + * VALUE, a NaN / infinite number, or a missing value. This is what IFERROR + * catches — deliberately NOT a look-alike string, which is ordinary text. */ +export function isErrorResult(value: unknown): boolean { + if (value === null || value === undefined) return true; + if (typeof value === "number") return Number.isNaN(value) || !Number.isFinite(value); + return isSpreadsheetErrorValue(value); +} diff --git a/src/engine/textFormat.ts b/src/engine/textFormat.ts new file mode 100644 index 0000000..c27880e --- /dev/null +++ b/src/engine/textFormat.ts @@ -0,0 +1,108 @@ +/** + * Excel number-format patterns for the TEXT function. + * + * Deliberately separate from `formatter.formatNumber`, which renders CELL + * display values: the two disagree on defaults — Excel's `TEXT(0.5,"0%")` is + * `"50%"`, while a cell carrying the format `0%` has always displayed + * `"50.00%"` — so sharing the whole path would move every formatted cell in + * every stored workbook. Only the digit-grouping primitive is shared. + */ + +import { groupThousands } from "./formatter"; + +const PERCENT_SCALE = 100; + +// Characters that introduce an Excel format feature this module does not +// render (quoted literals, fill/skip, fractions, dates, text placeholder, +// negative/zero sections). Seeing one means the caller keeps its own fallback. +const UNSUPPORTED_LITERAL_CHARS = `?\\*_[]"@;/`; + +// One run of digit placeholders, optionally carrying grouping commas and a +// decimal point: the `#,##0.00` family. +const NUMERIC_CORE_RE = /[#0][#0,.]*/; +const PLACEHOLDER_RE = /[#0]/; +const TRAILING_ZEROS_RE = /0+$/; + +export interface NumberPattern { + prefix: string; + suffix: string; + useGrouping: boolean; + integerMinDigits: number; + minDecimals: number; + maxDecimals: number; + isPercent: boolean; +} + +const countOf = (text: string, chars: string): number => Array.from(text).filter((char) => chars.includes(char)).length; + +const hasUnsupportedLiteral = (text: string): boolean => Array.from(text).some((char) => UNSUPPORTED_LITERAL_CHARS.includes(char)); + +/** + * Split a format code into a leading literal, one numeric core and a trailing + * literal, or `null` when the code uses a feature this module does not render. + */ +export const parseNumberPattern = (pattern: string): NumberPattern | null => { + const core = NUMERIC_CORE_RE.exec(pattern); + if (!core) return null; + // A comma AFTER the last placeholder scales by a thousand in Excel. + if (core[0].endsWith(",")) return null; + + const prefix = pattern.slice(0, core.index); + const suffix = pattern.slice(core.index + core[0].length); + // A second placeholder run (`0.00E+00`, `# ?/?`) is a format of its own kind. + if (PLACEHOLDER_RE.test(suffix)) return null; + if (hasUnsupportedLiteral(prefix) || hasUnsupportedLiteral(suffix)) return null; + + const [integerPart, decimalPart = "", ...extraParts] = core[0].split("."); + if (integerPart === undefined || extraParts.length > 0) return null; + + return { + prefix, + suffix, + useGrouping: integerPart.includes(","), + integerMinDigits: countOf(integerPart, "0"), + minDecimals: countOf(decimalPart, "0"), + maxDecimals: countOf(decimalPart, "#0"), + isPercent: prefix.includes("%") || suffix.includes("%"), + }; +}; + +// `0` keeps a decimal digit, `#` drops it once it is a trailing zero — so +// `0.0#` renders 0.5 as "0.5" but 0.25 as "0.25". +const trimOptionalZeros = (decimals: string, minDecimals: number): string => { + const kept = decimals.replace(TRAILING_ZEROS_RE, ""); + return kept.length >= minDecimals ? kept : decimals.slice(0, minDecimals); +}; + +const renderDigits = (absValue: number, pattern: NumberPattern): string => { + const fixed = absValue.toFixed(pattern.maxDecimals); + const point = fixed.indexOf("."); + const wholeDigits = point === -1 ? fixed : fixed.slice(0, point); + const decimalDigits = point === -1 ? "" : fixed.slice(point + 1); + const padded = wholeDigits.padStart(pattern.integerMinDigits, "0"); + const whole = pattern.useGrouping ? groupThousands(padded) : padded; + const decimals = trimOptionalZeros(decimalDigits, pattern.minDecimals); + return decimals ? `${whole}.${decimals}` : whole; +}; + +const isZeroText = (digits: string): boolean => Array.from(digits).every((char) => char === "0" || char === "," || char === "."); + +/** + * Render a number the way Excel's TEXT does for the common numeric format + * codes: `#,##0` grouping, `0.00` fixed decimals, `0.##` optional decimals, a + * literal prefix/suffix such as `$`, and a `%` that scales by 100. + * + * Returns `null` for a format code it does not render, so the caller can keep + * its own fallback rather than inventing a plausible wrong string. + */ +export const formatWithPattern = (value: number, pattern: string): string | null => { + if (!Number.isFinite(value)) return null; + const spec = parseNumberPattern(pattern); + if (!spec) return null; + + const scaled = spec.isPercent ? value * PERCENT_SCALE : value; + const digits = renderDigits(Math.abs(scaled), spec); + // The sign follows the ROUNDED digits: -0.001 under "0.00" reads "0.00", not "-0.00". + const sign = scaled < 0 && !isZeroText(digits) ? "-" : ""; + return `${sign}${spec.prefix}${digits}${spec.suffix}`; +}; diff --git a/src/engine/translateFormula.ts b/src/engine/translateFormula.ts new file mode 100644 index 0000000..9dfc0d3 --- /dev/null +++ b/src/engine/translateFormula.ts @@ -0,0 +1,69 @@ +/** + * Excel-formula → JS-expression translation + * + * Pure string transforms that turn the Excel-operator form of an + * already-substituted expression (cell references resolved, functions replaced) + * into the JavaScript form handed to `new Function`, plus the character + * allowlists that gate that evaluation. No engine or evaluator state is + * captured — every function here is input → output only. + */ + +/** Excel `^` exponentiation → JS `**`. + * + * Excel's `^` is left-associative (`2^3^2` = `(2^3)^2` = 64); JS `**` is + * right-associative (`2**3**2` = `2**(3**2)` = 512). This transform does not + * bridge that difference, so a chained `^` still evaluates the JS way — a known + * limitation tracked separately (#2359), pinned by the tests. */ +export function caretToPow(expr: string): string { + return expr.replace(/\^/g, "**"); +} + +/** Move the "inside a string literal" marker across one quote character. + * Empty marker means "not in a literal"; otherwise it holds the opening quote. */ +function toggleQuote(openQuote: string, char: string): string { + if (openQuote === "") return char; + return char === openQuote ? "" : openQuote; +} + +/** Excel `&` string concatenation → JS `+`, leaving any `&` inside a quoted + * string literal untouched. A quote toggles the "inside a literal" state only + * when it is not backslash-escaped. */ +export function replaceConcatOperator(expr: string): string { + const chars = expr.split(""); + const out: string[] = []; + let openQuote = ""; + chars.forEach((char, index) => { + const escaped = index > 0 && chars[index - 1] === "\\"; + if (!escaped && (char === '"' || char === "'")) { + openQuote = toggleQuote(openQuote, char); + } + out.push(char === "&" && openQuote === "" ? "+" : char); + }); + return out.join(""); +} + +/** Excel `=` equality → JS `==`, without disturbing `<=`, `>=` or `!=`. A single + * `=` is rewritten only when flanked by a non-`<>!` char on the left and a + * non-`=` on the right. + * + * Known limitations, pinned by the tests and tracked separately (#2359): the + * match consumes both flanking characters so replacements do not overlap + * (`5=6=7` → `5==6=7`, only the first rewritten); a `=` at either end of the + * expression has no left/right neighbour and is left alone (`5=` → `5=`); and a + * hand-typed `==` becomes `===` (its second `=` matches). Excel never emits the + * last two, so they only bite malformed input. */ +export function rewriteComparisonEq(expr: string): string { + return expr.replace(/([^<>!])=([^=])/g, "$1==$2"); +} + +/** Whether an expression contains only the characters an arithmetic evaluation + * may see: digits, ` + - * / ( ) . ` and spaces. Gates `new Function`. */ +export function isSafeArithmetic(expr: string): boolean { + return /^[\d+\-*/(). ]+$/.test(expr); +} + +/** Whether an expression contains only the characters a comparison evaluation + * may see: the arithmetic set plus ` < > ! = `. Gates `new Function`. */ +export function isSafeComparison(expr: string): boolean { + return /^[\d+\-*/(). <>!=]+$/.test(expr); +} diff --git a/src/engine/types.ts b/src/engine/types.ts index 747f60a..e530812 100644 --- a/src/engine/types.ts +++ b/src/engine/types.ts @@ -2,10 +2,17 @@ * Spreadsheet Engine Type Definitions */ -export type CellValue = number | string | boolean; +import type { SpreadsheetError } from "./spreadsheet-errors"; + +/** A value as it is STORED in a cell and serialized to the workbook JSON. A + * formula error is never stored — it only exists as a computed result. */ +export type StoredCellValue = number | string | boolean; + +/** A value as the engine COMPUTES it: a stored value, or a formula error. */ +export type CellValue = StoredCellValue | SpreadsheetError; export interface SpreadsheetCell { - v: CellValue; // Value or formula (formulas start with "=") + v: StoredCellValue; // Value or formula (formulas start with "=") f?: string; // Format code (e.g., "$#,##0.00") } @@ -57,7 +64,17 @@ export interface CalculationError { type: "circular" | "invalid_ref" | "div_zero" | "syntax" | "unknown"; } -export interface EngineOptions { +/** Per-calculation settings that change how cell CONTENT is read, as opposed + * to how the engine runs. Passed down explicitly so the engine stays pure — + * no ambient locale, no import-time state. */ +export interface CalculateOptions { + /** Read an ambiguous `A/B/YYYY` date as day-first. Only affects dates whose + * two leading numbers are both 12 or under; anything else decides itself. + * See engine/date-locale.ts for how a locale maps onto this. */ + preferDDMMYYYY?: boolean; +} + +export interface EngineOptions extends CalculateOptions { maxIterations?: number; // For circular reference detection enableCrossSheetRefs?: boolean; // Default: true strictMode?: boolean; // Throw on errors vs. return 0 diff --git a/tests/engine/cellAccess.ts b/tests/engine/cellAccess.ts new file mode 100644 index 0000000..4c90f2d --- /dev/null +++ b/tests/engine/cellAccess.ts @@ -0,0 +1,18 @@ +// Reading a cell out of a calculated grid. `0`, `""` and `false` are all +// legitimate cell values, so absence has to be tested explicitly — a truthiness +// check would reject a correct answer, and an assertion on `undefined` would let +// a missing row pass unnoticed. + +import type { CellValue } from "../../src/engine/types.ts"; + +export const rowAt = (grid: CellValue[][], row: number): CellValue[] => { + const line = grid[row]; + if (line === undefined) throw new Error(`calculated sheet has no row ${row}`); + return line; +}; + +export const cellAt = (grid: CellValue[][], row: number, col: number): CellValue => { + const cell = rowAt(grid, row)[col]; + if (cell === undefined) throw new Error(`calculated sheet has no cell (${row}, ${col})`); + return cell; +}; diff --git a/tests/engine/fixtures/README.md b/tests/engine/fixtures/README.md index b66c814..7f154ac 100644 --- a/tests/engine/fixtures/README.md +++ b/tests/engine/fixtures/README.md @@ -464,3 +464,12 @@ This creates one test case per fixture file automatically! ``` Done! No code changes, no configuration - just add two JSON files. + +## A note on the financial fixture + +`financial-functions` expects the sign convention Excel itself uses: a payment +leaves the account, so `PMT`, `IPMT` and `PPMT` are negative when the present +value is positive. Two cells in `expected/financial-functions.json` disagreed +with that and were corrected when the engine was replaced — the interest payment +had the wrong sign, and the principal payment was not `PMT - IPMT`. If a change +makes either go positive again, it is the change that is wrong. diff --git a/tests/engine/fixtures/expected/financial-functions.json b/tests/engine/fixtures/expected/financial-functions.json index 1429d2f..d4b338c 100644 --- a/tests/engine/fixtures/expected/financial-functions.json +++ b/tests/engine/fixtures/expected/financial-functions.json @@ -3,8 +3,8 @@ ["Monthly Payment", "PMT(6%/12, 30*12, 250000)", "-$1,498.88", "Mortgage"], ["Future Value", "FV(5%/12, 12*5, -200, -5000, 0)", "$20,018.01", "Savings"], ["Present Value", "PV(4%/12, 10*12, -300)", "$29,631.05", "Annuity"], - ["Interest Payment P1", "IPMT(5%/12,1,24,10000)", "$41.67", "Loan"], - ["Principal Payment P1", "PPMT(5%/12,1,24,10000)", "-$480.38", "Loan"], + ["Interest Payment P1", "IPMT(5%/12,1,24,10000)", "-$41.67", "Loan"], + ["Principal Payment P1", "PPMT(5%/12,1,24,10000)", "-$397.05", "Loan"], ["Periods Needed", "NPER(7%/12,-500,15000)", "33.07", "Goal"], ["Loan Rate", "RATE(36,-300,9000)", "0.0102", "Guess 10%"], ["", "", "", ""], diff --git a/tests/engine/parser.test.ts b/tests/engine/parser.test.ts index fd862c9..7e626d0 100644 --- a/tests/engine/parser.test.ts +++ b/tests/engine/parser.test.ts @@ -1,15 +1,17 @@ /** * Parser Unit Tests + * + * The A1-reference PARSING tests that used to live here went with + * parseCellRef / parseRangeRef / cellRefToA1: the engine resolves references + * through formulaRefs (expandRange / expandRangeOrCell) now, and those are + * covered in tests/engine/test_expandRangeOrCell.ts and + * test_cellRefSubstitution.ts. Column conversion stays because it is still the + * engine's own public helper, used by the Vue view. */ import { describe, test, expect } from "vitest"; -import { - columnToIndex, - indexToColumn, - parseCellRef, - parseRangeRef, - cellRefToA1, -} from "../../src/engine/parser"; +import { columnToIndex, indexToColumn } from "../../src/engine/parser"; + describe("Parser - Column Conversion", () => { describe("columnToIndex", () => { @@ -68,156 +70,3 @@ describe("Parser - Column Conversion", () => { } }); }); - -describe("Parser - Cell References", () => { - describe("parseCellRef - basic references", () => { - test("parses simple cell reference", () => { - expect(parseCellRef("A1")).toEqual({ row: 0, col: 0 }); - expect(parseCellRef("B2")).toEqual({ row: 1, col: 1 }); - expect(parseCellRef("Z26")).toEqual({ row: 25, col: 25 }); - }); - - test("parses double letter columns", () => { - expect(parseCellRef("AA1")).toEqual({ row: 0, col: 26 }); - expect(parseCellRef("AB10")).toEqual({ row: 9, col: 27 }); - }); - }); - - describe("parseCellRef - absolute references", () => { - test("parses fully absolute reference ($A$1)", () => { - expect(parseCellRef("$A$1")).toEqual({ - row: 0, - col: 0, - absolute: { row: true, col: true }, - }); - }); - - test("parses column-absolute reference ($A1)", () => { - expect(parseCellRef("$A1")).toEqual({ - row: 0, - col: 0, - absolute: { row: false, col: true }, - }); - }); - - test("parses row-absolute reference (A$1)", () => { - expect(parseCellRef("A$1")).toEqual({ - row: 0, - col: 0, - absolute: { row: true, col: false }, - }); - }); - - test("parses mixed absolute reference ($B$5)", () => { - expect(parseCellRef("$B$5")).toEqual({ - row: 4, - col: 1, - absolute: { row: true, col: true }, - }); - }); - }); - - describe("parseCellRef - cross-sheet references", () => { - test("parses simple cross-sheet reference", () => { - expect(parseCellRef("Sheet1!A1")).toEqual({ - row: 0, - col: 0, - sheet: "Sheet1", - }); - }); - - test("parses quoted sheet name with spaces", () => { - expect(parseCellRef("'My Sheet'!B2")).toEqual({ - row: 1, - col: 1, - sheet: "My Sheet", - }); - }); - - test("parses cross-sheet with absolute reference", () => { - expect(parseCellRef("Sheet1!$A$1")).toEqual({ - row: 0, - col: 0, - sheet: "Sheet1", - absolute: { row: true, col: true }, - }); - }); - }); - - describe("parseCellRef - error handling", () => { - test("throws on invalid reference", () => { - expect(() => parseCellRef("invalid")).toThrow("Invalid cell reference"); - expect(() => parseCellRef("123")).toThrow("Invalid cell reference"); - expect(() => parseCellRef("A")).toThrow("Invalid cell reference"); - }); - }); -}); - -describe("Parser - Range References", () => { - test("parses simple range", () => { - expect(parseRangeRef("A1:B2")).toEqual({ - start: { row: 0, col: 0 }, - end: { row: 1, col: 1 }, - }); - }); - - test("parses large range", () => { - expect(parseRangeRef("A1:Z100")).toEqual({ - start: { row: 0, col: 0 }, - end: { row: 99, col: 25 }, - }); - }); - - test("parses range with absolute references", () => { - expect(parseRangeRef("$A$1:$B$10")).toEqual({ - start: { row: 0, col: 0, absolute: { row: true, col: true } }, - end: { row: 9, col: 1, absolute: { row: true, col: true } }, - }); - }); - - test("throws on invalid range", () => { - expect(() => parseRangeRef("A1")).toThrow("Invalid range reference"); - expect(() => parseRangeRef("invalid:range")).toThrow( - "Invalid cell reference", - ); - }); -}); - -describe("Parser - Cell Reference to A1", () => { - test("converts basic cell ref to A1", () => { - expect(cellRefToA1({ row: 0, col: 0 })).toBe("A1"); - expect(cellRefToA1({ row: 1, col: 1 })).toBe("B2"); - expect(cellRefToA1({ row: 25, col: 25 })).toBe("Z26"); - }); - - test("converts absolute references to A1", () => { - expect( - cellRefToA1({ row: 0, col: 0, absolute: { row: true, col: true } }), - ).toBe("$A$1"); - expect( - cellRefToA1({ row: 0, col: 0, absolute: { row: false, col: true } }), - ).toBe("$A1"); - expect( - cellRefToA1({ row: 0, col: 0, absolute: { row: true, col: false } }), - ).toBe("A$1"); - }); - - test("converts cross-sheet references to A1", () => { - expect(cellRefToA1({ row: 0, col: 0, sheet: "Sheet1" })).toBe("Sheet1!A1"); - expect(cellRefToA1({ row: 1, col: 1, sheet: "My Sheet" })).toBe( - "'My Sheet'!B2", - ); - }); - - test("round-trip conversion works", () => { - const testRefs = ["A1", "$A$1", "$A1", "A$1", "Sheet1!A1", "'My Sheet'!B2"]; - - for (const ref of testRefs) { - const parsed = parseCellRef(ref); - const converted = cellRefToA1(parsed); - // Re-parse to normalize (e.g., Sheet1!A1 vs 'Sheet1'!A1) - const reparsed = parseCellRef(converted); - expect(reparsed).toEqual(parsed); - } - }); -}); diff --git a/tests/engine/run-evaluator-tests.ts b/tests/engine/run-evaluator-tests.ts index 7bcd9bd..923fbc8 100644 --- a/tests/engine/run-evaluator-tests.ts +++ b/tests/engine/run-evaluator-tests.ts @@ -50,12 +50,17 @@ function createContext( ranges: Record = {}, rawRanges?: Record, ): EvaluatorContext { + // A single cell answers the RANGE readers too, the way the real context does: + // it resolves every reference through expandRangeOrCell, so an aggregate like + // MAX(B1, A1:A3, 5) reads B1 via getRangeValues. A fake that answered nothing + // for "B1" dropped that argument without saying so. + const oneCell = (ref: string): (number | string)[] => (ref in cells ? [cells[ref]] : []); const context: EvaluatorContext = { getCellValue: (ref: string) => cells[ref] ?? 0, getRangeValues: (range: string) => - ranges[range] ?? rawRanges?.[range] ?? [], + ranges[range] ?? rawRanges?.[range] ?? oneCell(range), getRangeValuesRaw: (range: string) => - rawRanges?.[range] ?? ranges[range] ?? [], + rawRanges?.[range] ?? ranges[range] ?? oneCell(range), evaluateFormula: (formula: string) => evaluateFormula(formula, context), }; return context; diff --git a/tests/engine/test_argCountValidation.ts b/tests/engine/test_argCountValidation.ts new file mode 100644 index 0000000..383702c --- /dev/null +++ b/tests/engine/test_argCountValidation.ts @@ -0,0 +1,128 @@ +// #2397: the handler-side `if (args.length ...) throw` guards were removed for +// every function whose arity is fully expressed by the registry minArgs/maxArgs. +// The evaluator (evaluator.ts) validates arity from the registry BEFORE calling +// the handler, so those guards were unreachable dead code with a divergent +// message ("requires N" vs the evaluator's "requires at least N"). +// +// These tests pin that (a) invalid arity now surfaces the evaluator's single +// consistent message for the removed-guard functions — including the financial +// handlers, whose only remaining validation is the evaluator after #2394/#2442 — +// and (b) the shapes the registry CANNOT express keep their handler guard: IFS's +// even-count requirement and IRR's empty-range check. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { SpreadsheetEngine, type SheetData, type CellValue } from "../../src/engine/index.ts"; +import { cellAt } from "./cellAccess.ts"; + +/** Calculate `sheet` and report the value, error type, and recorded message for + * the cell at (row, col). */ +function cellResult(sheet: SheetData, row: number, col: number): { value: CellValue; type?: string | undefined; message?: string | undefined } { + const result = new SpreadsheetEngine().calculate(sheet); + const entry = result.errors.find((err) => err.cell.row === row && err.cell.col === col); + return { value: cellAt(result.data, row, col), type: entry?.type, message: entry?.error }; +} + +/** A one-cell sheet holding `formula` at A1 (for formulas that need no other cells). */ +const soleFormula = (formula: string) => cellResult({ name: "S", data: [[{ v: formula }]] }, 0, 0); + +describe("#2397 removed handler guards — arity now enforced by the evaluator", () => { + it("too few args surfaces the evaluator's 'requires at least N' message (not the handler's)", () => { + const { value, type, message } = soleFormula("=ROUND(1)"); + assert.equal(value, "#ERROR!"); + assert.equal(type, "unknown"); + assert.equal(message, "ROUND requires at least 2 arguments"); + }); + + it("too many args surfaces the evaluator's 'accepts at most N' message", () => { + const { value, type, message } = soleFormula("=ROUND(1, 2, 3)"); + assert.equal(value, "#ERROR!"); + assert.equal(type, "unknown"); + assert.equal(message, "ROUND accepts at most 2 arguments"); + }); + + it("singular wording for a one-argument bound (ABS accepts at most 1 argument)", () => { + assert.equal(soleFormula("=ABS(1, 2)").message, "ABS accepts at most 1 argument"); + }); + + it("singular wording for a one-argument minimum (UPPER requires at least 1 argument)", () => { + assert.equal(soleFormula("=UPPER()").message, "UPPER requires at least 1 argument"); + }); + + it("a zero-arg function rejects any argument (PI accepts at most 0 arguments)", () => { + assert.equal(soleFormula("=PI(1)").message, "PI accepts at most 0 arguments"); + }); + + // Financial handlers used to carry the ONLY validation (IPMT/PPMT once called + // pmtHandler/fvHandler directly). After #2394/#2442 they call the pure + // computeIpmt/computePpmt and are reached only through the evaluator, so the + // evaluator is now their sole arity gate. + it("financial: FV too few args is the evaluator error (handler no longer guards)", () => { + assert.equal(soleFormula("=FV(0.05, 10)").message, "FV requires at least 3 arguments"); + }); + + it("financial: IPMT too many args is the evaluator error", () => { + assert.equal(soleFormula("=IPMT(0.05, 1, 10, 1000, 0, 0, 9)").message, "IPMT accepts at most 6 arguments"); + }); +}); + +describe("#2397 SUMIF / AVERAGEIF — arity fully expressed by registry [2,3], enforced by evaluator", () => { + // Formula in C1 so it never self-references the A/B ranges it reads. + const withData = (formula: string): SheetData => ({ + name: "S", + data: [ + [{ v: 1 }, { v: 10 }, { v: formula }], + [{ v: 2 }, { v: 20 }], + [{ v: 3 }, { v: 30 }], + ], + }); + + it("SUMIF with 1 arg is rejected with the consistent 'at least 2' message", () => { + const { value, message } = cellResult(withData("=SUMIF(A1:A3)"), 0, 2); + assert.equal(value, "#ERROR!"); + assert.equal(message, "SUMIF requires at least 2 arguments"); + }); + + it("SUMIF with 4 args is rejected with the consistent 'at most 3' message", () => { + const { value, message } = cellResult(withData('=SUMIF(A1:A3, ">0", B1:B3, C1)'), 0, 2); + assert.equal(value, "#ERROR!"); + assert.equal(message, "SUMIF accepts at most 3 arguments"); + }); + + it("AVERAGEIF with 4 args is rejected by the evaluator", () => { + assert.equal(cellResult(withData('=AVERAGEIF(A1:A3, ">0", B1:B3, C1)'), 0, 2).message, "AVERAGEIF accepts at most 3 arguments"); + }); + + it("a valid 3-arg SUMIF still computes", () => { + assert.equal(cellResult(withData('=SUMIF(A1:A3, ">1", B1:B3)'), 0, 2).value, 50); + }); +}); + +describe("#2397 kept guards — shapes the registry cannot express", () => { + // IFS is minArgs:2 with no maxArgs; the EVEN-count requirement is inexpressible + // by min/max, so the handler guard stays. This is the red-on-break target: if + // the guard is removed, a 3-arg IFS no longer reports this arity error. + it("IFS with an odd number of args is rejected by the kept handler guard", () => { + const sheet: SheetData = { name: "S", data: [[{ v: 5 }, { v: 0 }, { v: "=IFS(A1>0, 1, A1>5)" }]] }; + const { value, type, message } = cellResult(sheet, 0, 2); + assert.equal(value, "#ERROR!"); + assert.equal(type, "unknown"); + assert.equal(message, "IFS requires an even number of arguments (condition-value pairs)"); + }); + + it("a valid even-arg IFS still returns the matched value", () => { + const sheet: SheetData = { name: "S", data: [[{ v: 5 }, { v: '=IFS(A1>0, "yes")' }]] }; + assert.equal(cellResult(sheet, 0, 1).value, "yes"); + }); + + // IRR is minArgs:1/maxArgs:2 — its arg-count guard was removed — but the + // "at least one numeric value in the range" rule is not an arg count and is + // kept. D1:D3 is an empty range, so the kept guard fires. + it("IRR over an empty range is rejected by the kept empty-range guard", () => { + const sheet: SheetData = { name: "S", data: [[{ v: 1 }, { v: 2 }, { v: 3 }, { v: "=IRR(F1:F3)" }]] }; + const { value, type, message } = cellResult(sheet, 0, 3); + assert.equal(value, "#ERROR!"); + assert.equal(type, "unknown"); + assert.equal(message, "IRR requires at least one value"); + }); +}); diff --git a/tests/engine/test_calculateDateLocale.ts b/tests/engine/test_calculateDateLocale.ts new file mode 100644 index 0000000..8782039 --- /dev/null +++ b/tests/engine/test_calculateDateLocale.ts @@ -0,0 +1,98 @@ +// The date-order setting reaching the cells. `prefersDayFirst` and `parseDate` +// are each covered on their own; what this file checks is that the flag +// actually travels from `EngineOptions` down to every place that reads a date — +// including the FORMAT the cell is given, because parsing day-first while +// rendering month-first would just move the confusion rather than fix it. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +// Imported through the barrel, not the class module: `engine/index.ts` is what +// registers the built-in functions, so a direct import leaves DAY() unknown. +import { SpreadsheetEngine, type SheetData } from "../../src/engine/index.ts"; +import { cellAt, rowAt } from "./cellAccess.ts"; + +const sheetWith = (value: string): SheetData => ({ name: "S", data: [[{ v: value }]] }); + +const renderedCell = (value: string, preferDDMMYYYY: boolean): unknown => + cellAt(new SpreadsheetEngine({ preferDDMMYYYY }).calculate(sheetWith(value)).data, 0, 0); + +describe("date order reaches the cell", () => { + // The whole point: the same text means different days in different places, + // and neither reading raises. + it("reads an ambiguous date month-first by default and day-first when asked", () => { + assert.equal(renderedCell("03/04/2025", false), "03/04/2025"); + assert.equal(renderedCell("03/04/2025", true), "03/04/2025"); + }); + + // Rendering has to follow the reading, or a day-first user sees their April 3 + // written back as "04/03" under an MM/DD label. + it("renders in the order it read, so the cell round-trips", () => { + const monthFirst = new SpreadsheetEngine({ preferDDMMYYYY: false }).calculate(sheetWith("03/04/2025")); + const dayFirst = new SpreadsheetEngine({ preferDDMMYYYY: true }).calculate(sheetWith("03/04/2025")); + // Same displayed text, different underlying dates — which is exactly what a + // user in each locale expects to see. + assert.equal(cellAt(monthFirst.data, 0, 0), "03/04/2025"); + assert.equal(cellAt(dayFirst.data, 0, 0), "03/04/2025"); + }); + + // Month-name formats are exercised in test_formatter.ts; this only needs the + // shapes the day/month decision can reach. + it("leaves unambiguous dates alone under either setting", () => { + for (const prefer of [false, true]) { + assert.equal(renderedCell("13/04/2025", prefer), "13/04/2025", "13 can only be a day"); + assert.equal(renderedCell("2025-03-04", prefer), "2025-03-04", "ISO is not affected"); + } + }); + + it("leaves non-dates alone", () => { + for (const prefer of [false, true]) { + assert.equal(renderedCell("hello", prefer), "hello"); + } + }); + + // The default must stay month-first, or every existing sheet silently + // reinterprets on the next render. + it("defaults to month-first when the option is omitted", () => { + const engine = new SpreadsheetEngine(); + assert.equal(engine.getOptions().preferDDMMYYYY, false); + }); + + it("can be changed after construction", () => { + const engine = new SpreadsheetEngine(); + engine.setOptions({ preferDDMMYYYY: true }); + assert.equal(engine.getOptions().preferDDMMYYYY, true); + }); +}); + +describe("date order reaches formulas", () => { + const formulaResult = (cells: string[], preferDDMMYYYY: boolean): unknown => + rowAt(new SpreadsheetEngine({ preferDDMMYYYY }).calculate({ name: "S", data: [cells.map((cell) => ({ v: cell }))] }).data, 0).at(-1); + + // `DAY()` reads the serial, so it reports which number the parser took as the + // day — the clearest observable difference between the two settings. + it("applies the setting to a date held in a cell", () => { + assert.equal(formulaResult(["03/04/2025", "=DAY(A1)"], false), 4, "month-first: the 4 is the day"); + assert.equal(formulaResult(["03/04/2025", "=DAY(A1)"], true), 3, "day-first: the 3 is the day"); + }); + + // A date written INSIDE a formula never passes through the cell + // preprocessing — the evaluator parses it on its own, so the setting has to + // reach there separately. + it("applies the setting to a date literal inside a formula", () => { + assert.equal(formulaResult(['=DAY("03/04/2025")'], false), 4); + assert.equal(formulaResult(['=DAY("03/04/2025")'], true), 3); + }); + + // Arithmetic takes a third route: quoted dates are substituted into the + // expression before it is computed. The sign flip makes the difference + // unmissable — the same subtraction is 30 days one way and -30 the other. + it("applies the setting to a date literal in an arithmetic expression", () => { + assert.equal(formulaResult(["04/03/2025", '=A1-"03/04/2025"'], false), 30, "Apr 3 minus Mar 4"); + assert.equal(formulaResult(["04/03/2025", '=A1-"03/04/2025"'], true), -30, "Mar 4 minus Apr 3"); + }); + + // Cross-sheet references are NOT covered here: `=Data!A1` currently returns + // 3 for a date cell on main, independent of this setting, because the + // cross-sheet path never sees the date preprocessing (#2332). Adding the + // coverage belongs with that fix, not here. +}); diff --git a/tests/engine/test_cellBuilder.ts b/tests/engine/test_cellBuilder.ts new file mode 100644 index 0000000..3758551 --- /dev/null +++ b/tests/engine/test_cellBuilder.ts @@ -0,0 +1,130 @@ +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { buildCellFromInput, looksLikeFormula, parseNonStringInput } from "../../src/engine/cellBuilder.js"; + +describe("looksLikeFormula", () => { + it("detects function calls at the start", () => { + assert.equal(looksLikeFormula("SUM(A1:A3)"), true); + assert.equal(looksLikeFormula("MAX(B1, C1)"), true); + assert.equal(looksLikeFormula("-IF(A1>0, 1, 0)"), true); + }); + + it("detects cell reference + operator", () => { + assert.equal(looksLikeFormula("A1+B1"), true); + assert.equal(looksLikeFormula("AA10 * 2"), true); + assert.equal(looksLikeFormula("A1/B1"), true); + }); + + it("detects numeric arithmetic", () => { + assert.equal(looksLikeFormula("6/100"), true); + assert.equal(looksLikeFormula("5 * 2"), true); + assert.equal(looksLikeFormula("3+4"), true); + }); + + it("rejects plain text", () => { + assert.equal(looksLikeFormula("hello world"), false); + assert.equal(looksLikeFormula("apple pie"), false); + assert.equal(looksLikeFormula(""), false); + }); + + it("rejects bare numbers", () => { + assert.equal(looksLikeFormula("42"), false); + assert.equal(looksLikeFormula("3.14"), false); + }); + + it("rejects bare cell refs (no operator)", () => { + assert.equal(looksLikeFormula("A1"), false); + assert.equal(looksLikeFormula("AA10"), false); + }); +}); + +describe("parseNonStringInput", () => { + it("empty input → empty string", () => { + assert.equal(parseNonStringInput(""), ""); + assert.equal(parseNonStringInput(" "), ""); + }); + + it("formula → prefixed with =", () => { + assert.equal(parseNonStringInput("SUM(A1:A3)"), "=SUM(A1:A3)"); + assert.equal(parseNonStringInput("A1+B1"), "=A1+B1"); + assert.equal(parseNonStringInput(" 6/100 "), "=6/100"); + }); + + it("numeric → parsed as number", () => { + assert.equal(parseNonStringInput("42"), 42); + assert.equal(parseNonStringInput("3.14"), 3.14); + assert.equal(parseNonStringInput("-5"), -5); + }); + + it("non-formula non-number → raw string", () => { + assert.equal(parseNonStringInput("hello"), "hello"); + assert.equal(parseNonStringInput("yes"), "yes"); + }); + + it("rejects trailing garbage that parseFloat would silently accept", () => { + // parseFloat("42abc") returns 42; we want the string preserved. + assert.equal(parseNonStringInput("42abc"), "42abc"); + assert.equal(parseNonStringInput("100 USD"), "100 USD"); + assert.equal(parseNonStringInput("3.14xyz"), "3.14xyz"); + }); + + it("accepts scientific notation", () => { + assert.equal(parseNonStringInput("1e3"), 1000); + assert.equal(parseNonStringInput("-2.5E-2"), -0.025); + }); +}); + +describe("buildCellFromInput", () => { + it('type "string" → v is coerced String', () => { + const cell = buildCellFromInput({ type: "string", value: 42 }); + assert.deepEqual(cell, { v: "42" }); + }); + + it('type "string" with null value', () => { + const cell = buildCellFromInput({ type: "string", value: null }); + assert.deepEqual(cell, { v: "null" }); + }); + + it('type "number" with numeric formula', () => { + const cell = buildCellFromInput({ + type: "number", + value: "", + formula: "42", + }); + assert.deepEqual(cell, { v: 42 }); + }); + + it("type object with formula input", () => { + const cell = buildCellFromInput({ + type: "formula", + value: "", + formula: "SUM(A1:A3)", + }); + assert.deepEqual(cell, { v: "=SUM(A1:A3)" }); + }); + + it("attaches format when provided", () => { + const cell = buildCellFromInput({ + type: "number", + value: "", + formula: "100", + format: "$#,##0.00", + }); + assert.deepEqual(cell, { v: 100, f: "$#,##0.00" }); + }); + + it("omits format when empty string", () => { + const cell = buildCellFromInput({ + type: "number", + value: "", + formula: "100", + format: "", + }); + assert.deepEqual(cell, { v: 100 }); + }); + + it("empty formula input → empty string value", () => { + const cell = buildCellFromInput({ type: "number", value: "", formula: "" }); + assert.deepEqual(cell, { v: "" }); + }); +}); diff --git a/tests/engine/test_cellEmpty.ts b/tests/engine/test_cellEmpty.ts new file mode 100644 index 0000000..cc68592 --- /dev/null +++ b/tests/engine/test_cellEmpty.ts @@ -0,0 +1,123 @@ +// Telling a blank cell apart from a stored 0. Get this wrong and an aggregate +// reports a plausible number computed over the wrong count — a blank counted as +// a value drags AVERAGE down and COUNT up, with nothing to show it happened. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { isEmptyCell } from "../../src/engine/cellEmpty.ts"; +import type { SpreadsheetCell } from "../../src/engine/types.ts"; +import { SpreadsheetEngine, type SheetData } from "../../src/engine/index.ts"; +import { cellAt } from "./cellAccess.ts"; + +describe("isEmptyCell — empty", () => { + it("treats an absent cell as empty", () => { + assert.equal(isEmptyCell(null), true); + assert.equal(isEmptyCell(undefined), true); + }); + + it("treats an object with no value as empty", () => { + assert.equal(isEmptyCell({}), true); + assert.equal(isEmptyCell({ f: "0.00" }), true, "a format without a value is still empty"); + }); + + it("treats a null or undefined stored value as empty", () => { + assert.equal(isEmptyCell({ v: null }), true); + assert.equal(isEmptyCell({ v: undefined }), true); + }); + + it("treats an empty or whitespace string as empty, bare or wrapped", () => { + assert.equal(isEmptyCell(""), true); + assert.equal(isEmptyCell(" "), true); + assert.equal(isEmptyCell({ v: "" }), true); + assert.equal(isEmptyCell({ v: " " }), true); + }); +}); + +describe("isEmptyCell — not empty", () => { + // The distinction the whole module exists for: a stored 0 is a value. + it("treats a stored zero as a value", () => { + assert.equal(isEmptyCell(0), false); + assert.equal(isEmptyCell({ v: 0 }), false); + }); + + it("treats false as a value", () => { + assert.equal(isEmptyCell(false), false); + assert.equal(isEmptyCell({ v: false }), false); + }); + + it("treats any non-empty text as a value", () => { + assert.equal(isEmptyCell("x"), false); + assert.equal(isEmptyCell({ v: "hello" }), false); + assert.equal(isEmptyCell({ v: "0" }), false, "a zero written as text is still a value"); + }); + + it("treats a number as a value", () => { + assert.equal(isEmptyCell(42), false); + assert.equal(isEmptyCell({ v: 42 }), false); + assert.equal(isEmptyCell(-1), false); + }); +}); + +// The same distinction driven through the engine: an aggregate over a range +// with blanks must count only the real values. +describe("blank cells are not values in an aggregate (#2358)", () => { + // 10, 20, 30 followed by two blanks. Excel divides by 3 and counts 3. + const withBlanks = (formula: string): SheetData => ({ + name: "S", + data: [[{ v: 10 }, { v: formula }], [{ v: 20 }], [{ v: 30 }], [{ v: "" }], [{ v: null } as unknown as SpreadsheetCell]], + }); + const run = (formula: string): unknown => cellAt(new SpreadsheetEngine().calculate(withBlanks(formula)).data, 0, 1); + + it("excludes blanks from AVERAGE's denominator", () => { + assert.equal(run("=AVERAGE(A1:A5)"), 20, "not 15, which counts the two blanks as 0"); + }); + + it("excludes blanks from COUNT", () => { + assert.equal(run("=COUNT(A1:A5)"), 3, "not 4"); + }); + + // A blank would have read as 0, and 0 does not change a sum — so SUM is the + // one aggregate the old behaviour got right, and it must stay right. + it("leaves SUM unchanged", () => { + assert.equal(run("=SUM(A1:A5)"), 60); + }); + + it("does not disturb MAX or MIN", () => { + assert.equal(run("=MAX(A1:A5)"), 30); + assert.equal(run("=MIN(A1:A5)"), 10); + }); + + // The line the fix walks: a stored 0 is a value and must still count, even + // though a blank does not. + it("still counts a stored zero", () => { + const sheet: SheetData = { name: "S", data: [[{ v: 10 }, { v: "=COUNT(A1:A3)" }], [{ v: 0 }], [{ v: 20 }]] }; + assert.equal(cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 1), 3); + }); + + it("averages a stored zero in, but not a blank", () => { + const sheet: SheetData = { name: "S", data: [[{ v: 6 }, { v: "=AVERAGE(A1:A3)" }], [{ v: 0 }], [{ v: "" }]] }; + assert.equal(cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 1), 3, "(6 + 0) / 2, the blank excluded"); + }); +}); + +describe("blanks stay in the raw range so criteria and values stay aligned", () => { + // SUMIF reads the criteria range and the sum range separately. Dropping + // blanks from the raw list would compact each independently and shift the + // rows out of alignment, aggregating the wrong values (Codex review on + // #2383). A blank in the criteria column must NOT desync the two ranges. + it("keeps SUMIF row-aligned when a criteria cell is blank", () => { + const sheet: SheetData = { + name: "S", + data: [ + [{ v: 10 }, { v: 100 }, { v: '=SUMIF(A1:A4,">5",B1:B4)' }], + [{ v: "" }, { v: 200 }], + [{ v: 20 }, { v: 300 }], + [{ v: 30 }, { v: 400 }], + ], + }; + // A1=10, A3=20, A4=30 are >5; their B values are 100, 300, 400 → 800. + // If the blank A2 shifted the value range, B would misalign and the sum + // would be wrong. + assert.equal(cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 2), 800); + }); +}); diff --git a/tests/engine/test_cellFormatting.ts b/tests/engine/test_cellFormatting.ts new file mode 100644 index 0000000..387252d --- /dev/null +++ b/tests/engine/test_cellFormatting.ts @@ -0,0 +1,73 @@ +// The display-formatting decision for a single cell. It runs only on the final +// output pass — cross-sheet reference resolution deliberately skips it — so a +// wrong branch here either hides a date or, worse, turns a raw serial into a +// "03/04/2025" string that a downstream parseFloat reads as 3 (issue #2332). + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { formatCellForDisplay, isLikelyDateSerial } from "../../src/engine/cellFormatting.ts"; +import { dateToSerial } from "../../src/engine/date-utils.ts"; + +const serial2025Mar4 = dateToSerial(new Date(Date.UTC(2025, 2, 4))); + +describe("isLikelyDateSerial", () => { + it("accepts integers inside the date-serial window", () => { + assert.equal(isLikelyDateSerial(serial2025Mar4), true); + }); + + it("accepts the exact window boundaries", () => { + assert.equal(isLikelyDateSerial(36000), true); + assert.equal(isLikelyDateSerial(63499), true); + }); + + it("rejects values just outside the window", () => { + assert.equal(isLikelyDateSerial(35999), false); + assert.equal(isLikelyDateSerial(63500), false); + }); + + it("rejects non-integers (a time component is not a bare date)", () => { + assert.equal(isLikelyDateSerial(45720.5), false); + }); + + it("rejects non-numbers", () => { + assert.equal(isLikelyDateSerial("45720" as unknown as number), false); + assert.equal(isLikelyDateSerial(true as unknown as number), false); + }); +}); + +describe("formatCellForDisplay — passthrough", () => { + it("returns the value unchanged when the original is not a cell", () => { + assert.equal(formatCellForDisplay(5, 5, false), 5); + assert.equal(formatCellForDisplay(null, 7, false), 7); + }); + + it("leaves text untouched", () => { + assert.equal(formatCellForDisplay({ v: "hello" }, "hello", false), "hello"); + }); + + it("leaves a plain formula number that is not a date serial", () => { + assert.equal(formatCellForDisplay({ v: "=A1+A2" }, 100, false), 100); + }); + + it("leaves an empty cell's zero as a number", () => { + assert.equal(formatCellForDisplay({ v: "" }, 0, false), 0); + }); +}); + +describe("formatCellForDisplay — formatting", () => { + it("applies an explicit currency format", () => { + assert.equal(formatCellForDisplay({ v: 1234.5, f: "$#,##0.00" }, 1234.5, false), "$1,234.50"); + }); + + it("auto-formats a formula's date serial (month-first by default)", () => { + assert.equal(formatCellForDisplay({ v: "=A1" }, serial2025Mar4, false), "03/04/2025"); + }); + + it("honours day-first preference for the auto date format", () => { + assert.equal(formatCellForDisplay({ v: "=A1" }, serial2025Mar4, true), "04/03/2025"); + }); + + it("does NOT auto-format a non-formula date serial (only formulas opt in)", () => { + assert.equal(formatCellForDisplay({ v: serial2025Mar4 }, serial2025Mar4, false), serial2025Mar4); + }); +}); diff --git a/tests/engine/test_cellRefSubstitution.ts b/tests/engine/test_cellRefSubstitution.ts new file mode 100644 index 0000000..c11f77a --- /dev/null +++ b/tests/engine/test_cellRefSubstitution.ts @@ -0,0 +1,179 @@ +// Substituting cell values into a formula. The failure this covers produced a +// NUMBER — `=A1+A10` came back 55 instead of 12 — so there was nothing in the +// sheet to suggest anything had gone wrong. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { SpreadsheetEngine, renderOperand, findCellRefs, endOfStringLiteral, type SheetData } from "../../src/engine/index.ts"; +import { cellAt, rowAt } from "./cellAccess.ts"; + +/** A single column of values, with `formula` in the cell beside the first. */ +function columnSheet(values: (string | number)[], formula: string): SheetData { + return { name: "S", data: values.map((value, index) => (index === 0 ? [{ v: value }, { v: formula }] : [{ v: value }])) }; +} + +const evaluate = (sheet: SheetData): unknown => cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 1); + +describe("cell reference substitution — prefix collisions", () => { + // A global string replace rewrote every occurrence of the shorter reference + // first, turning `A10` into `0`: 5 and 7 became "5+50" = 55. + it("does not let A1 rewrite A10", () => { + const values = [5, 0, 0, 0, 0, 0, 0, 0, 0, 7]; + assert.equal(evaluate(columnSheet(values, "=A1+A10")), 12); + }); + + it("does not let A1 rewrite A11 or A100", () => { + const values = [3, 0, 0, 0, 0, 0, 0, 0, 0, 0, 4]; + assert.equal(evaluate(columnSheet(values, "=A1+A11")), 7); + }); + + it("keeps the order of a reference used twice", () => { + const values = [2, 0, 0, 0, 0, 0, 0, 0, 0, 9]; + assert.equal(evaluate(columnSheet(values, "=A10-A1")), 7); + }); + + // The column letters collide the same way: `B1` is a prefix of `AB1` only in + // the substring sense, and the old replace did not care about boundaries. + it("does not let B1 rewrite AB1", () => { + const sheet: SheetData = { name: "S", data: [[{ v: 0 }, { v: 2 }, { v: "=B1+AB1" }]] }; + assert.equal(cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 2), 2, "AB1 is empty, so the sum is B1 alone"); + }); + + it("handles several colliding references in one formula", () => { + const values = [1, 2, 0, 0, 0, 0, 0, 0, 0, 10, 11]; + assert.equal(evaluate(columnSheet(values, "=A1+A2+A10+A11")), 24); + }); +}); + +describe("a lone reference returns the cell value unchanged", () => { + // `=A1` is not an expression to substitute into — it IS the cell. Rendering + // the value into expression text first would escape a string's quotes and + // backslashes, and those escapes would survive into the result (Codex + // review): `=A1` on `say "hi"` came back `say \"hi\"`. + it("returns text with quotes and backslashes intact", () => { + const sheet: SheetData = { name: "S", data: [[{ v: 'say "hi"' }, { v: "=A1" }, { v: "=$A$1" }]] }; + const row = rowAt(new SpreadsheetEngine().calculate(sheet).data, 0); + assert.equal(row[1], 'say "hi"'); + assert.equal(row[2], 'say "hi"', "absolute form too"); + }); + + // `= A1` and `=A1 ` are still nothing but one reference; the whitespace must + // not push them onto the substitution path, where the text would be escaped + // and the escapes kept (Codex review). + it("takes the fast path despite surrounding whitespace", () => { + for (const formula of ["= A1", "=A1 ", "= A1 "]) { + const sheet: SheetData = { name: "S", data: [[{ v: 'say "hi"' }, { v: formula }]] }; + assert.equal(cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 1), 'say "hi"', `${formula} should return the value verbatim`); + } + }); + + it("still substitutes when whitespace surrounds a reference inside an expression", () => { + const sheet: SheetData = { name: "S", data: [[{ v: 3 }, { v: "= A1 + 1" }]] }; + assert.equal(cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 1), 4); + }); + + it("returns a backslash-bearing string intact", () => { + const sheet: SheetData = { name: "S", data: [[{ v: "C:\\path" }, { v: "=A1" }]] }; + assert.equal(cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 1), "C:\\path"); + }); + + it("returns a number, not its string form", () => { + const sheet: SheetData = { name: "S", data: [[{ v: 42 }, { v: "=A1" }]] }; + assert.equal(cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 1), 42); + }); + + // The fast path is only for a formula that is EXACTLY one reference; the + // moment it is part of an expression the substitution path takes over. + it("does not take the fast path when the reference is part of an expression", () => { + const sheet: SheetData = { name: "S", data: [[{ v: 3 }, { v: "=A1+1" }]] }; + assert.equal(cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 1), 4); + }); +}); + +describe("a reference inside a string literal is a constant, not a reference", () => { + // `"A1"` is text the user typed, not the cell A1. The substitution used to + // scan inside literals, so `="A1"&"!"` returned A1's value (`5!`) instead of + // the constant (`A1!`) (Codex review). + it("does not substitute a double-quoted ref-looking constant", () => { + assert.equal(evaluate(columnSheet([5], '="A1"&"!"')), "A1!"); + }); + + it("substitutes a real reference beside a literal that looks like one", () => { + assert.equal(evaluate(columnSheet([7], '=A1&"B2"')), "7B2"); + }); + + // A single-quoted span that is NOT `'Sheet'!cell` is a string literal too, so + // its contents must not be read as references. + it("does not substitute a single-quoted ref-looking constant", () => { + assert.equal(evaluate(columnSheet([5], "='B2'&\"!\"")), "B2!"); + }); +}); + +describe("findCellRefs", () => { + it("finds every reference in an arithmetic expression", () => { + assert.deepEqual(findCellRefs("A1+A10"), [ + { ref: "A1", start: 0 }, + { ref: "A10", start: 3 }, + ]); + }); + + it("skips references inside a double-quoted literal", () => { + assert.deepEqual(findCellRefs('"A1"&"!"'), []); + assert.deepEqual(findCellRefs('A1&"B2"'), [{ ref: "A1", start: 0 }]); + }); + + it("keeps a quoted sheet reference but skips a plain single-quoted literal", () => { + assert.deepEqual(findCellRefs("'Sheet1'!A1"), [{ ref: "'Sheet1'!A1", start: 0 }]); + assert.deepEqual(findCellRefs("'B2'&\"!\""), []); + }); + + // An escaped quote must not end the literal early, or the tail would be + // scanned for references. + it("honours backslash escapes inside a literal", () => { + assert.deepEqual(findCellRefs('"say \\"A1\\""'), []); + }); +}); + +describe("endOfStringLiteral", () => { + it("returns the index just past the closing quote", () => { + assert.equal(endOfStringLiteral('"ab"cd', 0), 4); + }); + + it("does not close on an escaped quote", () => { + assert.equal(endOfStringLiteral('"a\\"b"x', 0), 6); + }); + + it("returns the length when the literal is never closed", () => { + assert.equal(endOfStringLiteral('"abc', 0), 4); + }); +}); + +describe("renderOperand", () => { + it("renders numbers and booleans verbatim", () => { + assert.equal(renderOperand(42), "42"); + assert.equal(renderOperand(-3.5), "-3.5"); + assert.equal(renderOperand(0), "0"); + assert.equal(renderOperand(true), "true"); + }); + + // Quoting is what keeps a text cell from being read as an identifier or an + // operator once it lands in the expression. + it("quotes strings", () => { + assert.equal(renderOperand("hello"), '"hello"'); + assert.equal(renderOperand(""), '""'); + }); + + // Without escaping, a cell containing a quote closes the literal early and + // the rest of its text becomes expression source. + it("escapes quotes and backslashes so the literal cannot be closed early", () => { + assert.equal(renderOperand('say "hi"'), '"say \\"hi\\""'); + assert.equal(renderOperand("back\\slash"), '"back\\\\slash"'); + assert.equal(renderOperand('"'), '"\\""'); + }); + + // Blanks are 0 here, as they are everywhere else in the engine. + it("renders a missing value as 0", () => { + assert.equal(renderOperand(null), "0"); + assert.equal(renderOperand(undefined), "0"); + }); +}); diff --git a/tests/engine/test_concatSafety.ts b/tests/engine/test_concatSafety.ts new file mode 100644 index 0000000..3e2f8d7 --- /dev/null +++ b/tests/engine/test_concatSafety.ts @@ -0,0 +1,109 @@ +// Deciding whether a string-concatenation expression is safe to evaluate. The +// bug this guards: the old check ran a character allowlist over the WHOLE +// expression, including the content of string literals — so a `!` inside a +// string, or the `\` an escaped operand produces, made a valid formula look +// unsafe and it was returned as raw text (#2376). Masking the literals first +// validates the structure without judging the content. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { maskStringLiterals, isSafeConcatExpression, SpreadsheetEngine, type SheetData } from "../../src/engine/index.ts"; +import { cellAt } from "./cellAccess.ts"; + +describe("maskStringLiterals", () => { + it("empties a double- or single-quoted literal, keeping the quotes", () => { + assert.equal(maskStringLiterals('"hello"'), '""'); + assert.equal(maskStringLiterals("'world'"), "''"); + }); + + it("keeps the structure around the literals", () => { + assert.equal(maskStringLiterals('"a"+"b"'), '""+""'); + assert.equal(maskStringLiterals('5+"x"'), '5+""'); + }); + + // The content is what must not leak into the structure check — punctuation, + // operators, whatever. + it("removes arbitrary content, including operators and punctuation", () => { + assert.equal(maskStringLiterals('"a>b!c"'), '""'); + assert.equal(maskStringLiterals('"1+2"+3'), '""+3'); + }); + + // An escaped quote does not end the literal, so its content — and the + // backslash — is masked away rather than leaking a stray quote. + it("honours backslash escapes inside a literal", () => { + assert.equal(maskStringLiterals('"a\\"b"'), '""'); + assert.equal(maskStringLiterals('"back\\\\slash"'), '""'); + }); + + it("leaves an expression with no literals unchanged", () => { + assert.equal(maskStringLiterals("1+2+3"), "1+2+3"); + assert.equal(maskStringLiterals(""), ""); + }); +}); + +describe("isSafeConcatExpression", () => { + it("accepts joined string literals", () => { + assert.equal(isSafeConcatExpression('"a"+"b"'), true); + assert.equal(isSafeConcatExpression('"hi"+"!"'), true, "a bang inside a string is content, not structure"); + }); + + it("accepts a literal carrying escapes and arbitrary characters", () => { + assert.equal(isSafeConcatExpression('"say \\"hi\\""+"!"'), true); + assert.equal(isSafeConcatExpression('"a\\\\b"+"c"'), true); + assert.equal(isSafeConcatExpression('"日本語"+"!"'), true); + }); + + it("accepts numbers and parentheses joining strings", () => { + assert.equal(isSafeConcatExpression('5+"x"'), true); + assert.equal(isSafeConcatExpression('("a")+("b")'), true); + }); + + // Once the literals are masked, an identifier in the STRUCTURE is not + // allowed — that would be an unresolved reference or injected code. + it("rejects an unresolved identifier in the structure", () => { + assert.equal(isSafeConcatExpression('foo+"a"'), false); + assert.equal(isSafeConcatExpression('"a"+process'), false); + }); + + // A boolean cell renders as a bare `true` / `false`; those two words must + // pass or a boolean operand's concat is returned as raw text (Codex review). + // Any other identifier — even one containing them as a substring — is still + // rejected, so the exemption cannot smuggle code into `new Function`. + it("accepts the boolean operand words but nothing else", () => { + assert.equal(isSafeConcatExpression('true+"!"'), true); + assert.equal(isSafeConcatExpression('false+"!"'), true); + assert.equal(isSafeConcatExpression('truthy+"!"'), false); + }); +}); + +describe("string concatenation through the engine (#2376)", () => { + const concat = (cellValue: string): unknown => + cellAt(new SpreadsheetEngine().calculate({ name: "S", data: [[{ v: cellValue }, { v: '=A1&"!"' }]] } satisfies SheetData).data, 0, 1); + + // The plain case that was already broken: a `!` in the appended string made + // the whole concat fail the allowlist and return the raw formula text. + it("appends a literal to a plain string", () => { + assert.equal(concat("hi"), "hi!"); + }); + + // The #2376 blocker: an escaped operand must survive the concat path. + it("appends to a string containing a quote", () => { + assert.equal(concat('say "hi"'), 'say "hi"!'); + }); + + it("appends to a string containing a backslash", () => { + assert.equal(concat("a\\b"), "a\\b!"); + }); + + it("appends to a numeric string", () => { + assert.equal(concat("5"), "5!"); + }); + + // A boolean operand (here A1 is the comparison `=1=1`) renders as `true`, + // which the stricter safety gate used to reject — the concat came back as the + // raw formula text instead of the joined value (Codex review). + it("appends a literal to a boolean cell value", () => { + const sheet: SheetData = { name: "S", data: [[{ v: "=1=1" }, { v: '=A1&"!"' }]] }; + assert.equal(cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 1), "true!"); + }); +}); diff --git a/tests/engine/test_condition.ts b/tests/engine/test_condition.ts new file mode 100644 index 0000000..7576645 --- /dev/null +++ b/tests/engine/test_condition.ts @@ -0,0 +1,292 @@ +// Reading a spreadsheet condition without running it. The grammar is one +// comparison or a bare value, and that narrowness is the safety property — +// this replaced an `eval` that executed whatever a cell happened to contain. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { + evaluateCondition, + isTruthyCondition, + readOperand, + renderConditionOperand, + splitComparison, + stripOuterParens, +} from "../../src/engine/condition.ts"; + +describe("splitComparison", () => { + it("splits on each operator", () => { + assert.deepEqual(splitComparison("A>1"), { left: "A", operator: ">", right: "1" }); + assert.deepEqual(splitComparison("A<1"), { left: "A", operator: "<", right: "1" }); + assert.deepEqual(splitComparison("A=1"), { left: "A", operator: "=", right: "1" }); + }); + + // Longest-first matching: a `>` that starts `>=` must not win. + it("prefers the two-character operators", () => { + assert.deepEqual(splitComparison("A>=1"), { left: "A", operator: ">=", right: "1" }); + assert.deepEqual(splitComparison("A<=1"), { left: "A", operator: "<=", right: "1" }); + assert.deepEqual(splitComparison("A<>1"), { left: "A", operator: "<>", right: "1" }); + assert.deepEqual(splitComparison("A!=1"), { left: "A", operator: "!=", right: "1" }); + assert.deepEqual(splitComparison("A==1"), { left: "A", operator: "==", right: "1" }); + }); + + it("trims both sides", () => { + assert.deepEqual(splitComparison(" A > 1 "), { left: "A", operator: ">", right: "1" }); + }); + + it("returns null when there is no comparison", () => { + assert.equal(splitComparison("A1"), null); + assert.equal(splitComparison("42"), null); + assert.equal(splitComparison(""), null); + }); + + // Only the first operator counts, so `a=b=c` is one comparison against the + // text `b=c` rather than a chain. A chain is what used to reach a JS parser. + it("takes only the first operator", () => { + assert.deepEqual(splitComparison("1=1=1"), { left: "1", operator: "=", right: "1=1" }); + }); + + // A blank cell substitutes to nothing, leaving the operator at position 0. + // That is a comparison against an empty left side, not a bare value — reading + // it as text made `IFS` pick branches for empty cells (Codex review). + it("treats a leading operator as a comparison with an empty left side", () => { + assert.deepEqual(splitComparison(">5"), { left: "", operator: ">", right: "5" }); + assert.deepEqual(splitComparison(">=1"), { left: "", operator: ">=", right: "1" }); + }); + + // A cell holding `a>b` substitutes as the literal `"a>b"`. Splitting on that + // `>` would compare two fragments of one string (Codex review). The same must + // hold for `<` and `=` inside the operand, not only `>`. + it("ignores operators inside quoted text", () => { + assert.deepEqual(splitComparison('"a>b"="a>b"'), { left: '"a>b"', operator: "=", right: '"a>b"' }); + assert.deepEqual(splitComparison('"ab"' }); + assert.deepEqual(splitComparison('A1="a { + assert.deepEqual(splitComparison('"a>b" = "c"'), { left: '"a>b"', operator: "=", right: '"c"' }); + }); +}); + +describe("stripOuterParens", () => { + // `IFS((A1>0), ...)` is valid, and the parser used to split it into `(1` and + // `0)` — two operands that compare as text and never match (Codex review). + it("removes parentheses that wrap the whole expression", () => { + assert.equal(stripOuterParens("(1>0)"), "1>0"); + assert.equal(stripOuterParens("((1>0))"), "1>0"); + assert.equal(stripOuterParens(" ( 1>0 ) "), "1>0"); + }); + + // The leading `(` closes before the end, so it wraps only its own operand. + // Removing the outer characters here would corrupt the expression into + // `A)=(B`. + it("keeps parentheses that wrap only part of the expression", () => { + assert.equal(stripOuterParens("(A)=(B)"), "(A)=(B)"); + assert.equal(stripOuterParens("(1)>(0)"), "(1)>(0)"); + }); + + it("leaves unbalanced input untouched rather than guessing", () => { + assert.equal(stripOuterParens("(1>0"), "(1>0"); + assert.equal(stripOuterParens("("), "("); + assert.equal(stripOuterParens(")("), ")("); + }); + + it("ignores parentheses inside quoted text", () => { + assert.equal(stripOuterParens('("a)b")'), '"a)b"'); + }); + + it("leaves an expression with no outer parentheses alone", () => { + assert.equal(stripOuterParens("1>0"), "1>0"); + assert.equal(stripOuterParens(""), ""); + }); +}); + +describe("readOperand", () => { + it("reads numbers", () => { + assert.equal(readOperand("42"), 42); + assert.equal(readOperand("-3.5"), -3.5); + assert.equal(readOperand("0"), 0); + }); + + it("reads booleans case-insensitively", () => { + assert.equal(readOperand("TRUE"), true); + assert.equal(readOperand("true"), true); + assert.equal(readOperand("FALSE"), false); + }); + + it("keeps quoted text as text, quotes removed", () => { + assert.equal(readOperand('"hello"'), "hello"); + assert.equal(readOperand("'hello'"), "hello"); + assert.equal(readOperand('"42"'), "42", "quoted digits stay text"); + }); + + // `Number` rather than `parseFloat`: trailing garbage makes the whole thing + // text instead of silently contributing its numeric prefix. + it("does not take a numeric prefix from mixed text", () => { + assert.equal(readOperand("12abc"), "12abc"); + assert.equal(readOperand("3.5kg"), "3.5kg"); + }); + + it("keeps unquoted text as text", () => { + assert.equal(readOperand("hello"), "hello"); + assert.equal(readOperand(""), ""); + }); +}); + +describe("evaluateCondition — comparisons", () => { + it("compares numbers", () => { + assert.equal(evaluateCondition("5>3"), true); + assert.equal(evaluateCondition("3>5"), false); + assert.equal(evaluateCondition("5>=5"), true, "the boundary counts for >="); + assert.equal(evaluateCondition("5>5"), false); + assert.equal(evaluateCondition("3<=3"), true); + }); + + it("compares for equality and inequality", () => { + assert.equal(evaluateCondition("5=5"), true); + assert.equal(evaluateCondition("5==5"), true); + assert.equal(evaluateCondition("5<>3"), true); + assert.equal(evaluateCondition("5!=5"), false); + }); + + // The regression the quote-aware scan exists for: both sides are one string + // each, so this is equality between them, not a comparison of fragments. + it("evaluates a parenthesised comparison the same as a bare one", () => { + assert.equal(evaluateCondition("(1>0)"), true); + assert.equal(evaluateCondition("((5>=5))"), true); + assert.equal(evaluateCondition("(3>5)"), false); + assert.equal(evaluateCondition('("a"="a")'), true); + }); + + it("evaluates a parenthesised bare value", () => { + assert.equal(evaluateCondition("(1)"), true); + assert.equal(evaluateCondition("(0)"), false); + }); + + it("compares text containing operator characters", () => { + assert.equal(evaluateCondition('"a>b"="a>b"'), true); + assert.equal(evaluateCondition('"a>b"="a>c"'), false); + assert.equal(evaluateCondition('"a { + assert.equal(evaluateCondition(">5"), false, "blank is not greater than 5"); + assert.equal(evaluateCondition(">=1"), false); + assert.equal(evaluateCondition("<>5"), true, "blank does differ from 5"); + assert.equal(evaluateCondition("=5"), false); + assert.equal(evaluateCondition('=""'), true, "blank equals blank"); + }); + + it("compares text", () => { + assert.equal(evaluateCondition('"abc"="abc"'), true); + assert.equal(evaluateCondition('"abc"="abd"'), false); + assert.equal(evaluateCondition('"abc"<"abd"'), true); + }); + + // A quoted number and a bare one are different types, so equality separates + // them — the same rule the rest of the engine follows. + it("distinguishes a quoted number from a bare one", () => { + assert.equal(evaluateCondition('42="42"'), false); + }); + + it("compares booleans for equality but refuses to order them", () => { + assert.equal(evaluateCondition("TRUE=TRUE"), true); + assert.equal(evaluateCondition("TRUE<>FALSE"), true); + assert.equal(evaluateCondition("TRUE>FALSE"), false, "no ordering is defined"); + }); +}); + +describe("evaluateCondition — bare values", () => { + // Spreadsheet truthiness, not JavaScript's: 0 and empty are false. + it("treats zero and empty as false, other values as true", () => { + assert.equal(evaluateCondition("0"), false); + assert.equal(evaluateCondition(""), false); + assert.equal(evaluateCondition('""'), false); + assert.equal(evaluateCondition("1"), true); + assert.equal(evaluateCondition("-1"), true, "a negative number is still a value"); + assert.equal(evaluateCondition("hello"), true); + }); + + it("reads bare booleans", () => { + assert.equal(isTruthyCondition("TRUE"), true); + assert.equal(isTruthyCondition("FALSE"), false); + }); +}); + +describe("evaluateCondition — code is data", () => { + // The point of the module. Each of these used to execute (#2360): the first + // two as a cell's substituted value, the third as text written straight into + // the formula. They must now be read as operands and nothing more. + it("does not execute an assignment", () => { + const marker = globalThis as Record; + marker.__conditionProbe = false; + assert.equal(evaluateCondition("globalThis.__conditionProbe=true"), false, "an assignment is text, and text is not a comparison match"); + assert.equal(marker.__conditionProbe, false, "nothing ran"); + }); + + it("does not execute a call or a sequence", () => { + const marker = globalThis as Record; + marker.__conditionProbe2 = false; + evaluateCondition("(globalThis.__conditionProbe2=true, 1)>0"); + assert.equal(marker.__conditionProbe2, false); + }); + + it("does not honour a logical operator smuggled into the condition", () => { + const marker = globalThis as Record; + marker.__conditionProbe3 = false; + evaluateCondition("1>0&&(globalThis.__conditionProbe3=true)"); + assert.equal(marker.__conditionProbe3, false); + }); + + it("never throws on syntactically broken input", () => { + for (const input of ["((((", '"unclosed', "1+", "}{", "throw 1"]) { + assert.equal(typeof evaluateCondition(input), "boolean", `${input} should still yield a boolean`); + } + }); +}); + +describe("renderConditionOperand", () => { + it("renders numbers and booleans as themselves", () => { + assert.equal(renderConditionOperand(42), "42"); + assert.equal(renderConditionOperand(0), "0"); + assert.equal(renderConditionOperand(true), "true"); + }); + + // A text cell must arrive as a quoted literal, so its own contents cannot be + // read as operators: `x>y` unquoted would make `A1="x>y"` parse as a + // comparison of fragments. + it("quotes strings", () => { + assert.equal(renderConditionOperand("x>y"), '"x>y"'); + assert.equal(renderConditionOperand("Yes"), '"Yes"'); + assert.equal(renderConditionOperand(""), '""'); + }); + + it("escapes quotes and backslashes so the literal cannot be closed early", () => { + assert.equal(renderConditionOperand('a"b'), '"a\\"b"'); + assert.equal(renderConditionOperand("a\\b"), '"a\\\\b"'); + }); + + it("renders a missing value as an empty quoted string", () => { + assert.equal(renderConditionOperand(null), '""'); + assert.equal(renderConditionOperand(undefined), '""'); + }); + + // Round-trip: whatever it renders, evaluateCondition reads back as the same + // value, so a quoted text operand compares equal to itself. + it("round-trips through evaluateCondition", () => { + assert.equal(evaluateCondition(`${renderConditionOperand("x>y")}="x>y"`), true); + assert.equal(evaluateCondition(`${renderConditionOperand("x>y")}="other"`), false); + }); + + // A literal backslash must survive the escape-on-render / unescape-on-read + // round-trip exactly once: the earlier substitution path escaped it twice + // (CodeQL js/double-escaping), which corrupted the operand. + it("round-trips a value holding a backslash without double-escaping", () => { + assert.equal(readOperand(renderConditionOperand("a\\b")), "a\\b"); + assert.equal(evaluateCondition(`${renderConditionOperand("a\\b")}=${renderConditionOperand("a\\b")}`), true); + assert.equal(evaluateCondition(`${renderConditionOperand("a\\b")}=${renderConditionOperand("a/b")}`), false); + }); +}); diff --git a/tests/engine/test_conditionalAggregates.ts b/tests/engine/test_conditionalAggregates.ts new file mode 100644 index 0000000..ab78e7e --- /dev/null +++ b/tests/engine/test_conditionalAggregates.ts @@ -0,0 +1,47 @@ +// SUMIF / AVERAGEIF must pair the criteria range and the value range by +// POSITION. Reading the value range in numeric-only mode dropped blanks, which +// shifted its indexes out of step with the (raw) criteria range and pulled a +// later row's number into an earlier match (#2358 Codex review). + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { SpreadsheetEngine, type SheetData } from "../../src/engine/index.ts"; +import { cellAt } from "./cellAccess.ts"; + +// A = criteria column, B = value column (with a blank at row 2), formula in C1. +const sheet = (formula: string): SheetData => ({ + name: "S", + data: [ + [{ v: 1 }, { v: 100 }, { v: formula }], + [{ v: 0 }, { v: "" }], + [{ v: 1 }, { v: 300 }], + ], +}); + +const evalFormula = (formula: string): unknown => cellAt(new SpreadsheetEngine().calculate(sheet(formula)).data, 0, 2); + +describe("SUMIF / AVERAGEIF stay row-aligned when the value range has a blank", () => { + // Rows 1 and 3 match (A > 0); their B values are 100 and 300. The blank B2 + // belongs to the non-matching row 2 and must not slide up into row 3. + it("sums the value range by position, not by compacted index", () => { + assert.equal(evalFormula('=SUMIF(A1:A3, ">0", B1:B3)'), 400); + }); + + it("averages the matching rows' values by position", () => { + assert.equal(evalFormula('=AVERAGEIF(A1:A3, ">0", B1:B3)'), 200); + }); + + // A blank inside the matched rows counts as 0 in SUMIF (not skipped), matching + // Excel: here rows 1 and 3 match, B1 is blank, so the sum is just 300. + it("treats a blank in a matched value cell as 0", () => { + const withBlankMatch: SheetData = { + name: "S", + data: [ + [{ v: 1 }, { v: "" }, { v: '=SUMIF(A1:A3, ">0", B1:B3)' }], + [{ v: 0 }, { v: 999 }], + [{ v: 1 }, { v: 300 }], + ], + }; + assert.equal(cellAt(new SpreadsheetEngine().calculate(withBlankMatch).data, 0, 2), 300); + }); +}); diff --git a/tests/engine/test_criteria.ts b/tests/engine/test_criteria.ts new file mode 100644 index 0000000..7360f8a --- /dev/null +++ b/tests/engine/test_criteria.ts @@ -0,0 +1,90 @@ +// Criteria matching for COUNTIF / SUMIF / AVERAGEIF. Both bugs undercounted +// silently: text was compared with `===`, so `"yes"` skipped a cell holding +// `Yes`, and `"A*"` was matched literally instead of as a wildcard (#2360). + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { parseCriteria } from "../../src/engine/registry.ts"; +import { SpreadsheetEngine, type SheetData } from "../../src/engine/index.ts"; +import { cellAt } from "./cellAccess.ts"; + +describe("parseCriteria — text is matched case-insensitively", () => { + it("matches regardless of case", () => { + const matches = parseCriteria("yes"); + assert.equal(matches("Yes"), true); + assert.equal(matches("YES"), true); + assert.equal(matches("yes"), true); + }); + + it("still rejects different text", () => { + const matches = parseCriteria("yes"); + assert.equal(matches("no"), false); + assert.equal(matches("yesterday"), false, "an exact match, not a prefix"); + }); +}); + +describe("parseCriteria — wildcards", () => { + it("treats * as any run of characters", () => { + const matches = parseCriteria("A*"); + assert.equal(matches("Axle"), true); + assert.equal(matches("A"), true, "* may match nothing"); + assert.equal(matches("Bar"), false); + }); + + it("treats ? as exactly one character", () => { + const matches = parseCriteria("b?t"); + assert.equal(matches("bat"), true); + assert.equal(matches("bt"), false); + assert.equal(matches("beat"), false); + }); + + it("escapes a wildcard with ~", () => { + const matches = parseCriteria("A~*"); + assert.equal(matches("A*"), true); + assert.equal(matches("Axle"), false); + }); + + it("does not let regex metacharacters act as a pattern", () => { + const matches = parseCriteria("a.c"); + assert.equal(matches("a.c"), true); + assert.equal(matches("abc"), false, "the dot is literal, not any-char"); + }); +}); + +describe("parseCriteria — numbers and operators", () => { + it("matches a numeric criteria against a number", () => { + const matches = parseCriteria("5"); + assert.equal(matches(5), true); + assert.equal(matches("5"), true); + assert.equal(matches(6), false); + }); + + it("keeps the comparison operators working", () => { + assert.equal(parseCriteria(">3")(5), true); + assert.equal(parseCriteria(">3")(2), false); + assert.equal(parseCriteria("<=3")(3), true); + }); + + it("applies case-insensitive text to = and <>", () => { + assert.equal(parseCriteria("=yes")("Yes"), true); + assert.equal(parseCriteria("<>yes")("Yes"), false); + assert.equal(parseCriteria("<>yes")("no"), true); + }); +}); + +describe("COUNTIF through the engine", () => { + const countif = (values: string[], criteria: string): unknown => { + const rows = values.map((value) => [{ v: value }]); + rows.push([{ v: `=COUNTIF(A1:A${values.length}, "${criteria}")` }]); + const sheet: SheetData = { name: "S", data: rows }; + return cellAt(new SpreadsheetEngine().calculate(sheet).data, values.length, 0); + }; + + it("counts a case-differing match", () => { + assert.equal(countif(["Yes", "no"], "yes"), 1); + }); + + it("counts a wildcard match", () => { + assert.equal(countif(["Axle", "Bar"], "A*"), 1); + }); +}); diff --git a/tests/engine/test_crossSheetReference.ts b/tests/engine/test_crossSheetReference.ts new file mode 100644 index 0000000..53194a8 --- /dev/null +++ b/tests/engine/test_crossSheetReference.ts @@ -0,0 +1,105 @@ +// Cross-sheet references (`=Data!A1`) must resolve a cell to the SAME value a +// same-sheet reference would. Regression for #2332: the target sheet was being +// resolved through its display-formatted output, so a date serial arrived as +// the string "03/04/2025" and parseFloat read it as 3 — `=Data!A1` returned 3 +// and `=DAY(Data!A1)` returned 2. Same-sheet was always correct; these tests +// pin cross-sheet to that same behaviour. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { SpreadsheetEngine, type SheetData } from "../../src/engine/index.ts"; +import { rowAt } from "./cellAccess.ts"; + +const engine = new SpreadsheetEngine(); + +const calcRow = (target: SheetData, all: SheetData[], row = 0) => rowAt(engine.calculate(target, all).data, row); + +describe("cross-sheet date reference (#2332 regression)", () => { + const data: SheetData = { name: "Data", data: [[{ v: "03/04/2025" }]] }; + const summary: SheetData = { name: "Summary", data: [[{ v: "=DAY(Data!A1)" }, { v: "=Data!A1" }]] }; + + it("=DAY(Data!A1) reads the date, not the leading digits", () => { + assert.equal(calcRow(summary, [data, summary])[0], 4); + }); + + it("=Data!A1 does not collapse to 3", () => { + assert.notEqual(calcRow(summary, [data, summary])[1], 3); + }); + + it("=Data!A1 matches the identical same-sheet reference", () => { + const sameSheet: SheetData = { name: "S", data: [[{ v: "03/04/2025" }, { v: "=DAY(A1)" }, { v: "=A1" }]] }; + const same = calcRow(sameSheet, [sameSheet]); + const cross = calcRow(summary, [data, summary]); + assert.equal(cross[0], same[1]); // =DAY + assert.equal(cross[1], same[2]); // =ref -> "03/04/2025" + }); +}); + +describe("cross-sheet reference — value types read straight across", () => { + const data: SheetData = { + name: "Data", + // date, number, text, empty, a formula that itself produces a date serial + data: [[{ v: "03/04/2025" }, { v: 42 }, { v: "hello" }, { v: "" }, { v: "=DATE(2025,3,4)" }]], + }; + const refs: SheetData = { + name: "Refs", + data: [[{ v: "=Data!A1" }, { v: "=Data!B1" }, { v: "=Data!C1" }, { v: "=Data!D1" }, { v: "=Data!E1" }]], + }; + + it("resolves each type the way the source cell holds it", () => { + assert.deepEqual(calcRow(refs, [data, refs]), ["03/04/2025", 42, "hello", 0, "03/04/2025"]); + }); + + it("feeds a cross-sheet date into a date function", () => { + const derived: SheetData = { name: "D2", data: [[{ v: "=DAY(Data!E1)" }, { v: "=Data!B1*2" }]] }; + assert.deepEqual(calcRow(derived, [data, derived]), [4, 84]); + }); +}); + +describe("cross-sheet range aggregation stays numeric", () => { + it("SUM over a cross-sheet range adds the raw numbers", () => { + const data: SheetData = { name: "D", data: [[{ v: 10 }, { v: 20 }, { v: 30 }]] }; + const sum: SheetData = { name: "S", data: [[{ v: "=SUM(D!A1:C1)" }]] }; + assert.deepEqual(calcRow(sum, [data, sum]), [60]); + }); +}); + +// resolveSheetData (#2482) folds the sheet-ref match -> cache check -> two-stage +// cache seed -> calculateSheet block that getCellValue and collectRangeValues +// shared. The two-stage seed is the cross-sheet infinite-loop guard: a cyclic +// reference must terminate with an error, not hang. If it hung, these tests would +// never return and the whole suite would time out. +describe("cyclic cross-sheet references terminate instead of hanging", () => { + it("a 2-sheet cycle (A!A1=B!A1, B!A1=A!A1) resolves to an error", () => { + const sheetA: SheetData = { name: "A", data: [[{ v: "=B!A1" }]] }; + const sheetB: SheetData = { name: "B", data: [[{ v: "=A!A1" }]] }; + assert.equal(String(calcRow(sheetA, [sheetA, sheetB])[0]), "#ERROR!"); + }); + + it("a 3-sheet cycle (A->B->C->A) also terminates with an error", () => { + const sheetA: SheetData = { name: "A", data: [[{ v: "=B!A1" }]] }; + const sheetB: SheetData = { name: "B", data: [[{ v: "=C!A1" }]] }; + const sheetC: SheetData = { name: "C", data: [[{ v: "=A!A1" }]] }; + assert.equal(String(calcRow(sheetA, [sheetA, sheetB, sheetC])[0]), "#ERROR!"); + }); + + it("a valid cross-sheet reference next to the cycle still resolves", () => { + const data: SheetData = { name: "D", data: [[{ v: 10 }, { v: 20 }]] }; + const main: SheetData = { name: "S", data: [[{ v: "=D!A1" }, { v: "=SUM(D!A1:B1)" }]] }; + assert.deepEqual(calcRow(main, [data, main]), [10, 30]); + }); +}); + +// A reference to a sheet that does not exist keeps each caller's terminal action +// after the fold: #REF! for a single cell, an empty range for an aggregate. +describe("missing-sheet reference keeps its per-caller terminal behaviour", () => { + it("a single cross-sheet cell to a missing sheet is #REF!", () => { + const main: SheetData = { name: "S", data: [[{ v: "=Ghost!A1" }]] }; + assert.equal(String(calcRow(main, [main])[0]), "#REF!"); + }); + + it("SUM over a range on a missing sheet contributes nothing (empty range)", () => { + const main: SheetData = { name: "S", data: [[{ v: "=SUM(Ghost!A1:B1)" }]] }; + assert.equal(calcRow(main, [main])[0], 0); + }); +}); diff --git a/tests/engine/test_dateLocale.ts b/tests/engine/test_dateLocale.ts new file mode 100644 index 0000000..77ee316 --- /dev/null +++ b/tests/engine/test_dateLocale.ts @@ -0,0 +1,95 @@ +// Which way `03/04/2025` reads. Getting this wrong does not throw — the cell +// holds a real date, just the wrong one, three days or eleven months off +// depending on the pair. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { prefersDayFirst } from "../../src/engine/date-locale.ts"; + +describe("prefersDayFirst — the locales the app ships", () => { + it("reads day-first for the European languages", () => { + for (const locale of ["es", "pt-BR", "fr", "de"]) { + assert.equal(prefersDayFirst(locale), true, `${locale} writes day-first`); + } + }); + + // Their conventional order is year-month-day, so in a two-part date the + // month still comes before the day — same as US order. + it("reads month-first for ja, zh and ko", () => { + for (const locale of ["ja", "zh", "ko"]) { + assert.equal(prefersDayFirst(locale), false, `${locale} puts the month first`); + } + }); +}); + +describe("prefersDayFirst — English splits on region, not language", () => { + // The app's own locale resolution folds `en-GB` to `en` before a plugin sees + // it, so a bare `en` cannot be resolved and keeps the US default. A caller + // that CAN supply the region gets the right answer. + it("keeps the US default for a bare en", () => { + assert.equal(prefersDayFirst("en"), false); + }); + + it("reads month-first for the month-first English regions", () => { + assert.equal(prefersDayFirst("en-US"), false); + assert.equal(prefersDayFirst("en-CA"), false); + assert.equal(prefersDayFirst("en-PH"), false); + }); + + it("reads day-first for the rest of the English-speaking world", () => { + for (const locale of ["en-GB", "en-AU", "en-NZ", "en-IE", "en-IN", "en-ZA"]) { + assert.equal(prefersDayFirst(locale), true, `${locale} writes day-first`); + } + }); +}); + +describe("prefersDayFirst — tag shapes", () => { + it("accepts underscores as well as hyphens", () => { + assert.equal(prefersDayFirst("en_GB"), true); + assert.equal(prefersDayFirst("pt_BR"), true); + }); + + it("ignores case", () => { + assert.equal(prefersDayFirst("EN-GB"), true); + assert.equal(prefersDayFirst("FR"), true); + assert.equal(prefersDayFirst("en-us"), false); + }); + + // A script subtag sits between the language and the region, so reading + // position 1 as the region flips `en-Latn-US` to day-first (Codex review). + // The earlier `zh-Hans-CN` case looked like it covered this and did not — + // non-English tags never consult the region at all. + it("reads past a script subtag to find the region", () => { + assert.equal(prefersDayFirst("en-Latn-US"), false, "en-Latn-US is month-first"); + assert.equal(prefersDayFirst("en-Latn-GB"), true, "en-Latn-GB is day-first"); + }); + + // An extension starts with a single-character subtag, and nothing after it + // is a region. `-u-nu-latn` is the case that distinguishes: "nu" is two + // letters and would otherwise be taken as a region, flipping a bare `en` to + // day-first. (`-u-ca-gregory` does NOT distinguish — "ca" happens to be a + // month-first region, so both readings agree by accident.) + it("does not read an extension subtag as a region", () => { + assert.equal(prefersDayFirst("en-u-nu-latn"), false, "no region: keeps the US default"); + assert.equal(prefersDayFirst("en-u-ca-gregory"), false); + assert.equal(prefersDayFirst("en-GB-u-nu-latn"), true, "a real region before the extension still wins"); + }); + + it("accepts a numeric UN M.49 region", () => { + assert.equal(prefersDayFirst("es-419"), true, "Latin American Spanish is still day-first by language"); + }); + + it("ignores the region for non-English tags", () => { + assert.equal(prefersDayFirst("zh-Hans-CN"), false); + assert.equal(prefersDayFirst("fr-CA"), true, "decided by language, not region"); + }); + + // Falling back to month-first keeps the existing behaviour for anything + // unrecognised, so a new locale cannot silently flip existing sheets. + it("falls back to month-first for unknown, empty and missing locales", () => { + assert.equal(prefersDayFirst("xx"), false); + assert.equal(prefersDayFirst(""), false); + assert.equal(prefersDayFirst(undefined), false); + assert.equal(prefersDayFirst(null), false); + }); +}); diff --git a/tests/engine/test_dateParser.ts b/tests/engine/test_dateParser.ts new file mode 100644 index 0000000..de861a0 --- /dev/null +++ b/tests/engine/test_dateParser.ts @@ -0,0 +1,211 @@ +// Turning a cell's text into a date. Every failure mode here is a wrong date +// rather than an error: March 4 read as April 3, 1930 read as 2030, a real +// date rejected as text. The cell still shows something plausible. +// +// Assertions compare against `dateToSerial(Date.UTC(...))` rather than literal +// serial numbers, so these tests are about how the STRING is interpreted; +// serial arithmetic itself is covered in test_dateUtils.ts. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { getDefaultDateFormat, isDateLike, parseDate } from "../../src/engine/date-parser.ts"; +import { dateToSerial } from "../../src/engine/date-utils.ts"; + +const serialOf = (year: number, month: number, day: number) => dateToSerial(new Date(Date.UTC(year, month - 1, day))); + +function assertParsesTo(input: string, year: number, month: number, day: number, preferDayFirst = false): void { + assert.equal(parseDate(input, preferDayFirst), serialOf(year, month, day), `${input} should read as ${year}-${month}-${day}`); +} + +describe("isDateLike", () => { + it("accepts the formats the parser handles", () => { + for (const input of ["03/04/2025", "2025-03-04", "2025/03/04", "4-Mar-2025", "Mar 4, 2025", "March 4, 2025", "4 Mar 2025"]) { + assert.equal(isDateLike(input), true, `${input} should look like a date`); + } + }); + + it("rejects text with no digits or no separator", () => { + assert.equal(isDateLike("hello world"), false); + assert.equal(isDateLike("20250304"), false); + }); + + // The length gate runs before any pattern match, so a short but perfectly + // well-formed date is rejected outright. + it("rejects a valid short date because of the 6-character floor", () => { + assert.equal(isDateLike("1/1/26"), true, "exactly 6 characters passes"); + assert.equal(isDateLike("1/1/6"), false, "5 characters is refused before any pattern runs"); + }); + + it("rejects anything longer than 30 characters", () => { + assert.equal(isDateLike(`${"September 30, 2025".padEnd(31, " ")}`), false); + }); + + it("rejects partial or malformed dates", () => { + assert.equal(isDateLike("03/2025"), false); + assert.equal(isDateLike("03-04-2025"), false, "hyphen-separated numerics are not a supported pattern"); + }); +}); + +describe("parseDate — ISO format", () => { + it("reads YYYY-MM-DD and YYYY/MM/DD", () => { + assertParsesTo("2025-03-04", 2025, 3, 4); + assertParsesTo("2025/03/04", 2025, 3, 4); + }); + + it("accepts unpadded month and day", () => { + assertParsesTo("2025-3-4", 2025, 3, 4); + }); + + // ISO is matched first, so a leading 4-digit group is never mistaken for a + // day even when it would be a legal day-first date. + it("takes the leading 4-digit group as the year", () => { + assertParsesTo("2025-01-02", 2025, 1, 2); + }); +}); + +describe("parseDate — slash format and the MM/DD vs DD/MM decision", () => { + // The default is US order. This is the single most consequential choice in + // the module: an ambiguous date silently becomes a different day. + it("defaults an ambiguous date to MM/DD", () => { + assertParsesTo("03/04/2025", 2025, 3, 4); + }); + + it("reads an ambiguous date as DD/MM when asked to prefer it", () => { + assertParsesTo("03/04/2025", 2025, 4, 3, true); + }); + + // Unambiguous cases ignore the preference entirely — the value decides. + it("reads day-first when the first number cannot be a month", () => { + assertParsesTo("13/04/2025", 2025, 4, 13); + assertParsesTo("13/04/2025", 2025, 4, 13, true); + }); + + it("reads month-first when the second number cannot be a day-of-month position", () => { + assertParsesTo("03/13/2025", 2025, 3, 13); + assertParsesTo("03/13/2025", 2025, 3, 13, true); + }); + + it("rejects a slash date where neither ordering is valid", () => { + assert.equal(parseDate("13/13/2025"), null); + }); +}); + +describe("parseDate — two-digit years", () => { + // The pivot is hardcoded at 30 and not configurable, so "01/01/30" is 1930. + it("maps years under 30 to the 2000s and 30 or over to the 1900s", () => { + assertParsesTo("01/01/29", 2029, 1, 1); + assertParsesTo("01/01/30", 1930, 1, 1); + assertParsesTo("01/01/99", 1999, 1, 1); + assertParsesTo("01/01/00", 2000, 1, 1); + }); + + it("applies the same pivot to the DD-MMM-YY form", () => { + assertParsesTo("1-Jan-29", 2029, 1, 1); + assertParsesTo("1-Jan-30", 1930, 1, 1); + }); +}); + +describe("parseDate — month-name formats", () => { + it("reads DD-MMM-YYYY", () => { + assertParsesTo("4-Mar-2025", 2025, 3, 4); + assertParsesTo("04-Mar-2025", 2025, 3, 4); + }); + + it("reads MMM D, YYYY and MMMM D, YYYY", () => { + assertParsesTo("Mar 4, 2025", 2025, 3, 4); + assertParsesTo("March 4, 2025", 2025, 3, 4); + }); + + it("reads D MMM YYYY", () => { + assertParsesTo("4 Mar 2025", 2025, 3, 4); + assertParsesTo("4 March 2025", 2025, 3, 4); + }); + + it("matches month names case-insensitively", () => { + assertParsesTo("4-MAR-2025", 2025, 3, 4); + assertParsesTo("march 4, 2025", 2025, 3, 4); + }); + + it("rejects a month name that is not a real month", () => { + assert.equal(parseDate("4-Foo-2025"), null); + assert.equal(parseDate("Smarch 4, 2025"), null); + }); + + it("makes the comma optional in MMM D YYYY", () => { + assertParsesTo("Mar 4 2025", 2025, 3, 4); + }); +}); + +describe("parseDate — validity and range", () => { + it("rejects a day that does not exist in its month", () => { + assert.equal(parseDate("2025-02-30"), null); + assert.equal(parseDate("2025-04-31"), null); + assert.equal(parseDate("02/30/2025"), null); + }); + + it("accepts Feb 29 in a leap year and rejects it otherwise", () => { + assertParsesTo("2024-02-29", 2024, 2, 29); + assert.equal(parseDate("2025-02-29"), null); + }); + + // The window is 1900–2100 inclusive; outside it a well-formed date is + // rejected rather than converted. + it("enforces the 1900–2100 year window", () => { + assertParsesTo("1900-01-01", 1900, 1, 1); + assertParsesTo("2100-12-31", 2100, 12, 31); + assert.equal(parseDate("1899-12-31"), null); + assert.equal(parseDate("2101-01-01"), null); + }); + + it("returns null for anything isDateLike rejects", () => { + assert.equal(parseDate("hello"), null); + assert.equal(parseDate(""), null); + assert.equal(parseDate("1/1/6"), null); + }); + + it("tolerates surrounding whitespace", () => { + assertParsesTo(" 2025-03-04 ", 2025, 3, 4); + }); +}); + +describe("getDefaultDateFormat", () => { + it("echoes the shape it was given", () => { + assert.equal(getDefaultDateFormat("2025-03-04"), "YYYY-MM-DD"); + assert.equal(getDefaultDateFormat("2025/03/04"), "YYYY/MM/DD"); + assert.equal(getDefaultDateFormat("4-Mar-2025"), "DD-MMM-YYYY"); + assert.equal(getDefaultDateFormat("Mar 4, 2025"), "MMM D, YYYY"); + assert.equal(getDefaultDateFormat("March 4, 2025"), "MMMM D, YYYY"); + }); + + // The three-vs-four letter split is what separates the two month-name + // formats; a four-letter month name takes the long form. + it("splits the month-name formats on name length", () => { + assert.equal(getDefaultDateFormat("Jun 4, 2025"), "MMM D, YYYY"); + assert.equal(getDefaultDateFormat("June 4, 2025"), "MMMM D, YYYY"); + }); + + // A slash date is labelled in the order the parser READ it, so the cell + // renders the halves the way the user typed them. + it("labels a slash date in its reading order", () => { + assert.equal(getDefaultDateFormat("03/04/2025"), "MM/DD/YYYY", "ambiguous: US default"); + assert.equal(getDefaultDateFormat("03/04/2025", true), "DD/MM/YYYY", "ambiguous: day-first when preferred"); + assert.equal(getDefaultDateFormat("13/04/2025"), "DD/MM/YYYY", "13 can only be a day, whatever the preference"); + assert.equal(getDefaultDateFormat("03/13/2025", true), "MM/DD/YYYY", "13 can only be a day here too"); + }); + + // Slash-separated ISO parses year-first, so it must be LABELLED year-first + // too. Falling through to the slash default re-rendered it as MM/DD or DD/MM + // — the same digits in a different order, which reads as a different date + // (Codex review). + it("keeps a year-first label for YYYY/MM/DD under either preference", () => { + assert.equal(getDefaultDateFormat("2025/03/04"), "YYYY/MM/DD"); + assert.equal(getDefaultDateFormat("2025/03/04", true), "YYYY/MM/DD"); + assert.equal(getDefaultDateFormat("2025/3/4", true), "YYYY/MM/DD", "unpadded too"); + }); + + it("falls back to the preference for anything unrecognised", () => { + assert.equal(getDefaultDateFormat("not a date"), "MM/DD/YYYY"); + assert.equal(getDefaultDateFormat("not a date", true), "DD/MM/YYYY"); + assert.equal(getDefaultDateFormat(""), "MM/DD/YYYY"); + }); +}); diff --git a/tests/engine/test_dateUtils.ts b/tests/engine/test_dateUtils.ts new file mode 100644 index 0000000..1ce2e55 --- /dev/null +++ b/tests/engine/test_dateUtils.ts @@ -0,0 +1,123 @@ +// Excel serial-number conversion. Every date function in the engine routes +// through this pair, and an error here is invisible: a date still renders, it +// is just the wrong day. The epoch choice is the subtle part — Excel's serial +// numbering embeds a 1900 leap-year bug, and the base date compensates for it. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { + dateToSerial, + serialToDate, + DAY_NAMES_FULL, + DAY_NAMES_SHORT, + MONTH_NAMES_FULL, + MONTH_NAMES_SHORT, +} from "../../src/engine/date-utils.ts"; + +const utc = (year: number, month: number, day: number, hour = 0, minute = 0, second = 0) => new Date(Date.UTC(year, month - 1, day, hour, minute, second)); + +describe("dateToSerial", () => { + it("anchors serial 0 at the Dec 30 1899 base", () => { + assert.equal(dateToSerial(utc(1899, 12, 30)), 0); + assert.equal(dateToSerial(utc(1899, 12, 31)), 1); + }); + + it("counts whole days forward", () => { + assert.equal(dateToSerial(utc(1900, 1, 2)), 3); + assert.equal(dateToSerial(utc(1900, 2, 1)), 33); + }); + + // The point of the Dec 30 1899 base: Excel counts a phantom 1900-02-29 that + // never existed, and starting two days early makes every serial from March + // 1900 onward — i.e. every date anyone actually uses — agree with Excel's. + it("agrees with Excel from March 1900 onward", () => { + assert.equal(dateToSerial(utc(1900, 3, 1)), 61); // Excel: 61 + assert.equal(dateToSerial(utc(2000, 1, 1)), 36526); // Excel: 36526 + assert.equal(dateToSerial(utc(2026, 7, 22)), 46225); + }); + + // The flip side of that choice, pinned so it is a known limitation rather + // than a surprise: for the first two months of 1900 the serials sit one + // ahead of Excel's, because the phantom leap day has not been passed yet. + it("sits one ahead of Excel for Jan and Feb 1900", () => { + assert.equal(dateToSerial(utc(1900, 1, 1)), 2); // Excel: 1 + assert.equal(dateToSerial(utc(1900, 2, 28)), 60); // Excel: 59 + }); + + it("represents a time of day as the fractional part", () => { + assert.equal(dateToSerial(utc(1899, 12, 31, 12)), 1.5); + assert.equal(dateToSerial(utc(1899, 12, 31, 6)), 1.25); + }); + + it("goes negative for dates before the base", () => { + assert.ok(dateToSerial(utc(1899, 12, 29)) < 0); + }); +}); + +describe("serialToDate", () => { + it("maps serial 1 back to the day after the base", () => { + assert.equal(serialToDate(1).toISOString().slice(0, 10), "1899-12-31"); + }); + + it("maps a modern serial back to its date", () => { + assert.equal(serialToDate(46225).toISOString().slice(0, 10), "2026-07-22"); + assert.equal(serialToDate(61).toISOString().slice(0, 10), "1900-03-01"); + }); + + it("restores the time of day from the fractional part", () => { + assert.equal(serialToDate(1.5).toISOString().slice(11, 19), "12:00:00"); + assert.equal(serialToDate(1.25).toISOString().slice(11, 19), "06:00:00"); + }); + + // The fraction is rounded to the nearest second, so a value that lands + // mid-second must not drift to the previous one. + it("rounds the time component to the nearest second", () => { + const almostOneMinute = 1 + 59.6 / 86400; + assert.equal(serialToDate(almostOneMinute).toISOString().slice(11, 19), "00:01:00"); + }); +}); + +describe("dateToSerial / serialToDate round-trip", () => { + it("round-trips whole days across month, year and leap boundaries", () => { + const dates = [utc(1900, 1, 1), utc(1900, 3, 1), utc(1999, 12, 31), utc(2000, 2, 29), utc(2024, 2, 29), utc(2026, 7, 22), utc(2100, 1, 1)]; + for (const date of dates) { + const back = serialToDate(dateToSerial(date)); + assert.equal(back.toISOString(), date.toISOString(), `round-trip failed for ${date.toISOString()}`); + } + }); + + it("round-trips a date carrying a time of day", () => { + const date = utc(2026, 7, 22, 13, 45, 30); + assert.equal(serialToDate(dateToSerial(date)).toISOString(), date.toISOString()); + }); +}); + +describe("name tables", () => { + // These are indexed by `getMonth()` / `getDay()` directly, so a wrong length + // or a shifted entry produces an off-by-one month or weekday with no error. + it("has twelve months starting at January", () => { + assert.equal(MONTH_NAMES_SHORT.length, 12); + assert.equal(MONTH_NAMES_FULL.length, 12); + assert.equal(MONTH_NAMES_SHORT[0], "Jan"); + assert.equal(MONTH_NAMES_FULL[0], "January"); + assert.equal(MONTH_NAMES_SHORT[11], "Dec"); + assert.equal(MONTH_NAMES_FULL[11], "December"); + }); + + it("has seven days starting at Sunday, matching Date#getDay", () => { + assert.equal(DAY_NAMES_SHORT.length, 7); + assert.equal(DAY_NAMES_FULL.length, 7); + assert.equal(DAY_NAMES_SHORT[0], "Sun"); + assert.equal(DAY_NAMES_FULL[0], "Sunday"); + assert.equal(DAY_NAMES_SHORT[6], "Sat"); + }); + + it("keeps the short names as prefixes of the full names", () => { + MONTH_NAMES_SHORT.forEach((short, index) => + assert.ok(MONTH_NAMES_FULL[index]?.startsWith(short), `${short} is not a prefix of ${String(MONTH_NAMES_FULL[index])}`), + ); + DAY_NAMES_SHORT.forEach((short, index) => + assert.ok(DAY_NAMES_FULL[index]?.startsWith(short), `${short} is not a prefix of ${String(DAY_NAMES_FULL[index])}`), + ); + }); +}); diff --git a/tests/engine/test_datedif.ts b/tests/engine/test_datedif.ts new file mode 100644 index 0000000..63af48c --- /dev/null +++ b/tests/engine/test_datedif.ts @@ -0,0 +1,133 @@ +// DATEDIF's per-unit elapsed-time math. Each unit has its own boundary handling +// — complete years/months back off when the day-of-month has not been reached, +// MD is the day remainder after the complete months (always non-negative), YD +// wraps into the end's year — and a wrong branch returns a plausible number +// rather than an error. +// +// Inputs are Excel serials, so tests build them from a known date via a helper +// rather than hardcoding the serial arithmetic (that is covered in +// test_dateUtils.ts). + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { computeDatedif } from "../../src/engine/datedif.ts"; +import { dateToSerial } from "../../src/engine/date-utils.ts"; +import { NUM_ERROR } from "../../src/engine/spreadsheet-errors.ts"; + +const serial = (year: number, month: number, day: number) => dateToSerial(new Date(Date.UTC(year, month - 1, day))); +const diff = (start: [number, number, number], end: [number, number, number], unit: string) => computeDatedif(serial(...start), serial(...end), unit); + +describe("computeDatedif — Y (complete years)", () => { + it("counts whole years between the same month and day", () => { + assert.equal(diff([2020, 6, 15], [2023, 6, 15], "Y"), 3); + }); + + // The end has not yet reached the start's month/day in its year, so the last + // year is incomplete. + it("backs off a year when the anniversary has not been reached", () => { + assert.equal(diff([2020, 6, 15], [2023, 6, 14], "Y"), 2); + assert.equal(diff([2020, 6, 15], [2023, 5, 20], "Y"), 2); + }); + + it("counts the year once the anniversary is reached exactly", () => { + assert.equal(diff([2020, 2, 29], [2024, 2, 29], "Y"), 4, "leap day to leap day"); + }); +}); + +describe("computeDatedif — M (complete months)", () => { + it("counts whole months", () => { + assert.equal(diff([2023, 1, 10], [2023, 4, 10], "M"), 3); + assert.equal(diff([2020, 1, 1], [2023, 1, 1], "M"), 36); + }); + + it("backs off a month when the day-of-month has not been reached", () => { + assert.equal(diff([2023, 1, 15], [2023, 4, 10], "M"), 2); + }); +}); + +describe("computeDatedif — D (calendar days)", () => { + it("counts the days between two dates", () => { + assert.equal(diff([2023, 1, 1], [2023, 1, 31], "D"), 30); + assert.equal(diff([2023, 1, 1], [2024, 1, 1], "D"), 365); + assert.equal(diff([2024, 1, 1], [2025, 1, 1], "D"), 366, "leap year"); + }); +}); + +describe("computeDatedif — MD (day-of-month diff, months ignored)", () => { + it("subtracts the days directly when the end day is later", () => { + assert.equal(diff([2023, 1, 10], [2023, 3, 25], "MD"), 15); + }); + + // Multi-month spans still measure the day remainder correctly: Jan 15 → Mar 10 + // has one complete month (to Feb 15), leaving 23 days to Mar 10. + it("measures the day remainder across several months", () => { + assert.equal(diff([2023, 1, 15], [2023, 3, 10], "MD"), 23); + }); + + // The day remainder is anchored on start-plus-complete-months, so it is never + // negative even when the start day outruns the month before `end`. Jan 30 → + // Mar 1 has one complete month (clamped to Feb 28/29), leaving one day — where + // the old borrow-the-previous-month math returned -1 (#2414). + it("stays non-negative when the start day outruns the preceding month", () => { + assert.equal(diff([2023, 1, 30], [2023, 3, 1], "MD"), 1, "Jan 30 + 1 month → Feb 28, then 1 day"); + assert.equal(diff([2024, 1, 30], [2024, 3, 1], "MD"), 1, "leap year: Jan 30 + 1 month → Feb 29, then 1 day"); + }); + + // The remainder counts whole days: a datetime serial's time-of-day must not + // change the result (an 18:00 fraction on `end` once rounded MD up to 2). + it("ignores the time-of-day of a datetime serial", () => { + const start = serial(2023, 1, 30); + const end = serial(2023, 3, 1); + assert.equal(computeDatedif(start, end + 0.75, "MD"), 1, "end at 18:00 still yields 1"); + assert.equal(computeDatedif(start + 0.75, end + 0.25, "MD"), 1, "start and end times both ignored"); + }); +}); + +describe("computeDatedif — YM (month diff, years ignored)", () => { + it("counts months within the year", () => { + assert.equal(diff([2020, 1, 10], [2023, 4, 10], "YM"), 3); + }); + + // Ignoring years can leave the month difference negative; it wraps into 0..11. + it("wraps a negative month difference into the 0..11 range", () => { + assert.equal(diff([2020, 11, 10], [2023, 2, 10], "YM"), 3); + }); + + it("backs off when the day has not been reached, then wraps", () => { + assert.equal(diff([2020, 11, 20], [2023, 2, 10], "YM"), 2); + }); +}); + +describe("computeDatedif — YD (day diff, years ignored)", () => { + it("counts days within the same year window", () => { + assert.equal(diff([2023, 1, 1], [2023, 3, 1], "YD"), 59, "Jan + Feb 2023"); + }); + + // Moving the start into the end's year would put it after the end, so it + // steps back a year and counts across the boundary. + it("crosses the year boundary when the start falls later in the end's year", () => { + assert.equal(diff([2020, 12, 20], [2023, 1, 5], "YD"), 16, "Dec 20 to Jan 5"); + }); +}); + +describe("computeDatedif — errors", () => { + it("returns #NUM! when start is after end", () => { + assert.equal(diff([2023, 6, 15], [2023, 6, 10], "D"), NUM_ERROR); + }); + + it("returns #NUM! for an unknown unit", () => { + assert.equal(diff([2020, 1, 1], [2023, 1, 1], "Q"), NUM_ERROR); + assert.equal(diff([2020, 1, 1], [2023, 1, 1], ""), NUM_ERROR); + }); + + it("matches the unit case-insensitively", () => { + assert.equal(diff([2020, 1, 1], [2023, 1, 1], "y"), 3); + assert.equal(diff([2023, 1, 1], [2023, 1, 31], "d"), 30); + }); + + it("returns 0 for identical dates in every unit", () => { + for (const unit of ["Y", "M", "D", "MD", "YM", "YD"]) { + assert.equal(diff([2023, 6, 15], [2023, 6, 15], unit), 0, `unit ${unit}`); + } + }); +}); diff --git a/tests/engine/test_errorReporting.ts b/tests/engine/test_errorReporting.ts new file mode 100644 index 0000000..3d892ca --- /dev/null +++ b/tests/engine/test_errorReporting.ts @@ -0,0 +1,101 @@ +// Failed formulas must surface as TYPED errors with an Excel-style error value in +// the cell — not silently become a bare string or a wrong number with an empty +// errors[] (issue #2359). The root cause was the over-broad top-level catch in +// evaluateFormula, which meant the function never threw, so calculator.ts's +// per-cell catch was unreachable and only "circular" was ever recorded. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { SpreadsheetEngine, type SheetData, type CalculatedSheet } from "../../src/engine/index.ts"; +import { cellAt } from "./cellAccess.ts"; + +const calc = (sheet: SheetData, all?: SheetData[]): CalculatedSheet => new SpreadsheetEngine().calculate(sheet, all ?? [sheet]); + +/** The single formula cell's value and the error type recorded for it, if any. */ +function cellAndError(sheet: SheetData, row: number, col: number, all?: SheetData[]) { + const result = calc(sheet, all); + const entry = result.errors.find((err) => err.cell.row === row && err.cell.col === col); + return { value: cellAt(result.data, row, col), errorType: entry?.type }; +} + +describe("#2359 typed error reporting", () => { + it("div_zero: =1/0 becomes #DIV/0!, not Infinity", () => { + const { value, errorType } = cellAndError({ name: "S", data: [[{ v: "=1/0" }]] }, 0, 0); + assert.equal(value, "#DIV/0!"); + assert.equal(errorType, "div_zero"); + }); + + it("div_zero: a reference division by zero (=A1/A2, 10/0) is #DIV/0!", () => { + const sheet: SheetData = { name: "S", data: [[{ v: 10 }], [{ v: 0 }], [{ v: "=A1/A2" }]] }; + const { value, errorType } = cellAndError(sheet, 2, 0); + assert.equal(value, "#DIV/0!"); + assert.equal(errorType, "div_zero"); + }); + + it("invalid_ref: a reference to a missing sheet is #REF!", () => { + const sheet: SheetData = { name: "S", data: [[{ v: "=Missing!A1" }]] }; + const { value, errorType } = cellAndError(sheet, 0, 0); + assert.equal(value, "#REF!"); + assert.equal(errorType, "invalid_ref"); + }); + + it("syntax: an unknown function is #NAME?, not a processed string", () => { + const sheet: SheetData = { name: "S", data: [[{ v: 7 }, { v: "=UNKNOWNFN(A1)" }]] }; + const { value, errorType } = cellAndError(sheet, 0, 1); + assert.equal(value, "#NAME?"); + assert.equal(errorType, "syntax"); + }); + + // IFS needs condition/value PAIRS — a count the registry's min/max cannot + // express, so the handler itself throws. (This used to use SUM over two + // ranges, which now legitimately sums them.) + it("unknown: a handler that throws (IFS with an odd argument count) is #ERROR!, not the formula text", () => { + const sheet: SheetData = { name: "S", data: [[{ v: 1 }, { v: '=IFS(A1>0, "yes", "orphan")' }]] }; + const { value, errorType } = cellAndError(sheet, 0, 1); + assert.equal(value, "#ERROR!"); + assert.equal(errorType, "unknown"); + }); + + it("propagates an error through an arithmetic reference (=A1+1 where A1 is #DIV/0!)", () => { + const sheet: SheetData = { name: "S", data: [[{ v: "=1/0" }, { v: "=A1+1" }]] }; + const { value, errorType } = cellAndError(sheet, 0, 1); + assert.equal(value, "#DIV/0!", "must not become the bare string 'Infinity+1'"); + assert.equal(errorType, "div_zero"); + }); + + it("a failed formula never lands in the cell as a bare string", () => { + const formulas = ["=1/0", "=UNKNOWNFN(A1)", '=IFS(A1>0, "yes", "orphan")']; + for (const formula of formulas) { + const value = cellAt(calc({ name: "S", data: [[{ v: formula }]] }).data, 0, 0); + assert.equal(typeof value === "string" && value.startsWith("#"), true, `${formula} → ${JSON.stringify(value)} should be an # error value`); + } + }); +}); + +describe("#2359 success paths are preserved", () => { + it("keeps circular-reference detection working", () => { + const sheet: SheetData = { name: "S", data: [[{ v: "=B1+1" }, { v: "=A1+1" }]] }; + const result = calc(sheet); + assert.equal( + result.errors.some((err) => err.type === "circular"), + true, + ); + }); + + it("=ZZ999 (an empty in-bounds cell) stays 0 and is NOT an error", () => { + const { value, errorType } = cellAndError({ name: "S", data: [[{ v: "=ZZ999" }]] }, 0, 0); + assert.equal(value, 0); + assert.equal(errorType, undefined); + }); + + it("valid SUM and arithmetic are unaffected", () => { + assert.equal(cellAt(calc({ name: "S", data: [[{ v: 1 }, { v: "=SUM(A1:A3)" }], [{ v: 2 }], [{ v: 3 }]] }).data, 0, 1), 6); + assert.equal(cellAt(calc({ name: "S", data: [[{ v: 2 }], [{ v: 3 }], [{ v: "=A1+A2" }]] }).data, 2, 0), 5); + }); + + it("a valid cross-sheet reference still resolves", () => { + const data: SheetData = { name: "Data", data: [[{ v: 100 }]] }; + const summary: SheetData = { name: "Summary", data: [[{ v: "=Data!A1*2" }]] }; + assert.equal(cellAt(calc(summary, [data, summary]).data, 0, 0), 200); + }); +}); diff --git a/tests/engine/test_errorValue.ts b/tests/engine/test_errorValue.ts new file mode 100644 index 0000000..fe01bca --- /dev/null +++ b/tests/engine/test_errorValue.ts @@ -0,0 +1,180 @@ +// Formula errors as a distinct VALUE (#2451). +// +// While errors were plain strings, `SQRT(-1)` and `CONCAT("#N","UM!")` both +// produced "#NUM!", so IFERROR could not tell a real error from text that +// merely spells one — the computed case was caught as an error and silently +// replaced by the fallback. An error is now its own value carrying the code; +// text stays text. The display pass renders the value back to `#NUM!`, so the +// cells look exactly as they did. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { SpreadsheetEngine, type SheetData } from "../../src/engine/index.ts"; +import { evaluateFormula } from "../../src/engine/evaluator.ts"; +import { formatCellForDisplay } from "../../src/engine/cellFormatting.ts"; +import { + DIV_ZERO_ERROR, + NA_ERROR, + NUM_ERROR, + SpreadsheetError, + isSpreadsheetErrorValue, + isErrorResult, + spreadsheetError, +} from "../../src/engine/spreadsheet-errors.ts"; +import type { CellValue } from "../../src/engine/types.ts"; +import { cellAt } from "./cellAccess.ts"; + +/** The raw computed value of a formula, before the display pass turns an error + * back into its code — this is where provenance is observable. */ +const evaluateRaw: (formula: string) => CellValue = (formula) => + evaluateFormula(formula, { + getCellValue: () => 0, + getRangeValues: () => [], + evaluateFormula: evaluateRaw, + }); + +/** What a single-formula sheet DISPLAYS, through the public engine API. */ +const displayed = (formula: string): CellValue => + cellAt(new SpreadsheetEngine().calculate({ name: "S", data: [[{ v: formula }]] } satisfies SheetData).data, 0, 0); + +describe("the error value and its guard", () => { + it("recognises an error value and rejects a string that spells the same code", () => { + assert.equal(isSpreadsheetErrorValue(NUM_ERROR), true); + assert.equal(isSpreadsheetErrorValue("#NUM!"), false); + assert.equal(isSpreadsheetErrorValue(0), false); + assert.equal(isSpreadsheetErrorValue(null), false); + }); + + it("carries the code and renders as it when coerced to text", () => { + assert.equal(NUM_ERROR.code, "#NUM!"); + assert.equal(String(NUM_ERROR), "#NUM!"); + assert.equal(`${DIV_ZERO_ERROR}`, "#DIV/0!"); + }); + + it("hands out one instance per code, so two errors of a kind compare equal", () => { + const fromLookup = spreadsheetError("#N/A"); + const fromLookupAgain = spreadsheetError("#N/A"); + assert.equal(fromLookup, NA_ERROR); + assert.equal(fromLookup === fromLookupAgain, true); + assert.equal(NA_ERROR instanceof SpreadsheetError, true); + }); + + it("serializes to its code rather than to an empty object", () => { + assert.equal(JSON.stringify({ cell: NUM_ERROR }), '{"cell":"#NUM!"}'); + }); +}); + +describe("isErrorResult keys off the value, not the text", () => { + it("catches an error value", () => { + assert.equal(isErrorResult(NUM_ERROR), true); + assert.equal(isErrorResult(DIV_ZERO_ERROR), true); + }); + + it("still catches NaN / infinity / missing", () => { + assert.equal(isErrorResult(NaN), true); + assert.equal(isErrorResult(Infinity), true); + assert.equal(isErrorResult(null), true); + }); + + it("does NOT catch a look-alike string — the whole point of #2451", () => { + assert.equal(isErrorResult("#NUM!"), false); + assert.equal(isErrorResult("#N/A"), false); + }); +}); + +describe("the display pass renders an error value to its code", () => { + it("returns the code for a formula cell", () => { + assert.equal(formatCellForDisplay({ v: "=SQRT(-1)" }, NUM_ERROR, false), "#NUM!"); + }); + + it("returns the code regardless of the cell's format code", () => { + assert.equal(formatCellForDisplay({ v: "=A1/A2", f: "$#,##0.00" }, DIV_ZERO_ERROR, false), "#DIV/0!"); + }); + + it("leaves ordinary values alone", () => { + assert.equal(formatCellForDisplay({ v: "=1+1" }, 2, false), 2); + assert.equal(formatCellForDisplay({ v: "text" }, "text", false), "text"); + }); +}); + +describe("functions return an error VALUE, and the cell still shows its code", () => { + it("SQRT(-1) computes to the #NUM! value", () => { + const value = evaluateRaw("SQRT(-1)"); + assert.equal(isSpreadsheetErrorValue(value), true); + assert.equal(value, NUM_ERROR); + }); + + it("MOD(5, 0) computes to the #DIV/0! value", () => { + assert.equal(evaluateRaw("MOD(5, 0)"), DIV_ZERO_ERROR); + }); + + it("computed text that spells an error stays a plain string", () => { + const value = evaluateRaw('CONCAT("#N","UM!")'); + assert.equal(isSpreadsheetErrorValue(value), false); + assert.equal(value, "#NUM!"); + }); + + it("displays the same codes end-to-end as before the refactor", () => { + assert.equal(displayed("=SQRT(-1)"), "#NUM!"); + assert.equal(displayed("=MOD(5, 0)"), "#DIV/0!"); + assert.equal(displayed("=1/0"), "#DIV/0!"); + assert.equal(displayed('=DATEDIF(45000, 44000, "D")'), "#NUM!"); + assert.equal(displayed('=VALUE("abc")'), "#VALUE!"); + }); + + it("propagates an error VALUE through a reference and shows the code", () => { + const sheet: SheetData = { name: "S", data: [[{ v: "=SQRT(-1)" }, { v: "=A1+1" }]] }; + assert.equal(cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 1), "#NUM!"); + }); +}); + +describe("IFERROR keys off provenance", () => { + it("catches a real error", () => { + assert.equal(displayed("=IFERROR(SQRT(-1), 42)"), 42); + assert.equal(displayed("=IFERROR(MOD(5, 0), -1)"), -1); + }); + + it("passes a non-error through untouched", () => { + assert.equal(displayed("=IFERROR(SQRT(4), 42)"), 2); + assert.equal(displayed('=IFERROR("hello", 42)'), "hello"); + }); + + it("does not catch a quoted literal that only looks like an error", () => { + assert.equal(displayed('=IFERROR("#NUM!", 42)'), "#NUM!"); + }); + + // THE headline case. Before #2451 this returned 42: the computed text was + // indistinguishable from a real #NUM!, so IFERROR swallowed it. + it("does not catch COMPUTED text that spells an error", () => { + assert.equal(displayed('=IFERROR(CONCAT("#N","UM!"), 42)'), "#NUM!"); + assert.equal(displayed('=IFERROR(CONCATENATE("#DIV/", "0!"), 42)'), "#DIV/0!"); + }); + + it("does not catch an error-looking string built with the & operator", () => { + assert.equal(displayed('=IFERROR("#N" & "UM!", 42)'), "#NUM!"); + }); + + it("still catches an error that reaches it through arithmetic", () => { + assert.equal(displayed("=IFERROR(SQRT(-1) + 1, 42)"), 42); + }); +}); + +describe("IFNA keys off the error value's code", () => { + it("substitutes the fallback for a real #N/A", () => { + const sheet: SheetData = { name: "S", data: [[{ v: 1 }, { v: '=IFNA(MATCH(99, A1:A1, 0), "missing")' }]] }; + assert.equal(cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 1), "missing"); + }); + + it("leaves a different error alone", () => { + assert.equal(displayed('=IFNA(SQRT(-1), "missing")'), "#NUM!"); + }); + + it("does not substitute for text that merely spells #N/A", () => { + assert.equal(displayed('=IFNA("#N/A", "missing")'), "#N/A"); + assert.equal(displayed('=IFNA(CONCAT("#N", "/A"), "missing")'), "#N/A"); + }); + + it("passes an ordinary value through", () => { + assert.equal(displayed('=IFNA(7, "missing")'), 7); + }); +}); diff --git a/tests/engine/test_expandRangeOrCell.ts b/tests/engine/test_expandRangeOrCell.ts new file mode 100644 index 0000000..437cc2b --- /dev/null +++ b/tests/engine/test_expandRangeOrCell.ts @@ -0,0 +1,83 @@ +// Turning a range or single-cell reference into coordinates. The calculator's +// old inline regex was range-only and case-sensitive and did not strip `$`, so +// three common reference shapes fell through to "no values" — and a function +// over an empty list is 0, not an error (#2356). + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { expandRangeOrCell } from "../../src/engine/formulaRefs.ts"; + +describe("expandRangeOrCell — ranges", () => { + it("expands a simple range top-to-bottom, left-to-right", () => { + assert.deepEqual(expandRangeOrCell("A1:B2"), [ + { row: 0, col: 0 }, + { row: 0, col: 1 }, + { row: 1, col: 0 }, + { row: 1, col: 1 }, + ]); + }); + + it("expands a single-column range", () => { + assert.deepEqual(expandRangeOrCell("A1:A3"), [ + { row: 0, col: 0 }, + { row: 1, col: 0 }, + { row: 2, col: 0 }, + ]); + }); + + // The fill-down form. The old regex left the `$` in and matched nothing. + it("strips absolute-reference dollar signs", () => { + assert.deepEqual(expandRangeOrCell("$A$1:$A$3"), expandRangeOrCell("A1:A3")); + assert.deepEqual(expandRangeOrCell("$A1:A$3"), expandRangeOrCell("A1:A3")); + }); + + // Spreadsheets accept lowercase and upcase it; the old regex was `[A-Z]` + // only, so a lowercase range silently produced nothing. + it("upcases lowercase references", () => { + assert.deepEqual(expandRangeOrCell("a1:b2"), expandRangeOrCell("A1:B2")); + assert.deepEqual(expandRangeOrCell("$a$1:$a$3"), expandRangeOrCell("A1:A3")); + }); + + it("tolerates surrounding whitespace", () => { + assert.deepEqual(expandRangeOrCell(" A1:A2 "), expandRangeOrCell("A1:A2")); + }); + + it("crosses the Z→AA column boundary", () => { + assert.deepEqual(expandRangeOrCell("Z1:AA1"), [ + { row: 0, col: 25 }, + { row: 0, col: 26 }, + ]); + }); +}); + +describe("expandRangeOrCell — single cells", () => { + // The case Excel sums as one value and the old regex refused for lack of a + // colon. + it("expands a bare cell to one coordinate", () => { + assert.deepEqual(expandRangeOrCell("A1"), [{ row: 0, col: 0 }]); + assert.deepEqual(expandRangeOrCell("B3"), [{ row: 2, col: 1 }]); + }); + + it("strips dollar signs and upcases a single cell", () => { + assert.deepEqual(expandRangeOrCell("$A$1"), [{ row: 0, col: 0 }]); + assert.deepEqual(expandRangeOrCell("a1"), [{ row: 0, col: 0 }]); + }); + + it("reads a multi-letter column", () => { + assert.deepEqual(expandRangeOrCell("AA10"), [{ row: 9, col: 26 }]); + }); +}); + +describe("expandRangeOrCell — non-references", () => { + // Null rather than an empty array: the caller distinguishes "not a reference" + // from "a valid but empty range", and returning [] for garbage would hide + // typos as zero-value sums. + it("returns null for text that is not a reference", () => { + assert.equal(expandRangeOrCell("hello"), null); + assert.equal(expandRangeOrCell(""), null); + assert.equal(expandRangeOrCell("A"), null); + assert.equal(expandRangeOrCell("1"), null); + assert.equal(expandRangeOrCell("A1:B"), null); + assert.equal(expandRangeOrCell("A1:"), null); + }); +}); diff --git a/tests/engine/test_financialMath.ts b/tests/engine/test_financialMath.ts new file mode 100644 index 0000000..05db2eb --- /dev/null +++ b/tests/engine/test_financialMath.ts @@ -0,0 +1,188 @@ +// The per-period interest/principal split of an annuity. The bug this covers +// returned a plausible NUMBER — IPMT came back +1250 for a -1250 interest +// payment (sign inverted), and PPMT (= PMT - IPMT) amplified it to -2748.88 +// instead of -248.88 (#2386). Values are checked against Excel. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { + computeFv, + computePmt, + computeIpmt, + computePpmt, + computePv, + computeNper, + computeRate, + computeNpv, + computeIrr, +} from "../../src/engine/financial-math.ts"; +import { NUM_ERROR, type SpreadsheetError } from "../../src/engine/spreadsheet-errors.ts"; + +const closeTo = (actual: number, expected: number, eps = 0.01): boolean => Math.abs(actual - expected) <= eps; + +/** RATE / IRR may answer `#NUM!`; the convergent cases assert on the number. */ +const converged = (result: number | SpreadsheetError): number => { + if (typeof result !== "number") throw new Error(`expected a rate, got ${String(result)}`); + return result; +}; + +// A 250,000 loan at 0.5%/period over 360 periods — the issue's worked example. +const RATE = 0.005; +const NPER = 360; +const PRINCIPAL = 250000; + +describe("computePmt", () => { + it("matches Excel's constant payment (negative outflow)", () => { + assert.ok(closeTo(computePmt(RATE, NPER, PRINCIPAL, 0, 0), -1498.88), "PMT ≈ -1498.88"); + }); + + it("splits a zero-interest loan evenly", () => { + assert.ok(closeTo(computePmt(0, 10, 1000, 0, 0), -100, 1e-9), "zero-rate PMT"); + }); +}); + +describe("computeIpmt", () => { + it("returns the first period's interest with Excel's sign", () => { + // Interest on the full 250,000 balance: 250000 * 0.005 = 1250, as a payment + // it is negative. The bug returned +1250. + assert.ok(closeTo(computeIpmt(RATE, 1, NPER, PRINCIPAL, 0, 0), -1250, 1e-9), "IPMT(1) = -1250"); + }); + + it("decreases in magnitude as the balance is paid down", () => { + assert.ok(closeTo(computeIpmt(RATE, 2, NPER, PRINCIPAL, 0, 0), -1248.76), "IPMT(2) ≈ -1248.76"); + }); + + it("has no interest in the first period of an annuity due", () => { + assert.equal(computeIpmt(RATE, 1, NPER, PRINCIPAL, 0, 1), 0); + }); +}); + +describe("computePpmt", () => { + it("returns the first period's principal, not a wildly wrong value", () => { + // PMT - IPMT = -1498.88 - (-1250) = -248.88. The sign bug made this -2748.88. + assert.ok(closeTo(computePpmt(RATE, 1, NPER, PRINCIPAL, 0, 0), -248.88), "PPMT(1) ≈ -248.88"); + }); +}); + +describe("the interest and principal split reconstitutes the payment", () => { + it("IPMT(per) + PPMT(per) == PMT for every period", () => { + const pmt = computePmt(RATE, NPER, PRINCIPAL, 0, 0); + for (const per of [1, 2, 12, 180, 360]) { + const split = computeIpmt(RATE, per, NPER, PRINCIPAL, 0, 0) + computePpmt(RATE, per, NPER, PRINCIPAL, 0, 0); + assert.ok(closeTo(split, pmt, 1e-9), `period ${per}: IPMT + PPMT == PMT`); + } + }); +}); + +describe("computeNpv", () => { + const NPV_RATE = 0.1; + + // Each flow discounts by its 1-based POSITION in the flattened list. The #2390 + // bug used the argument index, so a scalar after a 3-cell range landed at + // period 2 instead of 4 — the position is what makes 100/1.1 + 200/1.1^2 + + // 300/1.1^3 + 500/1.1^4 correct. + it("discounts each flow by its 1-based position", () => { + const expected = 100 / 1.1 + 200 / 1.1 ** 2 + 300 / 1.1 ** 3 + 500 / 1.1 ** 4; + assert.ok(closeTo(computeNpv(NPV_RATE, [100, 200, 300, 500]), expected, 1e-9)); + }); + + it("sums flows undiscounted at a zero rate", () => { + assert.ok(closeTo(computeNpv(0, [100, 200, 300, 500]), 1100, 1e-9)); + }); + + it("is zero for no cash flows", () => { + assert.equal(computeNpv(NPV_RATE, []), 0); + }); + + it("discounts a single flow by one period", () => { + assert.ok(closeTo(computeNpv(NPV_RATE, [100]), 100 / 1.1, 1e-9)); + }); +}); + +describe("computeFv", () => { + it("carries the payment-negative sign the interest split relies on", () => { + // Balance outstanding at the start of period 1 is the present value, which + // FV expresses as its negative. + assert.ok(closeTo(computeFv(RATE, 0, computePmt(RATE, NPER, PRINCIPAL, 0, 0), PRINCIPAL, 0), -PRINCIPAL, 1e-9), "FV of pv over 0 periods = -pv"); + }); + + it("sums a zero-rate stream directly", () => { + assert.ok(closeTo(computeFv(0, 10, -100, 0, 0), 1000, 1e-9), "zero-rate FV"); + }); +}); + +// PV / NPER / RATE / NPV / IRR were inline in the handlers before this refactor; +// the values below were captured from the pre-refactor formulas (verbatim) and +// cross-checked against Excel, so they double as regression pins. + +describe("computePv", () => { + it("matches Excel's present value of an annuity", () => { + // Excel PV(0.05, 10, -1000) = 7721.73 (paying out 1000/period is a positive PV). + assert.ok(closeTo(computePv(0.05, 10, -1000, 0, 0), 7721.73), "PV ≈ 7721.73"); + }); + + it("discounts a zero-rate stream to its undiscounted total", () => { + assert.ok(closeTo(computePv(0, 10, -100, 0, 0), 1000, 1e-9), "zero-rate PV"); + }); + + it("is the inverse of PMT — it recovers the principal from that payment", () => { + const payment = computePmt(RATE, NPER, PRINCIPAL, 0, 0); + assert.ok(closeTo(computePv(RATE, NPER, payment, 0, 0), PRINCIPAL, 1e-6), "PV(PMT(pv)) == pv"); + }); +}); + +describe("computeNper", () => { + it("counts the periods needed to pay off a loan (Excel value)", () => { + // Excel NPER(0.05, -1000, 8000) = 10.47. + assert.ok(closeTo(computeNper(0.05, -1000, 8000, 0, 0), 10.47), "NPER ≈ 10.47"); + }); + + it("splits a zero-rate balance into equal periods", () => { + assert.ok(closeTo(computeNper(0, -100, 1000, 0, 0), 10, 1e-9), "zero-rate NPER"); + }); +}); + +describe("computeRate", () => { + it("recovers the rate implied by a known payment via Newton-Raphson", () => { + const payment = computePmt(0.05, 12, 1000, 0, 0); + assert.ok(closeTo(converged(computeRate(12, payment, 1000, 0, 0, 0.1)), 0.05, 1e-6), "RATE recovers 0.05"); + }); + + it("reports #NUM! instead of a divergent rate when no root exists", () => { + assert.equal(computeRate(10, 100, 100, 100, 0, 0.1), NUM_ERROR); + }); +}); + +describe("computeNpv", () => { + it("discounts each cash flow one period further out (Excel NPV)", () => { + // Excel NPV(0.1, -10000, 3000, 4200, 6800) = 1188.44 — the first flow is + // discounted one period, unlike IRR which places element 0 at period 0. + assert.ok(closeTo(computeNpv(0.1, [-10000, 3000, 4200, 6800]), 1188.44), "NPV ≈ 1188.44"); + }); + + it("discounts a single flow by exactly one period", () => { + assert.ok(closeTo(computeNpv(0.1, [100]), 90.9090909, 1e-6), "100 / 1.1"); + }); + + it("returns zero for an empty cash-flow series", () => { + assert.equal(computeNpv(0.1, []), 0); + }); +}); + +describe("computeIrr", () => { + it("finds the rate that zeroes the NPV (Excel IRR)", () => { + assert.ok(closeTo(converged(computeIrr([-100, 60, 60], 0.1)), 0.130662, 1e-5), "IRR ≈ 0.130662"); + }); + + it("drives the period-0 discounted cash flows to zero at the returned rate", () => { + const irr = converged(computeIrr([-1000, 500, 400, 300, 100], 0.1)); + const npvFromPeriodZero = [-1000, 500, 400, 300, 100].reduce((sum, value, index) => sum + value / (1 + irr) ** index, 0); + assert.ok(closeTo(npvFromPeriodZero, 0, 1e-6), "NPV at IRR ≈ 0"); + }); + + // Same-sign cash flows have no internal rate of return: the derivative collapses + // and there is nowhere to step, which Excel reports as #NUM!. + it("reports #NUM! when the cash flows never change sign", () => { + assert.equal(computeIrr([100, 200, 300], 0.1), NUM_ERROR); + }); +}); diff --git a/tests/engine/test_financialPeriodicHandlers.ts b/tests/engine/test_financialPeriodicHandlers.ts new file mode 100644 index 0000000..c758ea7 --- /dev/null +++ b/tests/engine/test_financialPeriodicHandlers.ts @@ -0,0 +1,34 @@ +// IPMT and PPMT now share one arg-parsing factory, makePeriodicComponentHandler +// (#2482). These drive both THROUGH the engine — the layer the factory lives in — +// so a swapped compute call or a mis-parsed optional arg (fv / type) is caught. +// The pure computeIpmt / computePpmt tests exercise the math, not the handler. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { SpreadsheetEngine, type SheetData } from "../../src/engine/index.ts"; +import { cellAt } from "./cellAccess.ts"; + +const evalA1 = (formula: string): unknown => { + const sheet: SheetData = { name: "S", data: [[{ v: formula }]] }; + return cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 0); +}; + +const closeTo = (actual: unknown, expected: number, eps = 0.01): boolean => typeof actual === "number" && Math.abs(actual - expected) <= eps; + +describe("IPMT / PPMT through the engine (shared handler factory)", () => { + it("IPMT(0.005, 1, 360, 250000) is the interest-only first payment (-1250)", () => { + assert.ok(closeTo(evalA1("=IPMT(0.005, 1, 360, 250000)"), -1250, 1e-6), `got ${String(evalA1("=IPMT(0.005, 1, 360, 250000)"))}`); + }); + + it("PPMT(0.005, 1, 360, 250000) is the principal-only first payment (~ -248.88)", () => { + assert.ok(closeTo(evalA1("=PPMT(0.005, 1, 360, 250000)"), -248.88), `got ${String(evalA1("=PPMT(0.005, 1, 360, 250000)"))}`); + }); + + it("IPMT and PPMT stay distinct — the factory did not collapse them onto one compute", () => { + assert.notEqual(evalA1("=IPMT(0.005, 2, 360, 250000)"), evalA1("=PPMT(0.005, 2, 360, 250000)")); + }); + + it("parses the optional type arg: IPMT at period 1, begin-of-period, is 0", () => { + assert.equal(evalA1("=IPMT(0.005, 1, 360, 250000, 0, 1)"), 0); + }); +}); diff --git a/tests/engine/test_formatter.ts b/tests/engine/test_formatter.ts new file mode 100644 index 0000000..937d74b --- /dev/null +++ b/tests/engine/test_formatter.ts @@ -0,0 +1,157 @@ +// Excel format codes → display strings. Nothing here throws: a bad format code +// produces a bad-looking cell, and a wrong decimal count produces a number that +// is simply off. Both read as ordinary output. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { addThousandSeparators, formatNumber } from "../../src/engine/formatter.ts"; +import { dateToSerial } from "../../src/engine/date-utils.ts"; + +const serialOf = (year: number, month: number, day: number, hour = 0, minute = 0, second = 0) => + dateToSerial(new Date(Date.UTC(year, month - 1, day, hour, minute, second))); + +const MAR_4_2025 = serialOf(2025, 3, 4); + +describe("formatNumber — numeric date formats", () => { + it("renders the padded and unpadded numeric orders", () => { + assert.equal(formatNumber(MAR_4_2025, "MM/DD/YYYY"), "03/04/2025"); + assert.equal(formatNumber(MAR_4_2025, "M/D/YYYY"), "3/4/2025"); + assert.equal(formatNumber(MAR_4_2025, "YYYY-MM-DD"), "2025-03-04"); + assert.equal(formatNumber(MAR_4_2025, "DD/MM/YYYY"), "04/03/2025"); + }); + + it("renders a two-digit year", () => { + assert.equal(formatNumber(MAR_4_2025, "MM/DD/YY"), "03/04/25"); + }); +}); + +describe("formatNumber — month-name formats", () => { + // The regression from #2330: substitution used to run as a sequence of + // replaces, so `/M/g` fired AFTER the month name had been inserted and + // rewrote the M inside "Mar" / "March" — and "March" then lost its "h" to + // the hour token, giving "3arc0". + it("renders month names intact", () => { + assert.equal(formatNumber(MAR_4_2025, "DD-MMM-YYYY"), "04-Mar-2025"); + assert.equal(formatNumber(MAR_4_2025, "MMM D, YYYY"), "Mar 4, 2025"); + assert.equal(formatNumber(MAR_4_2025, "MMMM D, YYYY"), "March 4, 2025"); + }); + + // These are the shapes `getDefaultDateFormat` hands back for the matching + // input, so a user typing "4-Mar-2025" gets this format applied to their own + // cell without asking for it. + it("round-trips the formats getDefaultDateFormat infers", () => { + assert.equal(formatNumber(serialOf(2025, 9, 30), "MMMM D, YYYY"), "September 30, 2025"); + assert.equal(formatNumber(serialOf(2025, 12, 1), "DD-MMM-YYYY"), "01-Dec-2025"); + }); + + it("renders weekday names", () => { + assert.equal(formatNumber(MAR_4_2025, "dddd"), "Tuesday"); + assert.equal(formatNumber(MAR_4_2025, "ddd"), "Tue"); + }); +}); + +describe("formatNumber — time formats", () => { + it("renders 24-hour time", () => { + assert.equal(formatNumber(serialOf(2025, 3, 4, 13, 45, 30), "HH:mm:ss"), "13:45:30"); + assert.equal(formatNumber(serialOf(2025, 3, 4, 9, 5, 0), "HH:mm"), "09:05"); + }); + + it("renders 12-hour time with a meridiem", () => { + assert.equal(formatNumber(serialOf(2025, 3, 4, 13, 45), "h:mm AM/PM"), "1:45 PM"); + assert.equal(formatNumber(serialOf(2025, 3, 4, 9, 5), "h:mm AM/PM"), "9:05 AM"); + }); + + // Midnight and noon are the two values a `% 12` gets wrong without the + // `|| 12` fallback. + it("renders midnight as 12 AM and noon as 12 PM", () => { + assert.equal(formatNumber(serialOf(2025, 3, 4, 0, 0), "h:mm AM/PM"), "12:00 AM"); + assert.equal(formatNumber(serialOf(2025, 3, 4, 12, 0), "h:mm AM/PM"), "12:00 PM"); + }); +}); + +describe("formatNumber — currency", () => { + it("renders a currency amount with separators and decimals", () => { + assert.equal(formatNumber(1234.5, "$#,##0.00"), "$1,234.50"); + assert.equal(formatNumber(1234.5, "$#,##0"), "$1,235"); + assert.equal(formatNumber(1234.5, "$0.00"), "$1234.50"); + }); + + it("groups every three digits", () => { + assert.equal(formatNumber(1234567.89, "$#,##0.00"), "$1,234,567.89"); + assert.equal(formatNumber(100, "$#,##0"), "$100"); + assert.equal(formatNumber(1000, "$#,##0"), "$1,000"); + }); + + // The sign goes outside the symbol: "-$1,000.00", not "$-1,000.00". + it("puts the minus sign before the currency symbol", () => { + assert.equal(formatNumber(-1000, "$#,##0.00"), "-$1,000.00"); + }); + + it("renders zero", () => { + assert.equal(formatNumber(0, "$#,##0.00"), "$0.00"); + }); +}); + +describe("formatNumber — percentage", () => { + it("multiplies by 100 and appends the sign", () => { + assert.equal(formatNumber(0.5, "0.0%"), "50.0%"); + assert.equal(formatNumber(0.1234, "0.00%"), "12.34%"); + assert.equal(formatNumber(1, "0.00%"), "100.00%"); + }); + + // The decimal count is read from a `.0+` run in the format. A format with no + // such run falls back to 2 for percentages while currency falls back to 0 — + // an asymmetry worth knowing about, since "0%" renders as "50.00%". + it("falls back to two decimals when the format declares none", () => { + assert.equal(formatNumber(0.5, "0%"), "50.00%"); + }); +}); + +describe("formatNumber — plain numbers", () => { + it("renders a fixed number of decimals", () => { + assert.equal(formatNumber(1234.5678, "0.00"), "1234.57"); + assert.equal(formatNumber(1234.5678, "0.000"), "1234.568"); + }); + + it("renders thousands separators without a currency symbol", () => { + assert.equal(formatNumber(1234567, "#,##0"), "1,234,567"); + assert.equal(formatNumber(1234.5, "#,##0.00"), "1,234.50"); + assert.equal(formatNumber(-1234.5, "#,##0.00"), "-1,234.50"); + }); + + it("returns the raw number when there is no format", () => { + assert.equal(formatNumber(1234.5, ""), "1234.5"); + }); + + // `#` placeholders are not read at all — only a literal `.0+` run sets the + // decimal count — so a format built from them is ignored entirely. + it("ignores a format whose decimals are written with # placeholders", () => { + assert.equal(formatNumber(0.5, "0.##"), "0.5"); + assert.equal(formatNumber(1234.5678, "0.###"), "1234.5678"); + }); + + it("reads the decimal count from the leading .0 run of a mixed format", () => { + assert.equal(formatNumber(0.5, "0.0#"), "0.5"); + }); +}); + +// The split/group/join wrapper the currency and plain-comma branches of +// formatNumber both share: group the integer part, leave any fraction alone. +describe("addThousandSeparators", () => { + it("groups the integer part in threes", () => { + assert.equal(addThousandSeparators("1000"), "1,000"); + assert.equal(addThousandSeparators("1234567"), "1,234,567"); + }); + + it("leaves the fractional part untouched", () => { + assert.equal(addThousandSeparators("1234567.89"), "1,234,567.89"); + assert.equal(addThousandSeparators("12.5"), "12.5"); + assert.equal(addThousandSeparators("0.50"), "0.50"); + }); + + it("passes short and empty integer parts through unchanged", () => { + assert.equal(addThousandSeparators("100"), "100"); + assert.equal(addThousandSeparators("999"), "999"); + assert.equal(addThousandSeparators(""), ""); + }); +}); diff --git a/tests/engine/test_formulaError.ts b/tests/engine/test_formulaError.ts new file mode 100644 index 0000000..ca224e4 --- /dev/null +++ b/tests/engine/test_formulaError.ts @@ -0,0 +1,94 @@ +// Pure taxonomy + classification helpers behind #2359's typed error reporting. +// These decide which Excel error value and CalculationError type a failure maps +// to; a wrong mapping silently mislabels a cell, so each direction is pinned. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { + FORMULA_ERROR_VALUES, + FormulaError, + isFormulaError, + divZeroError, + invalidRefError, + nameError, + unknownError, + propagatedError, + classifyThrownError, +} from "../../src/engine/formulaError.ts"; + +describe("FORMULA_ERROR_VALUES", () => { + it("maps each kind to its Excel literal", () => { + assert.deepEqual(FORMULA_ERROR_VALUES, { + div_zero: "#DIV/0!", + invalid_ref: "#REF!", + syntax: "#NAME?", + unknown: "#ERROR!", + }); + }); +}); + +describe("factories carry the matching kind and display", () => { + it("divZeroError", () => { + const error = divZeroError(); + assert.equal(error.errorType, "div_zero"); + assert.equal(error.display, "#DIV/0!"); + assert.equal(isFormulaError(error), true); + }); + + it("invalidRefError includes the reference in the message", () => { + const error = invalidRefError("Missing!A1"); + assert.equal(error.errorType, "invalid_ref"); + assert.equal(error.display, "#REF!"); + assert.match(error.message, /Missing!A1/); + }); + + it("nameError includes the function name in the message", () => { + const error = nameError("UNKNOWNFN"); + assert.equal(error.errorType, "syntax"); + assert.equal(error.display, "#NAME?"); + assert.match(error.message, /UNKNOWNFN/); + }); + + it("unknownError defaults its message to the display value", () => { + assert.equal(unknownError().message, "#ERROR!"); + assert.equal(unknownError("boom").message, "boom"); + assert.equal(unknownError().errorType, "unknown"); + }); +}); + +describe("isFormulaError", () => { + it("accepts a FormulaError and rejects anything else", () => { + assert.equal(isFormulaError(new FormulaError("unknown", "#ERROR!")), true); + assert.equal(isFormulaError(new Error("plain")), false); + assert.equal(isFormulaError("#DIV/0!"), false); + assert.equal(isFormulaError(null), false); + }); +}); + +describe("propagatedError maps a value back to its kind", () => { + it("keeps the dedicated kind for values that have one", () => { + assert.equal(propagatedError("#DIV/0!").errorType, "div_zero"); + assert.equal(propagatedError("#REF!").errorType, "invalid_ref"); + assert.equal(propagatedError("#NAME?").errorType, "syntax"); + }); + + it("falls back to unknown for values without a dedicated kind", () => { + assert.equal(propagatedError("#N/A").errorType, "unknown"); + assert.equal(propagatedError("#NUM!").errorType, "unknown"); + }); + + it("preserves the error value as the display", () => { + assert.equal(propagatedError("#N/A").display, "#N/A"); + }); +}); + +describe("classifyThrownError", () => { + it("passes a FormulaError's own kind and display through", () => { + assert.deepEqual(classifyThrownError(divZeroError()), { type: "div_zero", display: "#DIV/0!" }); + }); + + it("maps any non-FormulaError throw to unknown / #ERROR!", () => { + assert.deepEqual(classifyThrownError(new Error("SUM accepts at most 1 argument")), { type: "unknown", display: "#ERROR!" }); + assert.deepEqual(classifyThrownError("weird"), { type: "unknown", display: "#ERROR!" }); + }); +}); diff --git a/tests/engine/test_formulaRefs.ts b/tests/engine/test_formulaRefs.ts new file mode 100644 index 0000000..a300347 --- /dev/null +++ b/tests/engine/test_formulaRefs.ts @@ -0,0 +1,339 @@ +// Unit tests for the pure formula-reference scanner extracted from +// `src/plugins/spreadsheet/View.vue` (the original was 70 lines of +// inline regex + nested loops with cognitive complexity 32). +// +// Per CLAUDE.md's Testing requirements, covers: +// - Happy path +// - Edge cases (empty, single cell, ranges of every shape) +// - Corner cases (absolute $ refs, large row/col indices) +// - Boundary cases (last letter columns, max-int-ish rows) +// - Invalid / malformed inputs +// - Regression fixtures for the exact shapes View.vue passed to +// the original function. +// +// Tracks #175. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { stripFormulaPrefix, expandRange, parseSingleCellRef, extractCellReferences } from "../../src/engine/formulaRefs.js"; + +describe("stripFormulaPrefix", () => { + it("strips a leading =", () => { + assert.equal(stripFormulaPrefix("=A1+B2"), "A1+B2"); + }); + + it("leaves a formula without = untouched", () => { + assert.equal(stripFormulaPrefix("A1+B2"), "A1+B2"); + }); + + it("returns empty for empty", () => { + assert.equal(stripFormulaPrefix(""), ""); + }); + + it("only strips the first = (documenting behaviour)", () => { + assert.equal(stripFormulaPrefix("==A1"), "=A1"); + }); + + it("handles a bare =", () => { + assert.equal(stripFormulaPrefix("="), ""); + }); +}); + +describe("expandRange", () => { + it("expands a small square range", () => { + assert.deepEqual(expandRange("A1:B2"), [ + { row: 0, col: 0 }, + { row: 0, col: 1 }, + { row: 1, col: 0 }, + { row: 1, col: 1 }, + ]); + }); + + it("expands a single-cell range (A1:A1)", () => { + assert.deepEqual(expandRange("A1:A1"), [{ row: 0, col: 0 }]); + }); + + it("expands a single-row range (A1:C1)", () => { + assert.deepEqual(expandRange("A1:C1"), [ + { row: 0, col: 0 }, + { row: 0, col: 1 }, + { row: 0, col: 2 }, + ]); + }); + + it("expands a single-column range (A1:A3)", () => { + assert.deepEqual(expandRange("A1:A3"), [ + { row: 0, col: 0 }, + { row: 1, col: 0 }, + { row: 2, col: 0 }, + ]); + }); + + it("strips $ on absolute refs ($A$1:$B$2)", () => { + assert.deepEqual(expandRange("$A$1:$B$2"), expandRange("A1:B2")); + }); + + it("handles partial absolute refs ($A1:B$2)", () => { + assert.deepEqual(expandRange("$A1:B$2"), expandRange("A1:B2")); + }); + + it("handles multi-letter columns (Z1:AA1)", () => { + // Z = col 25, AA = col 26 + assert.deepEqual(expandRange("Z1:AA1"), [ + { row: 0, col: 25 }, + { row: 0, col: 26 }, + ]); + }); + + it("returns [] for a reversed range (loops fall through)", () => { + // B2 to A1 — start > end, for-loops iterate 0 times. Matches + // the original inline behaviour. + assert.deepEqual(expandRange("B2:A1"), []); + }); + + it("returns [] for malformed input: no colon", () => { + assert.deepEqual(expandRange("A1"), []); + }); + + it("returns [] for malformed input: junk", () => { + assert.deepEqual(expandRange("foo:bar"), []); + }); + + it("returns [] for empty string", () => { + assert.deepEqual(expandRange(""), []); + }); + + it("returns [] for missing column letters", () => { + assert.deepEqual(expandRange("1:2"), []); + }); +}); + +describe("parseSingleCellRef", () => { + it("parses A1 (col 0, row 0)", () => { + assert.deepEqual(parseSingleCellRef("A1"), { row: 0, col: 0 }); + }); + + it("parses B3 (col 1, row 2)", () => { + assert.deepEqual(parseSingleCellRef("B3"), { row: 2, col: 1 }); + }); + + it("parses absolute $A$1 (same as A1)", () => { + assert.deepEqual(parseSingleCellRef("$A$1"), { row: 0, col: 0 }); + }); + + it("parses partial absolute $A1 and A$1", () => { + assert.deepEqual(parseSingleCellRef("$A1"), { row: 0, col: 0 }); + assert.deepEqual(parseSingleCellRef("A$1"), { row: 0, col: 0 }); + }); + + it("parses multi-letter column AA1 (col 26)", () => { + assert.deepEqual(parseSingleCellRef("AA1"), { row: 0, col: 26 }); + }); + + it("parses large row A100 (row 99)", () => { + assert.deepEqual(parseSingleCellRef("A100"), { row: 99, col: 0 }); + }); + + it("parses the last single-letter column Z1 (col 25)", () => { + assert.deepEqual(parseSingleCellRef("Z1"), { row: 0, col: 25 }); + }); + + it("returns null for lowercase (regex is case-sensitive)", () => { + assert.equal(parseSingleCellRef("a1"), null); + }); + + it("returns null for row-then-col order (1A)", () => { + assert.equal(parseSingleCellRef("1A"), null); + }); + + it("returns null for empty string", () => { + assert.equal(parseSingleCellRef(""), null); + }); + + it("returns null for garbage", () => { + assert.equal(parseSingleCellRef("not-a-ref"), null); + }); + + it("returns null when the row part is missing", () => { + assert.equal(parseSingleCellRef("A"), null); + }); + + it("returns null when the col part is missing", () => { + assert.equal(parseSingleCellRef("42"), null); + }); +}); + +describe("extractCellReferences — happy path", () => { + it("returns [] for empty formula", () => { + assert.deepEqual(extractCellReferences(""), []); + }); + + it("returns [] for a formula with no cell refs (only literals)", () => { + assert.deepEqual(extractCellReferences("=123+456"), []); + }); + + it("picks up a single cell ref", () => { + assert.deepEqual(extractCellReferences("=A1"), [{ row: 0, col: 0 }]); + }); + + it("picks up multiple single cell refs in order", () => { + assert.deepEqual(extractCellReferences("=A1+B2+C3"), [ + { row: 0, col: 0 }, + { row: 1, col: 1 }, + { row: 2, col: 2 }, + ]); + }); + + it("works without a leading = prefix", () => { + // The scanner is also used on partial text (during live edit). + assert.deepEqual(extractCellReferences("A1+B2"), [ + { row: 0, col: 0 }, + { row: 1, col: 1 }, + ]); + }); + + it("picks up absolute references", () => { + assert.deepEqual(extractCellReferences("=$A$1+$B2+C$3"), [ + { row: 0, col: 0 }, + { row: 1, col: 1 }, + { row: 2, col: 2 }, + ]); + }); + + it("works with common function syntax", () => { + assert.deepEqual(extractCellReferences("=SUM(A1, B2, C3)"), [ + { row: 0, col: 0 }, + { row: 1, col: 1 }, + { row: 2, col: 2 }, + ]); + }); +}); + +describe("extractCellReferences — range handling", () => { + it("expands a SUM(A1:B2) range into 4 cells", () => { + assert.deepEqual(extractCellReferences("=SUM(A1:B2)"), [ + { row: 0, col: 0 }, + { row: 0, col: 1 }, + { row: 1, col: 0 }, + { row: 1, col: 1 }, + ]); + }); + + it("does NOT emit the range endpoints as standalone cells", () => { + // The original code strips matched ranges before running the + // cell regex — regression pin so a future refactor doesn't + // accidentally double-count A1 and B2. + const refs = extractCellReferences("=A1:B2"); + // Should be exactly 4 cells from the range expansion, not 6 + // (4 range + 2 endpoints). + assert.equal(refs.length, 4); + }); + + it("combines a range with standalone cells", () => { + assert.deepEqual(extractCellReferences("=SUM(A1:A2)+C5"), [ + { row: 0, col: 0 }, + { row: 1, col: 0 }, + { row: 4, col: 2 }, + ]); + }); + + it("expands multiple ranges in one formula", () => { + const refs = extractCellReferences("=SUM(A1:A2)+SUM(C1:C2)"); + assert.equal(refs.length, 4); + assert.deepEqual(refs.slice().sort(cmpCoord), [ + { row: 0, col: 0 }, + { row: 0, col: 2 }, + { row: 1, col: 0 }, + { row: 1, col: 2 }, + ]); + }); + + it("handles absolute-reference ranges", () => { + assert.deepEqual(extractCellReferences("=SUM($A$1:$B$2)"), [ + { row: 0, col: 0 }, + { row: 0, col: 1 }, + { row: 1, col: 0 }, + { row: 1, col: 1 }, + ]); + }); +}); + +describe("extractCellReferences — deduplication", () => { + it("drops duplicate standalone cells", () => { + assert.deepEqual(extractCellReferences("=A1+A1+A1"), [{ row: 0, col: 0 }]); + }); + + it("drops duplicate cells across absolute / relative forms", () => { + // A1 and $A$1 refer to the same cell; the scanner normalises by + // stripping $, so only one entry appears. + assert.deepEqual(extractCellReferences("=A1+$A$1"), [{ row: 0, col: 0 }]); + }); + + it("drops cells already covered by a range", () => { + const refs = extractCellReferences("=SUM(A1:B2)+A1+B2"); + // Range contributes 4 cells; A1 and B2 are already among them. + assert.equal(refs.length, 4); + }); + + it("preserves first-occurrence order of unique cells", () => { + assert.deepEqual(extractCellReferences("=C3+A1+B2+C3+A1"), [ + { row: 2, col: 2 }, + { row: 0, col: 0 }, + { row: 1, col: 1 }, + ]); + }); +}); + +describe("extractCellReferences — malformed / edge input", () => { + it("ignores lowercase (regex is case-sensitive, matching Excel)", () => { + assert.deepEqual(extractCellReferences("=a1+b2"), []); + }); + + it("ignores partial tokens", () => { + // `A` alone isn't a ref, neither is `1`; no digits adjacent + // to letters means no match. + assert.deepEqual(extractCellReferences("=A + 1"), []); + }); + + it("doesn't match numbers embedded in text without letters", () => { + assert.deepEqual(extractCellReferences("=(100/4)"), []); + }); + + it("handles a formula of only an = sign", () => { + assert.deepEqual(extractCellReferences("="), []); + }); + + it("picks up cells inside parentheses and operators", () => { + assert.deepEqual(extractCellReferences("=((A1+B2)*C3)"), [ + { row: 0, col: 0 }, + { row: 1, col: 1 }, + { row: 2, col: 2 }, + ]); + }); +}); + +describe("extractCellReferences — boundary / precision", () => { + it("handles triple-letter columns (AAA1 = col 702)", () => { + // A=0, Z=25, AA=26, AZ=51, BA=52, ZZ=701, AAA=702 + assert.deepEqual(extractCellReferences("=AAA1"), [{ row: 0, col: 702 }]); + }); + + it("handles row numbers near Excel's 2^20 limit", () => { + // Excel's max row is 1048576. Our parser doesn't enforce that + // cap (and shouldn't — it's a pure scanner) but it should + // still produce a valid integer. + const refs = extractCellReferences("=A1048576"); + assert.equal(refs.length, 1); + const [ref] = refs; + assert.ok(ref); + assert.equal(ref.row, 1048575); + assert.equal(ref.col, 0); + }); +}); + +// --- helpers --- + +function cmpCoord(coordA: { row: number; col: number }, coordB: { row: number; col: number }): number { + if (coordA.row !== coordB.row) return coordA.row - coordB.row; + return coordA.col - coordB.col; +} diff --git a/tests/engine/test_ifBranchEvaluation.ts b/tests/engine/test_ifBranchEvaluation.ts new file mode 100644 index 0000000..d980a8f --- /dev/null +++ b/tests/engine/test_ifBranchEvaluation.ts @@ -0,0 +1,61 @@ +// What IF does with the branch it picks. Both bugs here returned a plausible +// value instead of an error, so a sheet looked fine while holding wrong data: +// a hard-coded list of nine function names meant every OTHER nested call came +// back as its own text (`ROUND(A1,1)` → the string "ROUND(4.567,1)" — IF's own +// registered example did not work), and the fallback read an arithmetic branch +// through `parseFloat("3+1")`, yielding 3 (#2360). + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { SpreadsheetEngine, type SheetData } from "../../src/engine/index.ts"; +import { cellAt } from "./cellAccess.ts"; + +const evaluate = (rows: (string | number)[][], row: number, col: number): unknown => { + const sheet: SheetData = { name: "S", data: rows.map((cells) => cells.map((value) => ({ v: value }))) }; + return cellAt(new SpreadsheetEngine().calculate(sheet).data, row, col); +}; + +describe("IF evaluates a nested function branch, whatever the function", () => { + // ROUND was outside the old whitelist, so this returned "ROUND(4.567,1)". + it("evaluates a function that the old whitelist omitted", () => { + assert.equal(evaluate([[4.567, "=IF(A1>0, ROUND(A1,1), 0)"]], 0, 1), 4.6); + }); + + it("evaluates a text function branch", () => { + assert.equal(evaluate([["hi", '=IF(TRUE, UPPER(A1), "x")']], 0, 1), "HI"); + }); + + it("still evaluates the functions the whitelist did cover", () => { + assert.equal(evaluate([[1, "=IF(A1>0, SUM(A1:A1), 0)"]], 0, 1), 1); + }); + + it("evaluates a nested IF", () => { + assert.equal(evaluate([[5, '=IF(A1>10, "big", IF(A1>3, "mid", "small"))']], 0, 1), "mid"); + }); + + it("takes the false branch without evaluating the true one", () => { + assert.equal(evaluate([[0, "=IF(A1>0, ROUND(9.99,1), 0)"]], 0, 1), 0); + }); +}); + +describe("IF evaluates an arithmetic branch", () => { + // The fallback substituted refs then called parseFloat, which stops at the + // operator: parseFloat("3+1") is 3. + it("computes a reference plus a literal", () => { + assert.equal(evaluate([[3, "=IF(A1>0, A1+1, 0)"]], 0, 1), 4); + }); + + it("returns a bare reference's value", () => { + assert.equal(evaluate([[7, 0, "=IF(A1>0, A1, B1)"]], 0, 2), 7); + }); + + it("returns a numeric literal branch", () => { + assert.equal(evaluate([[3, "=IF(A1>0, 42, 0)"]], 0, 1), 42); + }); +}); + +describe("IF still unwraps a quoted string branch", () => { + it("returns the text without its quotes", () => { + assert.equal(evaluate([[3, '=IF(A1>0, "yes", "no")']], 0, 1), "yes"); + }); +}); diff --git a/tests/engine/test_ifsInjection.ts b/tests/engine/test_ifsInjection.ts new file mode 100644 index 0000000..5ec95d4 --- /dev/null +++ b/tests/engine/test_ifsInjection.ts @@ -0,0 +1,221 @@ +// IFS reaching the condition evaluator rather than a JS engine. +// +// `test_condition.ts` covers the evaluator on its own; this file drives the +// whole path a user's data actually takes — a value typed into a cell, or text +// written into the formula — because that is what made #2360 reachable rather +// than theoretical. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { SpreadsheetEngine, type SheetData } from "../../src/engine/index.ts"; +import { cellAt } from "./cellAccess.ts"; + +const marker = globalThis as Record; + +const calculate = (cellValue: string, formula: string): unknown => + cellAt(new SpreadsheetEngine().calculate({ name: "S", data: [[{ v: cellValue }, { v: formula }]] } satisfies SheetData).data, 0, 1); + +describe("IFS — normal use", () => { + it("returns the first matching branch", () => { + assert.equal(calculate("5", '=IFS(A1>3, "big", A1>0, "small")'), "big"); + }); + + it("falls through to a later branch", () => { + assert.equal(calculate("1", '=IFS(A1>3, "big", A1>0, "small")'), "small"); + }); + + it("returns #N/A when no branch matches", () => { + assert.equal(calculate("-1", '=IFS(A1>3, "big", A1>0, "small")'), "#N/A"); + }); + + it("compares against text", () => { + assert.equal(calculate("Yes", '=IFS(A1="Yes", "confirmed", A1="No", "declined")'), "confirmed"); + assert.equal(calculate("No", '=IFS(A1="Yes", "confirmed", A1="No", "declined")'), "declined"); + }); + + // `IFS((A1>0), ...)` is ordinary usage; the parser used to split the + // parenthesised form into two text operands that never matched. + it("accepts a parenthesised condition", () => { + assert.equal(calculate("5", '=IFS((A1>3), "big")'), "big"); + assert.equal(calculate("1", '=IFS((A1>3), "big", (A1>0), "small")'), "small"); + assert.equal(calculate("5", '=IFS(((A1>3)), "big")'), "big"); + }); + + it("handles the boundary operators", () => { + assert.equal(calculate("3", '=IFS(A1>=3, "atLeast3")'), "atLeast3"); + assert.equal(calculate("3", '=IFS(A1>3, "over3")'), "#N/A"); + }); +}); + +describe("IFS — a cell's contents are data, not code", () => { + // Typing this string into a cell used to execute it, because the cell value + // was substituted into the condition and the result handed to `eval`. + it("does not execute an assignment stored in a cell", () => { + marker.__ifsProbe = false; + calculate("globalThis.__ifsProbe=true", '=IFS(A1>0, "hit")'); + assert.equal(marker.__ifsProbe, false, "the cell's contents must not run"); + }); + + it("does not execute a cell used as a bare condition", () => { + marker.__ifsProbe2 = false; + calculate("globalThis.__ifsProbe2=true", '=IFS(A1, "hit")'); + assert.equal(marker.__ifsProbe2, false); + }); + + // A payload starting with a digit was already neutralised by accident — + // `getRawValue` reads its numeric prefix — so it is NOT evidence the hole is + // closed. Pinned so nobody mistakes it for coverage. + it("also refuses a payload whose numeric prefix used to mask it", () => { + marker.__ifsProbe3 = false; + calculate("1)||(globalThis.__ifsProbe3=true", '=IFS(A1>0, "hit")'); + assert.equal(marker.__ifsProbe3, false); + }); +}); + +describe("IFS — a text cell's operators are data, not syntax", () => { + // A cell holding `x>y` used to substitute as bare `x>y`, so `A1="x>y"` became + // `x>y="x>y"` and never matched. Quoting the operand fixes it (Codex review). + it("compares against a cell whose text contains operators", () => { + assert.equal(calculate("x>y", '=IFS(A1="x>y", "match", TRUE, "no")'), "match"); + assert.equal(calculate("x>y", '=IFS(A1="other", "match", TRUE, "no")'), "no"); + }); + + it("treats a bare operator-bearing cell as truthy text, not a comparison", () => { + assert.equal(calculate("x>y", '=IFS(A1, "truthy", TRUE, "no")'), "truthy"); + assert.equal(calculate("", '=IFS(A1, "truthy", TRUE, "empty")'), "empty"); + }); + + // A cell holding a quote must not corrupt the comparison: `renderConditionOperand` + // escapes it, and the condition parser tracks the escape. + it("compares a quote-bearing cell without corrupting the parse", () => { + assert.equal(calculate('a"b', '=IFS(A1="z", "match", TRUE, "no")'), "no", 'a"b is not z'); + assert.equal(calculate('a"b', '=IFS(A1, "truthy", TRUE, "no")'), "truthy", "still non-empty text"); + }); +}); + +describe("IFS — operator characters inside a cell value stay data", () => { + // A cell holding `x>y` renders into the condition as a quoted literal, and the + // condition parser skips quoted regions when looking for the operator — so the + // inner `>` is never read as a comparison (Codex review flagged this path). + it("treats a bare reference to a string with an operator as truthy text", () => { + assert.equal(calculate("x>y", '=IFS(A1, "hit")'), "hit"); + }); + + it("compares equal against a string literal that contains an operator", () => { + assert.equal(calculate("x>y", '=IFS(A1="x>y", "hit")'), "hit"); + }); + + it("does not match when the operator-bearing strings differ", () => { + assert.equal(calculate("x>y", '=IFS(A1="a>b", "hit")'), "#N/A"); + }); +}); + +describe("IFS — absolute and mixed references resolve", () => { + // The ref used to be escaped twice before the RegExp, so `$A$1` never matched + // and was left as literal text in the condition (Codex review). + it("substitutes an absolute reference", () => { + assert.equal(calculate("5", '=IFS($A$1>0, "hit")'), "hit"); + assert.equal(calculate("5", '=IFS($A$1>10, "hit")'), "#N/A"); + }); + + it("substitutes a mixed reference", () => { + assert.equal(calculate("5", '=IFS(A$1>0, "hit")'), "hit"); + assert.equal(calculate("5", '=IFS($A1>0, "hit")'), "hit"); + }); +}); + +describe("IFS — a cell value ending in a backslash substitutes intact", () => { + // CodeQL js/double-escaping. `renderConditionOperand` escapes the backslash so + // a trailing `\` cannot escape the closing quote and corrupt the literal; + // substituting by position (not a regex) keeps it exactly once. A cell value + // `a\` is the sharp case — an unescaped one turns `"a\"` into an open literal. + const trailingBackslashSheet = (other: string): SheetData => ({ + name: "S", + data: [ + [{ v: "a\\" }, { v: '=IFS(A1=A2, "eq", TRUE, "ne")' }], + [{ v: other }, { v: 0 }], + ], + }); + + it("matches two cells that both end in a backslash", () => { + assert.equal(cellAt(new SpreadsheetEngine().calculate(trailingBackslashSheet("a\\")).data, 0, 1), "eq"); + }); + + it("does not match a backslash cell against different text", () => { + assert.equal(cellAt(new SpreadsheetEngine().calculate(trailingBackslashSheet("ab")).data, 0, 1), "ne"); + }); + + it("treats a bare backslash-bearing cell as truthy text", () => { + assert.equal(calculate("a\\b", '=IFS(A1, "truthy", TRUE, "no")'), "truthy"); + }); +}); + +describe("IFS — a reference inside a string literal stays literal text", () => { + // `A1="B2"` compares A1 to the TEXT "B2". The `"B2"` must not be read as a + // reference and replaced with cell B2's value (Codex review). + it("does not substitute a ref that sits inside quotes", () => { + const sheet: SheetData = { + name: "S", + data: [ + [{ v: "B2" }, { v: '=IFS(A1="B2", "hit")' }], + [{ v: 0 }, { v: 99 }], + ], + }; + assert.equal(cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 1), "hit", "A1's text equals the literal B2"); + }); +}); + +describe("IFS — bare TRUE/FALSE literals are booleans", () => { + // `evaluateFormula("FALSE")` returns the non-empty string "FALSE" (truthy); + // a bare logical literal in a condition must stay a boolean (Codex review). + it("skips a FALSE condition and matches a later TRUE", () => { + assert.equal(calculate("5", '=IFS(FALSE, "hit", TRUE, "miss")'), "miss"); + }); + + it("matches a bare TRUE condition", () => { + assert.equal(calculate("5", '=IFS(TRUE, "hit")'), "hit"); + }); + + it("uses TRUE as a catch-all after a false comparison", () => { + assert.equal(calculate("5", '=IFS(A1>10, "hit", TRUE, "miss")'), "miss"); + }); +}); + +describe("IFS — arithmetic operands are computed, not compared as text", () => { + // Removing eval left `A1+1>10` read as the string "5+1" vs 10, which flipped + // the branch. Each operand is now resolved by the engine's safe evaluator so + // the arithmetic is computed (Codex review). + it("computes an arithmetic left operand", () => { + assert.equal(calculate("5", '=IFS(A1+1>10, "hit", TRUE, "miss")'), "miss", "6 > 10 is false"); + assert.equal(calculate("5", '=IFS(A1+1>5, "hit", TRUE, "miss")'), "hit", "6 > 5 is true"); + }); + + it("computes arithmetic on both sides", () => { + assert.equal(calculate("4", '=IFS(A1*2 > 3+3, "hit", TRUE, "miss")'), "hit", "8 > 6 is true"); + }); +}); + +describe("IFS — the formula itself is data too", () => { + it("does not execute an expression written into the condition", () => { + marker.__ifsProbe4 = false; + calculate("1", '=IFS(A1>0&&(globalThis.__ifsProbe4=true), "hit")'); + assert.equal(marker.__ifsProbe4, false); + }); + + it("does not execute a call in the condition", () => { + marker.__ifsProbe5 = false; + calculate("1", '=IFS((globalThis.__ifsProbe5=true)>0, "hit")'); + assert.equal(marker.__ifsProbe5, false); + }); + + // A crash here would take the whole sheet's calculation with it. Broken + // input degrades to the formula text — that is the engine's existing + // swallow-everything behaviour (#2359), not something this change decides; + // what matters here is that it neither throws nor runs. + it("survives a syntactically broken condition", () => { + marker.__ifsProbe6 = false; + const result = calculate("1", '=IFS(((((globalThis.__ifsProbe6=true, "hit")'); + assert.equal(typeof result, "string"); + assert.equal(marker.__ifsProbe6, false); + }); +}); diff --git a/tests/engine/test_jsonCellLocator.ts b/tests/engine/test_jsonCellLocator.ts new file mode 100644 index 0000000..957ab9f --- /dev/null +++ b/tests/engine/test_jsonCellLocator.ts @@ -0,0 +1,118 @@ +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { findCellJsonPosition } from "../../src/engine/jsonCellLocator.js"; + +describe("findCellJsonPosition", () => { + const sample = JSON.stringify( + [ + { + name: "Sheet1", + data: [ + [{ v: "A1" }, { v: "B1" }, { v: 42 }], + [{ v: "A2" }, { v: "B2" }, { v: "=SUM(A1:B1)" }], + ], + }, + { + name: "Sheet2", + data: [[{ v: "X" }, { v: "Y" }]], + }, + ], + null, + 2, + ); + + // The locator returns the character offset of the cell's opening + // `{`. In a pretty-printed document, following lines expand the + // object so we only assert the starting char and that the + // substring contains the unique cell value. + function assertCellAt(pos: number, expectedSubstring: string) { + assert.ok(pos > 0, `expected a positive offset, got ${pos}`); + assert.equal(sample[pos], "{", `expected offset to land on '{', got '${sample[pos]}'`); + assert.ok(sample.substring(pos).includes(expectedSubstring), `expected substring ${JSON.stringify(expectedSubstring)} after position ${pos}`); + } + + it("locates cell (0,0) in the first sheet", () => { + assertCellAt(findCellJsonPosition(sample, "Sheet1", 0, 0), `"v": "A1"`); + }); + + it("locates cell (0,2) — the last column in row 0", () => { + assertCellAt(findCellJsonPosition(sample, "Sheet1", 0, 2), `"v": 42`); + }); + + it("locates cell (1,2) with a formula value", () => { + assertCellAt(findCellJsonPosition(sample, "Sheet1", 1, 2), `"v": "=SUM(A1:B1)"`); + }); + + it("locates cell in a non-first sheet by name", () => { + assertCellAt(findCellJsonPosition(sample, "Sheet2", 0, 1), `"v": "Y"`); + }); + + it("returns -1 for a sheet name that does not exist", () => { + assert.equal(findCellJsonPosition(sample, "Unknown", 0, 0), -1); + }); + + it("returns -1 for a row index past the end", () => { + assert.equal(findCellJsonPosition(sample, "Sheet1", 99, 0), -1); + }); + + it("returns -1 on completely empty input", () => { + assert.equal(findCellJsonPosition("", "Sheet1", 0, 0), -1); + }); + + it("handles strings containing brackets and commas without miscounting", () => { + const tricky = JSON.stringify( + [ + { + name: "Sheet1", + data: [[{ v: "has [bracket], and comma" }, { v: "second" }]], + }, + ], + null, + 2, + ); + const pos = findCellJsonPosition(tricky, "Sheet1", 0, 1); + assert.ok(pos > 0); + assert.equal(tricky[pos], "{"); + assert.ok(tricky.substring(pos).includes(`"v": "second"`)); + }); + + it("picks the correct row when an earlier row contains '[' inside a string", () => { + // Row 0 cell 0 contains a literal '[' — the naive counter would + // treat that as an extra row opener and shift all subsequent + // rowIndex lookups by one. + const withBracketInRow0 = JSON.stringify( + [ + { + name: "Sheet1", + data: [ + [{ v: "row0 has [bracket]" }, { v: "r0c1" }], + [{ v: "r1c0" }, { v: "TARGET" }], + ], + }, + ], + null, + 2, + ); + const pos = findCellJsonPosition(withBracketInRow0, "Sheet1", 1, 1); + assert.ok(pos > 0); + assert.ok(withBracketInRow0.substring(pos).includes(`"v": "TARGET"`)); + }); + + it("finds a sheet whose name contains a quote character", () => { + // Sheet names with embedded `"` need JSON-escaping when building + // the text marker, otherwise indexOf misses them entirely. + const text = JSON.stringify( + [ + { + name: 'Sheet "Q1"', + data: [[{ v: "FOUND" }]], + }, + ], + null, + 2, + ); + const pos = findCellJsonPosition(text, 'Sheet "Q1"', 0, 0); + assert.ok(pos > 0, `expected to locate the sheet, got ${pos}`); + assert.ok(text.substring(pos).includes(`"v": "FOUND"`)); + }); +}); diff --git a/tests/engine/test_locateSubstring.ts b/tests/engine/test_locateSubstring.ts new file mode 100644 index 0000000..58835fe --- /dev/null +++ b/tests/engine/test_locateSubstring.ts @@ -0,0 +1,52 @@ +// locateSubstring folds the shared body of FIND (case-sensitive) and SEARCH +// (case-insensitive) (#2482): identical 0-based start, identical 1-based hit +// index, identical #VALUE! miss. Case folding is the ONLY axis that may differ, +// so these pin both the common rule and that one deliberate asymmetry. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { locateSubstring } from "../../src/engine/functions/text.ts"; +import { isSpreadsheetErrorValue } from "../../src/engine/spreadsheet-errors.ts"; + +const CASE_SENSITIVE = { caseInsensitive: false }; +const CASE_INSENSITIVE = { caseInsensitive: true }; + +const missed = (result: ReturnType): boolean => isSpreadsheetErrorValue(result) && result.code === "#VALUE!"; + +describe("locateSubstring — 1-based hit index", () => { + it("returns the 1-based position of the first match", () => { + assert.equal(locateSubstring("b", "abc", 0, CASE_SENSITIVE), 2); + }); + + it("finds a match at the very start", () => { + assert.equal(locateSubstring("a", "abc", 0, CASE_SENSITIVE), 1); + }); + + it("honours a non-zero start, skipping an earlier match", () => { + assert.equal(locateSubstring("a", "banana", 2, CASE_SENSITIVE), 4); + }); +}); + +describe("locateSubstring — case sensitivity is the only difference", () => { + it("case-sensitive: a wrong-case needle misses with #VALUE!", () => { + assert.ok(missed(locateSubstring("O", "hello", 0, CASE_SENSITIVE))); + }); + + it("case-insensitive: the same wrong-case needle matches", () => { + assert.equal(locateSubstring("O", "hello", 0, CASE_INSENSITIVE), 5); + }); + + it("case-insensitive folds BOTH the needle and the haystack", () => { + assert.equal(locateSubstring("HELLO", "hello world", 0, CASE_INSENSITIVE), 1); + }); +}); + +describe("locateSubstring — misses and edge cases", () => { + it("a needle absent from the haystack is #VALUE!", () => { + assert.ok(missed(locateSubstring("z", "abc", 0, CASE_SENSITIVE))); + }); + + it("an empty needle matches at position 1 (indexOf semantics preserved)", () => { + assert.equal(locateSubstring("", "abc", 0, CASE_SENSITIVE), 1); + }); +}); diff --git a/tests/engine/test_logicalFunctions.ts b/tests/engine/test_logicalFunctions.ts new file mode 100644 index 0000000..9032bb2 --- /dev/null +++ b/tests/engine/test_logicalFunctions.ts @@ -0,0 +1,76 @@ +// Boolean coercion shared by the logical functions. The bug this covers: IF and +// AND/OR read the SAME value oppositely — IF("0") took the true branch while +// AND("0") was false — because each function coerced truthiness its own way +// (#2387). One shared rule keeps them in agreement. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { SpreadsheetEngine, type SheetData } from "../../src/engine/index.ts"; +import { coerceToBoolean } from "../../src/engine/coerce-boolean.ts"; +import { cellAt } from "./cellAccess.ts"; + +describe("coerceToBoolean", () => { + it("passes booleans through", () => { + assert.equal(coerceToBoolean(true), true); + assert.equal(coerceToBoolean(false), false); + }); + + it("treats only 0 as false among numbers", () => { + assert.equal(coerceToBoolean(0), false); + assert.equal(coerceToBoolean(1), true); + assert.equal(coerceToBoolean(-1), true); + assert.equal(coerceToBoolean(0.5), true); + }); + + it("treats blank and empty as false", () => { + assert.equal(coerceToBoolean(""), false); + assert.equal(coerceToBoolean(" "), false); + assert.equal(coerceToBoolean(null), false); + assert.equal(coerceToBoolean(undefined), false); + }); + + it("reads the words true/false case-insensitively", () => { + assert.equal(coerceToBoolean("true"), true); + assert.equal(coerceToBoolean("TRUE"), true); + assert.equal(coerceToBoolean("false"), false); + assert.equal(coerceToBoolean("False"), false); + }); + + // The crux of #2387: a numeric string follows its number, so "0" is false in + // every logical function — not true in IF and false in AND. + it("follows the number in a numeric string", () => { + assert.equal(coerceToBoolean("0"), false); + assert.equal(coerceToBoolean("0.0"), false); + assert.equal(coerceToBoolean("5"), true); + assert.equal(coerceToBoolean("-3"), true); + }); + + it("treats other non-empty text as true", () => { + assert.equal(coerceToBoolean("hello"), true); + assert.equal(coerceToBoolean("no"), true); + }); +}); + +describe("IF and AND/OR/NOT agree on the same value", () => { + const evalFormula = (formula: string): unknown => + cellAt(new SpreadsheetEngine().calculate({ name: "S", data: [[{ v: formula }]] } satisfies SheetData).data, 0, 0); + + // Each value should send IF down the false branch exactly when AND/OR/NOT read + // it as false. Previously IF("0") returned 1 while AND("0") returned false. + for (const [literal, truthy] of [ + ['"0"', false], + ['"false"', false], + ['""', false], + ["0", false], + ['"5"', true], + ['"hello"', true], + ["1", true], + ] as const) { + it(`agrees that ${literal} is ${truthy ? "true" : "false"}`, () => { + assert.equal(evalFormula(`=IF(${literal}, 1, 2)`), truthy ? 1 : 2, "IF branch"); + assert.equal(evalFormula(`=AND(${literal})`), truthy, "AND"); + assert.equal(evalFormula(`=OR(${literal})`), truthy, "OR"); + assert.equal(evalFormula(`=NOT(${literal})`), !truthy, "NOT"); + }); + } +}); diff --git a/tests/engine/test_lookupBounds.ts b/tests/engine/test_lookupBounds.ts new file mode 100644 index 0000000..14d0d0c --- /dev/null +++ b/tests/engine/test_lookupBounds.ts @@ -0,0 +1,135 @@ +// VLOOKUP's col_index_num and HLOOKUP's row_index_num were used unchecked, so an +// index past the table addressed a cell OUTSIDE the range and returned whatever +// lived there — usually a silent 0 where Excel reports #REF! (#2360). INDEX +// already had this guard; these two did not. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { resolveTableOffset } from "../../src/engine/formulaRefs.ts"; +import { SpreadsheetEngine, type SheetData } from "../../src/engine/index.ts"; +import { cellAt } from "./cellAccess.ts"; + +// A1:B2 = [a, 1] / [b, 2]; the formula sits in C1. +const table = (formula: string): unknown => { + const sheet: SheetData = { + name: "S", + data: [ + [{ v: "a" }, { v: 1 }, { v: formula }], + [{ v: "b" }, { v: 2 }], + ], + }; + return cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 2); +}; + +describe("resolveTableOffset", () => { + it("maps a 1-based position to a 0-based offset", () => { + assert.equal(resolveTableOffset(1, 2), 0); + assert.equal(resolveTableOffset(2, 2), 1); + }); + + it("rejects a position past the table", () => { + assert.equal(resolveTableOffset(3, 2), null); + assert.equal(resolveTableOffset(9, 2), null); + }); + + it("rejects zero and negative positions", () => { + assert.equal(resolveTableOffset(0, 2), null); + assert.equal(resolveTableOffset(-1, 2), null); + }); + + // INDEX reads a `0` position as "the whole line" and collapses it to the only + // cell when the line is one long. A lookup index has no such meaning — its + // columns are numbered from 1 — so `0` is out of range here too (Codex review). + it("rejects zero even for a single-line table, unlike INDEX", () => { + assert.equal(resolveTableOffset(0, 1), null); + }); + + it("rejects a non-finite position", () => { + assert.equal(resolveTableOffset(NaN, 2), null); + }); + + it("truncates a fractional position toward zero, as Excel does", () => { + assert.equal(resolveTableOffset(2.9, 2), 1); + }); +}); + +describe("VLOOKUP column bounds", () => { + it("is #REF! when the column index is past the table", () => { + assert.equal(table('=VLOOKUP("a",A1:B2,9,FALSE)'), "#REF!"); + }); + + it("is #REF! for a zero or negative column index", () => { + assert.equal(table('=VLOOKUP("a",A1:B2,0,FALSE)'), "#REF!"); + assert.equal(table('=VLOOKUP("a",A1:B2,-1,FALSE)'), "#REF!"); + }); + + it("still returns the value for an in-range column", () => { + assert.equal(table('=VLOOKUP("b",A1:B2,1,FALSE)'), "b", "column 1 is the key column"); + assert.equal(table('=VLOOKUP("b",A1:B2,2,FALSE)'), 2); + }); + + it("still reports #N/A when the key is not found", () => { + assert.equal(table('=VLOOKUP("zz",A1:B2,2,FALSE)'), "#N/A", "a missing key is not a #REF!"); + }); +}); + +// Excel treats an out-of-range index as an argument error, evaluated before the +// key is searched for. Validating it after the match let a missing key mask it as +// #N/A, so a typo'd index looked like "value not in the table" (Codex review). +describe("an out-of-range index outranks a missing key", () => { + it("is #REF! for VLOOKUP with a missing key and an index past the table", () => { + assert.equal(table('=VLOOKUP("zz",A1:B2,9,FALSE)'), "#REF!"); + }); + + it("is #REF! for VLOOKUP with a missing key and a zero or negative index", () => { + assert.equal(table('=VLOOKUP("zz",A1:B2,0,FALSE)'), "#REF!"); + assert.equal(table('=VLOOKUP("zz",A1:B2,-1,FALSE)'), "#REF!"); + }); + + it("is #REF! for HLOOKUP with a missing key and an index past the table", () => { + assert.equal(table('=HLOOKUP("zz",A1:B2,9,FALSE)'), "#REF!"); + }); + + // The approximate path reaches #N/A by a different route (no candidate <= the + // key) rather than by an absent exact match, so it needs its own case. + it("is #REF! on the approximate path when the key is below every candidate", () => { + assert.equal(table("=VLOOKUP(0,A1:B2,9,TRUE)"), "#REF!"); + }); +}); + +describe("single-line tables still reject index 0", () => { + // A one-column table is where INDEX's whole-line `0` would have slipped through. + const oneColumn = (formula: string): unknown => { + const sheet: SheetData = { + name: "S", + data: [[{ v: "a" }, { v: formula }], [{ v: "b" }]], + }; + return cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 1); + }; + + it("is #REF! for VLOOKUP with index 0 on a single-column table", () => { + assert.equal(oneColumn('=VLOOKUP("a",A1:A2,0,FALSE)'), "#REF!"); + }); + + it("still returns the key column for index 1", () => { + assert.equal(oneColumn('=VLOOKUP("b",A1:A2,1,FALSE)'), "b"); + }); + + it("is #REF! when the key is missing too, not #N/A", () => { + assert.equal(oneColumn('=VLOOKUP("zz",A1:A2,0,FALSE)'), "#REF!"); + }); + + it("still reports #N/A for a missing key with a valid index", () => { + assert.equal(oneColumn('=VLOOKUP("zz",A1:A2,1,FALSE)'), "#N/A"); + }); +}); + +describe("HLOOKUP row bounds", () => { + it("is #REF! when the row index is past the table", () => { + assert.equal(table('=HLOOKUP("a",A1:B2,9,FALSE)'), "#REF!"); + }); + + it("still returns the value for an in-range row", () => { + assert.equal(table('=HLOOKUP("a",A1:B2,2,FALSE)'), "b"); + }); +}); diff --git a/tests/engine/test_lookupFunctions.ts b/tests/engine/test_lookupFunctions.ts new file mode 100644 index 0000000..698be42 --- /dev/null +++ b/tests/engine/test_lookupFunctions.ts @@ -0,0 +1,181 @@ +// Lookup functions driven through the whole engine. The cross-sheet VLOOKUP case +// is the #2390 regression: a sheet-qualified table array (`Data!A1:B3`) used to +// throw because one of VLOOKUP's two range parses ran a sheet-unaware regex and +// rejected the prefix before the sheet-aware parse could run. Now a single +// `parseRangeBounds` handles both (#2396). + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { SpreadsheetEngine, type SheetData } from "../../src/engine/index.ts"; +import { cellAt } from "./cellAccess.ts"; + +/** Calculate `formula` in cell A-after-the-data of a single sheet built from `rows`. */ +const evalInSheet = (rows: (string | number)[][], formula: string): unknown => { + const data = rows.map((row) => row.map((value) => ({ v: value }))); + data.push([{ v: formula }]); + const result = new SpreadsheetEngine().calculate({ name: "S", data }); + return cellAt(result.data, data.length - 1, 0); +}; + +describe("VLOOKUP — same sheet", () => { + const table: (string | number)[][] = [ + ["Alice", 10], + ["Bob", 20], + ["Carol", 30], + ]; + + it("returns the result-column value for an exact match", () => { + assert.equal(evalInSheet(table, '=VLOOKUP("Carol", A1:B3, 2, FALSE)'), 30); + }); + + it("returns #N/A when the value is absent", () => { + assert.equal(evalInSheet(table, '=VLOOKUP("Zoe", A1:B3, 2, FALSE)'), "#N/A"); + }); +}); + +describe("HLOOKUP — same sheet", () => { + it("looks across the top row and returns the row below", () => { + const table: (string | number)[][] = [ + ["a", "b", "c"], + [1, 2, 3], + ]; + assert.equal(evalInSheet(table, '=HLOOKUP("b", A1:C2, 2, FALSE)'), 2); + }); +}); + +describe("VLOOKUP — cross-sheet table array (#2390: no longer throws)", () => { + it("resolves a sheet-qualified table array", () => { + const data: SheetData = { + name: "Data", + data: [ + [{ v: "Alice" }, { v: 10 }], + [{ v: "Bob" }, { v: 20 }], + [{ v: "Carol" }, { v: 30 }], + ], + }; + const main: SheetData = { name: "Main", data: [[{ v: '=VLOOKUP("Bob", Data!A1:B3, 2, FALSE)' }]] }; + const [mainResult] = new SpreadsheetEngine().calculateWorkbook([main, data]); + assert.ok(mainResult); + assert.equal(cellAt(mainResult.data, 0, 0), 20); + }); +}); + +// Approximate match (#2360): the 4th argument TRUE (and omitted) must do an +// approximate match — the largest first-column/row value <= the lookup key in a +// sorted range — not fall back to exact. The literal TRUE reaches the handler as +// the string "TRUE", which the old accept-only-`true|1|"1"` check missed, so +// `VLOOKUP(4, …, TRUE)` silently returned #N/A. +describe("VLOOKUP — approximate match with TRUE (#2360)", () => { + const sorted: (string | number)[][] = [ + [1, "a"], + [3, "b"], + [5, "c"], + ]; + + it("returns the largest value <= the lookup key", () => { + assert.equal(evalInSheet(sorted, "=VLOOKUP(4, A1:B3, 2, TRUE)"), "b"); // 3 is the largest <= 4 + assert.equal(evalInSheet(sorted, "=VLOOKUP(2, A1:B3, 2, TRUE)"), "a"); // 1 is the largest <= 2 + assert.equal(evalInSheet(sorted, "=VLOOKUP(9, A1:B3, 2, TRUE)"), "c"); // past the end -> last row + }); + + it("matches an exact key on the approximate path too", () => { + assert.equal(evalInSheet(sorted, "=VLOOKUP(3, A1:B3, 2, TRUE)"), "b"); + }); + + it("treats a lowercase true the same as TRUE", () => { + assert.equal(evalInSheet(sorted, "=VLOOKUP(4, A1:B3, 2, true)"), "b"); + }); + + it("approximates when the 4th argument is omitted (default TRUE)", () => { + assert.equal(evalInSheet(sorted, "=VLOOKUP(4, A1:B3, 2)"), "b"); + }); + + it("returns #N/A when the key is below the smallest value", () => { + assert.equal(evalInSheet(sorted, "=VLOOKUP(0, A1:B3, 2, TRUE)"), "#N/A"); + }); + + it("keeps the exact FALSE path unchanged", () => { + assert.equal(evalInSheet(sorted, "=VLOOKUP(3, A1:B3, 2, FALSE)"), "b"); + assert.equal(evalInSheet(sorted, "=VLOOKUP(4, A1:B3, 2, FALSE)"), "#N/A"); + }); +}); + +describe("HLOOKUP — approximate match with TRUE (#2360)", () => { + const sorted: (string | number)[][] = [ + [10, 20, 30], + ["x", "y", "z"], + ]; + + it("returns the row-2 value under the largest column <= the lookup key", () => { + assert.equal(evalInSheet(sorted, "=HLOOKUP(25, A1:C2, 2, TRUE)"), "y"); // 20 is the largest <= 25 + assert.equal(evalInSheet(sorted, "=HLOOKUP(30, A1:C2, 2, TRUE)"), "z"); + }); + + it("returns #N/A when the key is below the smallest value", () => { + assert.equal(evalInSheet(sorted, "=HLOOKUP(5, A1:C2, 2, TRUE)"), "#N/A"); + }); +}); + +describe("INDEX — bounds (#2390)", () => { + const grid: (string | number)[][] = [ + [10, 11], + [20, 21], + [30, 31], + ]; + + it("returns the addressed cell for an in-range position", () => { + assert.equal(evalInSheet(grid, "=INDEX(A1:B3, 2, 2)"), 21); // B2 + assert.equal(evalInSheet(grid, "=INDEX(A1:A3, 3)"), 30); // A3 + }); + + it("returns #REF! when the row is past the range (was reading A5)", () => { + assert.equal(evalInSheet(grid, "=INDEX(A1:A3, 5)"), "#REF!"); + }); + + it("returns #REF! for row 0 on a multi-row range (was reading A1 above the range)", () => { + assert.equal(evalInSheet(grid, "=INDEX(A2:B3, 0, 1)"), "#REF!"); + }); +}); + +// MATCH and XLOOKUP read their ranges through the NUMERIC-ONLY reader, which +// drops every text cell. #2358 moved SUMIF/AVERAGEIF onto the raw reader for +// exactly this reason — `calculator.ts` still explains it — and these two were +// missed, so they carry both halves of that failure: text keys are invisible, +// and a lookup/return pair filtered independently falls out of row alignment +// and answers with a DIFFERENT row's value, silently. +describe("MATCH / XLOOKUP over text (#2765)", () => { + const fruit: (string | number)[][] = [ + ["apple", 1], + ["banana", 2], + ["cherry", 3], + ]; + + it("MATCH finds a text key", () => { + assert.equal(evalInSheet(fruit, '=MATCH("banana", A1:A3, 0)'), 2); + }); + + it("XLOOKUP returns the paired value for a text key", () => { + assert.equal(evalInSheet(fruit, '=XLOOKUP("banana", A1:A3, B1:B3)'), 2); + }); + + // The dangerous one: no error, just a wrong number. `A2` is text, so the + // lookup column loses that row while the return column keeps all three — + // the match at index 1 then reads B2 instead of B3. + it("XLOOKUP stays row-aligned when the lookup column holds text", () => { + const mixed: (string | number)[][] = [ + [1, 100], + ["x", 200], + [3, 300], + ]; + assert.equal(evalInSheet(mixed, "=XLOOKUP(3, A1:A3, B1:B3)"), 300); + }); + + it("MATCH keeps its 1-based index when earlier rows hold text", () => { + const mixed: (string | number)[][] = [ + ["x", 0], + [7, 0], + [9, 0], + ]; + assert.equal(evalInSheet(mixed, "=MATCH(9, A1:A3, 0)"), 3); + }); +}); diff --git a/tests/engine/test_lookupMath.ts b/tests/engine/test_lookupMath.ts new file mode 100644 index 0000000..ad063b3 --- /dev/null +++ b/tests/engine/test_lookupMath.ts @@ -0,0 +1,43 @@ +// isApproximateMatch reads VLOOKUP/HLOOKUP's range_lookup argument (#2360). The +// literal TRUE arrives as the STRING "TRUE" (the evaluator leaves bare words +// unquoted), which the old accept-only-`true|1|"1"` check missed and so fell +// back to exact match. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { isApproximateMatch } from "../../src/engine/functions/lookup-math.ts"; + +describe("isApproximateMatch", () => { + it("treats a real boolean as itself", () => { + assert.equal(isApproximateMatch(true), true); + assert.equal(isApproximateMatch(false), false); + }); + + // The fix: the bare word TRUE evaluates to the string "TRUE", not a boolean. + it("reads the string forms of TRUE/FALSE, case-insensitively", () => { + assert.equal(isApproximateMatch("TRUE"), true); + assert.equal(isApproximateMatch("true"), true); + assert.equal(isApproximateMatch("True"), true); + assert.equal(isApproximateMatch("FALSE"), false); + assert.equal(isApproximateMatch("false"), false); + }); + + it("reads numeric logicals: 0 exact, non-zero approximate", () => { + assert.equal(isApproximateMatch(1), true); + assert.equal(isApproximateMatch(0), false); + assert.equal(isApproximateMatch(2), true); + assert.equal(isApproximateMatch("1"), true); + assert.equal(isApproximateMatch("0"), false); + }); + + it("treats blank or stray text as exact (FALSE), matching Excel coercion", () => { + assert.equal(isApproximateMatch(""), false); + assert.equal(isApproximateMatch(" "), false); + assert.equal(isApproximateMatch("yes"), false); + }); + + it("ignores surrounding whitespace on the string forms", () => { + assert.equal(isApproximateMatch(" TRUE "), true); + assert.equal(isApproximateMatch(" 0 "), false); + }); +}); diff --git a/tests/engine/test_mathematicalFunctions.ts b/tests/engine/test_mathematicalFunctions.ts new file mode 100644 index 0000000..957c758 --- /dev/null +++ b/tests/engine/test_mathematicalFunctions.ts @@ -0,0 +1,169 @@ +// Domain and boundary rules for the math functions. The bugs here returned a +// plausible NUMBER (FLOOR(-2.5,2) = -4, ROUND(-2.5,0) = -2, MOD(-3,2) = -1) or a +// silent NaN/∞ instead of an Excel error (#2389). The rounding direction, the +// modulo sign and the domain guards are checked directly on the pure helpers, +// with a few end-to-end checks that the handlers surface the error values. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { + roundTo, + roundUpTo, + roundDownTo, + floorToSignificance, + ceilingToSignificance, + modulo, + power, + safeLog, + safeLog10, + safeSqrt, + logWithBase, +} from "../../src/engine/math-ops.ts"; +import { SpreadsheetEngine, type SheetData } from "../../src/engine/index.ts"; +import { DIV_ZERO_ERROR, NUM_ERROR } from "../../src/engine/spreadsheet-errors.ts"; +import { cellAt } from "./cellAccess.ts"; + +const closeTo = (actual: number, expected: number, eps = 1e-9): boolean => Math.abs(actual - expected) <= eps; + +describe("roundTo / roundUpTo / roundDownTo — direction", () => { + it("rounds half away from zero, not toward +infinity", () => { + assert.equal(roundTo(-2.5, 0), -3); + assert.equal(roundTo(2.5, 0), 3); + assert.ok(closeTo(roundTo(0.125, 2), 0.13)); + }); + + it("rounds up away from zero", () => { + assert.ok(closeTo(roundUpTo(-3.14159, 2), -3.15)); + assert.ok(closeTo(roundUpTo(3.14159, 2), 3.15)); + }); + + it("rounds down toward zero", () => { + assert.ok(closeTo(roundDownTo(-3.14159, 2), -3.14)); + assert.ok(closeTo(roundDownTo(3.19, 1), 3.1)); + }); +}); + +describe("floorToSignificance / ceilingToSignificance — sign domain", () => { + it("is a #NUM! error value when number and significance disagree in sign", () => { + assert.equal(floorToSignificance(-2.5, 2), NUM_ERROR); + assert.equal(ceilingToSignificance(2.5, -2), NUM_ERROR); + }); + + it("rounds to the multiple when the signs match", () => { + assert.equal(floorToSignificance(2.5, 2), 2); + assert.equal(floorToSignificance(-2.5, -2), -2); + assert.equal(ceilingToSignificance(2.5, 2), 4); + assert.equal(ceilingToSignificance(-2.5, -2), -4); + }); + + it("returns 0 for a zero value", () => { + assert.equal(floorToSignificance(0, 2), 0); + assert.equal(ceilingToSignificance(0, 2), 0); + }); + + // Excel is deliberately asymmetric here: FLOOR(x, 0) is #DIV/0! while + // CEILING(x, 0) is 0. Both used to answer 0, so FLOOR swallowed a divide-by- + // zero (#2360). Keep the pair pinned so neither is "made consistent" later. + it("is #DIV/0! for FLOOR with a zero significance, but 0 for CEILING", () => { + assert.equal(floorToSignificance(3, 0), DIV_ZERO_ERROR); + assert.equal(ceilingToSignificance(3, 0), 0); + }); + + // The zero check has to win over the sign check, or a negative number with a + // zero significance would report the wrong error (#NUM! instead of #DIV/0!). + it("reports #DIV/0! rather than #NUM! for a negative number over a zero significance", () => { + assert.equal(floorToSignificance(-3, 0), DIV_ZERO_ERROR); + }); +}); + +describe("modulo — divisor sign and division by zero", () => { + it("takes the sign of the divisor", () => { + assert.equal(modulo(-3, 2), 1); + assert.equal(modulo(3, -2), -1); + assert.equal(modulo(-3, -2), -1); + assert.equal(modulo(5, 3), 2); + }); + + it("is #DIV/0! when the divisor is zero", () => { + assert.equal(modulo(5, 0), DIV_ZERO_ERROR); + }); +}); + +describe("power — negative base domain", () => { + it("is #NUM! for a negative base with a non-integer exponent", () => { + assert.equal(power(-8, 1 / 3), NUM_ERROR); + assert.equal(power(-2, 0.5), NUM_ERROR); + }); + + it("computes when the exponent is an integer or the base is non-negative", () => { + assert.equal(power(-2, 3), -8); + assert.equal(power(2, 10), 1024); + assert.ok(closeTo(power(9, 0.5) as number, 3)); + }); +}); + +describe("safeSqrt / safeLog / safeLog10 — domain", () => { + it("is #NUM! outside the domain", () => { + assert.equal(safeSqrt(-1), NUM_ERROR); + assert.equal(safeLog(0), NUM_ERROR); + assert.equal(safeLog(-1), NUM_ERROR); + assert.equal(safeLog10(0), NUM_ERROR); + }); + + it("computes inside the domain", () => { + assert.equal(safeSqrt(4), 2); + assert.ok(closeTo(safeLog(Math.E) as number, 1)); + assert.equal(safeLog10(1000), 3); + }); +}); + +describe("logWithBase — number and base domain", () => { + it("computes a valid base-N log", () => { + assert.ok(closeTo(logWithBase(8, 2) as number, 3)); + assert.ok(closeTo(logWithBase(100, 10) as number, 2)); + }); + + it("is #NUM! for a non-positive number", () => { + assert.equal(logWithBase(0, 10), NUM_ERROR); + assert.equal(logWithBase(-1, 10), NUM_ERROR); + }); + + it("is #NUM! for a base that is non-positive or exactly 1", () => { + assert.equal(logWithBase(8, 1), NUM_ERROR); + assert.equal(logWithBase(8, -2), NUM_ERROR); + assert.equal(logWithBase(8, 0), NUM_ERROR); + }); +}); + +describe("the handlers surface the errors end-to-end", () => { + const evalFormula = (formula: string): unknown => + cellAt(new SpreadsheetEngine().calculate({ name: "S", data: [[{ v: formula }]] } satisfies SheetData).data, 0, 0); + + it("displays the Excel error codes through the engine", () => { + assert.equal(evalFormula("=FLOOR(-2.5, 2)"), "#NUM!"); + assert.equal(evalFormula("=FLOOR(3, 0)"), "#DIV/0!"); + assert.equal(evalFormula("=CEILING(3, 0)"), 0); + assert.equal(evalFormula("=SQRT(-1)"), "#NUM!"); + assert.equal(evalFormula("=MOD(5, 0)"), "#DIV/0!"); + assert.equal(evalFormula("=ROUND(-2.5, 0)"), -3); + assert.equal(evalFormula("=MOD(-3, 2)"), 1); + assert.equal(evalFormula("=LOG(8, 1)"), "#NUM!"); + assert.equal(evalFormula("=LOG(8, -2)"), "#NUM!"); + assert.equal(evalFormula("=LOG(8, 2)"), 3); + }); + + // Domain misses are error VALUES (not NaN/∞), so IFERROR must still catch + // them or nested formulas would surface the raw error (#2389 review). + it("lets IFERROR catch the domain errors", () => { + assert.equal(evalFormula("=IFERROR(SQRT(-1), 42)"), 42); + assert.equal(evalFormula("=IFERROR(MOD(5, 0), -1)"), -1); + assert.equal(evalFormula("=IFERROR(SQRT(4), 42)"), 2, "a non-error passes through"); + }); + + // Text that only looks like an error is real text, not an error value, so + // IFERROR returns it rather than the fallback. + it("does not treat quoted error-looking text as an error", () => { + assert.equal(evalFormula('=IFERROR("#NUM!", 42)'), "#NUM!"); + assert.equal(evalFormula('=IFERROR("hello", 42)'), "hello"); + }); +}); diff --git a/tests/engine/test_midValue.ts b/tests/engine/test_midValue.ts new file mode 100644 index 0000000..3efa3f1 --- /dev/null +++ b/tests/engine/test_midValue.ts @@ -0,0 +1,115 @@ +// MID's bounds and VALUE's parsing. Both returned a plausible answer instead of +// an error: `substring` SWAPS reversed bounds, so a negative MID count read +// backwards and produced earlier characters, and `parseFloat` stops at the first +// unreadable character, so VALUE("12abc") came back 12 (#2360). + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { takeMid, parseValueText } from "../../src/engine/functions/text.ts"; +import { VALUE_ERROR } from "../../src/engine/spreadsheet-errors.ts"; +import { SpreadsheetEngine, type SheetData } from "../../src/engine/index.ts"; +import { cellAt } from "./cellAccess.ts"; + +const evaluate = (formula: string, cell: string | number = "Hello"): unknown => { + const sheet: SheetData = { name: "S", data: [[{ v: cell }, { v: formula }]] }; + return cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 1); +}; + +describe("takeMid — bounds", () => { + it("errors on a negative count instead of reading backwards", () => { + assert.equal(takeMid("Hello", 3, -1), VALUE_ERROR); + }); + + it("errors on a non-finite count", () => { + assert.equal(takeMid("Hello", 3, NaN), VALUE_ERROR); + assert.equal(takeMid("Hello", 3, Infinity), VALUE_ERROR); + }); + + // Excel's MID is 1-based; 0 and negatives are not positions. + it("errors on a start position below 1", () => { + assert.equal(takeMid("Hello", 0, 2), VALUE_ERROR); + assert.equal(takeMid("Hello", -1, 2), VALUE_ERROR); + }); + + it("takes the requested characters from a 1-based start", () => { + assert.equal(takeMid("Hello", 1, 2), "He"); + assert.equal(takeMid("Hello", 2, 3), "ell"); + }); + + it("stops at the end of the text when the count overruns", () => { + assert.equal(takeMid("Hello", 4, 99), "lo"); + }); + + it("returns an empty string for a zero count", () => { + assert.equal(takeMid("Hello", 2, 0), ""); + }); + + it("truncates a fractional count and start toward zero", () => { + assert.equal(takeMid("Hello", 2.9, 2.9), "el"); + }); +}); + +describe("parseValueText — the whole string must be a number", () => { + it("errors on trailing text rather than salvaging the prefix", () => { + assert.equal(parseValueText("12abc"), VALUE_ERROR); + assert.equal(parseValueText("3.5kg"), VALUE_ERROR); + }); + + it("errors on an empty or blank string", () => { + assert.equal(parseValueText(""), VALUE_ERROR); + assert.equal(parseValueText(" "), VALUE_ERROR); + }); + + it("reads a plain number, tolerating surrounding whitespace", () => { + assert.equal(parseValueText("42"), 42); + assert.equal(parseValueText(" 7 "), 7); + assert.equal(parseValueText("-3.5"), -3.5); + }); + + it("strips currency symbols and thousands separators", () => { + assert.equal(parseValueText("$1,234.5"), 1234.5); + }); + + it("reads a trailing percent as a fraction", () => { + assert.equal(parseValueText("50%"), 0.5); + assert.equal(parseValueText("12abc%"), VALUE_ERROR, "still a whole-string match"); + }); + + // `Number` accepts JS-only spellings a spreadsheet never should. + it("rejects JS-only numeric syntaxes", () => { + assert.equal(parseValueText("0x10"), VALUE_ERROR, "hex"); + assert.equal(parseValueText("0X10"), VALUE_ERROR, "hex, upper case"); + assert.equal(parseValueText("0b10"), VALUE_ERROR, "binary"); + assert.equal(parseValueText("0o17"), VALUE_ERROR, "octal"); + assert.equal(parseValueText("Infinity"), VALUE_ERROR); + assert.equal(parseValueText("-Infinity"), VALUE_ERROR); + assert.equal(parseValueText("1_000"), VALUE_ERROR, "numeric separator"); + }); + + // The decimal pattern matches these, so only the finiteness check rejects + // them — without it the guard would be dead code and could be dropped unseen. + it("rejects an exponent that overflows to infinity", () => { + assert.equal(parseValueText("1e999"), VALUE_ERROR); + assert.equal(parseValueText("-1e999"), VALUE_ERROR); + assert.equal(parseValueText("1e999%"), VALUE_ERROR, "also through the percent path"); + }); + + it("still reads decimal and scientific notation", () => { + assert.equal(parseValueText("1e3"), 1000); + assert.equal(parseValueText("-2.5E-2"), -0.025); + assert.equal(parseValueText(".5"), 0.5); + assert.equal(parseValueText("+7"), 7); + }); +}); + +describe("through the engine", () => { + it("surfaces the MID and VALUE errors in the cell", () => { + assert.equal(evaluate("=MID(A1,3,-1)"), "#VALUE!"); + assert.equal(evaluate('=VALUE("12abc")'), "#VALUE!"); + }); + + it("keeps the working cases working", () => { + assert.equal(evaluate("=MID(A1,2,3)"), "ell"); + assert.equal(evaluate('=VALUE("42")'), 42); + }); +}); diff --git a/tests/engine/test_multiRangeAggregates.ts b/tests/engine/test_multiRangeAggregates.ts new file mode 100644 index 0000000..748948b --- /dev/null +++ b/tests/engine/test_multiRangeAggregates.ts @@ -0,0 +1,134 @@ +// Aggregates over more than one argument. Excel takes up to 255 (`SUM(A1:A2, +// B1:B2)`, `SUM(A1:A2, 10)`), but eight of these functions were registered with +// `maxArgs: 1` and read only `args[0]`, so the second range made the whole +// formula fail — a loud `#ERROR!` on ordinary spreadsheet usage (#2360). + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { SpreadsheetEngine, type SheetData } from "../../src/engine/index.ts"; +import { cellAt } from "./cellAccess.ts"; + +// A = 1,2 ; B = 3,4 ; the formula sits in C1. +const evaluate = (formula: string): unknown => { + const sheet: SheetData = { + name: "S", + data: [ + [{ v: 1 }, { v: 3 }, { v: formula }], + [{ v: 2 }, { v: 4 }], + ], + }; + return cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 2); +}; + +describe("aggregates accept several ranges", () => { + it("sums two ranges", () => { + assert.equal(evaluate("=SUM(A1:A2,B1:B2)"), 10); + }); + + it("averages across two ranges", () => { + assert.equal(evaluate("=AVERAGE(A1:A2,B1:B2)"), 2.5); + }); + + it("counts numbers across two ranges", () => { + assert.equal(evaluate("=COUNT(A1:A2,B1:B2)"), 4); + }); + + it("counts non-empty cells across two ranges", () => { + assert.equal(evaluate("=COUNTA(A1:A2,B1:B2)"), 4); + }); + + it("takes the median across two ranges", () => { + assert.equal(evaluate("=MEDIAN(A1:A2,B1:B2)"), 2.5); + }); +}); + +describe("aggregates mix ranges with plain values", () => { + it("sums a range plus a literal", () => { + assert.equal(evaluate("=SUM(A1:A2,10)"), 13); + }); + + it("sums a range plus a single cell reference", () => { + assert.equal(evaluate("=SUM(A1:A2,B1)"), 6); + }); +}); + +describe("a single cell reference is read as a range, not a scalar", () => { + // The scalar path coerces a blank or text cell to 0, so COUNT(A999) counted an + // empty cell as a value once multi-argument collection was introduced (Codex + // review). A bare cell ref goes through the range path instead. + const countSheet = (formula: string): unknown => { + const sheet: SheetData = { + name: "S", + data: [[{ v: 5 }, { v: formula }], [{ v: "txt" }]], + }; + return cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 1); + }; + + it("does not count an out-of-bounds cell", () => { + assert.equal(countSheet("=COUNT(A999)"), 0); + assert.equal(countSheet("=COUNTA(A999)"), 0); + }); + + it("does not count a text cell as a number", () => { + assert.equal(countSheet("=COUNT(A2)"), 0); + assert.equal(countSheet("=COUNTA(A2)"), 1, "COUNTA does count text"); + }); + + it("counts a single numeric cell", () => { + assert.equal(countSheet("=COUNT(A1)"), 1); + }); +}); + +describe("COUNT counts only the arguments that hold a number", () => { + // Multi-argument collection reads a non-reference argument as a scalar, and the + // lenient `toNumber` turns anything unreadable into 0 — so every scalar looked + // like a value and COUNT("text") answered 1 (Codex review). Excel and the + // pre-change engine both answer 0. + it("does not count a text literal", () => { + assert.equal(evaluate('=COUNT("text")'), 0); + }); + + it("counts the numbers and skips the text when both are given", () => { + assert.equal(evaluate('=COUNT(1,"text")'), 1); + assert.equal(evaluate('=COUNT(A1:A2,"text")'), 2); + }); + + it("counts a number typed directly, quoted, or computed", () => { + assert.equal(evaluate("=COUNT(1)"), 1); + assert.equal(evaluate('=COUNT("1")'), 1, "Excel counts a quoted number typed as an argument"); + assert.equal(evaluate("=COUNT(1+1)"), 1); + }); + + // Excel counts a logical typed directly as an argument; this engine has PINNED + // booleans as non-numbers throughout (`toNumber(true)` is 0, see #2391), so + // COUNT follows the engine rather than splitting the difference. + it("does not count a logical literal", () => { + assert.equal(evaluate("=COUNT(TRUE)"), 0); + assert.equal(evaluate("=COUNT(FALSE)"), 0); + }); + + // Excel reads "12abc" as text and answers 0. The engine's shared numeric parser + // reads the leading number everywhere (SUM(1,"12abc") is 13), so COUNT stays + // consistent with the engine instead. + it("counts a leading-number string, as the rest of the engine reads it", () => { + assert.equal(evaluate('=COUNT("12abc")'), 1); + }); + + it("leaves the lenient aggregates coercing as before", () => { + assert.equal(evaluate('=SUM(1,"text")'), 1); + assert.equal(evaluate('=AVERAGE(1,"text")'), 0.5); + }); + + it("still lets COUNTA count a text literal", () => { + assert.equal(evaluate('=COUNTA(1,"text")'), 2); + assert.equal(evaluate('=COUNTA("")'), 0); + }); +}); + +describe("the single-range behaviour is unchanged", () => { + it("sums, averages and counts one range as before", () => { + assert.equal(evaluate("=SUM(A1:A2)"), 3); + assert.equal(evaluate("=AVERAGE(A1:A2)"), 1.5); + assert.equal(evaluate("=COUNT(A1:A2)"), 2); + }); +}); diff --git a/tests/engine/test_normalizeData.ts b/tests/engine/test_normalizeData.ts new file mode 100644 index 0000000..1e4ec04 --- /dev/null +++ b/tests/engine/test_normalizeData.ts @@ -0,0 +1,59 @@ +// Coercing whatever shape a model emitted into a 2D cell grid. It runs before +// every calculation, and a wrong reshape silently maps every A1-style reference +// onto a different cell — the sheet computes, just against the wrong layout. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { normalizeData } from "../../src/engine/calculator.ts"; + +describe("normalizeData — already valid", () => { + it("returns a 2D array unchanged", () => { + const grid = [[{ v: 1 }, { v: 2 }], [{ v: 3 }]]; + assert.equal(normalizeData(grid), grid, "same reference, not a copy"); + }); + + it("keeps an empty row structure", () => { + const grid = [[]]; + assert.equal(normalizeData(grid), grid); + }); +}); + +describe("normalizeData — nothing to normalise", () => { + it("returns empty for null, undefined and non-arrays", () => { + assert.deepEqual(normalizeData(null), []); + assert.deepEqual(normalizeData(undefined), []); + assert.deepEqual(normalizeData("A1"), []); + assert.deepEqual(normalizeData(42), []); + assert.deepEqual(normalizeData({ v: 1 }), []); + }); + + it("returns empty for an empty array", () => { + assert.deepEqual(normalizeData([]), []); + }); + + // A flat array of primitives is not a recognised shape — pairing them would + // invent structure, so it returns empty rather than guess. + it("returns empty for a flat array of primitives", () => { + assert.deepEqual(normalizeData([1, 2, 3]), []); + assert.deepEqual(normalizeData(["a", "b"]), []); + }); +}); + +describe("normalizeData — flat cell array to 2D", () => { + // A model that emits a flat list of cell objects is reshaped into rows of two. + it("pairs a flat cell array into two-column rows", () => { + assert.deepEqual(normalizeData([{ v: 1 }, { v: 2 }, { v: 3 }, { v: 4 }]), [ + [{ v: 1 }, { v: 2 }], + [{ v: 3 }, { v: 4 }], + ]); + }); + + // An odd length leaves a one-cell final row rather than dropping or padding. + it("leaves a lone final cell in its own row when the count is odd", () => { + assert.deepEqual(normalizeData([{ v: 1 }, { v: 2 }, { v: 3 }]), [[{ v: 1 }, { v: 2 }], [{ v: 3 }]]); + }); + + it("reshapes a single cell into one row", () => { + assert.deepEqual(normalizeData([{ v: 1 }]), [[{ v: 1 }]]); + }); +}); diff --git a/tests/engine/test_npvArgs.ts b/tests/engine/test_npvArgs.ts new file mode 100644 index 0000000..0fdf971 --- /dev/null +++ b/tests/engine/test_npvArgs.ts @@ -0,0 +1,38 @@ +// NPV period assignment through the handler. The pre-refactor handler used +// `period = argIndex + rangePosition`, so a scalar argument after a multi-cell +// range landed on the range's period rather than continuing the sequence (a +// latent off-by bug). The refactor normalizes this to strictly sequential +// periods — the Excel semantics — so a mixed `NPV(rate, range, scalar)` now +// discounts every flow by its position in the flattened series (#2394 / #2442). +// This pins the intended (sequential) behavior at the handler level. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { SpreadsheetEngine, type SheetData } from "../../src/engine/index.ts"; +import { cellAt } from "./cellAccess.ts"; + +const closeTo = (actual: number, expected: number, eps = 0.01): boolean => Math.abs(actual - expected) <= eps; + +describe("NPV — sequential periods across mixed range and scalar arguments", () => { + it("discounts a scalar after a range by the next period, not the range's", () => { + // A1:A3 = 100, 200, 300 (periods 1..3); B1 = 400 must be period 4. + // 100/1.1 + 200/1.1^2 + 300/1.1^3 + 400/1.1^4 = 754.7967… + // The old arg-index math discounted B1 at period 2 (= 812.17), which is wrong. + const sheet: SheetData = { + name: "S", + data: [[{ v: 100 }, { v: 400 }, { v: "=NPV(0.1, A1:A3, B1)" }], [{ v: 200 }], [{ v: 300 }]], + }; + const result = cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 2) as number; + assert.ok(closeTo(result, 754.7967), `NPV mixed args = ${result}, expected ~754.80 (sequential)`); + }); + + it("matches a single range read as consecutive periods", () => { + // 100/1.1 + 200/1.1^2 + 300/1.1^3 = 481.5928… + const sheet: SheetData = { + name: "S", + data: [[{ v: 100 }, { v: "=NPV(0.1, A1:A3)" }], [{ v: 200 }], [{ v: 300 }]], + }; + const result = cellAt(new SpreadsheetEngine().calculate(sheet).data, 0, 1) as number; + assert.ok(closeTo(result, 481.5928), `NPV single range = ${result}, expected ~481.59`); + }); +}); diff --git a/tests/engine/test_npvFunction.ts b/tests/engine/test_npvFunction.ts new file mode 100644 index 0000000..c1f878a --- /dev/null +++ b/tests/engine/test_npvFunction.ts @@ -0,0 +1,32 @@ +// NPV through the whole engine. The #2390 bug used each value's ARGUMENT index +// as its discount period, so a scalar after a range was discounted too little: +// NPV(0.1, A1:A3, 500) put 500 at period 2 instead of period 4. The period must +// count flattened values, not arguments. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { SpreadsheetEngine } from "../../src/engine/index.ts"; +import { cellAt } from "./cellAccess.ts"; + +const closeTo = (actual: unknown, expected: number, eps = 1e-6): boolean => typeof actual === "number" && Math.abs(actual - expected) <= eps; + +describe("NPV — range followed by a scalar (#2390)", () => { + it("discounts the trailing scalar at period 4, not period 2", () => { + const sheet = { + name: "S", + data: [[{ v: 100 }], [{ v: 200 }], [{ v: 300 }], [{ v: "=NPV(0.1, A1:A3, 500)" }]], + }; + const result = new SpreadsheetEngine().calculate(sheet); + const expected = 100 / 1.1 + 200 / 1.1 ** 2 + 300 / 1.1 ** 3 + 500 / 1.1 ** 4; + const actual = cellAt(result.data, 3, 0); + assert.ok(closeTo(actual, expected), `NPV ≈ ${expected}, got ${String(actual)}`); + }); + + it("matches the all-scalar form when there is no range", () => { + const sheet = { name: "S", data: [[{ v: "=NPV(0.1, 100, 200, 300, 500)" }]] }; + const result = new SpreadsheetEngine().calculate(sheet); + const expected = 100 / 1.1 + 200 / 1.1 ** 2 + 300 / 1.1 ** 3 + 500 / 1.1 ** 4; + const actual = cellAt(result.data, 0, 0); + assert.ok(closeTo(actual, expected), `NPV ≈ ${expected}, got ${String(actual)}`); + }); +}); diff --git a/tests/engine/test_numericCoercion.ts b/tests/engine/test_numericCoercion.ts new file mode 100644 index 0000000..7808122 --- /dev/null +++ b/tests/engine/test_numericCoercion.ts @@ -0,0 +1,152 @@ +// Two numeric reads share one string parser (#2391). `toNumber` stays lenient for +// range aggregation (unreadable → 0, booleans → 0 — PINNED, since changing it +// moves every SUM / AVERAGE / COUNTIF at once). `toScalarNumber` is the strict +// scalar read that ABS / SIGN now use: booleans are Excel's 1/0 and non-numeric +// text is #VALUE! instead of a silent 0. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { holdsNumber, parseNumericString, toScalarNumber } from "../../src/engine/numericCoercion.ts"; +import { DIV_ZERO_ERROR, VALUE_ERROR } from "../../src/engine/spreadsheet-errors.ts"; +import { toNumber } from "../../src/engine/registry.ts"; + +describe("parseNumericString — reads the formats the engine has always read", () => { + it("reads a percentage as its decimal value", () => { + assert.equal(parseNumericString("5%"), 0.05); + assert.equal(parseNumericString("100%"), 1); + }); + + it("reads currency and thousands-separated numbers", () => { + assert.equal(parseNumericString("$1,000"), 1000); + assert.equal(parseNumericString("$1,000.50"), 1000.5); + assert.equal(parseNumericString("1,234,567"), 1234567); + }); + + it("reads a plain numeric string, tolerating whitespace and exponents", () => { + assert.equal(parseNumericString("42"), 42); + assert.equal(parseNumericString(" 42 "), 42); + assert.equal(parseNumericString("-3.5"), -3.5); + assert.equal(parseNumericString("1e3"), 1000); + }); + + it("takes the leading number from a partly-numeric string", () => { + assert.equal(parseNumericString("12abc"), 12); + assert.equal(parseNumericString("3.5kg"), 3.5); + }); + + // The distinction that lets the two coercions differ: null, not 0, when there + // is no number at all. toNumber maps that to 0; toScalarNumber to #VALUE!. + it("returns null when nothing parses", () => { + assert.equal(parseNumericString("abc"), null); + assert.equal(parseNumericString(""), null); + assert.equal(parseNumericString(" "), null); + assert.equal(parseNumericString("N/A"), null); + assert.equal(parseNumericString("abc12"), null); + }); +}); + +describe("toNumber — PINNED lenient behaviour (#2391 does not change this)", () => { + it("returns a number unchanged", () => { + assert.equal(toNumber(42), 42); + assert.equal(toNumber(0), 0); + assert.equal(toNumber(-7.5), -7.5); + }); + + it("maps unreadable text to 0", () => { + assert.equal(toNumber("hello"), 0); + assert.equal(toNumber(""), 0); + assert.equal(toNumber("N/A"), 0); + }); + + // The high-blast-radius case the issue asked to pin: booleans read as 0 here, + // NOT Excel's 1/0, because SUM / AVERAGE / COUNTIF all lean on this. + it("maps booleans to 0, not Excel's 1 and 0", () => { + assert.equal(toNumber(true), 0); + assert.equal(toNumber(false), 0); + }); + + it("still reads formatted strings", () => { + assert.equal(toNumber("5%"), 0.05); + assert.equal(toNumber("$1,000"), 1000); + assert.equal(toNumber("12abc"), 12); + }); +}); + +describe("toScalarNumber — strict scalar read for ABS / SIGN (#2391)", () => { + it("returns a number unchanged", () => { + assert.equal(toScalarNumber(42), 42); + assert.equal(toScalarNumber(-7.5), -7.5); + assert.equal(toScalarNumber(0), 0); + }); + + // The boolean fix: TRUE=1, FALSE=0 (Excel), where toNumber gives 0 for both. + it("reads booleans as Excel's 1 and 0", () => { + assert.equal(toScalarNumber(true), 1); + assert.equal(toScalarNumber(false), 0); + }); + + it("parses numeric and formatted text", () => { + assert.equal(toScalarNumber("5"), 5); + assert.equal(toScalarNumber("$1,000"), 1000); + assert.equal(toScalarNumber("50%"), 0.5); + }); + + // The text fix: genuinely non-numeric text is an error, not a silent 0. + it("returns #VALUE! for non-numeric text and empty strings", () => { + assert.equal(toScalarNumber("abc"), VALUE_ERROR); + assert.equal(toScalarNumber(""), VALUE_ERROR); + assert.equal(toScalarNumber(" "), VALUE_ERROR); + }); + + // Deliberate leniency, pinned: a partly-numeric string keeps its leading number + // (matching the rest of the engine) rather than erroring as strict Excel would. + it("still takes the leading number from partly-numeric text", () => { + assert.equal(toScalarNumber("12abc"), 12); + }); +}); + +// The question `toNumber` cannot answer: it maps text, booleans and a genuine 0 +// to the same 0, so COUNT could not tell "no number here" from "the number zero" +// and counted COUNT("text") as a value (Codex review on #2360). +describe("holdsNumber — is there a number in this value at all", () => { + it("is true for numbers, including zero and negatives", () => { + assert.equal(holdsNumber(42), true); + assert.equal(holdsNumber(0), true); + assert.equal(holdsNumber(-7.5), true); + }); + + it("is true for text the engine reads as a number", () => { + assert.equal(holdsNumber("1"), true); + assert.equal(holdsNumber(" 42 "), true); + assert.equal(holdsNumber("5%"), true); + assert.equal(holdsNumber("$1,000"), true); + }); + + it("is false for text holding no number", () => { + assert.equal(holdsNumber("text"), false); + assert.equal(holdsNumber(""), false); + assert.equal(holdsNumber(" "), false); + assert.equal(holdsNumber("N/A"), false); + assert.equal(holdsNumber("abc12"), false); + }); + + // Same PINNED stance as toNumber: a boolean is not a number in this engine. + it("is false for booleans", () => { + assert.equal(holdsNumber(true), false); + assert.equal(holdsNumber(false), false); + }); + + it("is false for a formula error value", () => { + assert.equal(holdsNumber(DIV_ZERO_ERROR), false); + }); + + it("is false for NaN, which is a number that is no number", () => { + assert.equal(holdsNumber(NaN), false); + }); + + // Inherited from parseNumericString, pinned deliberately: the engine reads the + // leading number everywhere, so SUM and COUNT agree that "12abc" has one. + it("is true for a leading-number string, unlike Excel", () => { + assert.equal(holdsNumber("12abc"), true); + }); +}); diff --git a/tests/engine/test_numericFunctions2391.ts b/tests/engine/test_numericFunctions2391.ts new file mode 100644 index 0000000..28bbc51 --- /dev/null +++ b/tests/engine/test_numericFunctions2391.ts @@ -0,0 +1,72 @@ +// The four #2391 scenarios end-to-end through the engine, plus pins that the +// out-of-scope lenient aggregation paths (SUM / AVERAGE over text) are unchanged +// (those blanks-as-0 cases belong to #2383, not this PR). + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { SpreadsheetEngine, type SheetData } from "../../src/engine/index.ts"; +import { cellAt } from "./cellAccess.ts"; + +// Column A holds `colA` (one value per row); `formula` goes in a row below it so a +// range like A1:A3 never overlaps the formula cell. +const evalWithColumnA = (formula: string, colA: (string | number)[] = []): unknown => { + const data: { v: string | number }[][] = colA.map((value) => [{ v: value }]); + data.push([{ v: formula }]); + const result = new SpreadsheetEngine().calculate({ name: "S", data } as SheetData); + return cellAt(result.data, data.length - 1, 0); +}; + +describe("ABS / SIGN — scalar coercion (#2391)", () => { + it("reads a logical argument as 1 / 0 (was 0)", () => { + assert.equal(evalWithColumnA("=ABS(TRUE())"), 1); + assert.equal(evalWithColumnA("=ABS(FALSE())"), 0); + assert.equal(evalWithColumnA("=SIGN(TRUE())"), 1); + }); + + it("returns #VALUE! for non-numeric text (was 0)", () => { + assert.equal(evalWithColumnA('=ABS("abc")'), "#VALUE!"); + assert.equal(evalWithColumnA('=SIGN("abc")'), "#VALUE!"); + }); + + it("errors when the argument cell holds text", () => { + assert.equal(evalWithColumnA("=ABS(A1)", ["abc"]), "#VALUE!"); + }); + + it("still works for ordinary numbers", () => { + assert.equal(evalWithColumnA("=ABS(-5)"), 5); + assert.equal(evalWithColumnA("=SIGN(-5)"), -1); + assert.equal(evalWithColumnA("=ABS(A1)", [-42]), 42); + }); +}); + +describe("MODE — no repeat is #N/A (#2391)", () => { + it("returns #N/A when every value is distinct (was the first value)", () => { + assert.equal(evalWithColumnA("=MODE(A1:A3)", [1, 2, 3]), "#N/A"); + }); + + it("still returns the most frequent value when one repeats", () => { + assert.equal(evalWithColumnA("=MODE(A1:A4)", [1, 2, 2, 3]), 2); + }); +}); + +describe("AVERAGEIF — no match is #DIV/0! (#2391)", () => { + it("returns #DIV/0! when nothing matches (was 0)", () => { + assert.equal(evalWithColumnA('=AVERAGEIF(A1:A3,">100")', [1, 2, 3]), "#DIV/0!"); + }); + + it("still averages the matching cells", () => { + assert.equal(evalWithColumnA('=AVERAGEIF(A1:A3,">1")', [1, 2, 3]), 2.5); + }); +}); + +describe("out-of-scope lenient paths stay unchanged (#2383, not this PR)", () => { + // A text cell inside a SUM range leaves the total unchanged (30, not #VALUE!) — + // the aggregation path keeps its lenient reading. This PR must not touch it. + it("SUM leaves a text cell out of the total", () => { + assert.equal(evalWithColumnA("=SUM(A1:A3)", [10, "abc", 20]), 30); + }); + + it("AVERAGE over numeric cells is unaffected", () => { + assert.equal(evalWithColumnA("=AVERAGE(A1:A3)", [10, 20, 30]), 20); + }); +}); diff --git a/tests/engine/test_parseFunctionArgs.ts b/tests/engine/test_parseFunctionArgs.ts new file mode 100644 index 0000000..93136dc --- /dev/null +++ b/tests/engine/test_parseFunctionArgs.ts @@ -0,0 +1,79 @@ +// Splitting a function's argument string into arguments. Exported but never +// tested, and it decides where every multi-argument function's arguments begin +// and end — a wrong split feeds a formula the wrong operands with no error, just +// a wrong result. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { parseFunctionArgs } from "../../src/engine/evaluator.ts"; + +describe("parseFunctionArgs — plain splitting", () => { + it("splits comma-separated arguments and trims each", () => { + assert.deepEqual(parseFunctionArgs("A1, B2, C3"), ["A1", "B2", "C3"]); + assert.deepEqual(parseFunctionArgs("1,2,3"), ["1", "2", "3"]); + }); + + it("returns a single argument when there is no comma", () => { + assert.deepEqual(parseFunctionArgs("A1"), ["A1"]); + assert.deepEqual(parseFunctionArgs("A1:A10"), ["A1:A10"]); + }); + + it("returns nothing for an empty string", () => { + assert.deepEqual(parseFunctionArgs(""), []); + assert.deepEqual(parseFunctionArgs(" "), []); + }); +}); + +describe("parseFunctionArgs — nesting", () => { + // A comma inside a nested call belongs to that call, not the outer one: + // SUM(A1, MAX(B1, C1)) is two arguments, not three. + it("keeps a comma inside a nested function with its call", () => { + assert.deepEqual(parseFunctionArgs("A1, MAX(B1, C1)"), ["A1", "MAX(B1, C1)"]); + }); + + it("handles multiple and deeply nested calls", () => { + assert.deepEqual(parseFunctionArgs("SUM(A1,A2), COUNT(B1,B2,B3)"), ["SUM(A1,A2)", "COUNT(B1,B2,B3)"]); + assert.deepEqual(parseFunctionArgs("ROUND(SUM(A1,A2)/COUNT(A1,A2), 2)"), ["ROUND(SUM(A1,A2)/COUNT(A1,A2), 2)"]); + }); + + it("splits at the top level around a nested call", () => { + assert.deepEqual(parseFunctionArgs("IF(A1>0, 1, 0), B1"), ["IF(A1>0, 1, 0)", "B1"]); + }); +}); + +describe("parseFunctionArgs — strings", () => { + // A comma or a parenthesis inside a quoted string is text, not structure. + it("keeps a comma inside a quoted string", () => { + assert.deepEqual(parseFunctionArgs('"a, b", C1'), ['"a, b"', "C1"]); + assert.deepEqual(parseFunctionArgs("'x, y', 1"), ["'x, y'", "1"]); + }); + + it("keeps a parenthesis inside a quoted string from disturbing the depth", () => { + assert.deepEqual(parseFunctionArgs('"f(x)", A1'), ['"f(x)"', "A1"]); + assert.deepEqual(parseFunctionArgs('SUM(A1), "not )a close"'), ["SUM(A1)", '"not )a close"']); + }); + + it("preserves the quotes on a quoted argument", () => { + assert.deepEqual(parseFunctionArgs('"hello"'), ['"hello"']); + }); + + // A quote is only a boundary when it is not backslash-escaped, so an escaped + // quote inside a string does not close it early. + it("does not treat a backslash-escaped quote as a boundary", () => { + assert.deepEqual(parseFunctionArgs('"say \\"hi\\", ok", B1'), ['"say \\"hi\\", ok"', "B1"]); + }); +}); + +describe("parseFunctionArgs — edge behaviour worth pinning", () => { + // Documented current behaviour, not an endorsement: a trailing empty argument + // is dropped, so IF(A1>0,"yes",) reads as TWO arguments, not three (#2359). A + // leading or interior empty argument is kept. + it("drops a trailing empty argument", () => { + assert.deepEqual(parseFunctionArgs('A1, "yes",'), ["A1", '"yes"']); + }); + + it("keeps a leading or interior empty argument", () => { + assert.deepEqual(parseFunctionArgs(",B1"), ["", "B1"]); + assert.deepEqual(parseFunctionArgs("A1,,C1"), ["A1", "", "C1"]); + }); +}); diff --git a/tests/engine/test_parseRangeBounds.ts b/tests/engine/test_parseRangeBounds.ts new file mode 100644 index 0000000..b7f60d2 --- /dev/null +++ b/tests/engine/test_parseRangeBounds.ts @@ -0,0 +1,94 @@ +// `parseRangeBounds` is the single range parser the lookup functions now share +// (#2396). It carries the sheet-prefix split that one of the four former copies +// lacked — the copy that made cross-sheet VLOOKUP throw (#2390). Columns come +// back 0-based (A=0), rows stay 1-based (A1 notation). + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { parseRangeBounds, resolveIndexTarget, type RangeBounds } from "../../src/engine/formulaRefs.ts"; + +describe("parseRangeBounds — plain ranges", () => { + it("parses A1:B10 (cols 0-based, rows 1-based)", () => { + assert.deepEqual(parseRangeBounds("A1:B10"), { sheetPrefix: "", startCol: 0, startRow: 1, endCol: 1, endRow: 10 }); + }); + + it("parses an offset range A2:C10", () => { + assert.deepEqual(parseRangeBounds("A2:C10"), { sheetPrefix: "", startCol: 0, startRow: 2, endCol: 2, endRow: 10 }); + }); + + // Column boundaries: A is Excel column 1 → index 0, Z is 26 → 25, AA is 27 → 26. + it("parses the A / Z / AA column boundaries", () => { + assert.equal(parseRangeBounds("A1:A1")?.startCol, 0); + assert.equal(parseRangeBounds("Z1:Z1")?.startCol, 25); + assert.equal(parseRangeBounds("AA1:AB2")?.startCol, 26); + assert.equal(parseRangeBounds("AA1:AB2")?.endCol, 27); + }); + + it("keeps a single-cell-wide/tall range (start == end)", () => { + assert.deepEqual(parseRangeBounds("B3:B3"), { sheetPrefix: "", startCol: 1, startRow: 3, endCol: 1, endRow: 3 }); + }); +}); + +describe("parseRangeBounds — sheet-qualified ranges (the #2390 case)", () => { + it("splits an unquoted sheet prefix", () => { + assert.deepEqual(parseRangeBounds("Sheet1!A2:C10"), { sheetPrefix: "Sheet1!", startCol: 0, startRow: 2, endCol: 2, endRow: 10 }); + }); + + it("splits a quoted sheet name containing a space", () => { + assert.deepEqual(parseRangeBounds("'My Sheet'!A1:B2"), { sheetPrefix: "'My Sheet'!", startCol: 0, startRow: 1, endCol: 1, endRow: 2 }); + }); +}); + +describe("parseRangeBounds — non-ranges return null", () => { + it("rejects a single cell (no colon)", () => { + assert.equal(parseRangeBounds("A1"), null); + assert.equal(parseRangeBounds("Sheet1!A1"), null); + }); + + it("rejects malformed input", () => { + assert.equal(parseRangeBounds("not-a-range"), null); + assert.equal(parseRangeBounds(""), null); + }); + + // Deliberate limitation, pinned so it is not "fixed" by accident: the parser + // matches uppercase `[A-Z]` with no `$`, exactly as the four former copies did. + // The engine's own cell reader (calculator.getCellValue) is likewise + // uppercase-only, so accepting these here would not make them resolve. + it("rejects lowercase and $-absolute ranges (matches prior lookup behaviour)", () => { + assert.equal(parseRangeBounds("a1:b2"), null); + assert.equal(parseRangeBounds("$A$1:$B$2"), null); + }); +}); + +describe("resolveIndexTarget — INDEX bounds (#2390)", () => { + // A2:B5 → cols 0..1, rows 2..5 (4 rows × 2 cols). + const bounds: RangeBounds = { sheetPrefix: "", startCol: 0, startRow: 2, endCol: 1, endRow: 5 }; + + it("resolves an in-range 1-based position to an absolute cell", () => { + assert.deepEqual(resolveIndexTarget(bounds, 1, 1), { colIndex: 0, row: 2 }); // A2 + assert.deepEqual(resolveIndexTarget(bounds, 4, 2), { colIndex: 1, row: 5 }); // B5 + }); + + it("returns null (→ #REF!) when the row is past the range", () => { + const single: RangeBounds = { sheetPrefix: "", startCol: 0, startRow: 1, endCol: 0, endRow: 3 }; + assert.equal(resolveIndexTarget(single, 5, 1), null); // INDEX(A1:A3,5) + }); + + it("returns null for a row/col below 1", () => { + assert.equal(resolveIndexTarget(bounds, -1, 1), null); + assert.equal(resolveIndexTarget(bounds, 1, 3), null); // col past the 2-wide range + }); + + it("treats Excel's 0 (whole line) as out of range for a multi-line dimension", () => { + assert.equal(resolveIndexTarget(bounds, 0, 1), null); // INDEX(A2:B5,0,1) — must NOT read A1 + }); + + it("collapses 0 to the only line when that dimension is a single cell", () => { + const oneRow: RangeBounds = { sheetPrefix: "", startCol: 0, startRow: 3, endCol: 2, endRow: 3 }; + assert.deepEqual(resolveIndexTarget(oneRow, 0, 2), { colIndex: 1, row: 3 }); // whole (single) row, col 2 → B3 + }); + + it("truncates a fractional index toward zero, like Excel", () => { + assert.deepEqual(resolveIndexTarget(bounds, 2.9, 1), { colIndex: 0, row: 3 }); // 2.9 → row 2 → A3 + }); +}); diff --git a/tests/engine/test_parser.ts b/tests/engine/test_parser.ts new file mode 100644 index 0000000..098ee90 --- /dev/null +++ b/tests/engine/test_parser.ts @@ -0,0 +1,93 @@ +// A1-notation column conversion. Every other reference-handling path in the +// engine is built on these two, and both fail silently: `columnToIndex` does +// arithmetic on char codes with no validation, so a bad input returns a +// plausible-looking number instead of throwing, and the caller reads the wrong +// column. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { columnToIndex, indexToColumn } from "../../src/engine/parser.ts"; + +describe("columnToIndex", () => { + it("maps the single-letter columns", () => { + assert.equal(columnToIndex("A"), 0); + assert.equal(columnToIndex("B"), 1); + assert.equal(columnToIndex("Z"), 25); + }); + + // The bijective-base-26 boundary: AA follows Z, not "BA" or index 26+1. + it("maps the two-letter columns across the Z→AA boundary", () => { + assert.equal(columnToIndex("AA"), 26); + assert.equal(columnToIndex("AB"), 27); + assert.equal(columnToIndex("AZ"), 51); + assert.equal(columnToIndex("BA"), 52); + assert.equal(columnToIndex("ZZ"), 701); + }); + + it("maps three-letter columns", () => { + assert.equal(columnToIndex("AAA"), 702); + // XFD is Excel's last column (16384 columns, 0-based → 16383). + assert.equal(columnToIndex("XFD"), 16383); + }); + + // Documented current behaviour, NOT an endorsement: the function does no + // validation, so these return numbers rather than failing. Callers all + // pre-match `[A-Z]+`, which is what keeps the garbage out today — pinned so + // that if a caller's regex is ever relaxed, the consequence is visible here. + it("returns a wrong-but-plausible index for lowercase input (no validation)", () => { + // 'a' is 97; 97 - 64 - 1 = 32, i.e. column AG. + assert.equal(columnToIndex("a"), 32); + assert.equal(columnToIndex("z"), 57); + }); + + it("returns -1 for the empty string", () => { + assert.equal(columnToIndex(""), -1); + }); +}); + +describe("indexToColumn", () => { + it("maps the single-letter columns", () => { + assert.equal(indexToColumn(0), "A"); + assert.equal(indexToColumn(1), "B"); + assert.equal(indexToColumn(25), "Z"); + }); + + it("maps the two-letter columns across the Z→AA boundary", () => { + assert.equal(indexToColumn(26), "AA"); + assert.equal(indexToColumn(27), "AB"); + assert.equal(indexToColumn(51), "AZ"); + assert.equal(indexToColumn(52), "BA"); + assert.equal(indexToColumn(701), "ZZ"); + }); + + it("maps three-letter columns", () => { + assert.equal(indexToColumn(702), "AAA"); + assert.equal(indexToColumn(16383), "XFD"); + }); + + it("returns an empty string for a negative index", () => { + assert.equal(indexToColumn(-1), ""); + }); +}); + +describe("columnToIndex / indexToColumn round-trip", () => { + // The pair is used in both directions on the same value (range expansion + // walks indices, then renders refs back), so an asymmetry anywhere in the + // range silently shifts a whole range by one column. + it("round-trips every index across the single/double/triple letter boundaries", () => { + const boundaries = [0, 1, 24, 25, 26, 27, 50, 51, 52, 700, 701, 702, 703, 16382, 16383]; + for (const index of boundaries) { + assert.equal(columnToIndex(indexToColumn(index)), index, `round-trip failed at ${index}`); + } + }); + + it("round-trips a contiguous span with no gaps or repeats", () => { + const seen = new Set(); + for (let index = 0; index <= 1000; index++) { + const col = indexToColumn(index); + assert.equal(seen.has(col), false, `duplicate column label ${col} at index ${index}`); + seen.add(col); + assert.equal(columnToIndex(col), index); + } + }); +}); diff --git a/tests/engine/test_rangeFunctions.ts b/tests/engine/test_rangeFunctions.ts new file mode 100644 index 0000000..ecceaa9 --- /dev/null +++ b/tests/engine/test_rangeFunctions.ts @@ -0,0 +1,45 @@ +// Range-consuming functions through the whole engine. `expandRangeOrCell` is +// unit-tested on its own; this drives the shapes that used to return 0 with no +// error — the failure that is invisible because 0 is a plausible answer (#2356). + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { SpreadsheetEngine, type SheetData } from "../../src/engine/index.ts"; +import { cellAt } from "./cellAccess.ts"; + +/** A single column of numbers with `formula` beside the first cell. */ +const column = (values: number[], formula: string): SheetData => ({ + name: "S", + data: values.map((value, index) => (index === 0 ? [{ v: value }, { v: formula }] : [{ v: value }])), +}); + +const result = (values: number[], formula: string): unknown => cellAt(new SpreadsheetEngine().calculate(column(values, formula)).data, 0, 1); + +describe("range references that used to return 0", () => { + it("sums an absolute range", () => { + assert.equal(result([10, 20, 30], "=SUM($A$1:$A$3)"), 60); + }); + + it("sums a lowercase range", () => { + assert.equal(result([10, 20, 30], "=sum(a1:a3)"), 60); + }); + + it("sums a single-cell argument", () => { + assert.equal(result([42], "=SUM(A1)"), 42); + }); + + // MAX/MIN took a different path already, so they worked where SUM did not. + // The two must agree now that both go through the same expansion. + it("makes SUM and MAX agree on a single cell", () => { + assert.equal(result([42], "=SUM(A1)"), result([42], "=MAX(A1)")); + }); +}); + +describe("range functions over an absolute range", () => { + it("averages, counts and finds extremes", () => { + assert.equal(result([10, 20, 30], "=AVERAGE($A$1:$A$3)"), 20); + assert.equal(result([10, 20, 30], "=COUNT($A$1:$A$3)"), 3); + assert.equal(result([10, 20, 30], "=MAX($A$1:$A$3)"), 30); + assert.equal(result([10, 20, 30], "=MIN($A$1:$A$3)"), 10); + }); +}); diff --git a/tests/engine/test_registry.ts b/tests/engine/test_registry.ts new file mode 100644 index 0000000..98e7c0d --- /dev/null +++ b/tests/engine/test_registry.ts @@ -0,0 +1,220 @@ +// The two helpers every conditional and arithmetic function leans on. +// +// Neither can fail loudly: `toNumber` returns 0 for anything it cannot read, +// and `parseCriteria` returns a predicate that answers false. So a mistake here +// does not surface as an error — a SUM comes out smaller than it should, or a +// COUNTIF reports zero matches, and both look like ordinary answers. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { parseCriteria, toNumber, toString } from "../../src/engine/registry.ts"; +import type { CellValue } from "../../src/engine/types.ts"; + +const num = (value: CellValue) => toNumber(value); + +describe("toNumber — numbers pass through", () => { + it("returns a number unchanged, including 0 and negatives", () => { + assert.equal(num(42), 42); + assert.equal(num(0), 0); + assert.equal(num(-7.5), -7.5); + }); +}); + +describe("toNumber — formatted strings", () => { + it("reads a percentage as its decimal value", () => { + assert.equal(num("5%"), 0.05); + assert.equal(num("100%"), 1); + // Dividing by 100 is a binary-float operation, so a value with enough + // decimals lands a rounding step away from the exact literal. + assert.ok(Math.abs(num("0.4167%") - 0.004167) < 1e-12); + }); + + it("reads a currency string, stripping the symbol and separators", () => { + assert.equal(num("$1000"), 1000); + assert.equal(num("$1,000"), 1000); + assert.equal(num("$1,000.50"), 1000.5); + }); + + it("reads a comma-separated number", () => { + assert.equal(num("1,000"), 1000); + assert.equal(num("1,234,567"), 1234567); + }); + + it("reads a plain numeric string, tolerating surrounding whitespace", () => { + assert.equal(num("42"), 42); + assert.equal(num(" 42 "), 42); + assert.equal(num("-3.5"), -3.5); + assert.equal(num("1e3"), 1000); + }); +}); + +describe("toNumber — the branch order is load-bearing", () => { + // The checks run `%` → `$` → `,` → plain, and each strips only its OWN + // characters. A string carrying two of them therefore falls into the first + // branch and fails to parse there, yielding 0 rather than a number. + it("returns 0 for a string mixing a percent with a currency symbol", () => { + assert.equal(num("$1,000%"), 0); + }); + + it("reads a percent string with thousands separators off by three orders of magnitude", () => { + // The `%` branch strips only "%", so the comma survives and `parseFloat` + // stops there: "1,000" reads as 1, then /100 gives 0.01. The value a user + // means by "1,000%" is 10. Pinned as current behaviour, not as correct. + assert.equal(num("1,000%"), 0.01); + }); + + it("handles a currency string with a percent-free comma correctly", () => { + assert.equal(num("$2,500.25"), 2500.25); + }); +}); + +describe("toNumber — everything unreadable becomes 0", () => { + // This is the silent path: a text cell inside a SUM range contributes 0 + // instead of raising, so the total is quietly short. + it("returns 0 for non-numeric text", () => { + assert.equal(num("hello"), 0); + assert.equal(num(""), 0); + assert.equal(num(" "), 0); + assert.equal(num("N/A"), 0); + }); + + // `CellValue` is `number | string | boolean`, so an empty cell is never + // null here — `getRawValue` in the calculator maps blanks to 0 before this + // is reached. Booleans, though, are in the type and become 0 rather than + // 1/0 as Excel would have them. + it("returns 0 for booleans, not Excel's 1 and 0", () => { + assert.equal(num(true), 0); + assert.equal(num(false), 0); + }); + + // `parseFloat` stops at the first character it cannot read rather than + // rejecting the string, so a partly-numeric cell contributes its prefix. + it("takes the leading number from a partly-numeric string", () => { + assert.equal(num("12abc"), 12); + assert.equal(num("3.5kg"), 3.5); + }); + + it("returns 0 when the number does not come first", () => { + assert.equal(num("abc12"), 0); + }); +}); + +describe("toString", () => { + it("stringifies every shape `CellValue` allows", () => { + assert.equal(toString(42), "42"); + assert.equal(toString("text"), "text"); + assert.equal(toString(true), "true"); + assert.equal(toString(false), "false"); + assert.equal(toString(""), ""); + assert.equal(toString(0), "0"); + }); +}); + +describe("parseCriteria — comparison operators", () => { + it("compares greater-than and greater-or-equal at the boundary", () => { + const greater = parseCriteria(">5"); + assert.equal(greater(5), false); + assert.equal(greater(6), true); + + const gte = parseCriteria(">=5"); + assert.equal(gte(4), false); + assert.equal(gte(5), true, "the boundary value must count for >="); + assert.equal(gte(6), true); + }); + + it("compares less-than and less-or-equal at the boundary", () => { + const less = parseCriteria("<5"); + assert.equal(less(5), false); + assert.equal(less(4), true); + + const lte = parseCriteria("<=5"); + assert.equal(lte(6), false); + assert.equal(lte(5), true, "the boundary value must count for <="); + }); + + it("accepts both spellings of equality and inequality", () => { + for (const criteria of ["=5", "==5"]) { + assert.equal(parseCriteria(criteria)(5), true, `${criteria} should match 5`); + assert.equal(parseCriteria(criteria)(6), false); + } + for (const criteria of ["!=5", "<>5"]) { + assert.equal(parseCriteria(criteria)(5), false, `${criteria} should not match 5`); + assert.equal(parseCriteria(criteria)(6), true); + } + }); + + it("strips surrounding quotes before reading the operator", () => { + assert.equal(parseCriteria('">5"')(6), true); + assert.equal(parseCriteria("'>5'")(6), true); + }); + + it("tolerates whitespace around the criteria", () => { + assert.equal(parseCriteria(" >5 ")(6), true); + }); +}); + +describe("parseCriteria — the ways it silently matches nothing", () => { + // An unrecognised operator lands in the `default` arm, which answers false + // for every value. A COUNTIF written this way reports 0 matches and reads + // like a real answer. + it("matches nothing for an operator written backwards", () => { + const backwards = parseCriteria("=>5"); + assert.equal(backwards(5), false); + assert.equal(backwards(6), false); + assert.equal(backwards(4), false); + }); + + // A non-numeric comparand makes every numeric comparison false, since + // `NaN > x` and `NaN < x` are both false. + it("matches nothing when a comparison operator is given non-numeric text", () => { + const gtText = parseCriteria(">abc"); + assert.equal(gtText(0), false); + assert.equal(gtText(1000), false); + assert.equal(gtText("abc"), false); + }); + + // `*` and `?` are Excel wildcards; `~` escapes them back to literals. + it("treats a wildcard as a pattern, and ~ escapes it", () => { + const wildcard = parseCriteria("app*"); + assert.equal(wildcard("apple"), true); + assert.equal(wildcard("app"), true, "* may match nothing"); + assert.equal(wildcard("axe"), false); + assert.equal(parseCriteria("app~*")("app*"), true, "escaped, so only the literal matches"); + assert.equal(parseCriteria("app~*")("apple"), false); + }); +}); + +describe("parseCriteria — exact match", () => { + it("matches a plain string exactly", () => { + const apple = parseCriteria("apple"); + assert.equal(apple("apple"), true); + assert.equal(apple("Apple"), true, "matching is case-insensitive, as in Excel"); + assert.equal(apple("apples"), false); + }); + + it("matches a number written either as text or as a number", () => { + const five = parseCriteria("5"); + assert.equal(five(5), true); + assert.equal(five("5"), true); + assert.equal(five("5.0"), true, "numeric equality catches a different spelling"); + assert.equal(five(6), false); + }); + + // `toNumber` turns unreadable values into 0, so a criteria of "0" matches + // every text cell in the range. Pinned because it inflates a COUNTIF + // without any sign that something went wrong. + it("matches unreadable text when the criteria is 0", () => { + const zero = parseCriteria("0"); + assert.equal(zero(0), true); + assert.equal(zero("hello"), true, "toNumber('hello') is 0, so this counts"); + }); + + it("strips quotes around an exact-match criteria too", () => { + assert.equal(parseCriteria('"apple"')("apple"), true); + }); + + it("matches an empty criteria against an empty string", () => { + assert.equal(parseCriteria("")(""), true); + assert.equal(parseCriteria("")("x"), false); + }); +}); diff --git a/tests/engine/test_requiredArg.ts b/tests/engine/test_requiredArg.ts new file mode 100644 index 0000000..1cdce48 --- /dev/null +++ b/tests/engine/test_requiredArg.ts @@ -0,0 +1,102 @@ +// #2736: adopting `noUncheckedIndexedAccess` turned every `args[N]` in a +// function handler into `string | undefined`. `requiredArg` is the single reader +// that resolves it — and the reason it can be a *reader* rather than a default +// is that the evaluator validates the registry's `minArgs` before any handler +// runs. These tests pin both halves of that contract. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { functionRegistry, requiredArg, tooFewArgumentsError, type FunctionContext } from "../../src/engine/registry.ts"; +import "../../src/engine/functions/index.ts"; +import { substituteCellRefs, findCellRefs } from "../../src/engine/evaluator.ts"; + +const stubContext = (functionName: string): FunctionContext => ({ + functionName, + getCellValue: () => 1, + getRangeValues: () => [1, 2, 3], + getRangeValuesRaw: () => [1, 2, 3], + evaluateFormula: () => 1, +}); + +/** The wording the evaluator uses for an arity violation, which `requiredArg` + * must reproduce rather than invent a second message for. */ +const TOO_FEW_MESSAGE = /^[A-Z]+ requires at least \d+ arguments?$/; + +describe("requiredArg", () => { + it("returns the argument at the index", () => { + assert.equal(requiredArg(stubContext("SUM"), ["A1", "B1"], 1), "B1"); + }); + + it("throws the evaluator's own arity wording, naming the function and the count it needed", () => { + assert.throws(() => requiredArg(stubContext("ROUND"), ["A1"], 1), { message: "ROUND requires at least 2 arguments" }); + }); + + it("uses the singular for a one-argument minimum, like the evaluator does", () => { + assert.equal(tooFewArgumentsError("UPPER", 1).message, "UPPER requires at least 1 argument"); + assert.throws(() => requiredArg(stubContext("UPPER"), [], 0), { message: "UPPER requires at least 1 argument" }); + }); + + it("never substitutes a default — an absent argument is an error, not a 0 or an empty string", () => { + assert.throws(() => requiredArg(stubContext("MID"), ["A1", "1"], 2), { message: TOO_FEW_MESSAGE }); + }); +}); + +describe("every registered function's minArgs covers what its handler reads", () => { + // The audit behind #2736's spreadsheet pass, kept as a permanent guard: a + // registration whose minArgs is LOWER than the highest index its handler reads + // unguarded is a crash path reachable from a user formula, because the + // evaluator's arity gate would let that call through. Calling each handler + // with exactly minArgs arguments makes `requiredArg` the detector. + functionRegistry.getAllFunctions().forEach((definition) => { + it(`${definition.name} reads no argument beyond its declared minimum`, () => { + const minArgs = definition.minArgs ?? 0; + const args = Array.from({ length: minArgs }, () => "A1"); + try { + definition.handler(args, stubContext(definition.name)); + } catch (error) { + // A handler may fail for unrelated reasons (a stub range is not a real + // table); only the arity wording means minArgs under-declares. + const message = error instanceof Error ? error.message : String(error); + assert.doesNotMatch(message, TOO_FEW_MESSAGE, `${definition.name} read past its declared minArgs=${minArgs}`); + } + }); + }); +}); + +describe("substituteCellRefs", () => { + // #2357: a global string replace rewrote every occurrence of the SHORTER + // reference first, so A10 became "0" and the cell showed a + // plausible wrong number. Substituting back to front is what prevents it. + it("substitutes back to front, so a longer reference is not broken by a shorter prefix", () => { + const expr = "A1+A10"; + const result = substituteCellRefs(expr, findCellRefs(expr), (ref) => (ref === "A1" ? "5" : "7")); + assert.equal(result, "5+7"); + }); + + it("leaves the caller's span list untouched", () => { + const spans = findCellRefs("A1+B2"); + substituteCellRefs("A1+B2", spans, () => "0"); + assert.deepEqual( + spans.map((span) => span.ref), + ["A1", "B2"], + ); + }); + + it("returns the expression unchanged when there is nothing to substitute", () => { + assert.equal( + substituteCellRefs("1+2", [], () => "9"), + "1+2", + ); + }); + + it("propagates a throw from the renderer, so an errored reference still poisons the expression", () => { + const expr = "A1+1"; + assert.throws( + () => + substituteCellRefs(expr, findCellRefs(expr), () => { + throw new Error("#DIV/0!"); + }), + { message: "#DIV/0!" }, + ); + }); +}); diff --git a/tests/engine/test_responseDecoder.ts b/tests/engine/test_responseDecoder.ts new file mode 100644 index 0000000..4d46122 --- /dev/null +++ b/tests/engine/test_responseDecoder.ts @@ -0,0 +1,87 @@ +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { decodeSpreadsheetResponse } from "../../src/engine/responseDecoder.js"; + +describe("decodeSpreadsheetResponse", () => { + it("ok when kind is text with valid JSON array content", () => { + const sheets = [{ name: "Sheet1", data: [] }]; + const result = decodeSpreadsheetResponse({ + kind: "text", + content: JSON.stringify(sheets), + }); + assert.equal(result.kind, "ok"); + if (result.kind !== "ok") return; + assert.deepEqual(result.sheets, sheets); + }); + + it("ok when kind is missing (legacy response)", () => { + const sheets = [{ name: "A", data: [[{ v: 1 }]] }]; + const result = decodeSpreadsheetResponse({ + content: JSON.stringify(sheets), + }); + assert.equal(result.kind, "ok"); + }); + + it("error when kind is too-large", () => { + const result = decodeSpreadsheetResponse({ + kind: "too-large", + message: "File exceeds size limit", + }); + assert.equal(result.kind, "error"); + if (result.kind !== "error") return; + assert.equal(result.message, "File exceeds size limit"); + }); + + it("error when kind is binary", () => { + const result = decodeSpreadsheetResponse({ kind: "binary" }); + assert.equal(result.kind, "error"); + if (result.kind !== "error") return; + assert.match(result.message, /binary/); + }); + + it("error when content is missing", () => { + const result = decodeSpreadsheetResponse({ kind: "text" }); + assert.equal(result.kind, "error"); + if (result.kind !== "error") return; + assert.match(result.message, /no content/i); + }); + + it("error when content is not a string", () => { + // unsafely cast to exercise the runtime branch + const result = decodeSpreadsheetResponse({ + kind: "text", + content: 123 as unknown as string, + }); + assert.equal(result.kind, "error"); + }); + + it("error when JSON is malformed", () => { + const result = decodeSpreadsheetResponse({ + kind: "text", + content: "{not valid json", + }); + assert.equal(result.kind, "error"); + if (result.kind !== "error") return; + assert.match(result.message, /malformed/i); + }); + + it("error when content is valid JSON but not an array", () => { + const result = decodeSpreadsheetResponse({ + kind: "text", + content: '{"name": "Sheet1"}', + }); + assert.equal(result.kind, "error"); + if (result.kind !== "error") return; + assert.match(result.message, /not an array/i); + }); + + it("empty array is ok (new spreadsheet)", () => { + const result = decodeSpreadsheetResponse({ + kind: "text", + content: "[]", + }); + assert.equal(result.kind, "ok"); + if (result.kind !== "ok") return; + assert.deepEqual(result.sheets, []); + }); +}); diff --git a/tests/engine/test_serialFromParts.ts b/tests/engine/test_serialFromParts.ts new file mode 100644 index 0000000..db5764d --- /dev/null +++ b/tests/engine/test_serialFromParts.ts @@ -0,0 +1,48 @@ +// serialFromParts folds the validate -> Date.UTC -> dateToSerial tail that every +// dated branch of parseDate repeated (#2482). parseDate now delegates to it, so +// pinning the rule directly here is the only place a wrong month offset or a +// dropped validity check is caught independently of parseDate itself. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { serialFromParts } from "../../src/engine/date-parser.ts"; + +describe("serialFromParts — valid triples map to the Excel serial", () => { + it("March 4, 2025 is serial 45720", () => { + assert.equal(serialFromParts(2025, 3, 4), 45720); + }); + + it("Jan 1, 1900 is serial 2 (Excel's Dec-30-1899 base with the 1900 leap-year quirk)", () => { + assert.equal(serialFromParts(1900, 1, 1), 2); + }); + + it("accepts a real leap day (Feb 29, 2024)", () => { + assert.equal(serialFromParts(2024, 2, 29), 45351); + }); + + it("accepts the upper year boundary (Dec 31, 2100)", () => { + assert.equal(serialFromParts(2100, 12, 31), 73415); + }); +}); + +describe("serialFromParts — invalid triples are null", () => { + it("rejects Feb 29 on a non-leap year", () => { + assert.equal(serialFromParts(2025, 2, 29), null); + }); + + it("rejects a day past the month length (Feb 30)", () => { + assert.equal(serialFromParts(2025, 2, 30), null); + }); + + it("rejects a month above 12", () => { + assert.equal(serialFromParts(2025, 13, 1), null); + }); + + it("rejects a year below the 1900 floor", () => { + assert.equal(serialFromParts(1800, 1, 1), null); + }); + + it("rejects a zero day", () => { + assert.equal(serialFromParts(2025, 1, 0), null); + }); +}); diff --git a/tests/engine/test_spreadsheetErrors.ts b/tests/engine/test_spreadsheetErrors.ts new file mode 100644 index 0000000..1a71958 --- /dev/null +++ b/tests/engine/test_spreadsheetErrors.ts @@ -0,0 +1,71 @@ +// The error taxonomy: which CODES the engine knows, and which values count as +// an error RESULT. Since #2451 an error is its own value, so `isErrorResult` +// deliberately rejects a look-alike string — that is what lets IFERROR tell +// SQRT(-1) apart from CONCAT("#N","UM!"). The provenance behaviour itself is +// covered in test_errorValue.ts. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { + isSpreadsheetError, + isErrorResult, + errorCodeOf, + spreadsheetError, + SPREADSHEET_ERRORS, +} from "../../src/engine/spreadsheet-errors.ts"; + +describe("isSpreadsheetError", () => { + it("recognises the Excel error strings", () => { + assert.equal(isSpreadsheetError("#NUM!"), true); + assert.equal(isSpreadsheetError("#DIV/0!"), true); + assert.equal(isSpreadsheetError("#N/A"), true); + }); + + it("recognises every code in the taxonomy, including the engine's own #ERROR!", () => { + assert.deepEqual( + SPREADSHEET_ERRORS.filter((code) => !isSpreadsheetError(code)), + [], + ); + }); + + it("rejects ordinary text and non-strings", () => { + assert.equal(isSpreadsheetError("hello"), false); + assert.equal(isSpreadsheetError("#NOPE!"), false); + assert.equal(isSpreadsheetError(0), false); + assert.equal(isSpreadsheetError(null), false); + }); +}); + +describe("errorCodeOf", () => { + it("reads the code off an error value and off a string that spells one", () => { + assert.equal(errorCodeOf(spreadsheetError("#REF!")), "#REF!"); + assert.equal(errorCodeOf("#REF!"), "#REF!"); + }); + + it("is null for anything else", () => { + assert.equal(errorCodeOf("#OOPS!"), null); + assert.equal(errorCodeOf(7), null); + assert.equal(errorCodeOf(undefined), null); + }); +}); + +describe("isErrorResult", () => { + it("treats error values, NaN/∞ and missing values as errors", () => { + assert.equal(isErrorResult(spreadsheetError("#DIV/0!")), true); + assert.equal(isErrorResult(NaN), true); + assert.equal(isErrorResult(Infinity), true); + assert.equal(isErrorResult(null), true); + assert.equal(isErrorResult(undefined), true); + }); + + it("passes ordinary values through", () => { + assert.equal(isErrorResult(0), false); + assert.equal(isErrorResult(42), false); + assert.equal(isErrorResult("text"), false); + }); + + it("does NOT treat a string that merely spells an error as one (#2451)", () => { + assert.equal(isErrorResult("#DIV/0!"), false); + assert.equal(isErrorResult("#NUM!"), false); + }); +}); diff --git a/tests/engine/test_statisticalFunctions.ts b/tests/engine/test_statisticalFunctions.ts new file mode 100644 index 0000000..c5f8a98 --- /dev/null +++ b/tests/engine/test_statisticalFunctions.ts @@ -0,0 +1,101 @@ +// STDEV / VAR through the whole engine (#2360). Excel's STDEV / VAR are the +// SAMPLE estimators (divide by n-1); the engine used to divide by n, which is +// the POPULATION estimator (Excel's STDEVP / VARP) — a silent wrong answer. +// A single value has no n-1 to divide by, so Excel reports #DIV/0!. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { SpreadsheetEngine } from "../../src/engine/index.ts"; +import { cellAt } from "./cellAccess.ts"; + +/** Calculate `formula` in the cell just below a single column of `values`. */ +const evalOverColumn = (values: (string | number)[], formula: string): unknown => { + const data: { v: string | number }[][] = values.map((value) => [{ v: value }]); + data.push([{ v: formula }]); + const result = new SpreadsheetEngine().calculate({ name: "S", data }); + return cellAt(result.data, data.length - 1, 0); +}; + +// {2,4,4,4,5,5,7,9}: mean 5, Σ(x-μ)² = 32. +// Sample: 32 / (8-1) = 4.5714… → stdev 2.1380… +// Population: 32 / 8 = 4 → stdev 2.0 (the old wrong answer). +const SAMPLE_VALUES = [2, 4, 4, 4, 5, 5, 7, 9]; + +describe("STDEV — sample estimator (#2360)", () => { + it("divides by n-1, not n", () => { + const result = evalOverColumn(SAMPLE_VALUES, "=STDEV(A1:A8)"); + assert.equal(typeof result, "number"); + assert.ok(Math.abs((result as number) - 2.138089935) < 1e-6, `expected ~2.1381 sample stdev, got ${result}`); + }); + + it("returns #DIV/0! for a single value (no n-1 to divide by)", () => { + assert.equal(evalOverColumn([42], "=STDEV(A1:A1)"), "#DIV/0!"); + }); +}); + +describe("VAR — sample estimator (#2360)", () => { + it("divides by n-1, not n", () => { + const result = evalOverColumn(SAMPLE_VALUES, "=VAR(A1:A8)"); + assert.equal(typeof result, "number"); + assert.ok(Math.abs((result as number) - 32 / 7) < 1e-9, `expected 32/7 sample variance, got ${result}`); + }); + + it("returns #DIV/0! for a single value", () => { + assert.equal(evalOverColumn([42], "=VAR(A1:A1)"), "#DIV/0!"); + }); +}); + +// A range holding no numbers is not a range of zeros. AVERAGE and MEDIAN used +// to answer 0 for it, which reads like a genuine result (#2360). MAX / MIN / +// SUM / COUNT are NOT part of this: Excel really does answer 0 there. + +const BLANKS = ["", "", ""]; +const TEXTS = ["apple", "banana", "cherry"]; + +describe("AVERAGE — no numbers to average is #DIV/0! (#2360)", () => { + it("returns #DIV/0! for an all-blank range", () => { + assert.equal(evalOverColumn(BLANKS, "=AVERAGE(A1:A3)"), "#DIV/0!"); + }); + + it("returns #DIV/0! for a text-only range (Excel ignores text)", () => { + assert.equal(evalOverColumn(TEXTS, "=AVERAGE(A1:A3)"), "#DIV/0!"); + }); + + it("still averages when at least one number is present", () => { + assert.equal(evalOverColumn(["", 10, 20], "=AVERAGE(A1:A3)"), 15); + }); + + it("is an error VALUE, so IFERROR catches it", () => { + assert.equal(evalOverColumn(BLANKS, "=IFERROR(AVERAGE(A1:A3), 99)"), 99); + }); +}); + +describe("MEDIAN — no numbers has no middle, so #NUM! (#2360)", () => { + it("returns #NUM! for an all-blank range", () => { + assert.equal(evalOverColumn(BLANKS, "=MEDIAN(A1:A3)"), "#NUM!"); + }); + + it("returns #NUM! for a text-only range", () => { + assert.equal(evalOverColumn(TEXTS, "=MEDIAN(A1:A3)"), "#NUM!"); + }); + + it("still takes the median when numbers are present", () => { + assert.equal(evalOverColumn([3, 1, 2], "=MEDIAN(A1:A3)"), 2); + assert.equal(evalOverColumn(["", 1, 3], "=MEDIAN(A1:A3)"), 2, "blanks are ignored, not averaged in as 0"); + }); + + it("is an error VALUE, so IFERROR catches it", () => { + assert.equal(evalOverColumn(BLANKS, "=IFERROR(MEDIAN(A1:A3), 99)"), 99); + }); +}); + +// Excel's own boundary for these four is 0, so the engine's 0 is correct and +// must NOT be "fixed" into an error. +describe("MAX / MIN / SUM / COUNT over an empty range stay 0 (Excel agrees)", () => { + it("answers 0 rather than an error", () => { + assert.equal(evalOverColumn(BLANKS, "=MAX(A1:A3)"), 0); + assert.equal(evalOverColumn(BLANKS, "=MIN(A1:A3)"), 0); + assert.equal(evalOverColumn(BLANKS, "=SUM(A1:A3)"), 0); + assert.equal(evalOverColumn(BLANKS, "=COUNT(A1:A3)"), 0); + }); +}); diff --git a/tests/engine/test_statisticalMath.ts b/tests/engine/test_statisticalMath.ts new file mode 100644 index 0000000..0d3fbc6 --- /dev/null +++ b/tests/engine/test_statisticalMath.ts @@ -0,0 +1,130 @@ +// MODE's rule in isolation (#2391): the most frequent value, or #N/A when nothing +// repeats. The old handler returned the FIRST value for an all-distinct set, a +// silent wrong answer that reads like a real mode. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { + computeAverage, + computeMedian, + computeMode, + sampleVariance, + sampleStdev, +} from "../../src/engine/functions/statistical-math.ts"; +import { NA_ERROR, DIV_ZERO_ERROR, NUM_ERROR } from "../../src/engine/spreadsheet-errors.ts"; + +// AVERAGE and MEDIAN of nothing used to be 0 — a number a reader cannot tell +// from a real average of zeros (#2360). Excel reports the absence, and with +// DIFFERENT codes: AVERAGE divides by a count of zero (#DIV/0!) while MEDIAN +// has no division at all, just no middle element (#NUM!). + +describe("computeAverage", () => { + it("is the arithmetic mean", () => { + assert.equal(computeAverage([10, 20, 30]), 20); + assert.equal(computeAverage([1, 2]), 1.5); + }); + + it("averages a single value to itself", () => { + assert.equal(computeAverage([7]), 7); + }); + + it("keeps stored zeros in the denominator", () => { + assert.equal(computeAverage([0, 0, 3]), 1); + }); + + it("returns #DIV/0! for an empty set", () => { + assert.equal(computeAverage([]), DIV_ZERO_ERROR); + }); +}); + +describe("computeMedian", () => { + it("returns the middle value of an odd-sized set", () => { + assert.equal(computeMedian([3, 1, 2]), 2); + }); + + it("averages the two middle values of an even-sized set", () => { + assert.equal(computeMedian([4, 1, 3, 2]), 2.5); + }); + + it("returns the value itself for a single-value set", () => { + assert.equal(computeMedian([7]), 7); + }); + + it("sorts numerically, not lexicographically", () => { + assert.equal(computeMedian([10, 9, 100]), 10); + }); + + it("does not reorder the caller's array", () => { + const values = [3, 1, 2]; + computeMedian(values); + assert.deepEqual(values, [3, 1, 2]); + }); + + it("returns #NUM! for an empty set", () => { + assert.equal(computeMedian([]), NUM_ERROR); + }); +}); + +describe("computeMode", () => { + it("returns the single most frequent value", () => { + assert.equal(computeMode([1, 2, 2, 3]), 2); + assert.equal(computeMode([5, 5, 5, 1, 2]), 5); + }); + + // The fix: no value repeats, so the mode is undefined → #N/A, not values[0]. + it("returns #N/A when every value is distinct", () => { + assert.equal(computeMode([1, 2, 3]), NA_ERROR); + assert.equal(computeMode([9]), NA_ERROR); + }); + + it("returns #N/A for an empty set", () => { + assert.equal(computeMode([]), NA_ERROR); + }); + + // On a tie the earliest-appearing value wins (Map preserves insertion order). + it("breaks a frequency tie toward the first-appearing value", () => { + assert.equal(computeMode([2, 2, 1, 1]), 2); + assert.equal(computeMode([1, 1, 2, 2]), 1); + }); + + it("counts repeated negatives and zeros", () => { + assert.equal(computeMode([0, 0, 1]), 0); + assert.equal(computeMode([-3, -3, 4]), -3); + }); +}); + +// STDEV / VAR are Excel's SAMPLE estimators: divide by n-1, not n. The old +// handler divided by n (the POPULATION estimator, Excel's STDEVP / VARP), a +// silent understatement of the spread (#2360). + +describe("sampleVariance", () => { + // {2,4,4,4,5,5,7,9}: mean 5, Σ(x-μ)² = 32. Sample 32/7 vs population 32/8 = 4. + it("divides the summed squared deviations by n-1", () => { + assert.equal(sampleVariance([2, 4, 4, 4, 5, 5, 7, 9]), 32 / 7); + }); + + it("gives 0.5 for two values one apart, not the population 0.25", () => { + assert.equal(sampleVariance([1, 2]), 0.5); + }); + + it("returns #DIV/0! for a single value (no n-1)", () => { + assert.equal(sampleVariance([5]), DIV_ZERO_ERROR); + }); + + it("returns #DIV/0! for an empty set", () => { + assert.equal(sampleVariance([]), DIV_ZERO_ERROR); + }); +}); + +describe("sampleStdev", () => { + it("is the square root of the sample variance", () => { + const result = sampleStdev([2, 4, 4, 4, 5, 5, 7, 9]); + assert.equal(typeof result, "number"); + assert.ok(Math.abs((result as number) - Math.sqrt(32 / 7)) < 1e-12); + }); + + it("propagates #DIV/0! from the fewer-than-two boundary", () => { + assert.equal(sampleStdev([5]), DIV_ZERO_ERROR); + assert.equal(sampleStdev([]), DIV_ZERO_ERROR); + }); +}); diff --git a/tests/engine/test_textFormat.ts b/tests/engine/test_textFormat.ts new file mode 100644 index 0000000..6221961 --- /dev/null +++ b/tests/engine/test_textFormat.ts @@ -0,0 +1,254 @@ +// Excel number-format codes as read by the TEXT function. +// +// TEXT used to ignore the pattern's shape and hard-code its own: `#,##0` +// grouping was dropped entirely, and the `$` / `%` branches always produced two +// decimals — so `TEXT(1234.5,"$#,##0.00")` returned "$1234.50", `TEXT(0.5,"0%")` +// returned "50.00%" and `TEXT(5,"$0")` returned "$5.00". Every one of those is a +// plausible-looking string, which is why they survived (#2360). +// +// The pattern interpreter is tested directly; the end-to-end path through +// SpreadsheetEngine confirms the TEXT handler wires it up. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { formatWithPattern, parseNumberPattern } from "../../src/engine/textFormat.ts"; +import { groupThousands } from "../../src/engine/formatter.ts"; +import { SpreadsheetEngine } from "../../src/engine/index.ts"; +import { cellAt } from "./cellAccess.ts"; + +const engine = new SpreadsheetEngine(); +const evalFormula = (formula: string): unknown => cellAt(engine.calculate(engine.createSheet("S", [[`=${formula}`]])).data, 0, 0); + +describe("formatWithPattern — the cases reported in #2360", () => { + it("groups digits for a currency pattern", () => { + assert.equal(formatWithPattern(1234.5, "$#,##0.00"), "$1,234.50"); + }); + + it("takes the percent decimals from the pattern instead of forcing two", () => { + assert.equal(formatWithPattern(0.5, "0%"), "50%"); + }); + + it("takes the currency decimals from the pattern instead of forcing two", () => { + assert.equal(formatWithPattern(5, "$0"), "$5"); + }); +}); + +describe("formatWithPattern — grouping boundaries", () => { + it("leaves a three-digit number ungrouped", () => { + assert.equal(formatWithPattern(999, "#,##0"), "999"); + }); + + it("groups from four digits up", () => { + assert.equal(formatWithPattern(1000, "#,##0"), "1,000"); + }); + + it("groups every third digit of a large number", () => { + assert.equal(formatWithPattern(1234567, "#,##0"), "1,234,567"); + }); + + it("groups the carried digit when rounding crosses a boundary", () => { + assert.equal(formatWithPattern(999.5, "#,##0"), "1,000"); + }); + + it("omits grouping when the pattern has no comma", () => { + assert.equal(formatWithPattern(1234567, "0"), "1234567"); + }); +}); + +describe("formatWithPattern — negatives and zero", () => { + it("puts the sign in front of the whole rendering, currency included", () => { + assert.equal(formatWithPattern(-1234.5, "$#,##0.00"), "-$1,234.50"); + }); + + it("keeps the sign on a grouped plain number", () => { + assert.equal(formatWithPattern(-1234.5, "#,##0.00"), "-1,234.50"); + }); + + it("renders zero with the pattern's decimals", () => { + assert.equal(formatWithPattern(0, "$#,##0.00"), "$0.00"); + }); + + // The sign follows the ROUNDED digits: a value that rounds away to zero must + // not render as "-0.00", which reads as a negative amount that isn't there. + it("drops the sign when rounding leaves nothing but zeros", () => { + assert.equal(formatWithPattern(-0.001, "0.00"), "0.00"); + }); + + it("keeps the sign when rounding leaves a non-zero digit", () => { + assert.equal(formatWithPattern(-0.006, "0.00"), "-0.01"); + }); +}); + +describe("formatWithPattern — decimals", () => { + it("rounds to the pattern's fixed decimals", () => { + assert.equal(formatWithPattern(1234.5678, "0.00"), "1234.57"); + }); + + it("pads to the pattern's fixed decimals", () => { + assert.equal(formatWithPattern(2, "0.000"), "2.000"); + }); + + it("rounds to a whole number when the pattern has no decimal point", () => { + assert.equal(formatWithPattern(1234.5, "0"), "1235"); + }); + + it("drops an optional trailing decimal written as #", () => { + assert.equal(formatWithPattern(0.5, "0.0#"), "0.5"); + }); + + it("keeps an optional decimal that is not a trailing zero", () => { + assert.equal(formatWithPattern(0.25, "0.0#"), "0.25"); + }); + + it("drops every decimal when all of them are optional", () => { + assert.equal(formatWithPattern(2, "0.##"), "2"); + }); + + it("pads the whole part to the pattern's leading zeros", () => { + assert.equal(formatWithPattern(5, "000"), "005"); + }); +}); + +describe("formatWithPattern — percent", () => { + it("scales by 100 and keeps the pattern's two decimals", () => { + assert.equal(formatWithPattern(0.1234, "0.00%"), "12.34%"); + }); + + it("scales a value above one", () => { + assert.equal(formatWithPattern(1, "0%"), "100%"); + }); + + it("groups a large percentage", () => { + assert.equal(formatWithPattern(12.3456, "#,##0.0%"), "1,234.6%"); + }); + + it("keeps the sign of a negative percentage", () => { + assert.equal(formatWithPattern(-0.5, "0%"), "-50%"); + }); +}); + +describe("formatWithPattern — literals", () => { + it("keeps a leading literal", () => { + assert.equal(formatWithPattern(1234.5, "USD #,##0.00"), "USD 1,234.50"); + }); + + it("keeps a trailing literal", () => { + assert.equal(formatWithPattern(1234.5, "#,##0.0 kg"), "1,234.5 kg"); + }); +}); + +describe("formatWithPattern — formats it deliberately does not render", () => { + it("declines a pattern with no digit placeholder", () => { + assert.equal(formatWithPattern(1234.5, "MM/DD/YYYY"), null); + }); + + it("declines an empty pattern", () => { + assert.equal(formatWithPattern(1234.5, ""), null); + }); + + it("declines scientific notation", () => { + assert.equal(formatWithPattern(1234.5, "0.00E+00"), null); + }); + + it("declines a multi-section pattern, whose negative/zero sections it cannot honour", () => { + assert.equal(formatWithPattern(-1, "0.00;(0.00)"), null); + }); + + it("declines a quoted literal", () => { + assert.equal(formatWithPattern(1, '0" units"'), null); + }); + + it("declines a fraction pattern", () => { + assert.equal(formatWithPattern(1.5, "# ?/?"), null); + }); + + // A comma AFTER the last placeholder means "scale by a thousand" in Excel; + // rendering it as ordinary grouping would be off by 1000. + it("declines the thousands-scaling trailing comma", () => { + assert.equal(formatWithPattern(1234567, "#,##0,"), null); + }); + + it("declines a non-finite value", () => { + assert.equal(formatWithPattern(NaN, "0.00"), null); + assert.equal(formatWithPattern(Infinity, "0.00"), null); + }); +}); + +describe("parseNumberPattern", () => { + it("splits a currency pattern into literal, grouping and decimals", () => { + assert.deepEqual(parseNumberPattern("$#,##0.00"), { + prefix: "$", + suffix: "", + useGrouping: true, + integerMinDigits: 1, + minDecimals: 2, + maxDecimals: 2, + isPercent: false, + }); + }); + + it("marks a trailing percent sign and counts optional decimals separately", () => { + assert.deepEqual(parseNumberPattern("0.0#%"), { + prefix: "", + suffix: "%", + useGrouping: false, + integerMinDigits: 1, + minDecimals: 1, + maxDecimals: 2, + isPercent: true, + }); + }); + + it("returns null for a code it cannot render", () => { + assert.equal(parseNumberPattern("MMM D, YYYY"), null); + }); +}); + +describe("groupThousands", () => { + it("returns short runs unchanged", () => { + assert.equal(groupThousands(""), ""); + assert.equal(groupThousands("7"), "7"); + assert.equal(groupThousands("999"), "999"); + }); + + it("separates every third digit from the right", () => { + assert.equal(groupThousands("1000"), "1,000"); + assert.equal(groupThousands("1234567"), "1,234,567"); + assert.equal(groupThousands("100000000"), "100,000,000"); + }); +}); + +describe("TEXT — end to end through SpreadsheetEngine", () => { + it("groups digits for a currency pattern", () => { + assert.equal(evalFormula('TEXT(1234.5,"$#,##0.00")'), "$1,234.50"); + }); + + it("renders a bare percent pattern without inventing decimals", () => { + assert.equal(evalFormula('TEXT(0.5,"0%")'), "50%"); + }); + + it("renders a bare currency pattern without inventing decimals", () => { + assert.equal(evalFormula('TEXT(5,"$0")'), "$5"); + }); + + it("groups a large plain number", () => { + assert.equal(evalFormula('TEXT(1234567,"#,##0")'), "1,234,567"); + }); + + it("keeps the sign of a negative value", () => { + assert.equal(evalFormula('TEXT(-1234.5,"#,##0.00")'), "-1,234.50"); + }); + + it("formats a cell reference the same way", () => { + const sheet = engine.createSheet("S", [[{ v: 1234.5 }, { v: '=TEXT(A1,"$#,##0.00")' }]]); + assert.equal(cellAt(engine.calculate(sheet).data, 0, 1), "$1,234.50"); + }); + + it("returns the value's own text for a format code it does not render", () => { + assert.equal(evalFormula('TEXT(1234.5,"MM/DD/YYYY")'), "1234.5"); + }); + + it("passes non-numeric input through unchanged", () => { + assert.equal(evalFormula('TEXT("abc","0.00")'), "abc"); + }); +}); diff --git a/tests/engine/test_textFunctions.ts b/tests/engine/test_textFunctions.ts new file mode 100644 index 0000000..599abc5 --- /dev/null +++ b/tests/engine/test_textFunctions.ts @@ -0,0 +1,190 @@ +// Edge-case behaviour of the SUBSTITUTE / RIGHT / LEFT / PROPER text functions. +// +// Each fix targets a case that fails silently: SUBSTITUTE with an empty +// old_text inserted the replacement between every character, a non-positive +// instance was ignored instead of erroring, RIGHT/LEFT turned a negative count +// into an empty string, and PROPER only broke words on spaces — all returning a +// plausible-looking string rather than the Excel result or a #VALUE! error. +// +// The pure helpers are tested directly; the end-to-end path through +// SpreadsheetEngine confirms the handlers wire them up. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { substituteText, takeLeft, takeRight, toProperCase } from "../../src/engine/functions/text.ts"; +import { SpreadsheetEngine } from "../../src/engine/index.ts"; +import { VALUE_ERROR } from "../../src/engine/spreadsheet-errors.ts"; +import { cellAt } from "./cellAccess.ts"; + +// The engine's display pass renders an error VALUE back to its code, so the +// end-to-end assertions compare against the string a cell shows. +const VALUE_ERROR_TEXT = VALUE_ERROR.code; + +const engine = new SpreadsheetEngine(); +const evalFormula = (formula: string): unknown => cellAt(engine.calculate(engine.createSheet("S", [[`=${formula}`]])).data, 0, 0); + +describe("substituteText — empty old_text", () => { + it("returns the text unchanged instead of inserting between characters", () => { + assert.equal(substituteText("abc", "", "-"), "abc"); + }); + + it("stays unchanged even when an instance is supplied", () => { + assert.equal(substituteText("abc", "", "-", 1), "abc"); + }); + + it("stays unchanged for an empty text as well", () => { + assert.equal(substituteText("", "", "-"), ""); + }); +}); + +describe("substituteText — instance validation", () => { + it("errors when the instance is zero", () => { + assert.equal(substituteText("aa", "a", "b", 0), VALUE_ERROR); + }); + + it("errors when the instance is negative", () => { + assert.equal(substituteText("aa", "a", "b", -1), VALUE_ERROR); + }); + + it("errors when the instance is not a finite number", () => { + assert.equal(substituteText("aa", "a", "b", NaN), VALUE_ERROR); + }); + + it("truncates a fractional instance toward zero, so <1 errors", () => { + assert.equal(substituteText("aa", "a", "b", 0.9), VALUE_ERROR); + assert.equal(substituteText("aaa", "a", "b", 1.9), "baa", "1.9 truncates to the 1st occurrence"); + }); +}); + +describe("substituteText — replacing occurrences", () => { + it("replaces every occurrence when no instance is given", () => { + assert.equal(substituteText("Hello World", "World", "Earth"), "Hello Earth"); + assert.equal(substituteText("a-b-c", "-", "+"), "a+b+c"); + }); + + it("replaces only the requested 1-based occurrence", () => { + assert.equal(substituteText("a-b-c", "-", "+", 1), "a+b-c"); + assert.equal(substituteText("a-b-c", "-", "+", 2), "a-b+c"); + }); + + it("returns the text unchanged when the instance exceeds the occurrence count", () => { + assert.equal(substituteText("a-b-c", "-", "+", 3), "a-b-c"); + }); + + it("keeps matches non-overlapping like the replace-all path", () => { + // "aa" occurs once in "aaa" when scanned non-overlapping, so a 2nd instance + // does not exist and the text is returned unchanged. + assert.equal(substituteText("aaa", "aa", "b", 2), "aaa"); + assert.equal(substituteText("aaa", "aa", "b", 1), "ba"); + }); +}); + +describe("takeRight — negative count is an error", () => { + it("errors on a negative count instead of returning an empty string", () => { + assert.equal(takeRight("Hello", -1), VALUE_ERROR); + assert.equal(takeRight("Hello", -0.5), VALUE_ERROR); + }); + + it("returns an empty string for a zero count", () => { + assert.equal(takeRight("Hello", 0), ""); + }); + + it("returns the whole string when the count exceeds its length", () => { + assert.equal(takeRight("Hello", 10), "Hello"); + }); + + it("returns the rightmost characters for a normal count", () => { + assert.equal(takeRight("Hello", 2), "lo"); + }); + + it("truncates a fractional count toward zero", () => { + assert.equal(takeRight("Hello", 2.5), "lo"); + }); + + it("errors on a non-finite count", () => { + assert.equal(takeRight("Hello", NaN), VALUE_ERROR); + assert.equal(takeRight("Hello", Infinity), VALUE_ERROR); + }); + + it("returns an empty string from an empty text", () => { + assert.equal(takeRight("", 3), ""); + }); +}); + +describe("takeLeft — negative count is an error", () => { + it("errors on a negative count instead of returning an empty string", () => { + assert.equal(takeLeft("Hello", -1), VALUE_ERROR); + }); + + it("returns an empty string for a zero count", () => { + assert.equal(takeLeft("Hello", 0), ""); + }); + + it("returns the whole string when the count exceeds its length", () => { + assert.equal(takeLeft("Hello", 10), "Hello"); + }); + + it("returns the leftmost characters for a normal count", () => { + assert.equal(takeLeft("Hello", 2), "He"); + }); + + it("truncates a fractional count and errors on a non-finite one", () => { + assert.equal(takeLeft("Hello", 2.9), "He"); + assert.equal(takeLeft("Hello", NaN), VALUE_ERROR); + }); +}); + +describe("toProperCase — word boundaries include punctuation", () => { + it("capitalises after an apostrophe and a hyphen, not only spaces", () => { + assert.equal(toProperCase("o'neil-jr"), "O'Neil-Jr"); + }); + + it("capitalises the first letter of each space-separated word", () => { + assert.equal(toProperCase("hello world"), "Hello World"); + }); + + it("lowercases the remaining letters of an all-caps word", () => { + assert.equal(toProperCase("HELLO"), "Hello"); + }); + + it("treats a digit as a non-letter boundary", () => { + assert.equal(toProperCase("abc2def"), "Abc2Def"); + }); + + it("returns an empty string for empty input", () => { + assert.equal(toProperCase(""), ""); + }); + + // A decomposed accented letter is a base letter + a combining mark; the mark + // must not read as a word boundary and capitalize the next letter (#2388). + it("keeps a decomposed accented letter as one word", () => { + assert.equal(toProperCase("e\u0301clair"), "E\u0301clair", "NFD: e + combining acute"); + assert.equal(toProperCase("\u00e9clair"), "\u00c9clair", "NFC composed form"); + }); + + it("leaves leading punctuation in place and capitalises the first letter", () => { + assert.equal(toProperCase("'hello"), "'Hello"); + }); +}); + +describe("SpreadsheetEngine — text edge cases end to end", () => { + it("evaluates SUBSTITUTE with empty old_text and bad instance", () => { + assert.equal(evalFormula('SUBSTITUTE("abc","","-")'), "abc"); + assert.equal(evalFormula('SUBSTITUTE("aa","a","b",0)'), VALUE_ERROR_TEXT); + }); + + it("evaluates RIGHT / LEFT with a negative count as an error", () => { + assert.equal(evalFormula('RIGHT("Hello",-1)'), VALUE_ERROR_TEXT); + assert.equal(evalFormula('LEFT("Hello",-1)'), VALUE_ERROR_TEXT); + }); + + it("errors on a non-numeric count and truncates a fractional one", () => { + assert.equal(evalFormula('LEFT("Hello","x")'), VALUE_ERROR_TEXT); + assert.equal(evalFormula('RIGHT("Hello","x")'), VALUE_ERROR_TEXT); + assert.equal(evalFormula('RIGHT("Hello",2.5)'), "lo"); + }); + + it("evaluates PROPER across punctuation boundaries", () => { + assert.equal(evalFormula('PROPER("o\'neil-jr")'), "O'Neil-Jr"); + }); +}); diff --git a/tests/engine/test_translateFormula.ts b/tests/engine/test_translateFormula.ts new file mode 100644 index 0000000..531a794 --- /dev/null +++ b/tests/engine/test_translateFormula.ts @@ -0,0 +1,200 @@ +// Excel-formula → JS-expression translation. These are the pure string +// transforms the evaluator runs on an already-substituted expression before +// handing it to `new Function`. A mistranslated operator can still produce a +// plausible wrong NUMBER (`=2^3^2` = 512, not 64 — the associativity gap tracked +// by the sibling issues). What #2359 fixed: a formula that reaches `new Function` +// as invalid JS (`=5<>6`) no longer comes back as its raw text — it surfaces as +// an #ERROR!. This suite pins both the remaining known-wrong number cases and the +// post-#2359 error surfacing. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { + caretToPow, + replaceConcatOperator, + rewriteComparisonEq, + isSafeArithmetic, + isSafeComparison, + SpreadsheetEngine, + type SheetData, +} from "../../src/engine/index.ts"; +import { cellAt } from "./cellAccess.ts"; + +describe("caretToPow", () => { + it("rewrites a single caret to the JS exponentiation operator", () => { + assert.equal(caretToPow("2^3"), "2**3"); + }); + + it("rewrites every caret in the expression", () => { + assert.equal(caretToPow("2^3+4^5"), "2**3+4**5"); + assert.equal(caretToPow("A^B^C"), "A**B**C"); + }); + + // Excel's `^` is LEFT-associative (`2^3^2` = `(2^3)^2` = 64); JS `**` is + // RIGHT-associative (`2**3**2` = `2**(3**2)` = 512). This transform is a plain + // substitution and does NOT bridge that difference — a chained caret keeps + // producing the JS answer. Pinned as a known limitation (#2359). + it("does not correct the left-vs-right associativity of chained carets", () => { + assert.equal(caretToPow("2^3^2"), "2**3**2"); + assert.equal(new Function("return (2**3**2)")(), 512, "JS is right-associative, so 512 not 64"); + }); + + it("leaves an expression with no caret unchanged", () => { + assert.equal(caretToPow("2+3*4"), "2+3*4"); + assert.equal(caretToPow(""), ""); + }); +}); + +describe("replaceConcatOperator", () => { + it("rewrites a concatenation ampersand to +", () => { + assert.equal(replaceConcatOperator('"a"&"b"'), '"a"+"b"'); + assert.equal(replaceConcatOperator("5&6"), "5+6"); + }); + + it("rewrites every ampersand outside string literals", () => { + assert.equal(replaceConcatOperator('"a"&"b"&"c"'), '"a"+"b"+"c"'); + }); + + // An ampersand INSIDE a literal is text, not an operator — flipping it would + // corrupt the string content. + it("preserves an ampersand inside a double-quoted literal", () => { + assert.equal(replaceConcatOperator('"a&b"&"c"'), '"a&b"+"c"'); + }); + + it("preserves an ampersand inside a single-quoted literal", () => { + assert.equal(replaceConcatOperator("'a&b'&'c'"), "'a&b'+'c'"); + }); + + // A backslash-escaped quote does not end the literal, so an ampersand after it + // is still inside the string and must be preserved. + it("honours a backslash-escaped quote when tracking literal boundaries", () => { + assert.equal(replaceConcatOperator('"a\\"&b"&"c"'), '"a\\"&b"+"c"'); + }); + + it("leaves an expression with no ampersand unchanged", () => { + assert.equal(replaceConcatOperator("2+3"), "2+3"); + assert.equal(replaceConcatOperator('"plain"'), '"plain"'); + assert.equal(replaceConcatOperator(""), ""); + }); + + it("rewrites a leading or trailing ampersand outside a literal", () => { + assert.equal(replaceConcatOperator('&"x"'), '+"x"'); + assert.equal(replaceConcatOperator('"x"&'), '"x"+'); + }); +}); + +describe("rewriteComparisonEq", () => { + it("rewrites a single equality to the JS == operator", () => { + assert.equal(rewriteComparisonEq("5=5"), "5==5"); + assert.equal(rewriteComparisonEq("5=6"), "5==6"); + }); + + it("does not disturb <=, >= or != which are already valid JS", () => { + assert.equal(rewriteComparisonEq("5<=6"), "5<=6"); + assert.equal(rewriteComparisonEq("5>=6"), "5>=6"); + assert.equal(rewriteComparisonEq("5!=6"), "5!=6"); + }); + + // The regex matches the SECOND `=` of a `==` (its left neighbour, the first + // `=`, is not one of `<>!`), so an already-doubled `==` becomes `===`. Excel + // never emits `==`, so this only bites a hand-typed oddity — pinned as a known + // quirk (#2359) rather than relied upon. + it("turns an already-doubled == into === (known quirk)", () => { + assert.equal(rewriteComparisonEq("5==6"), "5===6"); + }); + + // The match consumes both flanking characters, so replacements cannot overlap: + // in `5=6=7` only the first `=` is rewritten. Pinned as a known limitation + // (#2359) — the non-overlapping replacement is a real bug to be fixed later. + it("rewrites only the first of two adjacent equalities (non-overlapping)", () => { + assert.equal(rewriteComparisonEq("5=6=7"), "5==6=7"); + }); + + // A `=` with nothing on one side has no flanking character to match, so it is + // left as a lone `=`. Pinned known limitation (#2359). + it("does not rewrite an = at the start or end of the expression", () => { + assert.equal(rewriteComparisonEq("=5"), "=5"); + assert.equal(rewriteComparisonEq("5="), "5="); + }); + + it("leaves an expression with no equals unchanged", () => { + assert.equal(rewriteComparisonEq("5<6"), "5<6"); + assert.equal(rewriteComparisonEq(""), ""); + }); +}); + +describe("isSafeArithmetic", () => { + it("accepts digits, arithmetic operators, parentheses, dot and space", () => { + assert.equal(isSafeArithmetic("2+3*4"), true); + assert.equal(isSafeArithmetic("(2 + 3) / 4.5"), true); + assert.equal(isSafeArithmetic("2**3"), true, "** survives the caret rewrite and must pass"); + }); + + it("rejects letters, quotes, ampersands and comparison characters", () => { + assert.equal(isSafeArithmetic("A1+2"), false); + assert.equal(isSafeArithmetic('"a"+"b"'), false); + assert.equal(isSafeArithmetic("5&6"), false); + assert.equal(isSafeArithmetic("5<6"), false); + assert.equal(isSafeArithmetic("5=6"), false); + }); + + // The allowlist requires at least one character, so the empty string is not + // "safe" — there is nothing to evaluate. + it("rejects the empty string", () => { + assert.equal(isSafeArithmetic(""), false); + }); +}); + +describe("isSafeComparison", () => { + it("accepts the arithmetic set plus < > ! =", () => { + assert.equal(isSafeComparison("5<=6"), true); + assert.equal(isSafeComparison("5<>6"), true, "the raw Excel <> passes the gate even though JS rejects it"); + assert.equal(isSafeComparison("(2**3) >= 6"), true); + }); + + it("rejects letters, quotes and ampersands", () => { + assert.equal(isSafeComparison("A1<6"), false); + assert.equal(isSafeComparison('"a"="b"'), false); + assert.equal(isSafeComparison("5&6"), false); + }); + + it("rejects the empty string", () => { + assert.equal(isSafeComparison(""), false); + }); +}); + +// End-to-end characterization: the pure functions compose inside the evaluator +// to the exact behaviour observed before the extraction. Includes the +// intentionally-wrong cases so the eventual #2359 fix shows up as a diff here. +describe("translation through the engine (characterization)", () => { + const calc = (formula: string): unknown => cellAt(new SpreadsheetEngine().calculate({ name: "S", data: [[{ v: formula }]] } satisfies SheetData).data, 0, 0); + + it("exponentiates (and keeps the JS-associativity quirk)", () => { + assert.equal(calc("=2^3"), 8); + assert.equal(calc("=2^3^2"), 512); + }); + + it("evaluates equality and the ordering comparisons", () => { + assert.equal(calc("=5=5"), true); + assert.equal(calc("=5=6"), false); + assert.equal(calc("=5<=6"), true); + assert.equal(calc("=5>=6"), false); + }); + + // `<>` is Excel's not-equal; it is not translated, so it reaches `new Function` + // as invalid JS and throws. Post-#2359 that throw surfaces as #ERROR! rather + // than the raw formula text (translating `<>` itself is a sibling issue). + it("surfaces the untranslated <> operator as #ERROR!, not raw text", () => { + assert.equal(calc("=5<>6"), "#ERROR!"); + }); + + it("concatenates strings and mixed operands", () => { + assert.equal(calc('="a"&"b"'), "ab"); + assert.equal(calc('="x"&5'), "x5"); + }); + + it("evaluates plain arithmetic", () => { + assert.equal(calc("=2*3+1"), 7); + assert.equal(calc("=(2+3)*4"), 20); + }); +});