From 81146d1ec22cf4aec23f4f0bedb70fa0b900d1e2 Mon Sep 17 00:00:00 2001 From: tobiadefami Date: Thu, 10 Sep 2026 20:33:26 +0100 Subject: [PATCH 01/11] feat(HF-221): add LINEST regression function --- CHANGELOG.md | 4 + docs/guide/known-limitations.md | 8 + docs/guide/list-of-differences.md | 6 + src/error-message.ts | 2 + src/i18n/languages/csCZ.ts | 1 + src/i18n/languages/daDK.ts | 1 + src/i18n/languages/deDE.ts | 1 + src/i18n/languages/enGB.ts | 1 + src/i18n/languages/esES.ts | 1 + src/i18n/languages/fiFI.ts | 1 + src/i18n/languages/frFR.ts | 1 + src/i18n/languages/huHU.ts | 1 + src/i18n/languages/idID.ts | 1 + src/i18n/languages/itIT.ts | 1 + src/i18n/languages/nbNO.ts | 1 + src/i18n/languages/nlNL.ts | 1 + src/i18n/languages/plPL.ts | 1 + src/i18n/languages/ptPT.ts | 1 + src/i18n/languages/ruRU.ts | 1 + src/i18n/languages/svSE.ts | 1 + src/i18n/languages/trTR.ts | 1 + .../categories/statistical.ts | 12 ++ src/interpreter/plugin/RegressionPlugin.ts | 199 +++++++++++++++++ src/interpreter/plugin/index.ts | 1 + .../plugin/regression/LinearRegression.ts | 202 ++++++++++++++++++ 25 files changed, 451 insertions(+) create mode 100644 src/interpreter/plugin/RegressionPlugin.ts create mode 100644 src/interpreter/plugin/regression/LinearRegression.ts diff --git a/CHANGELOG.md b/CHANGELOG.md index aeea8dd511..5bad1b8ddb 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,10 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/), ## [Unreleased] +### Added + +- Added `LINEST` for simple and multiple linear regression, with optional intercept and regression statistics. The `stats` argument must be constant because result dimensions are determined before evaluation. + ### Fixed - Fixed the `AVERAGEIF` function returning a division-by-zero error when the calculated average was `0`. [#1733](https://github.com/handsontable/hyperformula/pull/1733) diff --git a/docs/guide/known-limitations.md b/docs/guide/known-limitations.md index 91b161d94a..3a6ceb1e4c 100644 --- a/docs/guide/known-limitations.md +++ b/docs/guide/known-limitations.md @@ -61,6 +61,14 @@ a circular reference. * Ordering (including mixed types, empty cells, and text collation) follows HyperFormula's own comparison rules, which honor the `caseSensitive` and `accentSensitive` configuration options. Numbers sort before text, and text before logical values. +### LINEST function + +* The `stats` argument controls whether the result has one or five rows. It must be omitted or supplied as a constant: `TRUE()`, `FALSE()`, a number, or the text `"TRUE"` or `"FALSE"`. Parentheses and numeric unary signs are supported. Cell references, named expressions, and calculated expressions such as `1=1` return `#VALUE!`, including when `LINEST` is nested inside another function. + +* Result width is determined from the predictor dimensions before evaluation. Changes to values within fixed input ranges and to a referenced `const` argument recalculate normally. Changes to dependency values do not resize the result automatically; this is the existing array-sizing limitation. + +* If an input expression returns a different predictor count than predicted, `LINEST` returns `#VALUE!`. For example, filtering predictor columns can change the required result width. + ### OFFSET function HyperFormula resolves the OFFSET function at parse time rather than during evaluation. The parser inspects the arguments and rewrites the expression into a plain cell reference or range. This keeps the dependency graph accurate but imposes several restrictions. diff --git a/docs/guide/list-of-differences.md b/docs/guide/list-of-differences.md index a109f82616..f53af063dc 100644 --- a/docs/guide/list-of-differences.md +++ b/docs/guide/list-of-differences.md @@ -34,6 +34,12 @@ See a full list of differences between HyperFormula, Microsoft Excel, and Google **Contents:** [[toc]] +## LINEST + +Unlike Excel, HyperFormula requires a constant `stats` argument because the output dimensions are determined before evaluation. For example, `=LINEST(A1:A10,B1:B10,TRUE(),D1)` returns `#VALUE!`; use `TRUE()` or `FALSE()` directly for `stats`. See [LINEST limitations](known-limitations.md#linest-function) for supported constants and input-sizing requirements. + +Numerical results can differ for nearly dependent predictors and nearly perfect fits. Coefficient standard errors are sensitive to conditioning, and the F statistic is sensitive to residuals close to machine precision. For an effectively perfect multiple regression, Excel and HyperFormula can return different large finite F values even when the coefficients agree. + ## General functionalities | Functionality | Examples | HyperFormula | Google Sheets | Microsoft Excel | diff --git a/src/error-message.ts b/src/error-message.ts index 4b80ee06fa..b431bdbc6c 100644 --- a/src/error-message.ts +++ b/src/error-message.ts @@ -7,6 +7,8 @@ * This is a class for detailed error messages across HyperFormula. */ export class ErrorMessage { + public static LinestStaticSize = 'LINEST requires input dimensions that determine a fixed result size.' + public static LinestStaticStats = 'LINEST requires a constant stats argument to determine its result size.' public static DistinctSigns = 'Distinct signs.' public static WrongArgNumber = 'Wrong number of arguments.' public static EmptyArg = 'Empty function argument.' diff --git a/src/i18n/languages/csCZ.ts b/src/i18n/languages/csCZ.ts index fea286083c..d3e0575403 100644 --- a/src/i18n/languages/csCZ.ts +++ b/src/i18n/languages/csCZ.ts @@ -143,6 +143,7 @@ const dictionary: RawTranslationPackage = { ISTEXT: 'JE.TEXT', LEFT: 'ZLEVA', LEN: 'DÉLKA', + LINEST: 'LINREGRESE', LN: 'LN', LOG10: 'LOG', LOG: 'LOGZ', diff --git a/src/i18n/languages/daDK.ts b/src/i18n/languages/daDK.ts index 9264551249..f3737cefae 100644 --- a/src/i18n/languages/daDK.ts +++ b/src/i18n/languages/daDK.ts @@ -143,6 +143,7 @@ const dictionary: RawTranslationPackage = { ISTEXT: 'ER.TEKST', LEFT: 'VENSTRE', LEN: 'LÆNGDE', + LINEST: 'LINREGR', LN: 'LN', LOG10: 'LOG10', LOG: 'LOG', diff --git a/src/i18n/languages/deDE.ts b/src/i18n/languages/deDE.ts index 4d3d39319b..0b1a180f0b 100644 --- a/src/i18n/languages/deDE.ts +++ b/src/i18n/languages/deDE.ts @@ -143,6 +143,7 @@ const dictionary: RawTranslationPackage = { ISTEXT: 'ISTTEXT', LEFT: 'LINKS', LEN: 'LÄNGE', + LINEST: 'RGP', LN: 'LN', LOG10: 'LOG10', LOG: 'LOG', diff --git a/src/i18n/languages/enGB.ts b/src/i18n/languages/enGB.ts index d271b732cd..3a7185fdad 100644 --- a/src/i18n/languages/enGB.ts +++ b/src/i18n/languages/enGB.ts @@ -144,6 +144,7 @@ const dictionary: RawTranslationPackage = { ISTEXT: 'ISTEXT', LEFT: 'LEFT', LEN: 'LEN', + LINEST: 'LINEST', LN: 'LN', LOG10: 'LOG10', LOG: 'LOG', diff --git a/src/i18n/languages/esES.ts b/src/i18n/languages/esES.ts index 9c6b817b98..df2038ff36 100644 --- a/src/i18n/languages/esES.ts +++ b/src/i18n/languages/esES.ts @@ -143,6 +143,7 @@ export const dictionary: RawTranslationPackage = { ISTEXT: 'ESTEXTO', LEFT: 'IZQUIERDA', LEN: 'LARGO', + LINEST: 'ESTIMACION.LINEAL', LN: 'LN', LOG10: 'LOG10', LOG: 'LOG', diff --git a/src/i18n/languages/fiFI.ts b/src/i18n/languages/fiFI.ts index 99ff05708f..87ab6be3cd 100644 --- a/src/i18n/languages/fiFI.ts +++ b/src/i18n/languages/fiFI.ts @@ -143,6 +143,7 @@ const dictionary: RawTranslationPackage = { ISTEXT: 'ONTEKSTI', LEFT: 'VASEN', LEN: 'PITUUS', + LINEST: 'LINREGR', LN: 'LUONNLOG', LOG10: 'LOG10', LOG: 'LOG', diff --git a/src/i18n/languages/frFR.ts b/src/i18n/languages/frFR.ts index f7b9be45f7..bc962747b3 100644 --- a/src/i18n/languages/frFR.ts +++ b/src/i18n/languages/frFR.ts @@ -143,6 +143,7 @@ const dictionary: RawTranslationPackage = { ISTEXT: 'ESTTEXTE', LEFT: 'GAUCHE', LEN: 'NBCAR', + LINEST: 'DROITEREG', LN: 'LN', LOG10: 'LOG10', LOG: 'LOG', diff --git a/src/i18n/languages/huHU.ts b/src/i18n/languages/huHU.ts index e40219553a..245709fa3b 100644 --- a/src/i18n/languages/huHU.ts +++ b/src/i18n/languages/huHU.ts @@ -143,6 +143,7 @@ const dictionary: RawTranslationPackage = { ISTEXT: 'SZÖVEG.E', LEFT: 'BAL', LEN: 'HOSSZ', + LINEST: 'LIN.ILL', LN: 'LN', LOG10: 'LOG10', LOG: 'LOG', diff --git a/src/i18n/languages/idID.ts b/src/i18n/languages/idID.ts index 719cb11db6..532ee25e31 100644 --- a/src/i18n/languages/idID.ts +++ b/src/i18n/languages/idID.ts @@ -144,6 +144,7 @@ const dictionary: RawTranslationPackage = { ISTEXT: 'ADALAH.TEKS', LEFT: 'KIRI', LEN: 'PANJANG', + LINEST: 'LINEST', LN: 'LN', LOG10: 'LOG10', LOG: 'LOG', diff --git a/src/i18n/languages/itIT.ts b/src/i18n/languages/itIT.ts index 1a494628f8..1a863ea46a 100644 --- a/src/i18n/languages/itIT.ts +++ b/src/i18n/languages/itIT.ts @@ -143,6 +143,7 @@ const dictionary: RawTranslationPackage = { ISTEXT: 'VAL.TESTO', LEFT: 'SINISTRA', LEN: 'LUNGHEZZA', + LINEST: 'REGR.LIN', LN: 'LN', LOG10: 'LOG10', LOG: 'LOG', diff --git a/src/i18n/languages/nbNO.ts b/src/i18n/languages/nbNO.ts index ff15415a98..de4a1cf667 100644 --- a/src/i18n/languages/nbNO.ts +++ b/src/i18n/languages/nbNO.ts @@ -143,6 +143,7 @@ const dictionary: RawTranslationPackage = { ISTEXT: 'ERTEKST', LEFT: 'VENSTRE', LEN: 'LENGDE', + LINEST: 'RETTLINJE', LN: 'LN', LOG10: 'LOG10', LOG: 'LOG', diff --git a/src/i18n/languages/nlNL.ts b/src/i18n/languages/nlNL.ts index c491e2f44b..a7999b8913 100644 --- a/src/i18n/languages/nlNL.ts +++ b/src/i18n/languages/nlNL.ts @@ -143,6 +143,7 @@ const dictionary: RawTranslationPackage = { ISTEXT: 'ISTEKST', LEFT: 'LINKS', LEN: 'PITUUS', + LINEST: 'LIJNSCH', LN: 'LN', LOG10: 'LOG10', LOG: 'LOG', diff --git a/src/i18n/languages/plPL.ts b/src/i18n/languages/plPL.ts index 6fa4f016b4..7548158e88 100644 --- a/src/i18n/languages/plPL.ts +++ b/src/i18n/languages/plPL.ts @@ -143,6 +143,7 @@ const dictionary: RawTranslationPackage = { ISTEXT: 'CZY.TEKST', LEFT: 'LEWY', LEN: 'DŁ', + LINEST: 'REGLINP', LN: 'LN', LOG10: 'LOG10', LOG: 'LOG', diff --git a/src/i18n/languages/ptPT.ts b/src/i18n/languages/ptPT.ts index 6c1e68c328..a74fba28b5 100644 --- a/src/i18n/languages/ptPT.ts +++ b/src/i18n/languages/ptPT.ts @@ -143,6 +143,7 @@ const dictionary: RawTranslationPackage = { ISTEXT: 'ÉTEXTO', LEFT: 'ESQUERDA', LEN: 'NÚM.CARACT', + LINEST: 'PROJ.LIN', LN: 'LN', LOG10: 'LOG10', LOG: 'LOG', diff --git a/src/i18n/languages/ruRU.ts b/src/i18n/languages/ruRU.ts index 6b35c1ce2f..e808973691 100644 --- a/src/i18n/languages/ruRU.ts +++ b/src/i18n/languages/ruRU.ts @@ -143,6 +143,7 @@ const dictionary: RawTranslationPackage = { ISTEXT: 'ЕТЕКСТ', LEFT: 'ЛЕВСИМВ', LEN: 'ДЛСТР', + LINEST: 'ЛИНЕЙН', LN: 'LN', LOG10: 'LOG10', LOG: 'LOG', diff --git a/src/i18n/languages/svSE.ts b/src/i18n/languages/svSE.ts index 5f87580087..7d869a89f1 100644 --- a/src/i18n/languages/svSE.ts +++ b/src/i18n/languages/svSE.ts @@ -143,6 +143,7 @@ const dictionary: RawTranslationPackage = { ISTEXT: 'ÄRTEXT', LEFT: 'VÄNSTER', LEN: 'LÄNGD', + LINEST: 'REGR', LN: 'LN', LOG10: 'LOG10', LOG: 'LOG', diff --git a/src/i18n/languages/trTR.ts b/src/i18n/languages/trTR.ts index 93c2176d37..991b50a31f 100644 --- a/src/i18n/languages/trTR.ts +++ b/src/i18n/languages/trTR.ts @@ -143,6 +143,7 @@ const dictionary: RawTranslationPackage = { ISTEXT: 'EMETİNSE', LEFT: 'SOL', LEN: 'UZUNLUK', + LINEST: 'DOT', LN: 'LN', LOG10: 'LOG10', LOG: 'LOG', diff --git a/src/interpreter/functionMetadata/categories/statistical.ts b/src/interpreter/functionMetadata/categories/statistical.ts index 1752c3eaf6..6a404977f5 100644 --- a/src/interpreter/functionMetadata/categories/statistical.ts +++ b/src/interpreter/functionMetadata/categories/statistical.ts @@ -304,6 +304,18 @@ export const STATISTICAL_DOCS: Record = { documentationUrl: 'https://hyperformula.handsontable.com/docs/guide/built-in-functions.html', examples: ['=LARGE(A1:A10, 1)', '=LARGE(A1:A10, 3)'], }, + LINEST: { + category: 'Statistical', + shortDescription: 'Returns linear regression coefficients and optional statistics.', + parameters: [ + {name: 'known_y', description: 'A numeric range of observed dependent values.'}, + {name: 'known_x', description: 'Optional numeric predictors. If omitted, uses sequential values starting at 1 with the shape of known_y.'}, + {name: 'const', description: 'Whether to fit an intercept. Defaults to TRUE; FALSE fits through zero.'}, + {name: 'stats', description: 'Whether to return five rows including regression statistics. Defaults to FALSE. Must be a constant; cell references and computed expressions are unsupported.'}, + ], + documentationUrl: 'https://hyperformula.handsontable.com/docs/guide/built-in-functions.html', + examples: ['=LINEST(A1:A10, B1:C10)', '=LINEST(A1:A10, B1:C10, TRUE(), TRUE())'], + }, 'LOGNORM.DIST': { category: 'Statistical', shortDescription: 'Returns density of lognormal distribution.', diff --git a/src/interpreter/plugin/RegressionPlugin.ts b/src/interpreter/plugin/RegressionPlugin.ts new file mode 100644 index 0000000000..d64c392281 --- /dev/null +++ b/src/interpreter/plugin/RegressionPlugin.ts @@ -0,0 +1,199 @@ +/** + * @license + * Copyright (c) 2025 Handsoncode. All rights reserved. + */ + +import {ArraySize} from '../../ArraySize' +import {CellError, ErrorType} from '../../Cell' +import {ErrorMessage} from '../../error-message' +import {Ast, AstNodeType, ProcedureAst} from '../../parser' +import {SimpleRangeValue} from '../../SimpleRangeValue' +import {coerceScalarToBoolean, coerceToRange} from '../ArithmeticHelper' +import {InterpreterState} from '../InterpreterState' +import {EmptyValue, getRawValue, InternalScalarValue, InterpreterValue, isExtendedNumber} from '../InterpreterValue' +import {FunctionArgumentType, FunctionPlugin, FunctionPluginTypecheck, ImplementedFunctions} from './FunctionPlugin' +import {fitLinearRegression, LinearRegressionResult} from './regression/LinearRegression' + +/** The orientation and predictor count shared by size prediction and runtime validation. */ +interface RegressionShape { + predictors: number, + orientation: 'paired' | 'columns' | 'rows', +} + +/** Classifies the original dimensions before any values are flattened. */ +function regressionShape(y: ArraySize, x?: ArraySize): RegressionShape | undefined { + if (x === undefined || (y.width === x.width && y.height === x.height)) { + return {predictors: 1, orientation: 'paired'} + } + if (y.width === 1 && y.height === x.height) { + return {predictors: x.width, orientation: 'columns'} + } + if (y.height === 1 && y.width === x.width) { + return {predictors: x.height, orientation: 'rows'} + } + return undefined +} + +/** LINEST rejects empty strings as Boolean options, including formula-generated strings. */ +function regressionBoolean(value: InternalScalarValue): boolean | CellError | undefined { + return value === '' ? undefined : coerceScalarToBoolean(value) +} + +/** Recognizes a numeric literal with optional parentheses and unary signs. */ +function isNumericConstant(ast: Ast): boolean { + if (ast.type === AstNodeType.PARENTHESIS) { + return isNumericConstant(ast.expression) + } + if (ast.type === AstNodeType.PLUS_UNARY_OP || ast.type === AstNodeType.MINUS_UNARY_OP) { + return isNumericConstant(ast.value) + } + return ast.type === AstNodeType.NUMBER +} + +/** Resolves scalar constants without evaluating dependencies during array-size prediction. */ +function staticBoolean(ast: Ast | undefined): boolean | undefined { + if (ast === undefined || ast.type === AstNodeType.EMPTY) { + return false + } + if (ast.type === AstNodeType.PARENTHESIS) { + return staticBoolean(ast.expression) + } + if (ast.type === AstNodeType.NUMBER || ast.type === AstNodeType.STRING) { + const value = regressionBoolean(ast.value) + return typeof value === 'boolean' ? value : undefined + } + if (ast.type === AstNodeType.PLUS_UNARY_OP || ast.type === AstNodeType.MINUS_UNARY_OP) { + return isNumericConstant(ast.value) ? staticBoolean(ast.value) : undefined + } + if (ast.type === AstNodeType.FUNCTION_CALL && ast.args.length === 0) { + if (ast.procedureName === 'TRUE') { + return true + } + if (ast.procedureName === 'FALSE') { + return false + } + } + return undefined +} + +/** Converts numerical failures to spreadsheet errors, including inside an array result. */ +function finiteValue(value: number): number | CellError { + return Number.isFinite(value) ? value : new CellError(ErrorType.NUM, ErrorMessage.NaN) +} + +/** Formats the five-row result while preserving statistics errors independently of coefficients. */ +function regressionOutput(fit: LinearRegressionResult, statistics: boolean, fitIntercept: boolean): SimpleRangeValue { + const result: InternalScalarValue[][] = [[...fit.coefficients].reverse().concat(fit.intercept).map(finiteValue)] + if (!statistics) { + return SimpleRangeValue.onlyValues(result) + } + const regressionSumSquares = fit.totalSumSquares - fit.residualSumSquares + const variance = fit.degreesOfFreedom === 0 ? 0 : fit.residualSumSquares / fit.degreesOfFreedom + const rSquared = fit.totalSumSquares === 0 ? 1 : regressionSumSquares / fit.totalSumSquares + const f = finiteValue((regressionSumSquares / fit.retainedPredictorCount) / variance) + const interceptError = fitIntercept ? finiteValue(fit.interceptError) : new CellError(ErrorType.NA) + result.push([...fit.standardErrors].reverse().map(finiteValue).concat(interceptError)) + result.push([finiteValue(rSquared), finiteValue(Math.sqrt(variance))]) + result.push([f, fit.degreesOfFreedom]) + result.push([finiteValue(regressionSumSquares), finiteValue(fit.residualSumSquares)]) + for (const row of result) { + while (row.length < fit.coefficients.length + 1) { + row.push(new CellError(ErrorType.NA)) + } + } + return SimpleRangeValue.onlyValues(result) +} + +/** Implements linear regression with statically predictable result dimensions. */ +export class RegressionPlugin extends FunctionPlugin implements FunctionPluginTypecheck { + public static implementedFunctions: ImplementedFunctions = { + 'LINEST': { + method: 'linest', + sizeOfResultArrayMethod: 'linestArraySize', + vectorizationForbidden: true, + enableArrayArithmeticForArguments: true, + parameters: [ + {argumentType: FunctionArgumentType.RANGE}, + {argumentType: FunctionArgumentType.ANY, defaultValue: EmptyValue}, + {argumentType: FunctionArgumentType.SCALAR, defaultValue: true, emptyAsDefault: true}, + {argumentType: FunctionArgumentType.SCALAR, defaultValue: false, emptyAsDefault: true}, + ], + }, + } + + /** + * Evaluates LINEST(known_y, [known_x], [const], [stats]). + * Returns coefficients in reverse predictor order, followed by the intercept. + * With stats enabled, adds standard errors; R-squared and residual standard error; + * F and residual degrees of freedom; regression and residual sums of squares. + * Unused statistics cells and the intercept standard error for const=FALSE are #N/A. + * + * A dynamic stats option is rejected here as well as during size prediction so that + * nested consumers cannot accidentally bypass the fixed-size contract. + */ + public linest(ast: ProcedureAst, state: InterpreterState): InterpreterValue { + const generatedX = ast.args.length < 2 || ast.args[1].type === AstNodeType.EMPTY + return this.runFunction(ast.args, state, this.metadata('LINEST'), + (knownY: SimpleRangeValue, knownX: InterpreterValue, constArg: InternalScalarValue, statsArg: InternalScalarValue) => { + const fitIntercept = regressionBoolean(constArg) + const statistics = regressionBoolean(statsArg) + if (fitIntercept instanceof CellError) { + return fitIntercept + } + if (statistics instanceof CellError) { + return statistics + } + if (fitIntercept === undefined || statistics === undefined) { + return new CellError(ErrorType.VALUE, ErrorMessage.WrongType) + } + if (staticBoolean(ast.args[3]) === undefined) { + return new CellError(ErrorType.VALUE, ErrorMessage.LinestStaticStats) + } + const x = generatedX ? undefined : coerceToRange(knownX) + const shape = regressionShape(knownY.size, x?.size) + if (shape === undefined) { + return new CellError(ErrorType.REF, ErrorMessage.ArrayDimensions) + } + if (shape.predictors + 1 > this.config.maxColumns) { + return new CellError(ErrorType.VALUE, ErrorMessage.ValueLarge) + } + if (this.linestArraySize(ast, state).width !== shape.predictors + 1) { + return new CellError(ErrorType.VALUE, ErrorMessage.LinestStaticSize) + } + const yValues = Array.from(knownY.valuesFromTopLeftCorner()) + const xValues = x === undefined ? yValues.map((_, i) => i + 1) : Array.from(x.valuesFromTopLeftCorner()) + if (!yValues.every(isExtendedNumber) || !xValues.every(isExtendedNumber)) { + return new CellError(ErrorType.VALUE, ErrorMessage.NumberRange) + } + const observations = yValues.map(getRawValue) as number[] + const numericX = xValues.map(getRawValue) as number[] + const predictors = observations.map((_, i) => { + if (shape.orientation === 'paired') { + return [numericX[i]] + } + return Array.from({length: shape.predictors}, (_, j) => numericX[shape.orientation === 'columns' ? i * shape.predictors + j : j * observations.length + i]) + }) + const fit = fitLinearRegression(predictors, observations, fitIntercept, statistics) + return regressionOutput(fit, statistics, fitIntercept) + }) + } + + /** Predicts output width from predictor geometry and height from a constant stats option. */ + public linestArraySize(ast: ProcedureAst, state: InterpreterState): ArraySize { + if (ast.args.length < 1 || ast.args.length > 4) { + return ArraySize.error() + } + const statistics = staticBoolean(ast.args[3]) + if (statistics === undefined) { + return ArraySize.error() + } + const arrayState = new InterpreterState(state.formulaAddress, true) + const y = this.arraySizeForAst(ast.args[0], arrayState) + const x = ast.args.length < 2 || ast.args[1].type === AstNodeType.EMPTY ? undefined : this.arraySizeForAst(ast.args[1], arrayState) + const shape = regressionShape(y, x) + if (shape === undefined || shape.predictors + 1 > this.config.maxColumns) { + return ArraySize.error() + } + return new ArraySize(shape.predictors + 1, statistics ? 5 : 1) + } +} diff --git a/src/interpreter/plugin/index.ts b/src/interpreter/plugin/index.ts index 2a3005375b..505776ea2b 100644 --- a/src/interpreter/plugin/index.ts +++ b/src/interpreter/plugin/index.ts @@ -51,3 +51,4 @@ export {StatisticalPlugin} from './StatisticalPlugin' export {MathPlugin} from './MathPlugin' export {ComplexPlugin} from './ComplexPlugin' export {StatisticalAggregationPlugin} from './StatisticalAggregationPlugin' +export {RegressionPlugin} from './RegressionPlugin' diff --git a/src/interpreter/plugin/regression/LinearRegression.ts b/src/interpreter/plugin/regression/LinearRegression.ts new file mode 100644 index 0000000000..d3a286db0b --- /dev/null +++ b/src/interpreter/plugin/regression/LinearRegression.ts @@ -0,0 +1,202 @@ +/** + * @license + * Copyright (c) 2025 Handsoncode. All rights reserved. + */ + +/** Numerical fit and uncertainties, before spreadsheet output formatting. */ +export interface LinearRegressionResult { + coefficients: number[], + intercept: number, + standardErrors: number[], + interceptError: number, + residualSumSquares: number, + totalSumSquares: number, + degreesOfFreedom: number, + retainedPredictorCount: number, +} + +/** A predictor column with its original position and pre-factorization norm. */ +interface RegressionColumn { + values: number[], + index: number, + originalNorm: number, +} + +/** + * Rank threshold relative to a predictor's original centered norm. + * Excel Online retains a 1e-5 perturbation of a duplicated predictor and removes 1e-6. + * Keep the threshold separate from roundoff handling for the fitted statistics. + */ +const RANK_TOLERANCE = 1e-6 + +/** Computes a Euclidean norm without squaring large inputs or spreading an unbounded range. */ +function norm(values: number[], start = 0): number { + let result = 0 + for (let i = start; i < values.length; i++) { + result = Math.hypot(result, values[i]) + } + return result +} + +/** Computes the mean relative to the first value to preserve small changes on large offsets. */ +function mean(values: number[]): number { + const origin = values[0] + let sum = 0 + for (const value of values) { + sum += (value - origin) / values.length + } + return origin + sum +} + +/** Applies a Householder reflection to the trailing part of a column. */ +function reflect(values: number[], vector: number[], start: number, factor: number): void { + let product = 0 + for (let i = start; i < values.length; i++) { + product += vector[i - start] * values[i] + } + product *= factor + for (let i = start; i < values.length; i++) { + values[i] -= product * vector[i - start] + } +} + +/** Solves the retained triangular system, treating an exactly zero predictor as a zero coefficient. */ +function backSubstitute(columns: RegressionColumn[], count: number, right: number[]): number[] { + const solution = Array(count).fill(0) + for (let i = count - 1; i >= 0; i--) { + if (columns[i].values[i] === 0) { + continue + } + let value = right[i] + for (let j = i + 1; j < count; j++) { + value -= columns[j].values[i] * solution[j] + } + solution[i] = value / columns[i].values[i] + } + return solution +} + +/** + * Fits observations using pivoted Householder QR, with a fixed leading intercept column. + * Predictor centering preserves accuracy for large offsets. The factorization is column-major, + * uses O(n*k + k*k) storage, and costs O(n*k*k) for n >= k; Q is never materialized. + * + * Exactly zero predictors intentionally consume an available QR slot. Excel Online returns a + * zero coefficient but still excludes that transformed response entry from residual statistics. + * A dependent nonzero predictor is instead removed using its relative norm, increasing df. + * + * @param predictors - observation rows, with predictor columns in their original order + * @param observations - numeric response for each observation + * @param fitIntercept - whether to include a constant term + * @param statistics - whether to compute coefficient uncertainties + */ +export function fitLinearRegression(predictors: number[][], observations: number[], fitIntercept: boolean, statistics: boolean): LinearRegressionResult { + const n = observations.length + const k = predictors[0].length + const offset = fitIntercept ? 1 : 0 + const means = Array(k).fill(0) + const columns: RegressionColumn[] = [] + if (fitIntercept) { + columns.push({values: Array(n).fill(1), index: -1, originalNorm: Math.sqrt(n)}) + } + for (let j = 0; j < k; j++) { + const values = predictors.map(row => row[j]) + means[j] = fitIntercept ? mean(values) : 0 + const centered = values.map(value => value - means[j]) + columns.push({values: centered, index: j, originalNorm: norm(centered)}) + } + + // Treat the intercept as the last input column moved to the front by a swap. + // Retain this order when pivot norms tie, so duplicated predictors agree with Excel. + if (fitIntercept && k > 1) { + const firstPredictor = columns.splice(1, 1)[0] + columns.push(firstPredictor) + } + // Remove a common response offset before reflection to preserve small variations. + const responseOrigin = fitIntercept ? observations[0] : 0 + const transformedY = observations.map(value => value - responseOrigin) + let remaining = columns.length + let count = 0 + while (count < Math.min(n, remaining)) { + if (count >= offset) { + let selected = count + let largest = -1 + for (let j = count; j < remaining; j++) { + const magnitude = norm(columns[j].values, count) + if (magnitude > largest) { + largest = magnitude + selected = j + } + } + [columns[count], columns[selected]] = [columns[selected], columns[count]] + } + const column = columns[count] + const magnitude = norm(column.values, count) + if (count >= offset && magnitude < RANK_TOLERANCE * column.originalNorm) { + remaining-- + columns[count] = columns[remaining] + columns[remaining] = column + continue + } + if (magnitude !== 0) { + const sign = column.values[count] >= 0 ? 1 : -1 + const vector = column.values.slice(count).map(value => value / magnitude) + vector[0] += sign + const factor = 1 / (1 + Math.abs(column.values[count]) / magnitude) + for (let j = count + 1; j < remaining; j++) { + reflect(columns[j].values, vector, count, factor) + } + reflect(transformedY, vector, count, factor) + column.values[count] = -sign * magnitude + column.values.fill(0, count + 1) + } + count++ + } + + const solution = backSubstitute(columns, count, transformedY) + const coefficients = Array(k).fill(0) + for (let j = offset; j < count; j++) { + coefficients[columns[j].index] = solution[j] + } + const intercept = fitIntercept ? mean(observations) - coefficients.reduce((sum, value, j) => sum + value * means[j], 0) : 0 + const yMean = fitIntercept ? mean(observations) : 0 + const centeredY = observations.map(value => value - yMean) + const totalSumSquares = norm(centeredY) ** 2 + let residualSumSquares = norm(transformedY, count) ** 2 + // Simple exact fits have zero residual statistics in Excel. Discard only + // reflection roundoff, whose squared scale is O((n * machine epsilon)^2). + const simpleFitRoundoff = k === 1 && residualSumSquares <= totalSumSquares * (n * Number.EPSILON) ** 2 + if (totalSumSquares === 0 || count === n || simpleFitRoundoff) { + residualSumSquares = 0 + } + const degreesOfFreedom = n - count + const variance = degreesOfFreedom === 0 ? 0 : residualSumSquares / degreesOfFreedom + const standardErrors = Array(k).fill(0) + let interceptVariance = 0 + if (statistics && variance !== 0) { + // Each solve produces one column of R^-1. Accumulate only covariance diagonals + // and the intercept's quadratic form, avoiding a full covariance matrix. + for (let j = 0; j < count; j++) { + const right = Array(count).fill(0) + right[j] = 1 + const inverseColumn = backSubstitute(columns, count, right) + let interceptWeight = fitIntercept ? inverseColumn[0] : 0 + for (let i = offset; i < count; i++) { + const original = columns[i].index + standardErrors[original] += variance * inverseColumn[i] ** 2 + interceptWeight -= means[original] * inverseColumn[i] + } + interceptVariance += variance * interceptWeight ** 2 + } + } + return { + coefficients, + intercept, + standardErrors: standardErrors.map(Math.sqrt), + interceptError: Math.sqrt(interceptVariance), + residualSumSquares, + totalSumSquares, + degreesOfFreedom, + retainedPredictorCount: count - offset, + } +} From 64b6393a8b88a0029be729a6554173243cca1ff7 Mon Sep 17 00:00:00 2001 From: tobiadefami Date: Thu, 10 Sep 2026 20:37:24 +0100 Subject: [PATCH 02/11] docs(HF-221): link LINEST changelog to PR 1769 --- CHANGELOG.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 5bad1b8ddb..0b5938d44a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,7 +9,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/), ### Added -- Added `LINEST` for simple and multiple linear regression, with optional intercept and regression statistics. The `stats` argument must be constant because result dimensions are determined before evaluation. +- Added `LINEST` for simple and multiple linear regression, with optional intercept and regression statistics. The `stats` argument must be constant because result dimensions are determined before evaluation. [#1769](https://github.com/handsontable/hyperformula/pull/1769) ### Fixed From 65cb77397e606909f8245a7c3539a3e67f183645 Mon Sep 17 00:00:00 2001 From: tobiadefami Date: Thu, 10 Sep 2026 20:53:44 +0100 Subject: [PATCH 03/11] fix(HF-221): remove redundant LINEST type assertions --- src/interpreter/plugin/RegressionPlugin.ts | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/src/interpreter/plugin/RegressionPlugin.ts b/src/interpreter/plugin/RegressionPlugin.ts index d64c392281..7810eb2e7c 100644 --- a/src/interpreter/plugin/RegressionPlugin.ts +++ b/src/interpreter/plugin/RegressionPlugin.ts @@ -165,8 +165,8 @@ export class RegressionPlugin extends FunctionPlugin implements FunctionPluginTy if (!yValues.every(isExtendedNumber) || !xValues.every(isExtendedNumber)) { return new CellError(ErrorType.VALUE, ErrorMessage.NumberRange) } - const observations = yValues.map(getRawValue) as number[] - const numericX = xValues.map(getRawValue) as number[] + const observations = yValues.map(getRawValue) + const numericX = xValues.map(getRawValue) const predictors = observations.map((_, i) => { if (shape.orientation === 'paired') { return [numericX[i]] From a6ba94e5bd666b683594edc5d6d877ba94e30c85 Mon Sep 17 00:00:00 2001 From: tobiadefami Date: Fri, 11 Sep 2026 21:27:08 +0100 Subject: [PATCH 04/11] fix(HF-221): stabilize LINEST standard errors --- docs/guide/known-limitations.md | 2 ++ docs/guide/list-of-differences.md | 2 ++ .../plugin/regression/LinearRegression.ts | 16 +++++++++------- 3 files changed, 13 insertions(+), 7 deletions(-) diff --git a/docs/guide/known-limitations.md b/docs/guide/known-limitations.md index 3a6ceb1e4c..ba47769d71 100644 --- a/docs/guide/known-limitations.md +++ b/docs/guide/known-limitations.md @@ -69,6 +69,8 @@ a circular reference. * If an input expression returns a different predictor count than predicted, `LINEST` returns `#VALUE!`. For example, filtering predictor columns can change the required result width. +* Very large or very small input magnitudes can reduce numerical accuracy. Intermediate calculations, such as residual sums of squares, can still overflow or underflow, producing `#NUM!` or inaccurate statistics. Rescale inputs to more moderate units where possible. See [LINEST numerical differences](list-of-differences.md#linest) for compatibility considerations. + ### OFFSET function HyperFormula resolves the OFFSET function at parse time rather than during evaluation. The parser inspects the arguments and rewrites the expression into a plain cell reference or range. This keeps the dependency graph accurate but imposes several restrictions. diff --git a/docs/guide/list-of-differences.md b/docs/guide/list-of-differences.md index f53af063dc..e4a3b28fdc 100644 --- a/docs/guide/list-of-differences.md +++ b/docs/guide/list-of-differences.md @@ -40,6 +40,8 @@ Unlike Excel, HyperFormula requires a constant `stats` argument because the outp Numerical results can differ for nearly dependent predictors and nearly perfect fits. Coefficient standard errors are sensitive to conditioning, and the F statistic is sensitive to residuals close to machine precision. For an effectively perfect multiple regression, Excel and HyperFormula can return different large finite F values even when the coefficients agree. +Very large or very small input scales can also cause substantial differences in coefficients and statistics, including for a single predictor with a non-perfect fit. HyperFormula does not reproduce Excel's loss of predictors or zero standard errors observed at extreme scales. Rescaling inputs to more moderate units can reduce numerical errors in both engines. + ## General functionalities | Functionality | Examples | HyperFormula | Google Sheets | Microsoft Excel | diff --git a/src/interpreter/plugin/regression/LinearRegression.ts b/src/interpreter/plugin/regression/LinearRegression.ts index d3a286db0b..7b9d793927 100644 --- a/src/interpreter/plugin/regression/LinearRegression.ts +++ b/src/interpreter/plugin/regression/LinearRegression.ts @@ -172,10 +172,12 @@ export function fitLinearRegression(predictors: number[][], observations: number const degreesOfFreedom = n - count const variance = degreesOfFreedom === 0 ? 0 : residualSumSquares / degreesOfFreedom const standardErrors = Array(k).fill(0) - let interceptVariance = 0 + let interceptError = 0 if (statistics && variance !== 0) { - // Each solve produces one column of R^-1. Accumulate only covariance diagonals - // and the intercept's quadratic form, avoiding a full covariance matrix. + const residualStandardError = Math.sqrt(variance) + // Each solve produces one column of R^-1 without a full covariance matrix. + // Accumulate scaled uncertainty magnitudes with hypot: squaring first can + // overflow or underflow even when the final standard error is representable. for (let j = 0; j < count; j++) { const right = Array(count).fill(0) right[j] = 1 @@ -183,17 +185,17 @@ export function fitLinearRegression(predictors: number[][], observations: number let interceptWeight = fitIntercept ? inverseColumn[0] : 0 for (let i = offset; i < count; i++) { const original = columns[i].index - standardErrors[original] += variance * inverseColumn[i] ** 2 + standardErrors[original] = Math.hypot(standardErrors[original], residualStandardError * inverseColumn[i]) interceptWeight -= means[original] * inverseColumn[i] } - interceptVariance += variance * interceptWeight ** 2 + interceptError = Math.hypot(interceptError, residualStandardError * interceptWeight) } } return { coefficients, intercept, - standardErrors: standardErrors.map(Math.sqrt), - interceptError: Math.sqrt(interceptVariance), + standardErrors, + interceptError, residualSumSquares, totalSumSquares, degreesOfFreedom, From b8a20aa0aff87a38cc4e53b1fe28656e57f5d96f Mon Sep 17 00:00:00 2001 From: tobiadefami Date: Mon, 14 Sep 2026 19:57:01 +0100 Subject: [PATCH 05/11] refactor(HF-221): clarify LINEST regression flow --- src/interpreter/plugin/RegressionPlugin.ts | 45 ++- .../plugin/regression/LinearRegression.ts | 314 ++++++++++++------ 2 files changed, 241 insertions(+), 118 deletions(-) diff --git a/src/interpreter/plugin/RegressionPlugin.ts b/src/interpreter/plugin/RegressionPlugin.ts index 7810eb2e7c..9df3160c1e 100644 --- a/src/interpreter/plugin/RegressionPlugin.ts +++ b/src/interpreter/plugin/RegressionPlugin.ts @@ -16,24 +16,46 @@ import {fitLinearRegression, LinearRegressionResult} from './regression/LinearRe /** The orientation and predictor count shared by size prediction and runtime validation. */ interface RegressionShape { - predictors: number, + predictorCount: number, orientation: 'paired' | 'columns' | 'rows', } /** Classifies the original dimensions before any values are flattened. */ function regressionShape(y: ArraySize, x?: ArraySize): RegressionShape | undefined { if (x === undefined || (y.width === x.width && y.height === x.height)) { - return {predictors: 1, orientation: 'paired'} + return {predictorCount: 1, orientation: 'paired'} } if (y.width === 1 && y.height === x.height) { - return {predictors: x.width, orientation: 'columns'} + return {predictorCount: x.width, orientation: 'columns'} } if (y.height === 1 && y.width === x.width) { - return {predictors: x.height, orientation: 'rows'} + return {predictorCount: x.height, orientation: 'rows'} } return undefined } +/** + * Converts validated, flattened predictor values into one row per observation. + * In columns orientation, each source row is an observation; in rows orientation, + * each source row is a predictor. Paired ranges supply one predictor per observation. + */ +function buildPredictorRows(predictorValues: number[], observationCount: number, shape: RegressionShape): number[][] { + return Array.from({length: observationCount}, (_, observationIndex) => { + if (shape.orientation === 'paired') { + return [predictorValues[observationIndex]] + } + return Array.from({length: shape.predictorCount}, (_, predictorIndex) => { + let flatIndex: number + if (shape.orientation === 'columns') { + flatIndex = observationIndex * shape.predictorCount + predictorIndex + } else { + flatIndex = predictorIndex * observationCount + observationIndex + } + return predictorValues[flatIndex] + }) + }) +} + /** LINEST rejects empty strings as Boolean options, including formula-generated strings. */ function regressionBoolean(value: InternalScalarValue): boolean | CellError | undefined { return value === '' ? undefined : coerceScalarToBoolean(value) @@ -154,10 +176,10 @@ export class RegressionPlugin extends FunctionPlugin implements FunctionPluginTy if (shape === undefined) { return new CellError(ErrorType.REF, ErrorMessage.ArrayDimensions) } - if (shape.predictors + 1 > this.config.maxColumns) { + if (shape.predictorCount + 1 > this.config.maxColumns) { return new CellError(ErrorType.VALUE, ErrorMessage.ValueLarge) } - if (this.linestArraySize(ast, state).width !== shape.predictors + 1) { + if (this.linestArraySize(ast, state).width !== shape.predictorCount + 1) { return new CellError(ErrorType.VALUE, ErrorMessage.LinestStaticSize) } const yValues = Array.from(knownY.valuesFromTopLeftCorner()) @@ -167,12 +189,7 @@ export class RegressionPlugin extends FunctionPlugin implements FunctionPluginTy } const observations = yValues.map(getRawValue) const numericX = xValues.map(getRawValue) - const predictors = observations.map((_, i) => { - if (shape.orientation === 'paired') { - return [numericX[i]] - } - return Array.from({length: shape.predictors}, (_, j) => numericX[shape.orientation === 'columns' ? i * shape.predictors + j : j * observations.length + i]) - }) + const predictors = buildPredictorRows(numericX, observations.length, shape) const fit = fitLinearRegression(predictors, observations, fitIntercept, statistics) return regressionOutput(fit, statistics, fitIntercept) }) @@ -191,9 +208,9 @@ export class RegressionPlugin extends FunctionPlugin implements FunctionPluginTy const y = this.arraySizeForAst(ast.args[0], arrayState) const x = ast.args.length < 2 || ast.args[1].type === AstNodeType.EMPTY ? undefined : this.arraySizeForAst(ast.args[1], arrayState) const shape = regressionShape(y, x) - if (shape === undefined || shape.predictors + 1 > this.config.maxColumns) { + if (shape === undefined || shape.predictorCount + 1 > this.config.maxColumns) { return ArraySize.error() } - return new ArraySize(shape.predictors + 1, statistics ? 5 : 1) + return new ArraySize(shape.predictorCount + 1, statistics ? 5 : 1) } } diff --git a/src/interpreter/plugin/regression/LinearRegression.ts b/src/interpreter/plugin/regression/LinearRegression.ts index 7b9d793927..8cdf03b019 100644 --- a/src/interpreter/plugin/regression/LinearRegression.ts +++ b/src/interpreter/plugin/regression/LinearRegression.ts @@ -15,13 +15,31 @@ export interface LinearRegressionResult { retainedPredictorCount: number, } -/** A predictor column with its original position and pre-factorization norm. */ +/** + * Column-major storage: columns[columnIndex].values[rowIndex] is a matrix entry. + * Factorization overwrites the values with the triangular system. predictorIndex + * preserves the original predictor position through swaps; -1 identifies the intercept. + */ interface RegressionColumn { values: number[], - index: number, + predictorIndex: number, + /** Norm after optional centering, before any reflections. */ originalNorm: number, } +/** A reflection and the leading value it produces in the selected column tail. */ +interface HouseholderReflection { + vector: number[], + factor: number, + diagonalValue: number, +} + +/** Coefficient uncertainties restored to the original predictor order. */ +interface CoefficientStandardErrors { + standardErrors: number[], + interceptError: number, +} + /** * Rank threshold relative to a predictor's original centered norm. * Excel Online retains a 1e-5 perturbation of a duplicated predictor and removes 1e-6. @@ -30,10 +48,10 @@ interface RegressionColumn { const RANK_TOLERANCE = 1e-6 /** Computes a Euclidean norm without squaring large inputs or spreading an unbounded range. */ -function norm(values: number[], start = 0): number { +function norm(values: number[], startRow = 0): number { let result = 0 - for (let i = start; i < values.length; i++) { - result = Math.hypot(result, values[i]) + for (let rowIndex = startRow; rowIndex < values.length; rowIndex++) { + result = Math.hypot(result, values[rowIndex]) } return result } @@ -48,42 +66,143 @@ function mean(values: number[]): number { return origin + sum } -/** Applies a Householder reflection to the trailing part of a column. */ -function reflect(values: number[], vector: number[], start: number, factor: number): void { - let product = 0 - for (let i = start; i < values.length; i++) { - product += vector[i - start] * values[i] +/** + * Constructs a reflection for a nonzero column tail without modifying the column. + * + * Let x = values.slice(startRow), r = magnitude, and e0 = [1, 0, ...]. + * Choose s = 1 when x[0] >= 0, otherwise -1. + * The vector v = x / r + s * e0 reflects x to [-s * r, 0, ...]. Adding the same + * sign avoids cancellation when constructing v's first entry. + * Its squared length is v^T * v = 2 * (1 + abs(x[0]) / r), so the reflection + * factor 2 / (v^T * v) simplifies to 1 / (1 + abs(x[0]) / r). + * + * @param values - selected column, including any already processed rows + * @param startRow - first row of the column tail to reflect + * @param magnitude - nonzero norm of values starting at startRow, already computed by the caller + */ +function createHouseholderReflection(values: number[], startRow: number, magnitude: number): HouseholderReflection { + const sign = values[startRow] >= 0 ? 1 : -1 + const vector = values.slice(startRow).map(value => value / magnitude) + vector[0] += sign + const factor = 1 / (1 + Math.abs(values[startRow]) / magnitude) + return {vector, factor, diagonalValue: -sign * magnitude} +} + +/** + * Applies a Householder reflection to values starting at startRow, in place. + * + * For the active tail w and reflection vector v, with ^T denoting transpose: + * dotProduct = v^T * w + * reflectionScale = factor * dotProduct, where factor = 2 / (v^T * v) + * w_new = w - reflectionScale * v + * The subtracted vector is twice the projection onto v: this reverses that + * component while preserving the perpendicular component and the total length. + * + * startRow stays fixed during this call; rowIndex - startRow indexes the shorter v. + * Earlier rows remain unchanged. Compute the full dot product before overwriting w. + */ +function applyHouseholderReflection(values: number[], vector: number[], startRow: number, factor: number): void { + let dotProduct = 0 + for (let rowIndex = startRow; rowIndex < values.length; rowIndex++) { + dotProduct += vector[rowIndex - startRow] * values[rowIndex] + } + const reflectionScale = dotProduct * factor + for (let rowIndex = startRow; rowIndex < values.length; rowIndex++) { + values[rowIndex] -= reflectionScale * vector[rowIndex - startRow] } - product *= factor - for (let i = start; i < values.length; i++) { - values[i] -= product * vector[i - start] +} + +/** + * Selects the largest remaining column tail without changing column order. + * diagonalIndex is both the first candidate column and the first active row. + * Strict comparison keeps the first candidate on ties, preserving Excel's duplicate selection. + */ +function selectPivotColumn(columns: RegressionColumn[], diagonalIndex: number, activeColumnCount: number): number { + let selectedColumnIndex = diagonalIndex + let largestNorm = -1 + for (let columnIndex = diagonalIndex; columnIndex < activeColumnCount; columnIndex++) { + const trailingNorm = norm(columns[columnIndex].values, diagonalIndex) + if (trailingNorm > largestNorm) { + largestNorm = trailingNorm + selectedColumnIndex = columnIndex + } } + return selectedColumnIndex } -/** Solves the retained triangular system, treating an exactly zero predictor as a zero coefficient. */ -function backSubstitute(columns: RegressionColumn[], count: number, right: number[]): number[] { - const solution = Array(count).fill(0) - for (let i = count - 1; i >= 0; i--) { - if (columns[i].values[i] === 0) { +/** + * Solves R * solution = rightHandSide from the bottom row upward. + * R[row, column] is stored as columns[column].values[row]. Subtract the already + * solved terms to the right of the diagonal, then divide by the diagonal value. + * Exactly zero diagonals leave a zero coefficient for Excel-compatible zero columns. + */ +function backSubstitute(columns: RegressionColumn[], processedColumnCount: number, rightHandSide: number[]): number[] { + const solution = Array(processedColumnCount).fill(0) + for (let rowIndex = processedColumnCount - 1; rowIndex >= 0; rowIndex--) { + if (columns[rowIndex].values[rowIndex] === 0) { continue } - let value = right[i] - for (let j = i + 1; j < count; j++) { - value -= columns[j].values[i] * solution[j] + let remainingValue = rightHandSide[rowIndex] + for (let columnIndex = rowIndex + 1; columnIndex < processedColumnCount; columnIndex++) { + remainingValue -= columns[columnIndex].values[rowIndex] * solution[columnIndex] } - solution[i] = value / columns[i].values[i] + solution[rowIndex] = remainingValue / columns[rowIndex].values[rowIndex] } return solution } +/** + * Computes standard errors from the retained triangular system, in original predictor order. + * + * For nonsingular R, covariance = residualVariance * inverseR * transpose(inverseR). + * A coefficient's standard error is therefore sqrt(residualVariance) times the norm + * of its row of inverseR. Solve R * inverseColumn = e_j for each unit vector e_j + * to accumulate those row norms without building a full inverse or covariance matrix. + * + * With an intercept, centering changes it by -sum(coefficient * predictorMean), so each + * inverse column contributes interceptWeight = inverseColumn[0] - sum(mean * inverseEntry). + * Zero diagonals retain backSubstitute's Excel-compatible zero-coefficient behavior. + */ +function computeCoefficientStandardErrors( + columns: RegressionColumn[], + processedColumnCount: number, + predictorMeans: number[], + fitIntercept: boolean, + residualVariance: number, +): CoefficientStandardErrors { + const standardErrors = Array(predictorMeans.length).fill(0) + let interceptError = 0 + if (residualVariance === 0) { + return {standardErrors, interceptError} + } + + const interceptColumnCount = fitIntercept ? 1 : 0 + const residualStandardError = Math.sqrt(residualVariance) + for (let inverseColumnIndex = 0; inverseColumnIndex < processedColumnCount; inverseColumnIndex++) { + const rightHandSide = Array(processedColumnCount).fill(0) + rightHandSide[inverseColumnIndex] = 1 + const inverseColumn = backSubstitute(columns, processedColumnCount, rightHandSide) + let interceptWeight = fitIntercept ? inverseColumn[0] : 0 + for (let columnIndex = interceptColumnCount; columnIndex < processedColumnCount; columnIndex++) { + const predictorIndex = columns[columnIndex].predictorIndex + // Scale before hypot: squaring inverse entries first can overflow or underflow. + standardErrors[predictorIndex] = Math.hypot(standardErrors[predictorIndex], residualStandardError * inverseColumn[columnIndex]) + interceptWeight -= predictorMeans[predictorIndex] * inverseColumn[columnIndex] + } + interceptError = Math.hypot(interceptError, residualStandardError * interceptWeight) + } + return {standardErrors, interceptError} +} + /** * Fits observations using pivoted Householder QR, with a fixed leading intercept column. - * Predictor centering preserves accuracy for large offsets. The factorization is column-major, - * uses O(n*k + k*k) storage, and costs O(n*k*k) for n >= k; Q is never materialized. + * Prepares columns, factorizes them, restores coefficients, then computes fit statistics. + * Predictor centering preserves accuracy for large offsets. For n observations and k predictors, + * the factorization uses O(n*k + k*k) storage and costs O(n*k*k) for n >= k; Q is never materialized. * * Exactly zero predictors intentionally consume an available QR slot. Excel Online returns a * zero coefficient but still excludes that transformed response entry from residual statistics. - * A dependent nonzero predictor is instead removed using its relative norm, increasing df. + * A dependent nonzero predictor is instead removed using its relative norm, increasing degrees of freedom. * * @param predictors - observation rows, with predictor columns in their original order * @param observations - numeric response for each observation @@ -91,106 +210,93 @@ function backSubstitute(columns: RegressionColumn[], count: number, right: numbe * @param statistics - whether to compute coefficient uncertainties */ export function fitLinearRegression(predictors: number[][], observations: number[], fitIntercept: boolean, statistics: boolean): LinearRegressionResult { - const n = observations.length - const k = predictors[0].length - const offset = fitIntercept ? 1 : 0 - const means = Array(k).fill(0) + const observationCount = observations.length + const predictorCount = predictors[0].length + const interceptColumnCount = fitIntercept ? 1 : 0 + const predictorMeans = Array(predictorCount).fill(0) const columns: RegressionColumn[] = [] if (fitIntercept) { - columns.push({values: Array(n).fill(1), index: -1, originalNorm: Math.sqrt(n)}) + columns.push({values: Array(observationCount).fill(1), predictorIndex: -1, originalNorm: Math.sqrt(observationCount)}) } - for (let j = 0; j < k; j++) { - const values = predictors.map(row => row[j]) - means[j] = fitIntercept ? mean(values) : 0 - const centered = values.map(value => value - means[j]) - columns.push({values: centered, index: j, originalNorm: norm(centered)}) + for (let predictorIndex = 0; predictorIndex < predictorCount; predictorIndex++) { + const values = predictors.map(row => row[predictorIndex]) + predictorMeans[predictorIndex] = fitIntercept ? mean(values) : 0 + const centeredValues = values.map(value => value - predictorMeans[predictorIndex]) + columns.push({values: centeredValues, predictorIndex, originalNorm: norm(centeredValues)}) } - // Treat the intercept as the last input column moved to the front by a swap. - // Retain this order when pivot norms tie, so duplicated predictors agree with Excel. - if (fitIntercept && k > 1) { + // Excel tie ordering: [intercept, x1, x2, ...] -> [intercept, x2, ..., x1]. + // This matches swapping a trailing intercept into the first input position. + if (fitIntercept && predictorCount > 1) { const firstPredictor = columns.splice(1, 1)[0] columns.push(firstPredictor) } // Remove a common response offset before reflection to preserve small variations. const responseOrigin = fitIntercept ? observations[0] : 0 - const transformedY = observations.map(value => value - responseOrigin) - let remaining = columns.length - let count = 0 - while (count < Math.min(n, remaining)) { - if (count >= offset) { - let selected = count - let largest = -1 - for (let j = count; j < remaining; j++) { - const magnitude = norm(columns[j].values, count) - if (magnitude > largest) { - largest = magnitude - selected = j - } - } - [columns[count], columns[selected]] = [columns[selected], columns[count]] + const transformedResponse = observations.map(value => value - responseOrigin) + + // Columns are partitioned as [processed | candidates | rejected]: + // [0, processedColumnCount), [processedColumnCount, activeColumnCount), and the rest. + // processedColumnCount is also the current diagonal index, not the mathematical rank: + // exactly zero columns still consume a slot for Excel-compatible residual statistics. + let activeColumnCount = columns.length + let processedColumnCount = 0 + while (processedColumnCount < Math.min(observationCount, activeColumnCount)) { + if (processedColumnCount >= interceptColumnCount) { + const pivotColumnIndex = selectPivotColumn(columns, processedColumnCount, activeColumnCount) + const previousColumn = columns[processedColumnCount] + columns[processedColumnCount] = columns[pivotColumnIndex] + columns[pivotColumnIndex] = previousColumn } - const column = columns[count] - const magnitude = norm(column.values, count) - if (count >= offset && magnitude < RANK_TOLERANCE * column.originalNorm) { - remaining-- - columns[count] = columns[remaining] - columns[remaining] = column + const column = columns[processedColumnCount] + const trailingNorm = norm(column.values, processedColumnCount) + // Rejected predictors do not advance the diagonal. An originally zero column + // gives 0 < 0 here, so it is retained and advances processedColumnCount below. + if (processedColumnCount >= interceptColumnCount && trailingNorm < RANK_TOLERANCE * column.originalNorm) { + activeColumnCount-- + columns[processedColumnCount] = columns[activeColumnCount] + columns[activeColumnCount] = column continue } - if (magnitude !== 0) { - const sign = column.values[count] >= 0 ? 1 : -1 - const vector = column.values.slice(count).map(value => value / magnitude) - vector[0] += sign - const factor = 1 / (1 + Math.abs(column.values[count]) / magnitude) - for (let j = count + 1; j < remaining; j++) { - reflect(columns[j].values, vector, count, factor) + if (trailingNorm !== 0) { + const {vector, factor, diagonalValue} = createHouseholderReflection(column.values, processedColumnCount, trailingNorm) + for (let columnIndex = processedColumnCount + 1; columnIndex < activeColumnCount; columnIndex++) { + applyHouseholderReflection(columns[columnIndex].values, vector, processedColumnCount, factor) } - reflect(transformedY, vector, count, factor) - column.values[count] = -sign * magnitude - column.values.fill(0, count + 1) + applyHouseholderReflection(transformedResponse, vector, processedColumnCount, factor) + column.values[processedColumnCount] = diagonalValue + column.values.fill(0, processedColumnCount + 1) } - count++ + processedColumnCount++ } - const solution = backSubstitute(columns, count, transformedY) - const coefficients = Array(k).fill(0) - for (let j = offset; j < count; j++) { - coefficients[columns[j].index] = solution[j] + // Solve in pivot order, then restore the caller's predictor order. Rejected slots stay zero. + const solution = backSubstitute(columns, processedColumnCount, transformedResponse) + const coefficients = Array(predictorCount).fill(0) + for (let columnIndex = interceptColumnCount; columnIndex < processedColumnCount; columnIndex++) { + coefficients[columns[columnIndex].predictorIndex] = solution[columnIndex] } - const intercept = fitIntercept ? mean(observations) - coefficients.reduce((sum, value, j) => sum + value * means[j], 0) : 0 - const yMean = fitIntercept ? mean(observations) : 0 - const centeredY = observations.map(value => value - yMean) - const totalSumSquares = norm(centeredY) ** 2 - let residualSumSquares = norm(transformedY, count) ** 2 + // Recover the intercept in original coordinates: b = mean(y) - sum(coefficient * mean(x)). + const intercept = fitIntercept + ? mean(observations) - coefficients.reduce((sum, coefficient, predictorIndex) => sum + coefficient * predictorMeans[predictorIndex], 0) + : 0 + + const responseMean = fitIntercept ? mean(observations) : 0 + const centeredResponse = observations.map(value => value - responseMean) + const totalSumSquares = norm(centeredResponse) ** 2 + // Reflections preserve squared error; entries below the solved rows are the residual tail. + let residualSumSquares = norm(transformedResponse, processedColumnCount) ** 2 // Simple exact fits have zero residual statistics in Excel. Discard only // reflection roundoff, whose squared scale is O((n * machine epsilon)^2). - const simpleFitRoundoff = k === 1 && residualSumSquares <= totalSumSquares * (n * Number.EPSILON) ** 2 - if (totalSumSquares === 0 || count === n || simpleFitRoundoff) { + const simpleFitRoundoff = predictorCount === 1 && residualSumSquares <= totalSumSquares * (observationCount * Number.EPSILON) ** 2 + if (totalSumSquares === 0 || processedColumnCount === observationCount || simpleFitRoundoff) { residualSumSquares = 0 } - const degreesOfFreedom = n - count - const variance = degreesOfFreedom === 0 ? 0 : residualSumSquares / degreesOfFreedom - const standardErrors = Array(k).fill(0) - let interceptError = 0 - if (statistics && variance !== 0) { - const residualStandardError = Math.sqrt(variance) - // Each solve produces one column of R^-1 without a full covariance matrix. - // Accumulate scaled uncertainty magnitudes with hypot: squaring first can - // overflow or underflow even when the final standard error is representable. - for (let j = 0; j < count; j++) { - const right = Array(count).fill(0) - right[j] = 1 - const inverseColumn = backSubstitute(columns, count, right) - let interceptWeight = fitIntercept ? inverseColumn[0] : 0 - for (let i = offset; i < count; i++) { - const original = columns[i].index - standardErrors[original] = Math.hypot(standardErrors[original], residualStandardError * inverseColumn[i]) - interceptWeight -= means[original] * inverseColumn[i] - } - interceptError = Math.hypot(interceptError, residualStandardError * interceptWeight) - } - } + const degreesOfFreedom = observationCount - processedColumnCount + const residualVariance = degreesOfFreedom === 0 ? 0 : residualSumSquares / degreesOfFreedom + const {standardErrors, interceptError} = statistics + ? computeCoefficientStandardErrors(columns, processedColumnCount, predictorMeans, fitIntercept, residualVariance) + : {standardErrors: Array(predictorCount).fill(0), interceptError: 0} return { coefficients, intercept, @@ -199,6 +305,6 @@ export function fitLinearRegression(predictors: number[][], observations: number residualSumSquares, totalSumSquares, degreesOfFreedom, - retainedPredictorCount: count - offset, + retainedPredictorCount: processedColumnCount - interceptColumnCount, } } From 558bd2892edab46eeffe2243dccc72397fbf463a Mon Sep 17 00:00:00 2001 From: tobiadefami Date: Wed, 16 Sep 2026 21:07:11 +0100 Subject: [PATCH 06/11] refactor(HF-221): separate LINEST calculation stages --- .../plugin/regression/LinearRegression.ts | 126 ++++++++++++++---- 1 file changed, 101 insertions(+), 25 deletions(-) diff --git a/src/interpreter/plugin/regression/LinearRegression.ts b/src/interpreter/plugin/regression/LinearRegression.ts index 8cdf03b019..86b7b1400a 100644 --- a/src/interpreter/plugin/regression/LinearRegression.ts +++ b/src/interpreter/plugin/regression/LinearRegression.ts @@ -27,6 +27,21 @@ interface RegressionColumn { originalNorm: number, } +/** Working arrays owned by one fit, with centering offsets kept in the original predictor order. */ +interface PreparedRegression { + columns: RegressionColumn[], + predictorMeans: number[], + transformedResponse: number[], +} + +/** Residual statistics shared by coefficient uncertainty calculation and spreadsheet output. */ +interface ResidualStatistics { + totalSumSquares: number, + residualSumSquares: number, + degreesOfFreedom: number, + residualVariance: number, +} + /** A reflection and the leading value it produces in the selected column tail. */ interface HouseholderReflection { vector: number[], @@ -42,7 +57,7 @@ interface CoefficientStandardErrors { /** * Rank threshold relative to a predictor's original centered norm. - * Excel Online retains a 1e-5 perturbation of a duplicated predictor and removes 1e-6. + * Sampled Excel Online cases with proportional predictors retain a 1e-5 perturbation and remove 1e-6. * Keep the threshold separate from roundoff handling for the fitted statistics. */ const RANK_TOLERANCE = 1e-6 @@ -195,24 +210,20 @@ function computeCoefficientStandardErrors( } /** - * Fits observations using pivoted Householder QR, with a fixed leading intercept column. - * Prepares columns, factorizes them, restores coefficients, then computes fit statistics. - * Predictor centering preserves accuracy for large offsets. For n observations and k predictors, - * the factorization uses O(n*k + k*k) storage and costs O(n*k*k) for n >= k; Q is never materialized. + * Copies observations and predictors into working arrays without modifying the inputs. + * With an intercept, uses centeredX = x - mean(x) and shiftedY = y - y[0] to preserve + * small variations on large offsets. Without an intercept, leaves both coordinates unchanged. + * Predictor means retain their input order for recovering b = mean(y) - sum(m * mean(x)). + * When no intercept is fitted, predictorMeans contains zero offsets instead. * - * Exactly zero predictors intentionally consume an available QR slot. Excel Online returns a - * zero coefficient but still excludes that transformed response entry from residual statistics. - * A dependent nonzero predictor is instead removed using its relative norm, increasing degrees of freedom. - * - * @param predictors - observation rows, with predictor columns in their original order + * @param predictors - observation rows with predictor columns in their original order * @param observations - numeric response for each observation - * @param fitIntercept - whether to include a constant term - * @param statistics - whether to compute coefficient uncertainties + * @param fitIntercept - whether to add a column of ones and shift the coordinates + * @returns Working columns in Excel tie order, predictor centering offsets, and the shifted response. */ -export function fitLinearRegression(predictors: number[][], observations: number[], fitIntercept: boolean, statistics: boolean): LinearRegressionResult { +function prepareRegression(predictors: number[][], observations: number[], fitIntercept: boolean): PreparedRegression { const observationCount = observations.length const predictorCount = predictors[0].length - const interceptColumnCount = fitIntercept ? 1 : 0 const predictorMeans = Array(predictorCount).fill(0) const columns: RegressionColumn[] = [] if (fitIntercept) { @@ -234,7 +245,26 @@ export function fitLinearRegression(predictors: number[][], observations: number // Remove a common response offset before reflection to preserve small variations. const responseOrigin = fitIntercept ? observations[0] : 0 const transformedResponse = observations.map(value => value - responseOrigin) + return {columns, predictorMeans, transformedResponse} +} +/** + * Overwrites working columns with a pivoted triangular system and applies the same + * Householder reflections to the response. Retained columns store R; the transformed + * response supplies the right-hand side of R * solution = response and its residual tail. + * Original predictor positions remain available through each column's predictorIndex. + * + * Exactly zero predictors intentionally consume an available slot. Excel Online returns + * a zero coefficient but still excludes that response entry from residual statistics. + * A dependent nonzero predictor is removed without consuming a slot. + * + * @param columns - mutable working columns, including a fixed leading intercept when present + * @param transformedResponse - mutable shifted response, transformed in place alongside the columns + * @param interceptColumnCount - one when fitting an intercept, otherwise zero + * @returns Number of processed columns, including zero columns; this is not mathematical rank. + */ +function factorizeRegressionInPlace(columns: RegressionColumn[], transformedResponse: number[], interceptColumnCount: number): number { + const observationCount = transformedResponse.length // Columns are partitioned as [processed | candidates | rejected]: // [0, processedColumnCount), [processedColumnCount, activeColumnCount), and the rest. // processedColumnCount is also the current diagonal index, not the mathematical rank: @@ -269,18 +299,30 @@ export function fitLinearRegression(predictors: number[][], observations: number } processedColumnCount++ } + return processedColumnCount +} - // Solve in pivot order, then restore the caller's predictor order. Rejected slots stay zero. - const solution = backSubstitute(columns, processedColumnCount, transformedResponse) - const coefficients = Array(predictorCount).fill(0) - for (let columnIndex = interceptColumnCount; columnIndex < processedColumnCount; columnIndex++) { - coefficients[columns[columnIndex].predictorIndex] = solution[columnIndex] - } - // Recover the intercept in original coordinates: b = mean(y) - sum(coefficient * mean(x)). - const intercept = fitIntercept - ? mean(observations) - coefficients.reduce((sum, coefficient, predictorIndex) => sum + coefficient * predictorMeans[predictorIndex], 0) - : 0 - +/** + * Computes response variation and residual error without modifying either response array. + * Total variation is sum((y - mean(y))^2) with an intercept, otherwise sum(y^2). + * Residual error is the squared norm below the processed rows of the transformed response. + * Uses the factorization's Excel-compatible row boundary, including consumed zero columns. + * + * @param observations - response values in their original coordinates + * @param transformedResponse - shifted response after the factorization's reflections + * @param processedColumnCount - number of response entries assigned to the triangular solve + * @param predictorCount - original predictor count, used for the simple-fit roundoff rule + * @param fitIntercept - whether total variation is measured around the response mean + * @returns Sums of squares, remaining degrees of freedom, and residual variance (zero when no degrees remain). + */ +function computeResidualStatistics( + observations: number[], + transformedResponse: number[], + processedColumnCount: number, + predictorCount: number, + fitIntercept: boolean, +): ResidualStatistics { + const observationCount = observations.length const responseMean = fitIntercept ? mean(observations) : 0 const centeredResponse = observations.map(value => value - responseMean) const totalSumSquares = norm(centeredResponse) ** 2 @@ -294,6 +336,40 @@ export function fitLinearRegression(predictors: number[][], observations: number } const degreesOfFreedom = observationCount - processedColumnCount const residualVariance = degreesOfFreedom === 0 ? 0 : residualSumSquares / degreesOfFreedom + return {totalSumSquares, residualSumSquares, degreesOfFreedom, residualVariance} +} + +/** + * Fits observations using pivoted Householder QR, with a fixed leading intercept column. + * Prepares columns, factorizes them, restores coefficients, then computes fit statistics. + * Predictor centering preserves accuracy for large offsets. For n observations and k predictors, + * the factorization uses O(n*k + k*k) storage and costs O(n*k*k) for n >= k; Q is never materialized. + * + * @param predictors - observation rows, with predictor columns in their original order + * @param observations - numeric response for each observation + * @param fitIntercept - whether to include a constant term + * @param statistics - whether to compute coefficient uncertainties + */ +export function fitLinearRegression(predictors: number[][], observations: number[], fitIntercept: boolean, statistics: boolean): LinearRegressionResult { + const predictorCount = predictors[0].length + const interceptColumnCount = fitIntercept ? 1 : 0 + const {columns, predictorMeans, transformedResponse} = prepareRegression(predictors, observations, fitIntercept) + const processedColumnCount = factorizeRegressionInPlace(columns, transformedResponse, interceptColumnCount) + + // Solve in pivot order, then restore the caller's predictor order. Rejected slots stay zero. + const solution = backSubstitute(columns, processedColumnCount, transformedResponse) + const coefficients = Array(predictorCount).fill(0) + for (let columnIndex = interceptColumnCount; columnIndex < processedColumnCount; columnIndex++) { + coefficients[columns[columnIndex].predictorIndex] = solution[columnIndex] + } + // Recover the intercept in original coordinates: b = mean(y) - sum(coefficient * mean(x)). + const intercept = fitIntercept + ? mean(observations) - coefficients.reduce((sum, coefficient, predictorIndex) => sum + coefficient * predictorMeans[predictorIndex], 0) + : 0 + + const {totalSumSquares, residualSumSquares, degreesOfFreedom, residualVariance} = computeResidualStatistics( + observations, transformedResponse, processedColumnCount, predictorCount, fitIntercept, + ) const {standardErrors, interceptError} = statistics ? computeCoefficientStandardErrors(columns, processedColumnCount, predictorMeans, fitIntercept, residualVariance) : {standardErrors: Array(predictorCount).fill(0), interceptError: 0} From 16d881b6f140d978ed04411ad34aef1059975db6 Mon Sep 17 00:00:00 2001 From: tobiadefami Date: Thu, 8 Oct 2026 16:12:37 +0100 Subject: [PATCH 07/11] feat(HF-414): add INTERCEPT and FORECAST.LINEAR Add INTERCEPT and FORECAST.LINEAR, with FORECAST as an alias of FORECAST.LINEAR, next to SLOPE in StatisticalAggregationPlugin. Like SLOPE, they skip pairs with a non-numeric value, propagate errors from the ranges, and return #N/A when the ranges have a different number of cells. Fewer than two points, or x values that are all equal, return #DIV/0! (new ErrorMessage.ZeroVariance). The fit uses plain two-pass sums for the means and sums of squares, which reproduce Microsoft Excel's results to the last digits. In x, FORECAST.LINEAR coerces a blank cell, booleans and numeric text, and returns #VALUE! for an empty string, as Excel does. Both functions enable array arithmetic for their arguments, so computed ranges work in the default configuration. Includes the catalogue entries and the names in all language packs. Co-Authored-By: Claude Opus 5.5 --- src/error-message.ts | 1 + src/i18n/languages/csCZ.ts | 3 + src/i18n/languages/daDK.ts | 3 + src/i18n/languages/deDE.ts | 3 + src/i18n/languages/enGB.ts | 3 + src/i18n/languages/esES.ts | 3 + src/i18n/languages/fiFI.ts | 3 + src/i18n/languages/frFR.ts | 3 + src/i18n/languages/huHU.ts | 3 + src/i18n/languages/idID.ts | 3 + src/i18n/languages/itIT.ts | 3 + src/i18n/languages/nbNO.ts | 3 + src/i18n/languages/nlNL.ts | 3 + src/i18n/languages/plPL.ts | 3 + src/i18n/languages/ptPT.ts | 3 + src/i18n/languages/ruRU.ts | 3 + src/i18n/languages/svSE.ts | 3 + src/i18n/languages/trTR.ts | 3 + .../categories/statistical.ts | 21 +++++ .../plugin/StatisticalAggregationPlugin.ts | 80 +++++++++++++++++++ 20 files changed, 153 insertions(+) diff --git a/src/error-message.ts b/src/error-message.ts index 9f0ef381db..cff49fcea8 100644 --- a/src/error-message.ts +++ b/src/error-message.ts @@ -9,6 +9,7 @@ export class ErrorMessage { public static LinestStaticSize = 'LINEST requires input dimensions that determine a fixed result size.' public static LinestStaticStats = 'LINEST requires a constant stats argument to determine its result size.' + public static ZeroVariance = 'Values cannot all be equal.' public static DistinctSigns = 'Distinct signs.' public static WrongArgNumber = 'Wrong number of arguments.' public static EmptyArg = 'Empty function argument.' diff --git a/src/i18n/languages/csCZ.ts b/src/i18n/languages/csCZ.ts index d3e0575403..30c59eca2b 100644 --- a/src/i18n/languages/csCZ.ts +++ b/src/i18n/languages/csCZ.ts @@ -411,6 +411,9 @@ const dictionary: RawTranslationPackage = { FTEST: 'FTEST', STEYX: 'STEYX', SLOPE: 'SLOPE', + INTERCEPT: 'INTERCEPT', + 'FORECAST.LINEAR': 'FORECAST.LINEAR', + FORECAST: 'FORECAST', COVAR: 'COVAR', 'COVARIANCE.P': 'COVARIANCE.P', 'COVARIANCE.S': 'COVARIANCE.S', diff --git a/src/i18n/languages/daDK.ts b/src/i18n/languages/daDK.ts index f3737cefae..0fb1f76d54 100644 --- a/src/i18n/languages/daDK.ts +++ b/src/i18n/languages/daDK.ts @@ -411,6 +411,9 @@ const dictionary: RawTranslationPackage = { FTEST: 'FTEST', STEYX: 'STFYX', SLOPE: 'STIGNING', + INTERCEPT: 'SKÆRING', + 'FORECAST.LINEAR': 'FORECAST.LINEAR', + FORECAST: 'PROGNOSE', COVAR: 'KOVARIANS', 'COVARIANCE.P': 'KOVARIANS.P', 'COVARIANCE.S': 'KOVARIANS.S', diff --git a/src/i18n/languages/deDE.ts b/src/i18n/languages/deDE.ts index 0b1a180f0b..e824074cdb 100644 --- a/src/i18n/languages/deDE.ts +++ b/src/i18n/languages/deDE.ts @@ -411,6 +411,9 @@ const dictionary: RawTranslationPackage = { FTEST: 'FTEST', STEYX: 'STFEHLERYX', SLOPE: 'STEIGUNG', + INTERCEPT: 'ACHSENABSCHNITT', + 'FORECAST.LINEAR': 'FORECAST.LINEAR', + FORECAST: 'SCHÄTZER', COVAR: 'KOVAR', 'COVARIANCE.P': 'KOVARIANZ.P', 'COVARIANCE.S': 'KOVARIANZ.S', diff --git a/src/i18n/languages/enGB.ts b/src/i18n/languages/enGB.ts index 3a7185fdad..8216946733 100644 --- a/src/i18n/languages/enGB.ts +++ b/src/i18n/languages/enGB.ts @@ -414,6 +414,9 @@ const dictionary: RawTranslationPackage = { FTEST: 'FTEST', STEYX: 'STEYX', SLOPE: 'SLOPE', + INTERCEPT: 'INTERCEPT', + 'FORECAST.LINEAR': 'FORECAST.LINEAR', + FORECAST: 'FORECAST', 'CHISQ.TEST': 'CHISQ.TEST', CHITEST: 'CHITEST', 'T.TEST': 'T.TEST', diff --git a/src/i18n/languages/esES.ts b/src/i18n/languages/esES.ts index df2038ff36..875ebca977 100644 --- a/src/i18n/languages/esES.ts +++ b/src/i18n/languages/esES.ts @@ -411,6 +411,9 @@ export const dictionary: RawTranslationPackage = { FTEST: 'PRUEBA.F', STEYX: 'ERROR.TIPICO.XY', SLOPE: 'PENDIENTE', + INTERCEPT: 'INTERSECCION.EJE', + 'FORECAST.LINEAR': 'FORECAST.LINEAR', + FORECAST: 'PRONOSTICO', COVAR: 'COVAR', 'COVARIANCE.P': 'COVARIANCE.P', 'COVARIANCE.S': 'COVARIANZA.M', diff --git a/src/i18n/languages/fiFI.ts b/src/i18n/languages/fiFI.ts index 87ab6be3cd..57f3407790 100644 --- a/src/i18n/languages/fiFI.ts +++ b/src/i18n/languages/fiFI.ts @@ -411,6 +411,9 @@ const dictionary: RawTranslationPackage = { FTEST: 'FTESTI', STEYX: 'KESKIVIRHE', SLOPE: 'KULMAKERROIN', + INTERCEPT: 'LEIKKAUSPISTE', + 'FORECAST.LINEAR': 'FORECAST.LINEAR', + FORECAST: 'ENNUSTE', COVAR: 'KOVARIANSSI', 'COVARIANCE.P': 'KOVARIANSSI.P', 'COVARIANCE.S': 'KOVARIANSSI.S', diff --git a/src/i18n/languages/frFR.ts b/src/i18n/languages/frFR.ts index bc962747b3..84ae293469 100644 --- a/src/i18n/languages/frFR.ts +++ b/src/i18n/languages/frFR.ts @@ -411,6 +411,9 @@ const dictionary: RawTranslationPackage = { FTEST: 'TEST.F', STEYX: 'ERREUR.TYPE.XY', SLOPE: 'PENTE', + INTERCEPT: 'ORDONNEE.ORIGINE', + 'FORECAST.LINEAR': 'FORECAST.LINEAR', + FORECAST: 'PREVISION', COVAR: 'COVARIANCE', 'COVARIANCE.P': 'COVARIANCE.PEARSON', 'COVARIANCE.S': 'COVARIANCE.STANDARD', diff --git a/src/i18n/languages/huHU.ts b/src/i18n/languages/huHU.ts index 245709fa3b..0914d29b06 100644 --- a/src/i18n/languages/huHU.ts +++ b/src/i18n/languages/huHU.ts @@ -411,6 +411,9 @@ const dictionary: RawTranslationPackage = { FTEST: 'F.PRÓBA', STEYX: 'STHIBAYX', SLOPE: 'MEREDEKSÉG', + INTERCEPT: 'METSZ', + 'FORECAST.LINEAR': 'FORECAST.LINEAR', + FORECAST: 'ELŐREJELZÉS', COVAR: 'KOVAR', 'COVARIANCE.P': 'KOVARIANCIA.S', 'COVARIANCE.S': 'KOVARIANCIA.M', diff --git a/src/i18n/languages/idID.ts b/src/i18n/languages/idID.ts index 532ee25e31..cf348b89aa 100644 --- a/src/i18n/languages/idID.ts +++ b/src/i18n/languages/idID.ts @@ -414,6 +414,9 @@ const dictionary: RawTranslationPackage = { FTEST: 'FUJI', STEYX: 'STEYX', SLOPE: 'KEMIRINGAN', + INTERCEPT: 'INTERCEPT', + 'FORECAST.LINEAR': 'FORECAST.LINEAR', + FORECAST: 'FORECAST', 'CHISQ.TEST': 'CHISQ.UJI', CHITEST: 'CHIUJI', 'T.TEST': 'T.UJI', diff --git a/src/i18n/languages/itIT.ts b/src/i18n/languages/itIT.ts index 1a863ea46a..6d39b4ae78 100644 --- a/src/i18n/languages/itIT.ts +++ b/src/i18n/languages/itIT.ts @@ -411,6 +411,9 @@ const dictionary: RawTranslationPackage = { FTEST: 'TEST.F', STEYX: 'ERR.STD.YX', SLOPE: 'PENDENZA', + INTERCEPT: 'INTERCETTA', + 'FORECAST.LINEAR': 'FORECAST.LINEAR', + FORECAST: 'PREVISIONE', COVAR: 'COVARIANZA', 'COVARIANCE.P': 'COVARIANZA.P', 'COVARIANCE.S': 'COVARIANZA.C', diff --git a/src/i18n/languages/nbNO.ts b/src/i18n/languages/nbNO.ts index de4a1cf667..c6daf97ba2 100644 --- a/src/i18n/languages/nbNO.ts +++ b/src/i18n/languages/nbNO.ts @@ -411,6 +411,9 @@ const dictionary: RawTranslationPackage = { FTEST: 'FTEST', STEYX: 'STANDARDFEIL', SLOPE: 'STIGNINGSTALL', + INTERCEPT: 'SKJÆRINGSPUNKT', + 'FORECAST.LINEAR': 'FORECAST.LINEAR', + FORECAST: 'PROGNOSE', COVAR: 'KOVARIANS', 'COVARIANCE.P': 'KOVARIANS.P', 'COVARIANCE.S': 'KOVARIANS.S', diff --git a/src/i18n/languages/nlNL.ts b/src/i18n/languages/nlNL.ts index a7999b8913..04e9ef4614 100644 --- a/src/i18n/languages/nlNL.ts +++ b/src/i18n/languages/nlNL.ts @@ -411,6 +411,9 @@ const dictionary: RawTranslationPackage = { FTEST: 'F.TOETS', STEYX: 'STAND.FOUT.YX', SLOPE: 'RICHTING', + INTERCEPT: 'SNIJPUNT', + 'FORECAST.LINEAR': 'FORECAST.LINEAR', + FORECAST: 'VOORSPELLEN', COVAR: 'COVARIANTIE', 'COVARIANCE.P': 'COVARIANTIE.P', 'COVARIANCE.S': 'COVARIANTIE.S', diff --git a/src/i18n/languages/plPL.ts b/src/i18n/languages/plPL.ts index 7548158e88..71fa29bea0 100644 --- a/src/i18n/languages/plPL.ts +++ b/src/i18n/languages/plPL.ts @@ -411,6 +411,9 @@ const dictionary: RawTranslationPackage = { FTEST: 'TEST.F', STEYX: 'REGBŁSTD', SLOPE: 'NACHYLENIE', + INTERCEPT: 'ODCIĘTA', + 'FORECAST.LINEAR': 'FORECAST.LINEAR', + FORECAST: 'REGLINX', COVAR: 'KOWARIANCJA', 'COVARIANCE.P': 'KOWARIANCJA.POPUL', 'COVARIANCE.S': 'KOWARIANCJA.PRÓBKI', diff --git a/src/i18n/languages/ptPT.ts b/src/i18n/languages/ptPT.ts index a74fba28b5..5d9a3800db 100644 --- a/src/i18n/languages/ptPT.ts +++ b/src/i18n/languages/ptPT.ts @@ -411,6 +411,9 @@ const dictionary: RawTranslationPackage = { FTEST: 'TESTEF', STEYX: 'EPADYX', SLOPE: 'INCLINAÇÃO', + INTERCEPT: 'INTERCETAR', + 'FORECAST.LINEAR': 'FORECAST.LINEAR', + FORECAST: 'PREVISÃO', COVAR: 'COVAR', 'COVARIANCE.P': 'COVARIAÇÃO.P', 'COVARIANCE.S': 'COVARIAÇÃO.S', diff --git a/src/i18n/languages/ruRU.ts b/src/i18n/languages/ruRU.ts index e808973691..6b2bd491a7 100644 --- a/src/i18n/languages/ruRU.ts +++ b/src/i18n/languages/ruRU.ts @@ -411,6 +411,9 @@ const dictionary: RawTranslationPackage = { FTEST: 'ФТЕСТ', STEYX: 'СТОШYX', SLOPE: 'НАКЛОН', + INTERCEPT: 'ОТРЕЗОК', + 'FORECAST.LINEAR': 'FORECAST.LINEAR', + FORECAST: 'ПРЕДСКАЗ', COVAR: 'КОВАР', 'COVARIANCE.P': 'КОВАРИАЦИЯ.Г', 'COVARIANCE.S': 'КОВАРИАЦИЯ.В', diff --git a/src/i18n/languages/svSE.ts b/src/i18n/languages/svSE.ts index 7d869a89f1..2b2d88ceb4 100644 --- a/src/i18n/languages/svSE.ts +++ b/src/i18n/languages/svSE.ts @@ -411,6 +411,9 @@ const dictionary: RawTranslationPackage = { FTEST: 'FTEST', STEYX: 'STDFELYX', SLOPE: 'LUTNING', + INTERCEPT: 'SKÄRNINGSPUNKT', + 'FORECAST.LINEAR': 'FORECAST.LINEAR', + FORECAST: 'PREDIKTION', COVAR: 'KOVAR', 'COVARIANCE.P': 'KOVARIANS.P', 'COVARIANCE.S': 'KOVARIANS.S', diff --git a/src/i18n/languages/trTR.ts b/src/i18n/languages/trTR.ts index 991b50a31f..c1e1b2e907 100644 --- a/src/i18n/languages/trTR.ts +++ b/src/i18n/languages/trTR.ts @@ -411,6 +411,9 @@ const dictionary: RawTranslationPackage = { FTEST: 'FTEST', STEYX: 'STHYX', SLOPE: 'EĞİM', + INTERCEPT: 'KESMENOKTASI', + 'FORECAST.LINEAR': 'FORECAST.LINEAR', + FORECAST: 'TAHMİN', COVAR: 'KOVARYANS', 'COVARIANCE.P': 'KOVARYANS.P', 'COVARIANCE.S': 'KOVARYANS.S', diff --git a/src/interpreter/functionMetadata/categories/statistical.ts b/src/interpreter/functionMetadata/categories/statistical.ts index 6a404977f5..9519cf52b4 100644 --- a/src/interpreter/functionMetadata/categories/statistical.ts +++ b/src/interpreter/functionMetadata/categories/statistical.ts @@ -241,6 +241,17 @@ export const STATISTICAL_DOCS: Record = { documentationUrl: 'https://hyperformula.handsontable.com/docs/guide/built-in-functions.html', examples: ['=FISHERINV(0.5)'], }, + 'FORECAST.LINEAR': { + category: 'Statistical', + shortDescription: 'Returns the value at `x` of the least-squares line through the pairs of `known_y` and `known_x`. Pairs with a non-numeric value are skipped.', + parameters: [ + {name: 'x', description: 'The value of the independent variable at which to predict a value.'}, + {name: 'known_y', description: 'The range of dependent (y) values.'}, + {name: 'known_x', description: 'The range of independent (x) values, with the same number of cells as `known_y`.'}, + ], + documentationUrl: 'https://hyperformula.handsontable.com/docs/guide/built-in-functions.html', + examples: ['=FORECAST.LINEAR(7, A1:A6, B1:B6)'], + }, GAMMA: { category: 'Statistical', shortDescription: 'Returns value of Gamma function.', @@ -297,6 +308,16 @@ export const STATISTICAL_DOCS: Record = { documentationUrl: 'https://hyperformula.handsontable.com/docs/guide/built-in-functions.html', examples: ['=HYPGEOM.DIST(1, 4, 8, 20, FALSE())', '=HYPGEOM.DIST(1, 4, 8, 20, TRUE())'], }, + INTERCEPT: { + category: 'Statistical', + shortDescription: 'Returns the intercept of the least-squares line through the pairs of `known_y` and `known_x`. Pairs with a non-numeric value are skipped.', + parameters: [ + {name: 'known_y', description: 'The range of dependent (y) values.'}, + {name: 'known_x', description: 'The range of independent (x) values, with the same number of cells as `known_y`.'}, + ], + documentationUrl: 'https://hyperformula.handsontable.com/docs/guide/built-in-functions.html', + examples: ['=INTERCEPT(A1:A10, B1:B10)'], + }, LARGE: { category: 'Statistical', shortDescription: 'Returns k-th largest value in a range.', diff --git a/src/interpreter/plugin/StatisticalAggregationPlugin.ts b/src/interpreter/plugin/StatisticalAggregationPlugin.ts index c363bc1351..c85b3cfeef 100644 --- a/src/interpreter/plugin/StatisticalAggregationPlugin.ts +++ b/src/interpreter/plugin/StatisticalAggregationPlugin.ts @@ -122,6 +122,23 @@ export class StatisticalAggregationPlugin extends FunctionPlugin implements Func {argumentType: FunctionArgumentType.RANGE}, ], }, + 'INTERCEPT': { + method: 'intercept', + enableArrayArithmeticForArguments: true, + parameters: [ + {argumentType: FunctionArgumentType.RANGE}, + {argumentType: FunctionArgumentType.RANGE}, + ], + }, + 'FORECAST.LINEAR': { + method: 'forecastLinear', + enableArrayArithmeticForArguments: true, + parameters: [ + {argumentType: FunctionArgumentType.SCALAR}, + {argumentType: FunctionArgumentType.RANGE}, + {argumentType: FunctionArgumentType.RANGE}, + ], + }, 'CHISQ.TEST': { method: 'chisqtest', parameters: [ @@ -163,6 +180,7 @@ export class StatisticalAggregationPlugin extends FunctionPlugin implements Func COVARIANCEP: 'COVARIANCE.P', COVARIANCES: 'COVARIANCE.S', SKEWP: 'SKEW.P', + FORECAST: 'FORECAST.LINEAR', } public avedev(ast: ProcedureAst, state: InterpreterState): InterpreterValue { @@ -407,6 +425,37 @@ export class StatisticalAggregationPlugin extends FunctionPlugin implements Func }) } + /** + * Corresponds to INTERCEPT(known_y, known_x). + * + * Returns the intercept of the least-squares line through the numeric pairs of `known_y` and `known_x`. + */ + public intercept(ast: ProcedureAst, state: InterpreterState): InterpreterValue { + return this.runFunction(ast.args, state, this.metadata('INTERCEPT'), + (knownY: SimpleRangeValue, knownX: SimpleRangeValue) => { + const fit = simpleLinearFit(knownY, knownX) + return fit instanceof CellError ? fit : fit.intercept + }) + } + + /** + * Corresponds to FORECAST.LINEAR(x, known_y, known_x), also available as FORECAST. + * + * Returns the value at `x` of the least-squares line through the numeric pairs of `known_y` and `known_x`. + */ + public forecastLinear(ast: ProcedureAst, state: InterpreterState): InterpreterValue { + return this.runFunction(ast.args, state, this.metadata('FORECAST.LINEAR'), + (x: InternalScalarValue, knownY: SimpleRangeValue, knownX: SimpleRangeValue) => { + // Excel coerces a blank cell, a boolean and numeric text in x, but not an empty string. + const coercedX = x === '' ? new CellError(ErrorType.VALUE, ErrorMessage.NumberCoercion) : this.coerceScalarToNumberOrError(x) + if (coercedX instanceof CellError) { + return coercedX + } + const fit = simpleLinearFit(knownY, knownX) + return fit instanceof CellError ? fit : fit.intercept + fit.slope * getRawValue(coercedX) + }) + } + public chisqtest(ast: ProcedureAst, state: InterpreterState): InterpreterValue { return this.runFunction(ast.args, state, this.metadata('CHISQ.TEST'), (dataX: SimpleRangeValue, dataY: SimpleRangeValue) => { @@ -571,3 +620,34 @@ function parseTwoArrays(dataX: SimpleRangeValue, dataY: SimpleRangeValue): CellE } return [arrX, arrY] } + +/** + * Fits the least-squares line y = intercept + slope·x through the pairs of `knownY` and `knownX` in which both + * values are numbers. The ranges need the same number of cells; the first error met in a pair is returned. + * + * The means and sums are plain two-pass sums, which reproduce Excel's last digits more closely than jStat's + * `covariance` and `sumsqerr`. O(n). + */ +function simpleLinearFit(knownY: SimpleRangeValue, knownX: SimpleRangeValue): CellError | { slope: number, intercept: number } { + if (knownY.numberOfElements() !== knownX.numberOfElements()) { + return new CellError(ErrorType.NA, ErrorMessage.EqualLength) + } + const pairs = parseTwoArrays(knownY, knownX) + if (pairs instanceof CellError) { + return pairs + } + const [ys, xs] = pairs + const n = xs.length + if (n <= 1) { + return new CellError(ErrorType.DIV_BY_ZERO, ErrorMessage.TwoValues) + } + const meanX = xs.reduce((sum, x) => sum + x, 0) / n + const meanY = ys.reduce((sum, y) => sum + y, 0) / n + const sumSquaresX = xs.reduce((sum, x) => sum + (x - meanX) * (x - meanX), 0) + if (sumSquaresX === 0) { + return new CellError(ErrorType.DIV_BY_ZERO, ErrorMessage.ZeroVariance) + } + const sumProducts = xs.reduce((sum, x, i) => sum + (x - meanX) * (ys[i] - meanY), 0) + const slope = sumProducts / sumSquaresX + return {slope, intercept: meanY - slope * meanX} +} From ac2947b84921280e6549370f71134fcf410ec0b0 Mon Sep 17 00:00:00 2001 From: tobiadefami Date: Thu, 8 Oct 2026 16:12:52 +0100 Subject: [PATCH 08/11] feat(HF-414): add TREND, GROWTH and LOGEST Add TREND, GROWTH and LOGEST to RegressionPlugin, built on LINEST's shape detection and fitLinearRegression. - TREND returns the values of the least-squares fit at new_x (or at known_x when new_x is omitted). Its result size is predicted by the new trendArraySize: the shape of new_x for one predictor, one value per row or column of new_x for several. - GROWTH fits ln y with the same method and evaluates b * m^x in Microsoft Excel's product form, so it overflows and underflows where Excel does. Non-positive known_y values return #NUM!. - LOGEST returns LINEST's layout for the fit to ln y, with the first row exponentiated. It shares LINEST's size method and, like LINEST, requires a constant stats argument. As in Excel, any non-number in known_y or known_x returns #VALUE! (no pairs are skipped), dimension mismatches return #REF!, and const and stats treat a blank cell as FALSE and text as #VALUE!. An error read from a cell reference in const or stats returns #VALUE!; an error written in the formula propagates. New error messages: PositiveValues, StaticResultSize(name) and StaticStats(name). Includes the catalogue entries, the names in all language packs, list-of-differences and known-limitations sections, and the changelog entry. Co-Authored-By: Claude Opus 5.5 --- CHANGELOG.md | 1 + docs/guide/known-limitations.md | 6 + docs/guide/list-of-differences.md | 8 + src/error-message.ts | 3 + src/i18n/languages/csCZ.ts | 3 + src/i18n/languages/daDK.ts | 3 + src/i18n/languages/deDE.ts | 3 + src/i18n/languages/enGB.ts | 3 + src/i18n/languages/esES.ts | 3 + src/i18n/languages/fiFI.ts | 3 + src/i18n/languages/frFR.ts | 3 + src/i18n/languages/huHU.ts | 3 + src/i18n/languages/idID.ts | 3 + src/i18n/languages/itIT.ts | 3 + src/i18n/languages/nbNO.ts | 3 + src/i18n/languages/nlNL.ts | 3 + src/i18n/languages/plPL.ts | 3 + src/i18n/languages/ptPT.ts | 3 + src/i18n/languages/ruRU.ts | 3 + src/i18n/languages/svSE.ts | 3 + src/i18n/languages/trTR.ts | 3 + .../categories/statistical.ts | 36 +++ src/interpreter/plugin/RegressionPlugin.ts | 251 ++++++++++++++++++ 23 files changed, 356 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index da026b8a4e..d6c234cbf9 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -11,6 +11,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/), - Added `LINEST` for simple and multiple linear regression, with optional intercept and regression statistics. The `stats` argument must be constant because result dimensions are determined before evaluation. [#1769](https://github.com/handsontable/hyperformula/pull/1769) - Added support for the new license key format. A proprietary key can now grant a subset of the library: a function your key does not include evaluates to a `#LIC!` error, and the matching parts of the API throw a `LicenseCapabilityMissingError`. `getAvailableFunctions()` and `getFunctionDetails()` describe only the functions your key includes. [#1728](https://github.com/handsontable/hyperformula/pull/1728) +- Added the regression functions `FORECAST.LINEAR` (also available as `FORECAST`), `INTERCEPT`, `TREND`, `GROWTH`, and `LOGEST`. `LOGEST`, like `LINEST`, requires a constant `stats` argument. ### Changed diff --git a/docs/guide/known-limitations.md b/docs/guide/known-limitations.md index 8fd335900a..157da3f7b4 100644 --- a/docs/guide/known-limitations.md +++ b/docs/guide/known-limitations.md @@ -71,6 +71,12 @@ a circular reference. * Very large or very small input magnitudes can reduce numerical accuracy. Intermediate calculations, such as residual sums of squares, can still overflow or underflow, producing `#NUM!` or inaccurate statistics. Rescale inputs to more moderate units where possible. See [LINEST numerical differences](list-of-differences.md#linest) for compatibility considerations. +### TREND, GROWTH and LOGEST functions + +* `TREND` and `GROWTH` determine their result size before evaluation: from `new_x`, or from `known_y` when `new_x` is omitted. If an input expression returns a different size at evaluation, they return `#VALUE!`. + +* `LOGEST` returns the same layout as `LINEST`, and the `LINEST` limitations above apply to it: `stats` must be a constant, and the result width is determined from the predictor dimensions before evaluation. + ### OFFSET function HyperFormula resolves the OFFSET function at parse time rather than during evaluation. The parser inspects the arguments and rewrites the expression into a plain cell reference or range. This keeps the dependency graph accurate but imposes several restrictions. diff --git a/docs/guide/list-of-differences.md b/docs/guide/list-of-differences.md index e4a3b28fdc..96e85626ef 100644 --- a/docs/guide/list-of-differences.md +++ b/docs/guide/list-of-differences.md @@ -42,6 +42,14 @@ Numerical results can differ for nearly dependent predictors and nearly perfect Very large or very small input scales can also cause substantial differences in coefficients and statistics, including for a single predictor with a non-perfect fit. HyperFormula does not reproduce Excel's loss of predictors or zero standard errors observed at extreme scales. Rescaling inputs to more moderate units can reduce numerical errors in both engines. +## TREND, GROWTH and LOGEST + +Like `LINEST`, these functions determine their result size before evaluation. `LOGEST` requires a constant `stats` argument, as `LINEST` does: `=LOGEST(A1:A10,B1:B10,TRUE(),D1)` returns `#VALUE!`. The numerical differences described for `LINEST` also apply to the statistics returned by `LOGEST`, which are those of the fit to the natural logarithm of `known_y`. + +An array passed as `const` (or as `stats` in `LOGEST`) is not evaluated element by element. Microsoft Excel returns one result for each element, for example for `=TREND(A1:A6,B1:B6,C1:C3,{TRUE(),FALSE()})`; HyperFormula uses the first element as `const`, and returns `#VALUE!` for an array `stats`. + +A single cell that holds an error, passed as `known_x` or `new_x`, returns that error before the other arguments are checked. Microsoft Excel returns `#VALUE!`. For example, `=TREND(A1:A6,B1:B6,C1)` where C1 contains `=1/0` returns `#DIV/0!` in HyperFormula. An error inside a range of several cells returns `#VALUE!`, as in Microsoft Excel. + ## General functionalities | Functionality | Examples | HyperFormula | Google Sheets | Microsoft Excel | diff --git a/src/error-message.ts b/src/error-message.ts index cff49fcea8..a469b5b9e0 100644 --- a/src/error-message.ts +++ b/src/error-message.ts @@ -9,6 +9,7 @@ export class ErrorMessage { public static LinestStaticSize = 'LINEST requires input dimensions that determine a fixed result size.' public static LinestStaticStats = 'LINEST requires a constant stats argument to determine its result size.' + public static PositiveValues = 'Values need to be positive.' public static ZeroVariance = 'Values cannot all be equal.' public static DistinctSigns = 'Distinct signs.' public static WrongArgNumber = 'Wrong number of arguments.' @@ -83,4 +84,6 @@ export class ErrorMessage { public static NamedExpressionName = (arg: string) => `Named expression ${arg} not recognized.` public static LicenseKey = (arg: string) => `License key is ${arg}.` public static LicenseCapability = (functionName: string) => `Function ${functionName} is not included in your license.` + public static StaticResultSize = (functionName: string) => `${functionName} requires input dimensions that determine a fixed result size.` + public static StaticStats = (functionName: string) => `${functionName} requires a constant stats argument to determine its result size.` } diff --git a/src/i18n/languages/csCZ.ts b/src/i18n/languages/csCZ.ts index 30c59eca2b..4d0e05ef2f 100644 --- a/src/i18n/languages/csCZ.ts +++ b/src/i18n/languages/csCZ.ts @@ -144,6 +144,9 @@ const dictionary: RawTranslationPackage = { LEFT: 'ZLEVA', LEN: 'DÉLKA', LINEST: 'LINREGRESE', + LOGEST: 'LOGLINREGRESE', + TREND: 'LINTREND', + GROWTH: 'LOGLINTREND', LN: 'LN', LOG10: 'LOG', LOG: 'LOGZ', diff --git a/src/i18n/languages/daDK.ts b/src/i18n/languages/daDK.ts index 0fb1f76d54..3672b708f3 100644 --- a/src/i18n/languages/daDK.ts +++ b/src/i18n/languages/daDK.ts @@ -144,6 +144,9 @@ const dictionary: RawTranslationPackage = { LEFT: 'VENSTRE', LEN: 'LÆNGDE', LINEST: 'LINREGR', + LOGEST: 'LOGREGR', + TREND: 'TENDENS', + GROWTH: 'FORØGELSE', LN: 'LN', LOG10: 'LOG10', LOG: 'LOG', diff --git a/src/i18n/languages/deDE.ts b/src/i18n/languages/deDE.ts index e824074cdb..cc7d06d984 100644 --- a/src/i18n/languages/deDE.ts +++ b/src/i18n/languages/deDE.ts @@ -144,6 +144,9 @@ const dictionary: RawTranslationPackage = { LEFT: 'LINKS', LEN: 'LÄNGE', LINEST: 'RGP', + LOGEST: 'RKP', + TREND: 'TREND', + GROWTH: 'VARIATION', LN: 'LN', LOG10: 'LOG10', LOG: 'LOG', diff --git a/src/i18n/languages/enGB.ts b/src/i18n/languages/enGB.ts index 8216946733..7dfaf96210 100644 --- a/src/i18n/languages/enGB.ts +++ b/src/i18n/languages/enGB.ts @@ -145,6 +145,9 @@ const dictionary: RawTranslationPackage = { LEFT: 'LEFT', LEN: 'LEN', LINEST: 'LINEST', + LOGEST: 'LOGEST', + TREND: 'TREND', + GROWTH: 'GROWTH', LN: 'LN', LOG10: 'LOG10', LOG: 'LOG', diff --git a/src/i18n/languages/esES.ts b/src/i18n/languages/esES.ts index 875ebca977..1c8fc3360b 100644 --- a/src/i18n/languages/esES.ts +++ b/src/i18n/languages/esES.ts @@ -144,6 +144,9 @@ export const dictionary: RawTranslationPackage = { LEFT: 'IZQUIERDA', LEN: 'LARGO', LINEST: 'ESTIMACION.LINEAL', + LOGEST: 'ESTIMACION.LOGARITMICA', + TREND: 'TENDENCIA', + GROWTH: 'CRECIMIENTO', LN: 'LN', LOG10: 'LOG10', LOG: 'LOG', diff --git a/src/i18n/languages/fiFI.ts b/src/i18n/languages/fiFI.ts index 57f3407790..19cfd05dd5 100644 --- a/src/i18n/languages/fiFI.ts +++ b/src/i18n/languages/fiFI.ts @@ -144,6 +144,9 @@ const dictionary: RawTranslationPackage = { LEFT: 'VASEN', LEN: 'PITUUS', LINEST: 'LINREGR', + LOGEST: 'LOGREGR', + TREND: 'SUUNTAUS', + GROWTH: 'KASVU', LN: 'LUONNLOG', LOG10: 'LOG10', LOG: 'LOG', diff --git a/src/i18n/languages/frFR.ts b/src/i18n/languages/frFR.ts index 84ae293469..c6c14cd275 100644 --- a/src/i18n/languages/frFR.ts +++ b/src/i18n/languages/frFR.ts @@ -144,6 +144,9 @@ const dictionary: RawTranslationPackage = { LEFT: 'GAUCHE', LEN: 'NBCAR', LINEST: 'DROITEREG', + LOGEST: 'LOGREG', + TREND: 'TENDANCE', + GROWTH: 'CROISSANCE', LN: 'LN', LOG10: 'LOG10', LOG: 'LOG', diff --git a/src/i18n/languages/huHU.ts b/src/i18n/languages/huHU.ts index 0914d29b06..e7fee15732 100644 --- a/src/i18n/languages/huHU.ts +++ b/src/i18n/languages/huHU.ts @@ -144,6 +144,9 @@ const dictionary: RawTranslationPackage = { LEFT: 'BAL', LEN: 'HOSSZ', LINEST: 'LIN.ILL', + LOGEST: 'LOG.ILL', + TREND: 'TREND', + GROWTH: 'NÖV', LN: 'LN', LOG10: 'LOG10', LOG: 'LOG', diff --git a/src/i18n/languages/idID.ts b/src/i18n/languages/idID.ts index cf348b89aa..7d6ae0ef41 100644 --- a/src/i18n/languages/idID.ts +++ b/src/i18n/languages/idID.ts @@ -145,6 +145,9 @@ const dictionary: RawTranslationPackage = { LEFT: 'KIRI', LEN: 'PANJANG', LINEST: 'LINEST', + LOGEST: 'LOGEST', + TREND: 'TREND', + GROWTH: 'GROWTH', LN: 'LN', LOG10: 'LOG10', LOG: 'LOG', diff --git a/src/i18n/languages/itIT.ts b/src/i18n/languages/itIT.ts index 6d39b4ae78..7668333008 100644 --- a/src/i18n/languages/itIT.ts +++ b/src/i18n/languages/itIT.ts @@ -144,6 +144,9 @@ const dictionary: RawTranslationPackage = { LEFT: 'SINISTRA', LEN: 'LUNGHEZZA', LINEST: 'REGR.LIN', + LOGEST: 'REGR.LOG', + TREND: 'TENDENZA', + GROWTH: 'CRESCITA', LN: 'LN', LOG10: 'LOG10', LOG: 'LOG', diff --git a/src/i18n/languages/nbNO.ts b/src/i18n/languages/nbNO.ts index c6daf97ba2..a105d266b0 100644 --- a/src/i18n/languages/nbNO.ts +++ b/src/i18n/languages/nbNO.ts @@ -144,6 +144,9 @@ const dictionary: RawTranslationPackage = { LEFT: 'VENSTRE', LEN: 'LENGDE', LINEST: 'RETTLINJE', + LOGEST: 'KURVE', + TREND: 'TREND', + GROWTH: 'VEKST', LN: 'LN', LOG10: 'LOG10', LOG: 'LOG', diff --git a/src/i18n/languages/nlNL.ts b/src/i18n/languages/nlNL.ts index 04e9ef4614..16194549c7 100644 --- a/src/i18n/languages/nlNL.ts +++ b/src/i18n/languages/nlNL.ts @@ -144,6 +144,9 @@ const dictionary: RawTranslationPackage = { LEFT: 'LINKS', LEN: 'PITUUS', LINEST: 'LIJNSCH', + LOGEST: 'LOGSCH', + TREND: 'TREND', + GROWTH: 'GROEI', LN: 'LN', LOG10: 'LOG10', LOG: 'LOG', diff --git a/src/i18n/languages/plPL.ts b/src/i18n/languages/plPL.ts index 71fa29bea0..5d9971290b 100644 --- a/src/i18n/languages/plPL.ts +++ b/src/i18n/languages/plPL.ts @@ -144,6 +144,9 @@ const dictionary: RawTranslationPackage = { LEFT: 'LEWY', LEN: 'DŁ', LINEST: 'REGLINP', + LOGEST: 'REGEXPP', + TREND: 'REGLINW', + GROWTH: 'REGEXPW', LN: 'LN', LOG10: 'LOG10', LOG: 'LOG', diff --git a/src/i18n/languages/ptPT.ts b/src/i18n/languages/ptPT.ts index 5d9a3800db..cbceda5aae 100644 --- a/src/i18n/languages/ptPT.ts +++ b/src/i18n/languages/ptPT.ts @@ -144,6 +144,9 @@ const dictionary: RawTranslationPackage = { LEFT: 'ESQUERDA', LEN: 'NÚM.CARACT', LINEST: 'PROJ.LIN', + LOGEST: 'PROJ.LOG', + TREND: 'TENDÊNCIA', + GROWTH: 'CRESCIMENTO', LN: 'LN', LOG10: 'LOG10', LOG: 'LOG', diff --git a/src/i18n/languages/ruRU.ts b/src/i18n/languages/ruRU.ts index 6b2bd491a7..4c717a8516 100644 --- a/src/i18n/languages/ruRU.ts +++ b/src/i18n/languages/ruRU.ts @@ -144,6 +144,9 @@ const dictionary: RawTranslationPackage = { LEFT: 'ЛЕВСИМВ', LEN: 'ДЛСТР', LINEST: 'ЛИНЕЙН', + LOGEST: 'ЛГРФПРИБЛ', + TREND: 'ТЕНДЕНЦИЯ', + GROWTH: 'РОСТ', LN: 'LN', LOG10: 'LOG10', LOG: 'LOG', diff --git a/src/i18n/languages/svSE.ts b/src/i18n/languages/svSE.ts index 2b2d88ceb4..ab3a603526 100644 --- a/src/i18n/languages/svSE.ts +++ b/src/i18n/languages/svSE.ts @@ -144,6 +144,9 @@ const dictionary: RawTranslationPackage = { LEFT: 'VÄNSTER', LEN: 'LÄNGD', LINEST: 'REGR', + LOGEST: 'EXPREGR', + TREND: 'TREND', + GROWTH: 'EXPTREND', LN: 'LN', LOG10: 'LOG10', LOG: 'LOG', diff --git a/src/i18n/languages/trTR.ts b/src/i18n/languages/trTR.ts index c1e1b2e907..a0842d4416 100644 --- a/src/i18n/languages/trTR.ts +++ b/src/i18n/languages/trTR.ts @@ -144,6 +144,9 @@ const dictionary: RawTranslationPackage = { LEFT: 'SOL', LEN: 'UZUNLUK', LINEST: 'DOT', + LOGEST: 'LOT', + TREND: 'EĞİLİM', + GROWTH: 'BÜYÜME', LN: 'LN', LOG10: 'LOG10', LOG: 'LOG', diff --git a/src/interpreter/functionMetadata/categories/statistical.ts b/src/interpreter/functionMetadata/categories/statistical.ts index 9519cf52b4..32600cb194 100644 --- a/src/interpreter/functionMetadata/categories/statistical.ts +++ b/src/interpreter/functionMetadata/categories/statistical.ts @@ -294,6 +294,18 @@ export const STATISTICAL_DOCS: Record = { documentationUrl: 'https://hyperformula.handsontable.com/docs/guide/built-in-functions.html', examples: ['=GEOMEAN(1, 2, 3)', '=GEOMEAN(A1:A10)'], }, + GROWTH: { + category: 'Statistical', + shortDescription: 'Returns values of the exponential curve y = b·m^x fitted by least squares to `known_y`, at the points `new_x`. `known_y` values must be positive.', + parameters: [ + {name: 'known_y', description: 'A numeric range of observed dependent values.'}, + {name: 'known_x', description: 'Optional numeric predictors. If omitted, uses sequential values starting at 1 with the shape of `known_y`.'}, + {name: 'new_x', description: 'Optional points at which to predict values, with one column (or row) per predictor. If omitted, uses `known_x`.'}, + {name: 'const', description: 'Whether to fit the constant b. Defaults to TRUE; FALSE sets b to 1.'}, + ], + documentationUrl: 'https://hyperformula.handsontable.com/docs/guide/built-in-functions.html', + examples: ['=GROWTH(A1:A6, B1:B6, B7:B9)', '=GROWTH(A1:A6, B1:C6, B7:C9, FALSE())'], + }, HARMEAN: { category: 'Statistical', shortDescription: 'Returns the harmonic average.', @@ -337,6 +349,18 @@ export const STATISTICAL_DOCS: Record = { documentationUrl: 'https://hyperformula.handsontable.com/docs/guide/built-in-functions.html', examples: ['=LINEST(A1:A10, B1:C10)', '=LINEST(A1:A10, B1:C10, TRUE(), TRUE())'], }, + LOGEST: { + category: 'Statistical', + shortDescription: 'Returns the bases and constant of the exponential curve y = b·m1^x1·…·mk^xk fitted by least squares, and optional statistics. `known_y` values must be positive.', + parameters: [ + {name: 'known_y', description: 'A numeric range of observed dependent values.'}, + {name: 'known_x', description: 'Optional numeric predictors. If omitted, uses sequential values starting at 1 with the shape of `known_y`.'}, + {name: 'const', description: 'Whether to fit the constant b. Defaults to TRUE; FALSE sets b to 1.'}, + {name: 'stats', description: 'Whether to return five rows including regression statistics of the fit to the natural logarithm of `known_y`. Defaults to FALSE. Must be a constant; cell references and computed expressions are unsupported.'}, + ], + documentationUrl: 'https://hyperformula.handsontable.com/docs/guide/built-in-functions.html', + examples: ['=LOGEST(A1:A10, B1:C10)', '=LOGEST(A1:A10, B1:C10, TRUE(), TRUE())'], + }, 'LOGNORM.DIST': { category: 'Statistical', shortDescription: 'Returns density of lognormal distribution.', @@ -603,6 +627,18 @@ export const STATISTICAL_DOCS: Record = { documentationUrl: 'https://hyperformula.handsontable.com/docs/guide/built-in-functions.html', examples: ['=TDIST(1, 10, 1)', '=TDIST(1, 10, 2)'], }, + TREND: { + category: 'Statistical', + shortDescription: 'Returns values of the least-squares line fitted to `known_y`, at the points `new_x`.', + parameters: [ + {name: 'known_y', description: 'A numeric range of observed dependent values.'}, + {name: 'known_x', description: 'Optional numeric predictors. If omitted, uses sequential values starting at 1 with the shape of `known_y`.'}, + {name: 'new_x', description: 'Optional points at which to predict values, with one column (or row) per predictor. If omitted, uses `known_x`.'}, + {name: 'const', description: 'Whether to fit an intercept. Defaults to TRUE; FALSE fits through zero.'}, + ], + documentationUrl: 'https://hyperformula.handsontable.com/docs/guide/built-in-functions.html', + examples: ['=TREND(A1:A6, B1:B6, B7:B9)', '=TREND(A1:A6, B1:C6, B7:C9, FALSE())'], + }, 'VAR.P': { category: 'Statistical', shortDescription: 'Returns variance of a population.', diff --git a/src/interpreter/plugin/RegressionPlugin.ts b/src/interpreter/plugin/RegressionPlugin.ts index 9df3160c1e..c3409bdf8b 100644 --- a/src/interpreter/plugin/RegressionPlugin.ts +++ b/src/interpreter/plugin/RegressionPlugin.ts @@ -126,6 +126,84 @@ function regressionOutput(fit: LinearRegressionResult, statistics: boolean, fitI return SimpleRangeValue.onlyValues(result) } +/** The validated values of a regression: one observation and one predictor row per data point. */ +interface RegressionData { + observations: number[], + predictors: number[][], +} + +/** Tells whether an optional argument was left out or passed empty, as in `TREND(y,,new_x)`. */ +function isOmittedArgument(ast: ProcedureAst, index: number): boolean { + return ast.args.length <= index || ast.args[index].type === AstNodeType.EMPTY +} + +/** + * Reads the `const` or `stats` option of TREND, GROWTH and LOGEST as Excel does: a blank cell is FALSE, numbers and + * booleans are coerced, and text (also numeric text and an empty string) is #VALUE!. An error read from a cell + * reference is #VALUE!; any other error, such as `NA()` written in the formula, is returned as is. + */ +function regressionOption(value: InternalScalarValue, ast: Ast | undefined): boolean | CellError { + if (value instanceof CellError) { + return ast?.type === AstNodeType.CELL_REFERENCE ? new CellError(ErrorType.VALUE, ErrorMessage.WrongType) : value + } + const option = typeof value === 'string' ? undefined : coerceScalarToBoolean(value) + return typeof option === 'boolean' ? option : new CellError(ErrorType.VALUE, ErrorMessage.WrongType) +} + +/** + * Checks that every known value is a number and arranges the predictors one row per observation. + * Omitted `known_x` values are 1, 2, 3, … in the order of `known_y`. + */ +function regressionData(knownY: SimpleRangeValue, knownX: SimpleRangeValue | undefined, shape: RegressionShape): RegressionData | CellError { + const yValues = Array.from(knownY.valuesFromTopLeftCorner()) + const xValues = knownX === undefined ? yValues.map((_, i) => i + 1) : Array.from(knownX.valuesFromTopLeftCorner()) + if (!yValues.every(isExtendedNumber) || !xValues.every(isExtendedNumber)) { + return new CellError(ErrorType.VALUE, ErrorMessage.NumberRange) + } + const observations = yValues.map(getRawValue) + return {observations, predictors: buildPredictorRows(xValues.map(getRawValue), observations.length, shape)} +} + +/** + * Returns the size of a TREND or GROWTH result for `new_x` values of the given size, or undefined when `new_x` + * does not have one column (columns orientation) or one row (rows orientation) per predictor. + * One predictor: one value per `new_x` cell, in the shape of `new_x`. + */ +function predictionSize(shape: RegressionShape, newX: ArraySize): ArraySize | undefined { + if (shape.orientation === 'paired') { + return new ArraySize(newX.width, newX.height) + } + if (shape.orientation === 'columns') { + return newX.width === shape.predictorCount ? new ArraySize(1, newX.height) : undefined + } + return newX.height === shape.predictorCount ? new ArraySize(newX.width, 1) : undefined +} + +/** Arranges `new_x` values one predictor row per predicted point, in the result's row-major order. */ +function buildNewPointRows(newXValues: number[], newX: ArraySize, shape: RegressionShape): number[][] { + if (shape.orientation === 'paired') { + return newXValues.map(value => [value]) + } + if (shape.orientation === 'columns') { + return Array.from({length: newX.height}, (_, row) => newXValues.slice(row * newX.width, (row + 1) * newX.width)) + } + return Array.from({length: newX.width}, (_, column) => + Array.from({length: newX.height}, (_, row) => newXValues[row * newX.width + column])) +} + +/** Evaluates the fitted line b + m1·x1 + … + mk·xk at one point. */ +function linearPrediction(fit: LinearRegressionResult, point: number[]): number { + return point.reduce((sum, value, i) => sum + fit.coefficients[i] * value, fit.intercept) +} + +/** + * Evaluates the fitted exponential curve b · m1^x1 · … · mk^xk at one point, where b and m are the exponentials of + * the coefficients fitted to ln y. Excel's product form is kept: it overflows and underflows where Excel does. + */ +function exponentialPrediction(fit: LinearRegressionResult, point: number[]): number { + return point.reduce((product, value, i) => product * Math.exp(fit.coefficients[i]) ** value, Math.exp(fit.intercept)) +} + /** Implements linear regression with statically predictable result dimensions. */ export class RegressionPlugin extends FunctionPlugin implements FunctionPluginTypecheck { public static implementedFunctions: ImplementedFunctions = { @@ -141,6 +219,43 @@ export class RegressionPlugin extends FunctionPlugin implements FunctionPluginTy {argumentType: FunctionArgumentType.SCALAR, defaultValue: false, emptyAsDefault: true}, ], }, + 'LOGEST': { + method: 'logest', + // LOGEST shares LINEST's size method because both return the same coefficient and statistics layout. + sizeOfResultArrayMethod: 'linestArraySize', + vectorizationForbidden: true, + enableArrayArithmeticForArguments: true, + parameters: [ + {argumentType: FunctionArgumentType.RANGE}, + {argumentType: FunctionArgumentType.ANY, defaultValue: EmptyValue}, + {argumentType: FunctionArgumentType.SCALAR, defaultValue: true, emptyAsDefault: true}, + {argumentType: FunctionArgumentType.SCALAR, defaultValue: false, emptyAsDefault: true}, + ], + }, + 'TREND': { + method: 'trend', + sizeOfResultArrayMethod: 'trendArraySize', + vectorizationForbidden: true, + enableArrayArithmeticForArguments: true, + parameters: [ + {argumentType: FunctionArgumentType.RANGE}, + {argumentType: FunctionArgumentType.ANY, defaultValue: EmptyValue}, + {argumentType: FunctionArgumentType.ANY, defaultValue: EmptyValue}, + {argumentType: FunctionArgumentType.SCALAR, defaultValue: true, emptyAsDefault: true}, + ], + }, + 'GROWTH': { + method: 'growth', + sizeOfResultArrayMethod: 'trendArraySize', + vectorizationForbidden: true, + enableArrayArithmeticForArguments: true, + parameters: [ + {argumentType: FunctionArgumentType.RANGE}, + {argumentType: FunctionArgumentType.ANY, defaultValue: EmptyValue}, + {argumentType: FunctionArgumentType.ANY, defaultValue: EmptyValue}, + {argumentType: FunctionArgumentType.SCALAR, defaultValue: true, emptyAsDefault: true}, + ], + }, } /** @@ -213,4 +328,140 @@ export class RegressionPlugin extends FunctionPlugin implements FunctionPluginTy } return new ArraySize(shape.predictorCount + 1, statistics ? 5 : 1) } + + /** + * Evaluates LOGEST(known_y, [known_x], [const], [stats]): LINEST fitted to ln y, with the first row + * exponentiated into the bases m1 … mk and the constant b of y = b · m1^x1 · … · mk^xk. + * The statistics rows are LINEST's statistics of the ln y fit. As in LINEST, `stats` must be a constant. + */ + public logest(ast: ProcedureAst, state: InterpreterState): InterpreterValue { + return this.runFunction(ast.args, state, this.metadata('LOGEST'), + (knownY: SimpleRangeValue, knownXArg: InterpreterValue, constArg: InternalScalarValue, statsArg: InternalScalarValue) => { + const fitIntercept = regressionOption(constArg, ast.args[2]) + if (fitIntercept instanceof CellError) { + return fitIntercept + } + const statistics = regressionOption(statsArg, ast.args[3]) + if (statistics instanceof CellError) { + return statistics + } + if (staticBoolean(ast.args[3]) === undefined) { + return new CellError(ErrorType.VALUE, ErrorMessage.StaticStats('LOGEST')) + } + const knownX = isOmittedArgument(ast, 1) ? undefined : coerceToRange(knownXArg) + const shape = regressionShape(knownY.size, knownX?.size) + if (shape === undefined) { + return new CellError(ErrorType.REF, ErrorMessage.ArrayDimensions) + } + if (shape.predictorCount + 1 > this.config.maxColumns) { + return new CellError(ErrorType.VALUE, ErrorMessage.ValueLarge) + } + if (this.linestArraySize(ast, state).width !== shape.predictorCount + 1) { + return new CellError(ErrorType.VALUE, ErrorMessage.StaticResultSize('LOGEST')) + } + const data = regressionData(knownY, knownX, shape) + if (data instanceof CellError) { + return data + } + if (data.observations.some(value => value <= 0)) { + return new CellError(ErrorType.NUM, ErrorMessage.PositiveValues) + } + const fit = fitLinearRegression(data.predictors, data.observations.map(Math.log), fitIntercept, statistics) + const exponentiated = {...fit, coefficients: fit.coefficients.map(Math.exp), intercept: Math.exp(fit.intercept)} + return regressionOutput(exponentiated, statistics, fitIntercept) + }) + } + + /** + * Evaluates TREND(known_y, [known_x], [new_x], [const]): the values of the least-squares line at `new_x`, + * or at `known_x` when `new_x` is omitted. + */ + public trend(ast: ProcedureAst, state: InterpreterState): InterpreterValue { + return this.runFunction(ast.args, state, this.metadata('TREND'), + (knownY: SimpleRangeValue, knownX: InterpreterValue, newX: InterpreterValue, constArg: InternalScalarValue) => + this.predict(ast, state, 'TREND', knownY, knownX, newX, constArg)) + } + + /** + * Evaluates GROWTH(known_y, [known_x], [new_x], [const]): the values of the exponential curve fitted to `known_y` + * at `new_x`, or at `known_x` when `new_x` is omitted. Every `known_y` value must be positive. + */ + public growth(ast: ProcedureAst, state: InterpreterState): InterpreterValue { + return this.runFunction(ast.args, state, this.metadata('GROWTH'), + (knownY: SimpleRangeValue, knownX: InterpreterValue, newX: InterpreterValue, constArg: InternalScalarValue) => + this.predict(ast, state, 'GROWTH', knownY, knownX, newX, constArg)) + } + + /** + * Predicts the result size of TREND and GROWTH: the size of `known_y` when `new_x` is omitted, otherwise one value + * per `new_x` point (see `predictionSize`). + */ + public trendArraySize(ast: ProcedureAst, state: InterpreterState): ArraySize { + if (ast.args.length < 1 || ast.args.length > 4) { + return ArraySize.error() + } + const arrayState = new InterpreterState(state.formulaAddress, true) + const y = this.arraySizeForAst(ast.args[0], arrayState) + const x = isOmittedArgument(ast, 1) ? undefined : this.arraySizeForAst(ast.args[1], arrayState) + const shape = regressionShape(y, x) + if (shape === undefined) { + return ArraySize.error() + } + if (isOmittedArgument(ast, 2)) { + return new ArraySize(y.width, y.height) + } + return predictionSize(shape, this.arraySizeForAst(ast.args[2], arrayState)) ?? ArraySize.error() + } + + /** + * Shared body of TREND and GROWTH. Errors follow Excel's precedence: a bad `const`, then incompatible + * dimensions (#REF!), then non-numeric values (#VALUE!), then (GROWTH) a non-positive `known_y` (#NUM!). + * GROWTH fits ln y and evaluates the exponential curve. O(n·k² + m·k) for n observations, k predictors + * and m predicted points. + */ + private predict( + ast: ProcedureAst, + state: InterpreterState, + functionName: 'TREND' | 'GROWTH', + knownY: SimpleRangeValue, + knownXArg: InterpreterValue, + newXArg: InterpreterValue, + constArg: InternalScalarValue, + ): InterpreterValue { + const fitIntercept = regressionOption(constArg, ast.args[3]) + if (fitIntercept instanceof CellError) { + return fitIntercept + } + const knownX = isOmittedArgument(ast, 1) ? undefined : coerceToRange(knownXArg) + const newX = isOmittedArgument(ast, 2) ? undefined : coerceToRange(newXArg) + const shape = regressionShape(knownY.size, knownX?.size) + if (shape === undefined) { + return new CellError(ErrorType.REF, ErrorMessage.ArrayDimensions) + } + const resultSize = newX === undefined ? knownY.size : predictionSize(shape, newX.size) + if (resultSize === undefined) { + return new CellError(ErrorType.REF, ErrorMessage.ArrayDimensions) + } + const data = regressionData(knownY, knownX, shape) + if (data instanceof CellError) { + return data + } + const newXValues = newX === undefined ? [] : Array.from(newX.valuesFromTopLeftCorner()) + if (!newXValues.every(isExtendedNumber)) { + return new CellError(ErrorType.VALUE, ErrorMessage.NumberRange) + } + const exponential = functionName === 'GROWTH' + if (exponential && data.observations.some(value => value <= 0)) { + return new CellError(ErrorType.NUM, ErrorMessage.PositiveValues) + } + const predictedSize = this.trendArraySize(ast, state) + if (predictedSize.width !== resultSize.width || predictedSize.height !== resultSize.height) { + return new CellError(ErrorType.VALUE, ErrorMessage.StaticResultSize(functionName)) + } + const fit = fitLinearRegression(data.predictors, exponential ? data.observations.map(Math.log) : data.observations, fitIntercept, false) + const points = newX === undefined ? data.predictors : buildNewPointRows(newXValues.map(getRawValue), newX.size, shape) + const values = points.map(point => finiteValue(exponential ? exponentialPrediction(fit, point) : linearPrediction(fit, point))) + return SimpleRangeValue.onlyValues(Array.from({length: resultSize.height}, (_, row) => + values.slice(row * resultSize.width, (row + 1) * resultSize.width))) + } } From d028c67da8e5ed5174b4587baa3b622f43b68d65 Mon Sep 17 00:00:00 2001 From: tobiadefami Date: Thu, 8 Oct 2026 22:30:40 +0100 Subject: [PATCH 09/11] docs(HF-414): link the pull request in the changelog Co-Authored-By: Claude Opus 5.5 --- CHANGELOG.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index d6c234cbf9..ec7e026221 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -11,7 +11,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/), - Added `LINEST` for simple and multiple linear regression, with optional intercept and regression statistics. The `stats` argument must be constant because result dimensions are determined before evaluation. [#1769](https://github.com/handsontable/hyperformula/pull/1769) - Added support for the new license key format. A proprietary key can now grant a subset of the library: a function your key does not include evaluates to a `#LIC!` error, and the matching parts of the API throw a `LicenseCapabilityMissingError`. `getAvailableFunctions()` and `getFunctionDetails()` describe only the functions your key includes. [#1728](https://github.com/handsontable/hyperformula/pull/1728) -- Added the regression functions `FORECAST.LINEAR` (also available as `FORECAST`), `INTERCEPT`, `TREND`, `GROWTH`, and `LOGEST`. `LOGEST`, like `LINEST`, requires a constant `stats` argument. +- Added the regression functions `FORECAST.LINEAR` (also available as `FORECAST`), `INTERCEPT`, `TREND`, `GROWTH`, and `LOGEST`. `LOGEST`, like `LINEST`, requires a constant `stats` argument. [#1802](https://github.com/handsontable/hyperformula/pull/1802) ### Changed From d9fba12f5a9ecd148f21141788cb588db30cc02e Mon Sep 17 00:00:00 2001 From: tobiadefami Date: Thu, 8 Oct 2026 23:50:16 +0100 Subject: [PATCH 10/11] fix(HF-414): accept the text TRUE and FALSE as const and stats TREND, GROWTH and LOGEST returned #VALUE! for the text "TRUE" or "FALSE" as const or stats, while LINEST accepts it and LOGEST's size prediction already treated it as a constant. Read the options the way LINEST does, which matches Microsoft Excel: "TRUE" and "FALSE" in any letter case are booleans; other text, numeric text and an empty string are #VALUE!. An error read from a cell reference is still #VALUE!. Also state in LOGEST's short description that stats must be a constant, with a link to the known limitations. Co-Authored-By: Claude Opus 5.5 --- .../functionMetadata/categories/statistical.ts | 2 +- src/interpreter/plugin/RegressionPlugin.ts | 9 +++++---- 2 files changed, 6 insertions(+), 5 deletions(-) diff --git a/src/interpreter/functionMetadata/categories/statistical.ts b/src/interpreter/functionMetadata/categories/statistical.ts index 32600cb194..5d3be749ae 100644 --- a/src/interpreter/functionMetadata/categories/statistical.ts +++ b/src/interpreter/functionMetadata/categories/statistical.ts @@ -351,7 +351,7 @@ export const STATISTICAL_DOCS: Record = { }, LOGEST: { category: 'Statistical', - shortDescription: 'Returns the bases and constant of the exponential curve y = b·m1^x1·…·mk^xk fitted by least squares, and optional statistics. `known_y` values must be positive.', + shortDescription: 'Returns the bases and constant of the exponential curve y = b·m1^x1·…·mk^xk fitted by least squares, and optional statistics. `known_y` values must be positive.
`stats` must be a constant: a cell reference or a calculated `stats` returns `#VALUE!`; see the [known limitations](https://hyperformula.handsontable.com/docs/guide/known-limitations.html#trend-growth-and-logest-functions).', parameters: [ {name: 'known_y', description: 'A numeric range of observed dependent values.'}, {name: 'known_x', description: 'Optional numeric predictors. If omitted, uses sequential values starting at 1 with the shape of `known_y`.'}, diff --git a/src/interpreter/plugin/RegressionPlugin.ts b/src/interpreter/plugin/RegressionPlugin.ts index c3409bdf8b..fa78e498f9 100644 --- a/src/interpreter/plugin/RegressionPlugin.ts +++ b/src/interpreter/plugin/RegressionPlugin.ts @@ -138,15 +138,16 @@ function isOmittedArgument(ast: ProcedureAst, index: number): boolean { } /** - * Reads the `const` or `stats` option of TREND, GROWTH and LOGEST as Excel does: a blank cell is FALSE, numbers and - * booleans are coerced, and text (also numeric text and an empty string) is #VALUE!. An error read from a cell - * reference is #VALUE!; any other error, such as `NA()` written in the formula, is returned as is. + * Reads the `const` or `stats` option of TREND, GROWTH and LOGEST as Excel does: a blank cell is FALSE, numbers, + * booleans and the text "TRUE" or "FALSE" (in any letter case) are coerced as in LINEST, and other text (also numeric + * text and an empty string) is #VALUE!. An error read from a cell reference is #VALUE!; any other error, such as + * `NA()` written in the formula, is returned as is. */ function regressionOption(value: InternalScalarValue, ast: Ast | undefined): boolean | CellError { if (value instanceof CellError) { return ast?.type === AstNodeType.CELL_REFERENCE ? new CellError(ErrorType.VALUE, ErrorMessage.WrongType) : value } - const option = typeof value === 'string' ? undefined : coerceScalarToBoolean(value) + const option = regressionBoolean(value) return typeof option === 'boolean' ? option : new CellError(ErrorType.VALUE, ErrorMessage.WrongType) } From 130c9dc658d40b4b20e138d8401820f2a70beb2e Mon Sep 17 00:00:00 2001 From: tobiadefami Date: Thu, 8 Oct 2026 23:52:53 +0100 Subject: [PATCH 11/11] feat(HF-414): list the regression functions in the license catalog Add FORECAST.LINEAR, GROWTH, INTERCEPT, LOGEST and TREND to the ungrouped functions of the function capability table, so they are covered by fun:all and by their single-function tokens, like SLOPE and RSQ. FORECAST is an alias and travels with FORECAST.LINEAR. Co-Authored-By: Claude Opus 5.5 --- src/license/functionCapabilities.ts | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/src/license/functionCapabilities.ts b/src/license/functionCapabilities.ts index a26733ec5f..7f20338e33 100644 --- a/src/license/functionCapabilities.ts +++ b/src/license/functionCapabilities.ts @@ -84,18 +84,18 @@ const UNGROUPED_FUNCTIONS = [ 'DEC2BIN', 'DEC2OCT', 'DECIMAL', 'DEGREES', 'DELTA', 'DEVSQ', 'DGET', 'DMAX', 'DMIN', 'DOLLARDE', 'DOLLARFR', 'DPRODUCT', 'DSTDEV', 'DSTDEVP', 'DSUM', 'DVAR', 'DVARP', 'EFFECT', 'ERF', 'ERFC', 'EXPON.DIST', 'F.DIST', 'F.DIST.RT', 'F.INV', 'F.INV.RT', 'F.TEST', 'FACT', 'FACTDOUBLE', 'FISHER', - 'FISHERINV', 'FLOOR.MATH', 'FLOOR.PRECISE', 'FORMULATEXT', 'FVSCHEDULE', 'GAMMA', 'GAMMA.DIST', - 'GAMMA.INV', 'GAMMALN', 'GAUSS', 'GCD', 'GEOMEAN', 'HARMEAN', 'HEX2BIN', 'HEX2OCT', 'HYPGEOM.DIST', + 'FISHERINV', 'FLOOR.MATH', 'FLOOR.PRECISE', 'FORECAST.LINEAR', 'FORMULATEXT', 'FVSCHEDULE', 'GAMMA', 'GAMMA.DIST', + 'GAMMA.INV', 'GAMMALN', 'GAUSS', 'GCD', 'GEOMEAN', 'GROWTH', 'HARMEAN', 'HEX2BIN', 'HEX2OCT', 'HYPGEOM.DIST', 'IMABS', 'IMAGINARY', 'IMARGUMENT', 'IMCONJUGATE', 'IMCOS', 'IMCOSH', 'IMCOT', 'IMCSC', 'IMCSCH', 'IMDIV', 'IMEXP', 'IMLN', 'IMLOG10', 'IMLOG2', 'IMPOWER', 'IMPRODUCT', 'IMREAL', 'IMSEC', 'IMSECH', 'IMSIN', - 'IMSINH', 'IMSQRT', 'IMSUB', 'IMSUM', 'IMTAN', 'INTERVAL', 'ISBINARY', 'ISFORMULA', 'ISNONTEXT', 'ISPMT', - 'ISREF', 'LCM', 'LOG10', 'LOGNORM.DIST', 'LOGNORM.INV', 'MAXA', 'MAXPOOL', 'MEDIANPOOL', 'MINA', 'MIRR', + 'IMSINH', 'IMSQRT', 'IMSUB', 'IMSUM', 'IMTAN', 'INTERCEPT', 'INTERVAL', 'ISBINARY', 'ISFORMULA', 'ISNONTEXT', 'ISPMT', + 'ISREF', 'LCM', 'LOG10', 'LOGEST', 'LOGNORM.DIST', 'LOGNORM.INV', 'MAXA', 'MAXPOOL', 'MEDIANPOOL', 'MINA', 'MIRR', 'MMULT', 'MULTINOMIAL', 'NEGBINOM.DIST', 'NETWORKDAYS.INTL', 'NOMINAL', 'NORM.DIST', 'NORM.INV', 'NORM.S.DIST', 'NORM.S.INV', 'NPER', 'OCT2BIN', 'OCT2DEC', 'OCT2HEX', 'PDURATION', 'PERCENTILE.EXC', 'PHI', 'POISSON.DIST', 'QUARTILE.EXC', 'QUARTILE.INC', 'RADIANS', 'ROMAN', 'RRI', 'RSQ', 'SEC', 'SECH', 'SERIESSUM', 'SHEET', 'SHEETS', 'SINH', 'SKEW', 'SKEW.P', 'SLOPE', 'SPLIT', 'SQRTPI', 'STANDARDIZE', 'STEYX', 'SUMX2MY2', 'SUMX2PY2', 'SYD', 'T.DIST', 'T.DIST.2T', 'T.DIST.RT', 'T.INV', 'T.INV.2T', 'T.TEST', - 'TANH', 'TBILLEQ', 'TBILLPRICE', 'TBILLYIELD', 'TDIST', 'TIMEVALUE', 'UNICODE', 'VARA', 'VARPA', + 'TANH', 'TBILLEQ', 'TBILLPRICE', 'TBILLYIELD', 'TDIST', 'TIMEVALUE', 'TREND', 'UNICODE', 'VARA', 'VARPA', 'WEIBULL.DIST', 'WORKDAY.INTL', 'Z.TEST', ]