diff --git a/CHANGELOG.md b/CHANGELOG.md index 787c642ad9..50e78a865e 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -10,6 +10,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/), ### Added - Added support for the new license key format. A proprietary key can now grant a subset of the library: a function your key does not include evaluates to a `#LIC!` error, and the matching parts of the API throw a `LicenseCapabilityMissingError`. `getAvailableFunctions()` and `getFunctionDetails()` describe only the functions your key includes. [#1728](https://github.com/handsontable/hyperformula/pull/1728) +- Added new functions: `TEXTBEFORE`, `TEXTAFTER`, `TEXTSPLIT`. [#1792](https://github.com/handsontable/hyperformula/pull/1792) ### Changed diff --git a/docs/guide/list-of-differences.md b/docs/guide/list-of-differences.md index a109f82616..3d0e1cca5b 100644 --- a/docs/guide/list-of-differences.md +++ b/docs/guide/list-of-differences.md @@ -127,3 +127,5 @@ A few of the rows above share a root cause worth stating once: - **Rounding toward zero, not down.** `INT` discards the fractional part rather than rounding toward negative infinity, so it differs from Excel and Google Sheets for negative input only. `ROUNDDOWN`/`ROUNDUP` are unaffected — they are defined in terms of zero in all three. - **`ISEVEN`/`ISODD` do not truncate.** They test the remainder of the value as given, so a value with a fractional part returns `FALSE` from *both*. Excel and Google Sheets truncate to an integer first, so exactly one of the two is always `TRUE`. - **`CEILING.MATH`/`FLOOR.MATH` honour only `mode` = 1.** Excel and Google Sheets switch the negative-number rounding direction for any non-zero `mode`. +- **`TEXTSPLIT` returns `#N/A` where Excel returns `#CALC!`** when `ignore_empty` is on and nothing is left after the split (`=TEXTSPLIT(",,", ",", , TRUE)`), because HyperFormula has no `#CALC!` error. +- **`TEXTSPLIT` needs literal arguments to size its result.** The size of the spilled array depends on the text and the delimiters, and HyperFormula predicts it only when the text, the delimiters, `ignore_empty` and `match_mode` are literals (a number may carry a sign). With anything else, such as `=TEXTSPLIT(1+1, "")` or a cell reference, the formula is treated as returning one value and a longer result is `#VALUE!`, the same limitation as `SEQUENCE` with non-literal dimensions (`=SEQUENCE(1+1)`). Wrapped in another function, for example `=INDEX(TEXTSPLIT(A1, ","), 1, 2)`, it works for any argument. diff --git a/src/i18n/languages/csCZ.ts b/src/i18n/languages/csCZ.ts index fea286083c..e2e8d53f2f 100644 --- a/src/i18n/languages/csCZ.ts +++ b/src/i18n/languages/csCZ.ts @@ -231,7 +231,10 @@ const dictionary: RawTranslationPackage = { TBILLPRICE: 'TBILLPRICE', TBILLYIELD: 'TBILLYIELD', TEXT: 'HODNOTA.NA.TEXT', + TEXTAFTER: 'TEXTAFTER', + TEXTBEFORE: 'TEXTBEFORE', TEXTJOIN: 'TEXTJOIN', + TEXTSPLIT: 'TEXTSPLIT', TIME: 'ČAS', TIMEVALUE: 'ČASHODN', TODAY: 'DNES', diff --git a/src/i18n/languages/daDK.ts b/src/i18n/languages/daDK.ts index 9264551249..aba6e93175 100644 --- a/src/i18n/languages/daDK.ts +++ b/src/i18n/languages/daDK.ts @@ -231,7 +231,10 @@ const dictionary: RawTranslationPackage = { TBILLPRICE: 'STATSOBLIGATION.KURS', TBILLYIELD: 'STATSOBLIGATION.AFKAST', TEXT: 'TEKST', + TEXTAFTER: 'TEXTAFTER', + TEXTBEFORE: 'TEXTBEFORE', TEXTJOIN: 'TEKST.KOMBINER', + TEXTSPLIT: 'TEXTSPLIT', TIME: 'TID', TIMEVALUE: 'TIDSVÆRDI', TODAY: 'IDAG', diff --git a/src/i18n/languages/deDE.ts b/src/i18n/languages/deDE.ts index 4d3d39319b..3293619c49 100644 --- a/src/i18n/languages/deDE.ts +++ b/src/i18n/languages/deDE.ts @@ -231,7 +231,10 @@ const dictionary: RawTranslationPackage = { TBILLPRICE: 'TBILLKURS', TBILLYIELD: 'TBILLRENDITE', TEXT: 'TEXT', + TEXTAFTER: 'TEXTAFTER', + TEXTBEFORE: 'TEXTBEFORE', TEXTJOIN: 'TEXTVERKETTEN', + TEXTSPLIT: 'TEXTSPLIT', TIME: 'ZEIT', TIMEVALUE: 'ZEITWERT', TODAY: 'HEUTE', diff --git a/src/i18n/languages/enGB.ts b/src/i18n/languages/enGB.ts index d271b732cd..5b3fa5070f 100644 --- a/src/i18n/languages/enGB.ts +++ b/src/i18n/languages/enGB.ts @@ -233,7 +233,10 @@ const dictionary: RawTranslationPackage = { TBILLPRICE: 'TBILLPRICE', TBILLYIELD: 'TBILLYIELD', TEXT: 'TEXT', + TEXTAFTER: 'TEXTAFTER', + TEXTBEFORE: 'TEXTBEFORE', TEXTJOIN: 'TEXTJOIN', + TEXTSPLIT: 'TEXTSPLIT', TIME: 'TIME', TIMEVALUE: 'TIMEVALUE', TODAY: 'TODAY', diff --git a/src/i18n/languages/esES.ts b/src/i18n/languages/esES.ts index 9c6b817b98..d15b36c348 100644 --- a/src/i18n/languages/esES.ts +++ b/src/i18n/languages/esES.ts @@ -231,7 +231,10 @@ export const dictionary: RawTranslationPackage = { TBILLPRICE: 'LETRA.DE.TES.PRECIO', TBILLYIELD: 'LETRA.DE.TES.RENDTO', TEXT: 'TEXTO', + TEXTAFTER: 'TEXTAFTER', + TEXTBEFORE: 'TEXTBEFORE', TEXTJOIN: 'UNIRCADENAS', + TEXTSPLIT: 'TEXTSPLIT', TIME: 'NSHORA', TIMEVALUE: 'HORANUMERO', TODAY: 'HOY', diff --git a/src/i18n/languages/fiFI.ts b/src/i18n/languages/fiFI.ts index 99ff05708f..cc40d92b52 100644 --- a/src/i18n/languages/fiFI.ts +++ b/src/i18n/languages/fiFI.ts @@ -231,7 +231,10 @@ const dictionary: RawTranslationPackage = { TBILLPRICE: 'OBLIG.HINTA', TBILLYIELD: 'OBLIG.TUOTTO', TEXT: 'TEKSTI', + TEXTAFTER: 'TEXTAFTER', + TEXTBEFORE: 'TEXTBEFORE', TEXTJOIN: 'TEKSTI.YHDISTÄ', + TEXTSPLIT: 'TEXTSPLIT', TIME: 'AIKA', TIMEVALUE: 'AIKA_ARVO', TODAY: 'TÄMÄ.PÄIVÄ', diff --git a/src/i18n/languages/frFR.ts b/src/i18n/languages/frFR.ts index f7b9be45f7..fa7f67f001 100644 --- a/src/i18n/languages/frFR.ts +++ b/src/i18n/languages/frFR.ts @@ -231,7 +231,10 @@ const dictionary: RawTranslationPackage = { TBILLPRICE: 'PRIX.BON.TRESOR', TBILLYIELD: 'RENDEMENT.BON.TRESOR', TEXT: 'TEXTE', + TEXTAFTER: 'TEXTAFTER', + TEXTBEFORE: 'TEXTBEFORE', TEXTJOIN: 'JOINDRE.TEXTE', + TEXTSPLIT: 'TEXTSPLIT', TIME: 'TEMPS', TIMEVALUE: 'TEMPSVAL', TODAY: 'AUJOURDHUI', diff --git a/src/i18n/languages/huHU.ts b/src/i18n/languages/huHU.ts index e40219553a..a0a96d5362 100644 --- a/src/i18n/languages/huHU.ts +++ b/src/i18n/languages/huHU.ts @@ -231,7 +231,10 @@ const dictionary: RawTranslationPackage = { TBILLPRICE: 'KJEGY.ÁR', TBILLYIELD: 'KJEGY.HOZAM', TEXT: 'SZÖVEG', + TEXTAFTER: 'TEXTAFTER', + TEXTBEFORE: 'TEXTBEFORE', TEXTJOIN: 'SZÖVEGÖSSZEFŰZÉS', + TEXTSPLIT: 'TEXTSPLIT', TIME: 'IDŐ', TIMEVALUE: 'IDŐÉRTÉK', TODAY: 'MA', diff --git a/src/i18n/languages/idID.ts b/src/i18n/languages/idID.ts index 719cb11db6..a2c7495463 100644 --- a/src/i18n/languages/idID.ts +++ b/src/i18n/languages/idID.ts @@ -233,7 +233,10 @@ const dictionary: RawTranslationPackage = { TBILLPRICE: 'TBILL.HARGA', TBILLYIELD: 'TBILL.HASIL', TEXT: 'TEKS', + TEXTAFTER: 'TEXTAFTER', + TEXTBEFORE: 'TEXTBEFORE', TEXTJOIN: 'TEXTJOIN', + TEXTSPLIT: 'TEXTSPLIT', TIME: 'WAKTU', TIMEVALUE: 'NILAI.WAKTU', TODAY: 'HARI.INI', diff --git a/src/i18n/languages/itIT.ts b/src/i18n/languages/itIT.ts index 1a494628f8..a51d9004e7 100644 --- a/src/i18n/languages/itIT.ts +++ b/src/i18n/languages/itIT.ts @@ -231,7 +231,10 @@ const dictionary: RawTranslationPackage = { TBILLPRICE: 'BOT.PREZZO', TBILLYIELD: 'BOT.REND', TEXT: 'TESTO', + TEXTAFTER: 'TEXTAFTER', + TEXTBEFORE: 'TEXTBEFORE', TEXTJOIN: 'UNISCI.TESTO', + TEXTSPLIT: 'TEXTSPLIT', TIME: 'ORARIO', TIMEVALUE: 'ORARIO.VALORE', TODAY: 'OGGI', diff --git a/src/i18n/languages/nbNO.ts b/src/i18n/languages/nbNO.ts index ff15415a98..a281c7a6a6 100644 --- a/src/i18n/languages/nbNO.ts +++ b/src/i18n/languages/nbNO.ts @@ -231,7 +231,10 @@ const dictionary: RawTranslationPackage = { TBILLPRICE: 'TBILLPRIS', TBILLYIELD: 'TBILLAVKASTNING', TEXT: 'TEKST', + TEXTAFTER: 'TEXTAFTER', + TEXTBEFORE: 'TEXTBEFORE', TEXTJOIN: 'TEKST.KOMBINER', + TEXTSPLIT: 'TEXTSPLIT', TIME: 'TID', TIMEVALUE: 'TIDSVERDI', TODAY: 'IDAG', diff --git a/src/i18n/languages/nlNL.ts b/src/i18n/languages/nlNL.ts index c491e2f44b..b7581f22b5 100644 --- a/src/i18n/languages/nlNL.ts +++ b/src/i18n/languages/nlNL.ts @@ -231,7 +231,10 @@ const dictionary: RawTranslationPackage = { TBILLPRICE: 'SCHATK.PRIJS', TBILLYIELD: 'SCHATK.REND', TEXT: 'TEKST', + TEXTAFTER: 'TEXTAFTER', + TEXTBEFORE: 'TEXTBEFORE', TEXTJOIN: 'TEKST.KOPPELEN', + TEXTSPLIT: 'TEXTSPLIT', TIME: 'TIJD', TIMEVALUE: 'TIJDWAARDE', TODAY: 'VANDAAG', diff --git a/src/i18n/languages/plPL.ts b/src/i18n/languages/plPL.ts index 6fa4f016b4..b23e51465a 100644 --- a/src/i18n/languages/plPL.ts +++ b/src/i18n/languages/plPL.ts @@ -231,7 +231,10 @@ const dictionary: RawTranslationPackage = { TBILLPRICE: 'CENA.BS', TBILLYIELD: 'RENT.BS', TEXT: 'TEKST', + TEXTAFTER: 'TEXTAFTER', + TEXTBEFORE: 'TEXTBEFORE', TEXTJOIN: 'POŁĄCZ.TEKSTY', + TEXTSPLIT: 'TEXTSPLIT', TIME: 'CZAS', TIMEVALUE: 'CZAS.WARTOŚĆ', TODAY: 'DZIŚ', diff --git a/src/i18n/languages/ptPT.ts b/src/i18n/languages/ptPT.ts index 6c1e68c328..fc1fa00431 100644 --- a/src/i18n/languages/ptPT.ts +++ b/src/i18n/languages/ptPT.ts @@ -231,7 +231,10 @@ const dictionary: RawTranslationPackage = { TBILLPRICE: 'OTNVALOR', TBILLYIELD: 'OTNLUCRO', TEXT: 'TEXTO', + TEXTAFTER: 'TEXTAFTER', + TEXTBEFORE: 'TEXTBEFORE', TEXTJOIN: 'UNIRTEXTO', + TEXTSPLIT: 'TEXTSPLIT', TIME: 'TEMPO', TIMEVALUE: 'VALOR.TEMPO', TODAY: 'HOJE', diff --git a/src/i18n/languages/ruRU.ts b/src/i18n/languages/ruRU.ts index 6b35c1ce2f..bc27314d15 100644 --- a/src/i18n/languages/ruRU.ts +++ b/src/i18n/languages/ruRU.ts @@ -231,7 +231,10 @@ const dictionary: RawTranslationPackage = { TBILLPRICE: 'ЦЕНАКЧЕК', TBILLYIELD: 'ДОХОДКЧЕК', TEXT: 'ТЕКСТ', + TEXTAFTER: 'TEXTAFTER', + TEXTBEFORE: 'TEXTBEFORE', TEXTJOIN: 'ОБЪЕДИНИТЬ', + TEXTSPLIT: 'TEXTSPLIT', TIME: 'ВРЕМЯ', TIMEVALUE: 'ВРЕМЗНАЧ', TODAY: 'СЕГОДНЯ', diff --git a/src/i18n/languages/svSE.ts b/src/i18n/languages/svSE.ts index 5f87580087..2e4225e1ca 100644 --- a/src/i18n/languages/svSE.ts +++ b/src/i18n/languages/svSE.ts @@ -231,7 +231,10 @@ const dictionary: RawTranslationPackage = { TBILLPRICE: 'SSVXPRIS', TBILLYIELD: 'SSVXRÄNTA', TEXT: 'TEXT', + TEXTAFTER: 'TEXTAFTER', + TEXTBEFORE: 'TEXTBEFORE', TEXTJOIN: 'TEXTJOIN', + TEXTSPLIT: 'TEXTSPLIT', TIME: 'KLOCKSLAG', TIMEVALUE: 'TIDVÄRDE', TODAY: 'IDAG', diff --git a/src/i18n/languages/trTR.ts b/src/i18n/languages/trTR.ts index 93c2176d37..72a02ec106 100644 --- a/src/i18n/languages/trTR.ts +++ b/src/i18n/languages/trTR.ts @@ -231,7 +231,10 @@ const dictionary: RawTranslationPackage = { TBILLPRICE: 'HTAHDEĞER', TBILLYIELD: 'HTAHÖDEME', TEXT: 'METNEÇEVİR', + TEXTAFTER: 'TEXTAFTER', + TEXTBEFORE: 'TEXTBEFORE', TEXTJOIN: 'METİNBİRLEŞTİR', + TEXTSPLIT: 'TEXTSPLIT', TIME: 'ZAMAN', TIMEVALUE: 'ZAMANSAYISI', TODAY: 'BUGÜN', diff --git a/src/interpreter/functionMetadata/categories/text.ts b/src/interpreter/functionMetadata/categories/text.ts index 9672db3045..fecabe6cb2 100644 --- a/src/interpreter/functionMetadata/categories/text.ts +++ b/src/interpreter/functionMetadata/categories/text.ts @@ -143,6 +143,20 @@ export const TEXT_DOCS: Record = { documentationUrl: 'https://hyperformula.handsontable.com/docs/guide/built-in-functions.html', examples: ['=TEXT(1234.5, "0.00")', '=TEXT(TODAY(), "YYYY-MM-DD")'], }, + TEXTAFTER: { + category: 'Text', + shortDescription: 'Returns the text that occurs after the instance_num-th occurrence of delimiter. A negative instance_num counts occurrences from the end of text. Returns if_not_found, or #N/A when it is omitted, if the delimiter is not found.', + parameters: [{name: 'text', description: 'The text to search.'}, {name: 'delimiter', description: 'The text (or array of texts) that marks the point after which the result starts.'}, {name: 'instance_num', description: 'Which occurrence of delimiter to use, truncated to an integer. A negative value counts from the end of text. Defaults to 1.'}, {name: 'match_mode', description: '0 (default) for case-sensitive matching, 1 for case-insensitive matching.'}, {name: 'match_end', description: '1 to treat the end of text as a delimiter, 0 (default) otherwise.'}, {name: 'if_not_found', description: 'The value returned when delimiter is not found. Defaults to #N/A.'}], + documentationUrl: 'https://hyperformula.handsontable.com/docs/guide/built-in-functions.html', + examples: ['=TEXTAFTER("a-b-c", "-")', '=TEXTAFTER("a-b-c", "-", -1)'], + }, + TEXTBEFORE: { + category: 'Text', + shortDescription: 'Returns the text that occurs before the instance_num-th occurrence of delimiter. A negative instance_num counts occurrences from the end of text. Returns if_not_found, or #N/A when it is omitted, if the delimiter is not found.', + parameters: [{name: 'text', description: 'The text to search.'}, {name: 'delimiter', description: 'The text (or array of texts) that marks the point before which the result ends.'}, {name: 'instance_num', description: 'Which occurrence of delimiter to use, truncated to an integer. A negative value counts from the end of text. Defaults to 1.'}, {name: 'match_mode', description: '0 (default) for case-sensitive matching, 1 for case-insensitive matching.'}, {name: 'match_end', description: '1 to treat the end of text as a delimiter, 0 (default) otherwise.'}, {name: 'if_not_found', description: 'The value returned when delimiter is not found. Defaults to #N/A.'}], + documentationUrl: 'https://hyperformula.handsontable.com/docs/guide/built-in-functions.html', + examples: ['=TEXTBEFORE("a-b-c", "-")', '=TEXTBEFORE("a-b-c", "-", 2)'], + }, TEXTJOIN: { category: 'Text', shortDescription: 'Joins text from multiple strings and/or ranges with a delimiter. Supports array/range delimiters that cycle through gaps. When ignore_empty is TRUE, empty strings are skipped. Returns #VALUE! if result exceeds 32,767 characters.', @@ -150,6 +164,13 @@ export const TEXT_DOCS: Record = { documentationUrl: 'https://hyperformula.handsontable.com/docs/guide/built-in-functions.html', examples: ['=TEXTJOIN(", ", TRUE(), A1:A3)', '=TEXTJOIN("-", FALSE(), "a", "b", "c")'], }, + TEXTSPLIT: { + category: 'Text', + shortDescription: 'Splits text into an array: col_delimiter separates columns and row_delimiter separates rows. Rows shorter than the widest one are padded with pad_with. The result spills only when its arguments are literals.', + parameters: [{name: 'text', description: 'The text to split.'}, {name: 'col_delimiter', description: 'The text (or array of texts) that separates columns. Leave empty to split into rows only.'}, {name: 'row_delimiter', description: 'The text (or array of texts) that separates rows. When omitted, the result has a single row.'}, {name: 'ignore_empty', description: 'When TRUE, empty pieces are skipped. Defaults to FALSE.'}, {name: 'match_mode', description: '0 (default) for case-sensitive matching, 1 for case-insensitive matching.'}, {name: 'pad_with', description: 'The value that fills the missing cells of short rows. Defaults to #N/A.'}], + documentationUrl: 'https://hyperformula.handsontable.com/docs/guide/built-in-functions.html', + examples: ['=TEXTSPLIT("a,b,c", ",")', '=TEXTSPLIT("a,b;c,d", ",", ";")'], + }, TRIM: { category: 'Text', shortDescription: 'Strips extra spaces from text.', diff --git a/src/interpreter/plugin/TextPlugin.ts b/src/interpreter/plugin/TextPlugin.ts index 4326a57d67..195097557f 100644 --- a/src/interpreter/plugin/TextPlugin.ts +++ b/src/interpreter/plugin/TextPlugin.ts @@ -3,16 +3,25 @@ * Copyright (c) 2025 Handsoncode. All rights reserved. */ +import {ArraySize} from '../../ArraySize' import {CellError, ErrorType} from '../../Cell' import {ErrorMessage} from '../../error-message' import {Maybe} from '../../Maybe' -import {ProcedureAst} from '../../parser' +import {Ast, AstNodeType, ProcedureAst} from '../../parser' import {coerceScalarToString} from '../ArithmeticHelper' import {InterpreterState} from '../InterpreterState' import {SimpleRangeValue} from '../../SimpleRangeValue' -import {ExtendedNumber, InterpreterValue, isExtendedNumber, RawScalarValue, InternalScalarValue} from '../InterpreterValue' +import {EmptyValue, ExtendedNumber, InterpreterValue, isExtendedNumber, RawScalarValue, InternalScalarValue} from '../InterpreterValue' import {FunctionArgumentType, FunctionPlugin, FunctionPluginTypecheck, ImplementedFunctions} from './FunctionPlugin' +/** + * A single occurrence of a delimiter in a text: the delimiter occupies the characters from `start` (inclusive) to `end` (exclusive). + */ +interface DelimiterMatch { + start: number, + end: number, +} + /** * Interpreter plugin containing text-specific functions */ @@ -166,6 +175,41 @@ export class TextPlugin extends FunctionPlugin implements FunctionPluginTypechec {argumentType: FunctionArgumentType.ANY}, ], }, + 'TEXTBEFORE': { + method: 'textbefore', + parameters: [ + {argumentType: FunctionArgumentType.STRING}, + {argumentType: FunctionArgumentType.ANY}, + {argumentType: FunctionArgumentType.NUMBER, defaultValue: 1, emptyAsDefault: true}, + {argumentType: FunctionArgumentType.NUMBER, defaultValue: 0, emptyAsDefault: true}, + {argumentType: FunctionArgumentType.NUMBER, defaultValue: 0, emptyAsDefault: true}, + {argumentType: FunctionArgumentType.SCALAR, optionalArg: true}, + ], + }, + 'TEXTAFTER': { + method: 'textafter', + parameters: [ + {argumentType: FunctionArgumentType.STRING}, + {argumentType: FunctionArgumentType.ANY}, + {argumentType: FunctionArgumentType.NUMBER, defaultValue: 1, emptyAsDefault: true}, + {argumentType: FunctionArgumentType.NUMBER, defaultValue: 0, emptyAsDefault: true}, + {argumentType: FunctionArgumentType.NUMBER, defaultValue: 0, emptyAsDefault: true}, + {argumentType: FunctionArgumentType.SCALAR, optionalArg: true}, + ], + }, + 'TEXTSPLIT': { + method: 'textsplit', + sizeOfResultArrayMethod: 'textsplitArraySize', + parameters: [ + {argumentType: FunctionArgumentType.STRING}, + {argumentType: FunctionArgumentType.ANY}, + {argumentType: FunctionArgumentType.ANY, optionalArg: true}, + {argumentType: FunctionArgumentType.BOOLEAN, defaultValue: false, emptyAsDefault: true}, + {argumentType: FunctionArgumentType.NUMBER, defaultValue: 0, emptyAsDefault: true}, + {argumentType: FunctionArgumentType.SCALAR, optionalArg: true}, + ], + vectorizationForbidden: true, + }, } /** @@ -483,6 +527,461 @@ export class TextPlugin extends FunctionPlugin implements FunctionPluginTypechec ) } + /** + * Corresponds to TEXTBEFORE(text, delimiter, [instance_num], [match_mode], [match_end], [if_not_found]) + * + * Returns the text that occurs before the instance_num-th occurrence of delimiter. + * See {@link textAroundDelimiter} for the meaning of the optional arguments. + * + * @param {ProcedureAst} ast - The procedure AST node + * @param {InterpreterState} state - The interpreter state + */ + public textbefore(ast: ProcedureAst, state: InterpreterState): InterpreterValue { + return this.runFunction(ast.args, state, this.metadata('TEXTBEFORE'), + (text: string, delimiterArg: InternalScalarValue | SimpleRangeValue, instanceNum: number, matchMode: number, matchEnd: number, ifNotFound: Maybe) => + this.textAroundDelimiter(text, delimiterArg, instanceNum, matchMode, matchEnd, ifNotFound, (match) => text.substring(0, match.start)) + ) + } + + /** + * Corresponds to TEXTAFTER(text, delimiter, [instance_num], [match_mode], [match_end], [if_not_found]) + * + * Returns the text that occurs after the instance_num-th occurrence of delimiter. + * See {@link textAroundDelimiter} for the meaning of the optional arguments. + * + * @param {ProcedureAst} ast - The procedure AST node + * @param {InterpreterState} state - The interpreter state + */ + public textafter(ast: ProcedureAst, state: InterpreterState): InterpreterValue { + return this.runFunction(ast.args, state, this.metadata('TEXTAFTER'), + (text: string, delimiterArg: InternalScalarValue | SimpleRangeValue, instanceNum: number, matchMode: number, matchEnd: number, ifNotFound: Maybe) => + this.textAroundDelimiter(text, delimiterArg, instanceNum, matchMode, matchEnd, ifNotFound, (match) => text.substring(match.end)) + ) + } + + /** + * Corresponds to TEXTSPLIT(text, col_delimiter, [row_delimiter], [ignore_empty], [match_mode], [pad_with]) + * + * Splits text into rows on row_delimiter, then splits each row into columns on col_delimiter. + * Both delimiters may be arrays of delimiters; an empty col_delimiter or row_delimiter argument + * means that the text is not split along that axis. When ignore_empty is TRUE, empty pieces + * (and rows left with no pieces) are skipped. match_mode 1 makes the matching case-insensitive. + * Rows shorter than the widest one are padded with pad_with (#N/A by default). + * + * @param {ProcedureAst} ast - The procedure AST node + * @param {InterpreterState} state - The interpreter state + */ + public textsplit(ast: ProcedureAst, state: InterpreterState): InterpreterValue { + return this.runFunction(ast.args, state, this.metadata('TEXTSPLIT'), + (text: string, + colDelimiterArg: InternalScalarValue | SimpleRangeValue, + rowDelimiterArg: Maybe, + ignoreEmpty: boolean, + matchMode: number, + padWith: Maybe) => { + + const colDelimiters = this.delimitersFromArg(colDelimiterArg) + if (colDelimiters instanceof CellError) { + return colDelimiters + } + + const rowDelimiters = this.delimitersFromArg(rowDelimiterArg) + if (rowDelimiters instanceof CellError) { + return rowDelimiters + } + + if (!TextPlugin.isZeroOrOne(matchMode)) { + return new CellError(ErrorType.VALUE, ErrorMessage.BadMode) + } + + const rows = TextPlugin.splitText(text, colDelimiters, rowDelimiters, ignoreEmpty, matchMode === 1) + if (rows instanceof CellError) { + return rows + } + + const padValue = (padWith === undefined || padWith === EmptyValue) + ? new CellError(ErrorType.NA, ErrorMessage.ValueNotFound) + : padWith + const width = rows.reduce((max, row) => Math.max(max, row.length), 0) + + return SimpleRangeValue.onlyValues(rows.map(row => [...row, ...Array(width - row.length).fill(padValue)])) + } + ) + } + + /** + * Predicts the size of the TEXTSPLIT result at parse time. + * + * The size depends on the text and the delimiters, so it can be predicted only when the text, + * the delimiters, ignore_empty and match_mode are literals. Otherwise, the function is treated + * as returning a scalar, and a result larger than one cell is reported as #VALUE! + * (the same limitation as SEQUENCE with non-literal dimensions). + * + * @param {ProcedureAst} ast - The procedure AST node + * @param {InterpreterState} _state - The interpreter state (unused) + */ + public textsplitArraySize(ast: ProcedureAst, _state: InterpreterState): ArraySize { + if (ast.args.length < 2 || ast.args.length > 6) { + return ArraySize.error() + } + + const [textArg, colDelimiterArg, rowDelimiterArg, ignoreEmptyArg, matchModeArg] = ast.args + const text = TextPlugin.literalText(textArg) + const colDelimiters = TextPlugin.literalDelimiters(colDelimiterArg) + const rowDelimiters = TextPlugin.literalDelimiters(rowDelimiterArg) + const ignoreEmpty = TextPlugin.literalNumber(ignoreEmptyArg, 0) + const matchMode = TextPlugin.literalNumber(matchModeArg, 0) + + if (text === undefined || colDelimiters === undefined || rowDelimiters === undefined || ignoreEmpty === undefined || matchMode === undefined) { + return ArraySize.error() + } + + if (!TextPlugin.isZeroOrOne(matchMode)) { + return ArraySize.scalar() + } + + const rows = TextPlugin.splitText(text, colDelimiters, rowDelimiters, ignoreEmpty !== 0, matchMode === 1) + if (rows instanceof CellError) { + return ArraySize.scalar() + } + + return new ArraySize(rows.reduce((max, row) => Math.max(max, row.length), 0), rows.length) + } + + /** + * Shared logic of TEXTBEFORE and TEXTAFTER. + * + * - delimiter may be an array of delimiters; occurrences of any of them are counted in text order. + * - instance_num is truncated to an integer; a negative instance_num counts occurrences from the end of text. + * 0, or an absolute value larger than the length of a non-empty text, is #VALUE!. + * - match_mode 1 makes the matching case-insensitive. + * - match_end 1 treats the end of text (or its start, for a negative instance_num) as one more delimiter. + * - When the delimiter is not found, if_not_found is returned, or #N/A when it is omitted. + * + * @param {string} text - The text to search + * @param {InternalScalarValue | SimpleRangeValue} delimiterArg - The delimiter or array of delimiters + * @param {number} instanceNum - Which occurrence of the delimiter to use + * @param {number} matchMode - 0 for case-sensitive, 1 for case-insensitive matching + * @param {number} matchEnd - 1 to treat the end of text as a delimiter + * @param {Maybe} ifNotFound - The value returned when the delimiter is not found + * @param {(match: DelimiterMatch) => string} extractText - Picks the result from the found occurrence + */ + private textAroundDelimiter( + text: string, + delimiterArg: InternalScalarValue | SimpleRangeValue, + instanceNum: number, + matchMode: number, + matchEnd: number, + ifNotFound: Maybe, + extractText: (match: DelimiterMatch) => string, + ): InternalScalarValue { + const delimiters = this.flattenArgToStrings(delimiterArg) + if (delimiters instanceof CellError) { + return delimiters + } + + if (!TextPlugin.isZeroOrOne(matchMode) || !TextPlugin.isZeroOrOne(matchEnd)) { + return new CellError(ErrorType.VALUE, ErrorMessage.BadMode) + } + + const instance = Math.trunc(instanceNum) + if (instance === 0) { + return new CellError(ErrorType.VALUE, ErrorMessage.IndexBounds) + } + + // An empty delimiter matches at once: at the start of the text for a positive instance_num and at its end for a negative one. + if (delimiters.length > 0 && delimiters.every(delimiter => delimiter === '')) { + const position = instance > 0 ? 0 : text.length + return extractText({start: position, end: position}) + } + + if (text.length > 0 && Math.abs(instance) > text.length) { + return new CellError(ErrorType.VALUE, ErrorMessage.IndexBounds) + } + + const match = TextPlugin.findDelimiterOccurrence(text, delimiters, instance, matchMode === 1, matchEnd === 1) + if (match !== undefined) { + return extractText(match) + } + + return (ifNotFound === undefined || ifNotFound === EmptyValue) + ? new CellError(ErrorType.NA, ErrorMessage.PatternNotFound) + : ifNotFound + } + + /** + * Converts a TEXTSPLIT delimiter argument into a list of delimiters. + * An omitted or empty argument yields no delimiters, so the text is not split along that axis. + * + * @param {Maybe} arg - The delimiter argument + * @returns {string[] | CellError} - The delimiters, or the first error encountered + */ + private delimitersFromArg(arg: Maybe): string[] | CellError { + if (arg === undefined || arg === EmptyValue) { + return [] + } + return this.flattenArgToStrings(arg) + } + + /** + * Finds the instance-th occurrence of any of the delimiters in text. + * A positive instance scans from the start of text, a negative one scans backwards from the end; + * in both directions, occurrences do not overlap. + * When matchEnd is set and text holds exactly one occurrence too few, the end of text + * (or its start, for a negative instance) is returned as a zero-length occurrence. + * + * @param {string} text - The text to search + * @param {string[]} delimiters - The delimiters to look for + * @param {number} instance - Non-zero integer: which occurrence to return + * @param {boolean} ignoreCase - Whether the matching is case-insensitive + * @param {boolean} matchEnd - Whether the end of text counts as a delimiter + * @returns {Maybe} - The occurrence, or undefined when there is none + */ + private static findDelimiterOccurrence(text: string, delimiters: string[], instance: number, ignoreCase: boolean, matchEnd: boolean): Maybe { + const count = Math.abs(instance) + const matches = instance > 0 + ? TextPlugin.findForwardMatches(text, delimiters, ignoreCase, count) + : TextPlugin.findBackwardMatches(text, delimiters, ignoreCase, count) + + if (matches.length === count) { + return matches[count - 1] + } + + if (matchEnd && matches.length === count - 1) { + const boundary = instance > 0 ? text.length : 0 + return {start: boundary, end: boundary} + } + + return undefined + } + + /** + * Finds up to maxCount non-overlapping occurrences of the delimiters, scanning text from the start. + * When several delimiters match at the same position, the longest one wins. + * + * @param {string} text - The text to search + * @param {string[]} delimiters - The delimiters to look for + * @param {boolean} ignoreCase - Whether the matching is case-insensitive + * @param {number} maxCount - The maximum number of occurrences to find + * @returns {DelimiterMatch[]} - The occurrences, in text order + */ + private static findForwardMatches(text: string, delimiters: string[], ignoreCase: boolean, maxCount: number = Infinity): DelimiterMatch[] { + const matches: DelimiterMatch[] = [] + let position = 0 + + while (position <= text.length && matches.length < maxCount) { + const length = TextPlugin.longestDelimiterLength(delimiters, delimiter => TextPlugin.matchesAt(text, position, delimiter, ignoreCase)) + if (length === undefined) { + position++ + continue + } + matches.push({start: position, end: position + length}) + position += Math.max(length, 1) + } + + return matches + } + + /** + * Finds up to maxCount non-overlapping occurrences of the delimiters, scanning text backwards from the end. + * When several delimiters end at the same position, the longest one wins. + * + * @param {string} text - The text to search + * @param {string[]} delimiters - The delimiters to look for + * @param {boolean} ignoreCase - Whether the matching is case-insensitive + * @param {number} maxCount - The maximum number of occurrences to find + * @returns {DelimiterMatch[]} - The occurrences, from the last one in text to the first + */ + private static findBackwardMatches(text: string, delimiters: string[], ignoreCase: boolean, maxCount: number): DelimiterMatch[] { + const matches: DelimiterMatch[] = [] + let end = text.length + + while (end >= 0 && matches.length < maxCount) { + const length = TextPlugin.longestDelimiterLength(delimiters, delimiter => + delimiter.length <= end && TextPlugin.matchesAt(text, end - delimiter.length, delimiter, ignoreCase)) + if (length === undefined) { + end-- + continue + } + matches.push({start: end - length, end}) + end -= Math.max(length, 1) + } + + return matches + } + + /** + * Returns the length of the longest delimiter that satisfies the predicate, or undefined if none does. + * + * @param {string[]} delimiters - The delimiters to check + * @param {(delimiter: string) => boolean} isMatching - Tells whether a delimiter matches + * @returns {Maybe} - The length of the longest matching delimiter + */ + private static longestDelimiterLength(delimiters: string[], isMatching: (delimiter: string) => boolean): Maybe { + return delimiters + .filter(isMatching) + .reduce((longest: Maybe, delimiter) => (longest === undefined || delimiter.length > longest) ? delimiter.length : longest, undefined) + } + + /** + * Tells whether delimiter occurs in text at the given position. + * The comparison is done on the original text, so positions always refer to it. + * + * @param {string} text - The text to search + * @param {number} position - The position in text + * @param {string} delimiter - The delimiter to compare + * @param {boolean} ignoreCase - Whether the comparison is case-insensitive + */ + private static matchesAt(text: string, position: number, delimiter: string, ignoreCase: boolean): boolean { + if (ignoreCase) { + return text.slice(position, position + delimiter.length).toLowerCase() === delimiter.toLowerCase() + } + return text.startsWith(delimiter, position) + } + + /** + * Splits text at every occurrence of any of the delimiters. With no delimiters, text is not split. + * + * @param {string} text - The text to split + * @param {string[]} delimiters - The non-empty delimiters to split on + * @param {boolean} ignoreCase - Whether the matching is case-insensitive + * @returns {string[]} - The pieces between the delimiters + */ + private static splitByDelimiters(text: string, delimiters: string[], ignoreCase: boolean): string[] { + if (delimiters.length === 0) { + return [text] + } + + const pieces: string[] = [] + let pieceStart = 0 + for (const match of TextPlugin.findForwardMatches(text, delimiters, ignoreCase)) { + pieces.push(text.substring(pieceStart, match.start)) + pieceStart = match.end + } + pieces.push(text.substring(pieceStart)) + + return pieces + } + + /** + * Splits text into rows and columns, as TEXTSPLIT does (without padding). + * Used both at evaluation time and to predict the result size at parse time, so that the two always agree. + * + * @param {string} text - The text to split + * @param {string[]} colDelimiters - The delimiters separating columns + * @param {string[]} rowDelimiters - The delimiters separating rows + * @param {boolean} ignoreEmpty - Whether to skip empty pieces and rows left with no pieces + * @param {boolean} ignoreCase - Whether the matching is case-insensitive + * @returns {string[][] | CellError} - The rows of pieces, or an error for empty text, an empty delimiter, or an empty result + */ + private static splitText(text: string, colDelimiters: string[], rowDelimiters: string[], ignoreEmpty: boolean, ignoreCase: boolean): string[][] | CellError { + if (text === '' || colDelimiters.includes('') || rowDelimiters.includes('')) { + return new CellError(ErrorType.VALUE, ErrorMessage.EmptyString) + } + + const rows = TextPlugin.splitByDelimiters(text, rowDelimiters, ignoreCase) + .map(row => TextPlugin.splitByDelimiters(row, colDelimiters, ignoreCase)) + .map(pieces => ignoreEmpty ? pieces.filter(piece => piece !== '') : pieces) + .filter(pieces => pieces.length > 0) + + if (rows.length === 0) { + return new CellError(ErrorType.NA, ErrorMessage.EmptyRange) + } + + return rows + } + + /** + * Reads a literal text from an AST node at parse time: a string, or a number converted to text. + * + * @param {Ast} node - The AST node + * @returns {Maybe} - The text, or undefined if the node is not such a literal + */ + private static literalText(node: Ast): Maybe { + if (node.type === AstNodeType.STRING) { + return node.value + } + if (node.type === AstNodeType.NUMBER) { + return node.value.toString() + } + const signed = TextPlugin.signedNumberLiteral(node) + if (signed !== undefined) { + return signed.toString() + } + return undefined + } + + /** + * Reads a number literal with a leading sign (`-12.5`, `+12.5`), which the parser keeps as a unary operator node. + * + * @param {Ast} node - The AST node + * @returns {Maybe} - The signed number, or undefined if the node is not such a literal + */ + private static signedNumberLiteral(node: Ast): Maybe { + if ((node.type === AstNodeType.MINUS_UNARY_OP || node.type === AstNodeType.PLUS_UNARY_OP) && node.value.type === AstNodeType.NUMBER) { + return node.type === AstNodeType.MINUS_UNARY_OP ? -node.value.value : node.value.value + } + return undefined + } + + /** + * Reads literal TEXTSPLIT delimiters from an AST node at parse time. + * An omitted or empty argument yields no delimiters; an array literal yields all its elements. + * + * @param {Maybe} node - The AST node, or undefined if the argument is omitted + * @returns {Maybe} - The delimiters, or undefined if the node is not a literal + */ + private static literalDelimiters(node: Maybe): Maybe { + if (node === undefined || node.type === AstNodeType.EMPTY) { + return [] + } + + const elements = node.type === AstNodeType.ARRAY + ? node.args.reduce((all: Ast[], row) => all.concat(row), []) + : [node] + const delimiters = elements.map(element => TextPlugin.literalText(element)) + + return delimiters.every(delimiter => delimiter !== undefined) ? delimiters as string[] : undefined + } + + /** + * Reads a literal number from an AST node at parse time: a number, or TRUE()/FALSE() as 1/0. + * + * @param {Maybe} node - The AST node, or undefined if the argument is omitted + * @param {number} defaultValue - The value of an omitted or empty argument + * @returns {Maybe} - The number, or undefined if the node is not such a literal + */ + private static literalNumber(node: Maybe, defaultValue: number): Maybe { + if (node === undefined || node.type === AstNodeType.EMPTY) { + return defaultValue + } + if (node.type === AstNodeType.NUMBER) { + return node.value + } + const signed = TextPlugin.signedNumberLiteral(node) + if (signed !== undefined) { + return signed + } + if (node.type === AstNodeType.FUNCTION_CALL && node.args.length === 0) { + if (node.procedureName === 'TRUE') { + return 1 + } + if (node.procedureName === 'FALSE') { + return 0 + } + } + return undefined + } + + /** + * Tells whether a mode argument (match_mode, match_end) has one of its two valid values. + * + * @param {number} value - The argument value + */ + private static isZeroOrOne(value: number): boolean { + return value === 0 || value === 1 + } + /** * Flattens a scalar or range argument into an array of coerced strings. * Returns a CellError immediately if any value in the argument is an error or cannot be coerced. diff --git a/src/license/functionCapabilities.ts b/src/license/functionCapabilities.ts index a26733ec5f..b7d1113a8e 100644 --- a/src/license/functionCapabilities.ts +++ b/src/license/functionCapabilities.ts @@ -95,7 +95,7 @@ const UNGROUPED_FUNCTIONS = [ 'PHI', 'POISSON.DIST', 'QUARTILE.EXC', 'QUARTILE.INC', 'RADIANS', 'ROMAN', 'RRI', 'RSQ', 'SEC', 'SECH', 'SERIESSUM', 'SHEET', 'SHEETS', 'SINH', 'SKEW', 'SKEW.P', 'SLOPE', 'SPLIT', 'SQRTPI', 'STANDARDIZE', 'STEYX', 'SUMX2MY2', 'SUMX2PY2', 'SYD', 'T.DIST', 'T.DIST.2T', 'T.DIST.RT', 'T.INV', 'T.INV.2T', 'T.TEST', - 'TANH', 'TBILLEQ', 'TBILLPRICE', 'TBILLYIELD', 'TDIST', 'TIMEVALUE', 'UNICODE', 'VARA', 'VARPA', + 'TANH', 'TBILLEQ', 'TBILLPRICE', 'TBILLYIELD', 'TDIST', 'TEXTAFTER', 'TEXTBEFORE', 'TEXTSPLIT', 'TIMEVALUE', 'UNICODE', 'VARA', 'VARPA', 'WEIBULL.DIST', 'WORKDAY.INTL', 'Z.TEST', ]