From 3636965a35c7474e3324754f5c9ec950e6c181a9 Mon Sep 17 00:00:00 2001 From: Ismail El Korchi Date: Tue, 21 Jul 2026 23:21:43 +0100 Subject: [PATCH 1/3] fix: harden public parser boundaries --- .github/workflows/ci.yml | 49 +- .github/workflows/release-audit.yml | 4 +- .github/workflows/runtime-latest.yml | 11 +- CHANGELOG.md | 17 + README.md | 18 +- docs/api.md | 13 +- docs/data-model.md | 19 +- docs/getting-started.md | 5 +- docs/limits-errors-and-safety.md | 6 +- docs/maintainers/testing.md | 17 +- docs/parsing.md | 12 +- docs/streams-and-encoding.md | 24 +- scripts/mutation/config.json | 90 +++- scripts/oracles/document-browser-baseline.mjs | 51 ++ scripts/oracles/run-document-browser-diff.mjs | 21 +- .../run-public-fragment-browser-diff.mjs | 2 +- .../qualification/run-public-fragments.mjs | 2 +- .../run-public-serialization-roundtrip.mjs | 10 +- .../run-public-serialization.mjs | 2 +- scripts/qualification/run-resource-limits.mjs | 71 ++- scripts/smoke/browser-smoke.mjs | 22 +- scripts/smoke/control.mjs | 31 +- src/internal/encoding/sniff.ts | 100 ++-- .../foundation/internal-state-error.ts | 4 +- src/internal/foundation/name-validation.ts | 21 + src/public/chunking.ts | 2 + src/public/html-input.ts | 212 ++++++-- src/public/model.ts | 108 +++- src/public/operation.ts | 1 + ...-registry.ts => parsed-output-registry.ts} | 19 +- src/public/parsing.ts | 162 ++++-- src/public/patching.ts | 2 +- src/public/querying.ts | 69 ++- src/public/serialization.ts | 5 +- src/public/text-extraction.ts | 8 +- src/public/tree-validation.ts | 478 ++++++++++++++++++ src/public/types.ts | 78 ++- test/behavior/fragment-context.test.js | 22 +- test/behavior/namespaces-stack.test.js | 8 +- test/behavior/parse-result-metadata.test.js | 33 +- test/behavior/public-integration.test.js | 6 +- test/behavior/roundtrip.test.js | 4 +- test/behavior/serialization.test.js | 63 +++ test/behavior/streaming.test.js | 97 +++- test/behavior/trace-schema.test.js | 2 +- test/behavior/traversal-extraction.test.js | 84 +++ test/contracts/consumers/npm/runtime.mjs | 2 +- .../html-product-contract.type-test.ts | 13 +- .../document-browser-baseline.json | 95 +++- ...blic-serialization-qualification-cases.mjs | 5 +- .../tooling/document-browser-baseline.test.js | 50 ++ test/tooling/registry-integrity.test.js | 2 + 52 files changed, 1968 insertions(+), 284 deletions(-) create mode 100644 scripts/oracles/document-browser-baseline.mjs rename src/public/{parsed-document-registry.ts => parsed-output-registry.ts} (65%) create mode 100644 src/public/tree-validation.ts create mode 100644 test/tooling/document-browser-baseline.test.js diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 20c3863..7159452 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -29,12 +29,6 @@ jobs: node-version: ${{ matrix.node-version }} cache: npm - - name: Setup Deno - if: matrix.node-version == 20 - uses: denoland/setup-deno@22d081ff2d3a40755e97629de92e3bcbfa7cf2ed # v2.0.5 - with: - deno-version: v2.x - - name: Install run: npm ci @@ -57,8 +51,19 @@ jobs: - name: Test run: npm test - deno: - runs-on: ubuntu-latest + platform-runtime: + strategy: + fail-fast: false + matrix: + include: + - os: ubuntu-latest + name: linux + - os: macos-latest + name: macos + - os: windows-latest + name: windows + runs-on: ${{ matrix.os }} + name: runtime-${{ matrix.name }} steps: - name: Checkout uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0 @@ -74,35 +79,11 @@ jobs: with: deno-version: v2.x - - name: Install - run: npm ci - - - name: Build - run: npm run build - - - name: Control smoke (Deno) - run: npm run smoke:deno - - bun: - runs-on: ubuntu-latest - steps: - - name: Checkout - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0 - - - name: Setup Node - uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0 - with: - node-version: 20 - cache: npm - - name: Setup Bun uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2.2.0 - name: Install run: npm ci - - name: Build - run: npm run build - - - name: Control smoke (Bun) - run: npm run smoke:bun + - name: Runtime agreement + run: npm run qualification:runtime diff --git a/.github/workflows/release-audit.yml b/.github/workflows/release-audit.yml index 7e4b6e4..39694d1 100644 --- a/.github/workflows/release-audit.yml +++ b/.github/workflows/release-audit.yml @@ -57,10 +57,10 @@ jobs: run: deno publish --dry-run - name: Inspect npm publication state - run: node scripts/release/check-registry-version.mjs --registry=npm + run: node scripts/release/check-registry-version.mjs --registry=npm --require-present - name: Inspect JSR publication state - run: node scripts/release/check-registry-version.mjs --registry=jsr + run: node scripts/release/check-registry-version.mjs --registry=jsr --require-present - name: Report outdated dependencies continue-on-error: true diff --git a/.github/workflows/runtime-latest.yml b/.github/workflows/runtime-latest.yml index fdde7a9..f9e8b4b 100644 --- a/.github/workflows/runtime-latest.yml +++ b/.github/workflows/runtime-latest.yml @@ -14,7 +14,12 @@ concurrency: jobs: latest-runtime-smoke: - runs-on: ubuntu-latest + strategy: + fail-fast: false + matrix: + os: [ubuntu-latest, macos-latest, windows-latest] + runs-on: ${{ matrix.os }} + name: latest-${{ matrix.os }} timeout-minutes: 60 continue-on-error: true steps: @@ -41,17 +46,21 @@ jobs: run: npm ci - name: Lint + if: matrix.os == 'ubuntu-latest' run: npm run lint - name: Typecheck + if: matrix.os == 'ubuntu-latest' run: npm run typecheck - name: Test + if: matrix.os == 'ubuntu-latest' run: npm run test - name: Runtime smoke matrix run: npm run smoke:runtimes - name: Report outdated dependencies + if: matrix.os == 'ubuntu-latest' continue-on-error: true run: npm run deps:outdated diff --git a/CHANGELOG.md b/CHANGELOG.md index d972495..820fe32 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,23 @@ All notable changes are documented in this file. ## Unreleased +- Validate complete caller-constructed serialization graphs before emitting + markup, including names, namespaces, prefixes, attributes, document + placement, template ownership, unique ids, unique ownership, and acyclicity; + apply the same ownership safety to chunking and public tree traversals. +- Separate mandatory BOM detection from optional meta-encoding prescan, + initialize streaming decode as soon as encoding evidence is final, and feed + byte-array decoding directly into the parser without retaining decoded source + when source retention is disabled. +- Return resource metadata from fragment parsing and eager stream tokenization, + and expose and enforce deterministic `maxSteps` for eager tokenization. +- Validate all public query-helper arguments consistently with + `HtmlConfigurationError`. +- Require scheduled registry audits to find the released version on both npm + and JSR, and record every accepted browser difference with its exact case id, + classification, explanation, and result hashes. +- Run the blocking runtime contract on Linux, macOS, and Windows. + ## [0.2.0] - 2026-07-21 - Validate the complete runtime edit shape in source patching, normalize HTML diff --git a/README.md b/README.md index 8fa75d7..5c4b57d 100644 --- a/README.md +++ b/README.md @@ -60,16 +60,21 @@ ParsedDocument └── metadata input, encoding, and observed resource usage ``` -`parseFragment()` returns a `FragmentTree` directly and requires an explicit -namespace-aware context: +`parseFragment()` returns a `ParsedFragment` with the immutable tree and the +same successful resource evidence available for document parsing. It requires +an explicit namespace-aware context: ```ts import { HTML_NAMESPACE_URI, parseFragment } from "@ismail-elkorchi/html-parser"; -const rows = parseFragment("AB", { +const { tree: rows, metadata } = parseFragment("AB", { namespaceUri: HTML_NAMESPACE_URI, localName: "tbody" +}, { + budgets: { maxSteps: 10_000 } }); + +console.log(rows.kind, metadata.resourceUsage.steps); ``` ## Find what you need @@ -89,9 +94,10 @@ architecture, testing, corpus, and source-policy notes. ## Runtime support -The npm surface supports Node.js 20, 22, and 24. Deno, Bun, and evergreen -browsers are covered by smoke tests. npm/Node and JSR expose the same runtime -and TypeScript API, documented in [the API guide](./docs/api.md). +The npm surface supports Node.js 20, 22, and 24. Linux, macOS, and Windows run +the cross-runtime contract in CI; Deno, Bun, and evergreen browsers are also +covered by smoke tests. npm/Node and JSR expose the same runtime and TypeScript +API, documented in [the API guide](./docs/api.md). ## Safety diff --git a/docs/api.md b/docs/api.md index 653723a..7a52a73 100644 --- a/docs/api.md +++ b/docs/api.md @@ -11,12 +11,14 @@ the package remain the exact source of truth. `ParsedDocument`. - `parseStream(stream, options?)` reads, decodes, and parses a byte stream. - `parseFragment(input, context, options?)` parses with an explicit - namespace-aware `HtmlFragmentContextInput` and returns a `FragmentTree`. + namespace-aware `HtmlFragmentContextInput` and returns a `ParsedFragment`. - `serialize(input, options?)` serializes a document, fragment, or complete node representation. `SerializeOptions.scriptingMode` controls the conditional `noscript` rule. A document or fragment inherits the mode retained from parsing; an individual node defaults to `"inert"`. -- `tokenizeByteStreamEager(stream, options?)` returns logical tokens after EOF. +- `tokenizeByteStreamEager(stream, options?)` returns a + `TokenizeByteStreamEagerResult` containing logical tokens plus encoding and + resource evidence after EOF. Its budgets include deterministic `maxSteps`. - `getParseErrorSpecRef(parseErrorId)` maps a named parser diagnostic to its dedicated HTML Standard anchor and other identifiers to the general parse-errors section. @@ -75,9 +77,9 @@ Key exported type groups include: `ParseStreamOptions`, `TokenizeByteStreamEagerBudgetOptions`, `TokenizeByteStreamEagerOptions`, `SerializeOptions`, `HtmlScriptingMode`, `OperationOptions`, `SourceRetention`; -- trees and metadata: `ParsedDocument`, `ParsedDocumentMetadata`, +- trees and metadata: `ParsedDocument`, `ParsedFragment`, `ParseMetadata`, `ParseEncodingMetadata`, `ParseResourceUsage`, `DocumentTree`, - `FragmentTree`, `HtmlNode`, `NodeKind`, `ElementNode`, + `FragmentTree`, `HtmlNode`, `NodeKind`, `ElementNode`, `SerializableNode`, `TemplateContentNode`, `TextNode`, `CommentNode`, `ProcessingInstructionNode`, `DoctypeNode`, `DoctypeExternalId`, `Attribute`, `Span`, `SpanProvenance`, `ParseError`, `NodeId`, `NodeVisitor`, and @@ -88,7 +90,8 @@ Key exported type groups include: `TraceParseErrorEvent`, `TraceStreamEvent`, `TraceTokenEvent`, `TraceTreeMutationEvent`, `Token`, `TokenAttribute`, `StartTagToken`, `EndTagToken`, `CharsToken`, `CommentToken`, `ProcessingInstructionToken`, - `DoctypeToken`, and `EofToken`; + `DoctypeToken`, `EofToken`, `TokenizationResourceUsage`, + `TokenizeByteStreamEagerMetadata`, and `TokenizeByteStreamEagerResult`; - extraction: `TextExtractionPolicy`, `TextExtractionOptions`, `TextExtractionOptionsBase`, `VisibleTextExtractionOptions`, `TextContentExtractionOptions`, diff --git a/docs/data-model.md b/docs/data-model.md index 40aedca..6682fde 100644 --- a/docs/data-model.md +++ b/docs/data-model.md @@ -8,14 +8,15 @@ - `tree: DocumentTree` contains children, parse diagnostics, and optional trace; - `sourceText: string | null` contains the exact decoded input only when `sourceRetention: "text"` was selected; -- `metadata: ParsedDocumentMetadata` records input kind, transport size, +- `metadata: ParseMetadata` records input kind, transport size, encoding evidence, and successful resource observations. `DocumentTree` retains the effective `scriptingMode` used for parsing. -`parseFragment()` returns a frozen `FragmentTree` with its normalized -namespace-aware `context`, effective `scriptingMode`, `documentMode`, -`hasFormInContextChain` decision, children, diagnostics, and optional trace. It -has no source-retention wrapper. +`parseFragment()` returns a frozen `ParsedFragment`. Its `tree` is a +`FragmentTree` with the normalized namespace-aware `context`, effective +`scriptingMode`, `documentMode`, `hasFormInContextChain` decision, children, +diagnostics, and optional trace; its `metadata` reports successful resource +use. Fragments do not retain source text. Resource observations describe the exact successful parse; they are not limits. `steps` is the exception because counting it has a hot-path cost: it is @@ -106,3 +107,11 @@ which is an intentional extension beyond the HTML fragment algorithm's name-only doctype output. Serialization is deterministic, but arbitrary or parser-recovered trees are not guaranteed to survive a serialize/reparse cycle unchanged. + +Caller-constructed serialization inputs are validated completely before any +markup is emitted. Node ids and object ownership must be unique; the graph must +be acyclic; names, namespaces, prefixes, attributes, document placement, and +template-content ownership must satisfy the public model. Invalid graphs throw +`HtmlConfigurationError`. Traversal, querying, outlining, text extraction, and +chunking also reject cyclic or multiply-owned caller graphs rather than relying +on a deadline to terminate them. diff --git a/docs/getting-started.md b/docs/getting-started.md index 478f220..4e1afff 100644 --- a/docs/getting-started.md +++ b/docs/getting-started.md @@ -30,8 +30,9 @@ console.log(serialize(document.tree)); `parse()`, `parseBytes()`, and `parseStream()` return a frozen `ParsedDocument`. Its `tree` is the parsed document, `metadata` describes the exact parse, and `sourceText` is `null` unless source retention was requested. -`parseFragment()` instead returns a `FragmentTree` directly and requires the -namespace and local name of the context element. See [fragments](./parsing.md#fragments). +`parseFragment()` returns a frozen `ParsedFragment`: its `tree` is the fragment +and its `metadata` records successful resource use. It requires the namespace +and local name of the context element. See [fragments](./parsing.md#fragments). HTML recovery diagnostics do not normally throw. Invalid configuration, exceeded budgets, cancellation, stream failures, and invalid patch operations diff --git a/docs/limits-errors-and-safety.md b/docs/limits-errors-and-safety.md index 8de987b..848d60f 100644 --- a/docs/limits-errors-and-safety.md +++ b/docs/limits-errors-and-safety.md @@ -21,8 +21,10 @@ reader. | `maxTraceBytes` | Canonical UTF-8 bytes of retained events; valid only with `trace: "events"` | | `maxTimeMs` | Elapsed monotonic time across the operation | -Stream budgets also accept `maxEncodingPrescanBytes`, a prefix-retention cap -rather than a throwing total-input budget. Extraction has separate required +Stream budgets also accept `maxEncodingPrescanBytes`, an optional meta-encoding +prefix-retention cap rather than a throwing total-input budget. BOM detection +remains active when this value is zero. Eager stream tokenization accepts +`maxSteps` and reports the observed count when it is enabled. Extraction has separate required `maxOutputBytes` and `maxTokens` limits; visible-text extraction also requires `maxFallbackInputBytes` and `maxFallbackNodes`. diff --git a/docs/maintainers/testing.md b/docs/maintainers/testing.md index 069a8b9..71a38ab 100644 --- a/docs/maintainers/testing.md +++ b/docs/maintainers/testing.md @@ -74,10 +74,12 @@ classifications. Corpus details and refresh constraints are in `npm run oracle:documents` runs every one of the 1,698 scripting-invariant document cases in the pinned WPT tree-construction corpus plus 26 focused -product probes against Chromium, Firefox, and WebKit. The complete outcome and -known-difference inventories are fingerprinted per pinned browser version; a -browser update or any parser/browser result change requires an explicit -baseline review. +product probes against Chromium, Firefox, and WebKit. The complete outcome is +fingerprinted per pinned browser version. Every known difference also has an +exact case identifier, classification, explanation, and public/browser result +hashes; aggregate counts and hashes are secondary drift guards. A browser +update or any parser/browser result change requires an explicit baseline +review. `npm run qualification:serialization` verifies the exact pinned WPT serialization inventory, calls `dist/mod.js#serialize` for every applicable @@ -115,7 +117,8 @@ an oracle's behavior blindly. immediate-regression thresholds. - `npm run qualification:resources` compares bounded and unbounded parser work in isolated processes and requires every limit to fail at its first - unavailable unit while retaining less heap. + unavailable unit while retaining less heap. It also verifies that byte + parsing without source retention does not retain the complete decoded source. - `npm run qualification:mutation` requires every configured mutation to be killed; invalid or surviving mutations fail the command. @@ -132,3 +135,7 @@ three-browser oracles, fuzzing, resource and mutation checks, supply-chain evidence, and cross-revision performance. Both profiles fail at the first failed command and retain diagnostic reports under `reports/`; they do not assign an artificial quality score. + +The blocking CI runtime contract runs on Linux, macOS, and Windows. Linux also +executes the separate Node, Deno, Bun, and browser jobs so cross-runtime output +agreement and platform coverage remain distinct signals. diff --git a/docs/parsing.md b/docs/parsing.md index 66736ed..9269806 100644 --- a/docs/parsing.md +++ b/docs/parsing.md @@ -52,8 +52,8 @@ document or fragment and inherited by serialization and chunking. `parseFragment()` interprets input in the parsing context of an external HTML, SVG, or MathML element. The context is a descriptor, not a tag-name shortcut, because namespace and selected attributes affect tokenization and tree -construction. It returns a `FragmentTree`, not a `ParsedDocument`, and does -not support source retention. +construction. It returns a `ParsedFragment` containing a `FragmentTree` and +successful parse metadata. Fragment parsing does not support source retention. ```ts import { @@ -62,14 +62,14 @@ import { serialize } from "@ismail-elkorchi/html-parser"; -const fragment = parseFragment("AB", { +const { tree: fragment, metadata } = parseFragment("AB", { namespaceUri: HTML_NAMESPACE_URI, localName: "tbody" }, { budgets: { maxNodes: 32, maxDepth: 8 } }); -console.log(fragment.kind); // "fragment" +console.log(fragment.kind, metadata.resourceUsage.nodes); // "fragment", observed nodes console.log(serialize(fragment)); ``` @@ -92,7 +92,7 @@ parsing environment: descriptor is an HTML `form`. A context that is itself an HTML `form` is recognized directly and does not require a redundant option. -These values are retained on the returned fragment. `serialize(fragment)` and +These values are retained on the returned fragment tree. `serialize(fragment)` and `chunk(fragment)` inherit its scripting mode; an explicit serialization option still overrides it. Supply environment values from the real context document when parity with browser `innerHTML` parsing matters. @@ -100,7 +100,7 @@ when parity with browser `innerHTML` parsing matters. ## Parse diagnostics Malformed HTML is normally recovered according to HTML parsing rules. The -result keeps non-fatal diagnostics in `tree.errors` or `fragment.errors`; each +result keeps non-fatal diagnostics in its tree's `errors` array; each entry has a stable `parseErrorId`, location data when available, and a short message. `getParseErrorSpecRef()` returns the dedicated HTML Standard anchor for a named tokenizer or input-stream error. Unnamed tree-construction errors diff --git a/docs/streams-and-encoding.md b/docs/streams-and-encoding.md index 97f4ad7..98ccd06 100644 --- a/docs/streams-and-encoding.md +++ b/docs/streams-and-encoding.md @@ -48,8 +48,10 @@ During conversion to the immutable public result, already-converted internal children and attributes are released so the mutable and immutable trees are not both retained in full at the peak. -`tokenizeByteStreamEager()` has the same eager boundary: it returns all tokens -after EOF and does not expose tokens progressively. +`tokenizeByteStreamEager()` has the same eager boundary: it returns a frozen +`{ tokens, metadata }` result after EOF and does not expose tokens +progressively. Set `budgets.maxSteps` to impose deterministic tokenizer work; +the observed count is then returned in `metadata.resourceUsage.steps`. ## Encoding selection @@ -57,11 +59,19 @@ Byte and stream entry points use the same HTML sniffing and decoding pipeline. `metadata.encoding.source` reports `"bom"`, `"transport"`, `"meta"`, or `"default"`; `metadata.encoding.name` reports the selected WHATWG encoding. -`maxEncodingPrescanBytes` limits how much transport prefix is retained for -encoding prescan. It is not a total-input budget and later bytes do not make it -throw. The implementation never retains more than 16,384 bytes for prescan, -even when a larger value is configured. Use `maxInputBytes` for total transport -bytes and `maxDecodedUtf8Bytes` for decoded UTF-8 size. +Mandatory BOM detection always examines the required prefix of up to three +bytes and is independent of `maxEncodingPrescanBytes`. That option limits only +the prefix retained for optional `` prescanning; zero disables +meta prescanning without disabling UTF-8 or UTF-16 BOM recognition. It is not a +total-input budget and later bytes do not make it throw. The implementation +never retains more than 16,384 bytes for optional meta prescan, even when a +larger value is configured. + +Decoding starts as soon as the higher-priority evidence is final: after a BOM, +after BOM absence plus a recognized transport label, or after a complete early +meta declaration. Otherwise the default is selected at the configured cap or +EOF. Use `maxInputBytes` for total transport bytes and +`maxDecodedUtf8Bytes` for decoded UTF-8 size. ## Cancellation and stream failures diff --git a/scripts/mutation/config.json b/scripts/mutation/config.json index 2828128..2849bec 100644 --- a/scripts/mutation/config.json +++ b/scripts/mutation/config.json @@ -31,7 +31,12 @@ }, { "targetFile": "dist/public/serialization.js", - "testCommand": ["node", "--test", "test/behavior/fragment-context.test.js"], + "testCommand": [ + "node", + "--test", + "test/behavior/fragment-context.test.js", + "test/behavior/serialization.test.js" + ], "mutants": [ { "id": "parsed-tree-serializer-scripting-inheritance", @@ -41,6 +46,54 @@ } ] }, + { + "targetFile": "dist/public/tree-validation.js", + "testCommand": ["node", "--test", "test/behavior/serialization.test.js"], + "mutants": [ + { + "id": "serializer-element-name-preflight", + "description": "Allow an unsafe caller-constructed element name through serializer preflight.", + "search": "if (!isHtmlElementName(localName)) {", + "replace": "if (false && !isHtmlElementName(localName)) {" + }, + { + "id": "serializer-serialized-attribute-uniqueness", + "description": "Allow two caller attributes to emit the same serialized HTML name.", + "search": "if (serializedNames.has(identity.serializedName)) {", + "replace": "if (false && serializedNames.has(identity.serializedName)) {" + }, + { + "id": "serializer-template-content-ownership", + "description": "Allow an HTML template element without its owned template-content node.", + "search": "if (templateContent === undefined) {\n invalid(`${entry.path}.templateContent`, \"must be present on an HTML template element\");\n }\n pending.push({\n value: templateContent,\n path: `${entry.path}.templateContent`,\n placement: \"template\"\n });", + "replace": "if (templateContent !== undefined) {\n pending.push({\n value: templateContent,\n path: `${entry.path}.templateContent`,\n placement: \"template\"\n });\n }" + } + ] + }, + { + "targetFile": "dist/public/html-input.js", + "testCommand": ["node", "--test", "test/behavior/streaming.test.js"], + "mutants": [ + { + "id": "stream-early-meta-decision", + "description": "Delay a complete early meta-encoding decision until the prescan cap or EOF.", + "search": "pendingBytes <= 3 || lastBufferedByte === 0x3e ||", + "replace": "pendingBytes <= 3 || false ||" + } + ] + }, + { + "targetFile": "dist/public/model.js", + "testCommand": ["node", "--test", "test/behavior/traversal-extraction.test.js"], + "mutants": [ + { + "id": "query-element-attribute-shape", + "description": "Skip attribute-record validation at the shared query and traversal boundary.", + "search": "for (const attribute of attributes)\n validateAttributeShape(attribute, option);", + "replace": "for (const attribute of [])\n validateAttributeShape(attribute, option);" + } + ] + }, { "targetFile": "dist/public/chunking.js", "testCommand": [ @@ -185,7 +238,12 @@ }, { "targetFile": "dist/internal/encoding/sniff.js", - "testCommand": ["node", "--test", "test/behavior/encoding-sniff.test.js"], + "testCommand": [ + "node", + "--test", + "test/behavior/encoding-sniff.test.js", + "test/behavior/streaming.test.js" + ], "mutants": [ { "id": "encoding-alias-windows1252", @@ -210,6 +268,12 @@ "description": "Disable UTF-8 BOM detection.", "search": "if (bytes.length >= 3 && bytes[0] === 0xef && bytes[1] === 0xbb && bytes[2] === 0xbf) {", "replace": "if (false && bytes.length >= 3 && bytes[0] === 0xef && bytes[1] === 0xbb && bytes[2] === 0xbf) {" + }, + { + "id": "stream-bom-prefix-wait", + "description": "Stop waiting for the complete mandatory BOM prefix before lower-priority encoding evidence.", + "search": "if (bomState === \"pending\")\n return Object.freeze({ status: \"pending\" });", + "replace": "if (false && bomState === \"pending\")\n return Object.freeze({ status: \"pending\" });" } ] }, @@ -768,8 +832,8 @@ { "id": "product-trace-byte-accounting", "description": "Discard retained trace byte accounting from public resource observations.", - "search": "traceUtf8Bytes: trace.eventUtf8Bytes", - "replace": "traceUtf8Bytes: 0" + "search": "encodingPrescanBytes: input.encodingPrescanBytes,\n traceEvents: trace.eventCount,\n traceUtf8Bytes: trace.eventUtf8Bytes", + "replace": "encodingPrescanBytes: input.encodingPrescanBytes,\n traceEvents: trace.eventCount,\n traceUtf8Bytes: 0" }, { "id": "product-step-budget-forwarding", @@ -800,6 +864,24 @@ "description": "Discard decoded chunks instead of feeding tree construction before EOF.", "search": "requireInternalValue(engine, \"PUBLIC_PARSER_STREAM_ENGINE_NOT_INITIALIZED\").session.write(chunk);", "replace": "void chunk;" + }, + { + "id": "eager-tokenization-step-limit", + "description": "Drop the public eager-tokenization step limit before engine construction.", + "search": ": { maxSteps: normalized.budgets.maxSteps }),", + "replace": ": {})," + }, + { + "id": "eager-tokenization-step-evidence", + "description": "Hide the observed eager-tokenization step count after successful work.", + "search": "steps: normalized.budgets?.maxSteps === undefined ? null : usage.steps,", + "replace": "steps: null," + }, + { + "id": "fragment-step-evidence", + "description": "Hide the observed fragment parser step count after successful work.", + "search": "function fragmentResourceUsage(inputBytes, decodedCodeUnits, budgets, result, trace) {\n return Object.freeze({\n inputBytes,\n decodedUtf8Bytes: inputBytes,\n decodedCodeUnits,\n steps: budgets?.maxSteps === undefined ? null : result.resources.steps,", + "replace": "function fragmentResourceUsage(inputBytes, decodedCodeUnits, budgets, result, trace) {\n return Object.freeze({\n inputBytes,\n decodedUtf8Bytes: inputBytes,\n decodedCodeUnits,\n steps: null," } ] } diff --git a/scripts/oracles/document-browser-baseline.mjs b/scripts/oracles/document-browser-baseline.mjs new file mode 100644 index 0000000..8812ec9 --- /dev/null +++ b/scripts/oracles/document-browser-baseline.mjs @@ -0,0 +1,51 @@ +/** Expands reviewed known-difference groups into an exact per-engine case inventory. */ +export function expandKnownDifferenceGroups(groups, engineNames) { + if (!Array.isArray(groups)) { + throw new Error("Document browser baseline requires reviewed known-difference groups"); + } + const byEngine = new Map(engineNames.map((name) => [name, new Map()])); + for (const group of groups) { + if (typeof group?.classification !== "string" || group.classification.length === 0 || + typeof group.explanation !== "string" || group.explanation.length === 0 || + !Array.isArray(group.engines) || group.engines.length === 0) { + throw new Error("Document browser baseline has an invalid known-difference classification"); + } + if (group.caseIds !== undefined && !Array.isArray(group.caseIds)) { + throw new Error("Document browser baseline has an invalid case-id inventory"); + } + const caseIds = [...(group.caseIds ?? [])]; + if (group.caseIdPrefix !== undefined || group.caseNumbers !== undefined) { + if (typeof group.caseIdPrefix !== "string" || !Array.isArray(group.caseNumbers)) { + throw new Error("Document browser baseline has an invalid case-id expansion"); + } + for (const caseNumber of group.caseNumbers) { + if (!Number.isSafeInteger(caseNumber) || caseNumber < 1) { + throw new Error("Document browser baseline case numbers must be positive integers"); + } + caseIds.push(`${group.caseIdPrefix}${String(caseNumber)}`); + } + } + if (caseIds.length === 0 || caseIds.some((id) => typeof id !== "string" || id.length === 0)) { + throw new Error("Document browser baseline classification must identify every case"); + } + if (new Set(group.engines).size !== group.engines.length) { + throw new Error("Document browser baseline repeats an engine in one classification"); + } + for (const engine of group.engines) { + const inventory = byEngine.get(engine); + if (inventory === undefined) { + throw new Error(`Document browser baseline names unsupported engine ${String(engine)}`); + } + for (const id of caseIds) { + if (inventory.has(id)) { + throw new Error(`Document browser baseline classifies ${id} twice for ${engine}`); + } + inventory.set(id, Object.freeze({ + classification: group.classification, + explanation: group.explanation + })); + } + } + } + return byEngine; +} diff --git a/scripts/oracles/run-document-browser-diff.mjs b/scripts/oracles/run-document-browser-diff.mjs index 86f9e94..192c655 100644 --- a/scripts/oracles/run-document-browser-diff.mjs +++ b/scripts/oracles/run-document-browser-diff.mjs @@ -8,6 +8,7 @@ import { parse } from "../../dist/mod.js"; import { parseTreeDatFixtures } from "../../test/support/tree-dat.mjs"; import { verifyWptTreeCorpus } from "../../test/support/wpt-tree-corpus.mjs"; import { writeJson } from "../lib/report.mjs"; +import { expandKnownDifferenceGroups } from "./document-browser-baseline.mjs"; const BASELINE_PATH = "test/fixtures/qualification/document-browser-baseline.json"; @@ -150,6 +151,10 @@ if (baseline.schemaVersion !== 1 || typeof baseline.engines !== "object" || baseline.engines === null) { throw new Error("Document browser baseline does not match the pinned qualification inputs"); } +const knownDifferencesByEngine = expandKnownDifferenceGroups( + baseline.knownDifferenceGroups, + Object.keys(AVAILABLE_ENGINES) +); const results = []; for (const [name, launcher] of engines) { @@ -186,8 +191,11 @@ for (const [name, launcher] of engines) { const differenceSha256 = sha256(differenceRecords); const version = browser.version(); const expected = baseline.engines[name]; + const expectedKnownDifferences = knownDifferencesByEngine.get(name); const baselineMismatches = []; - if (expected === undefined) baselineMismatches.push("engine-not-baselined"); + if (expected === undefined || expectedKnownDifferences === undefined) { + baselineMismatches.push("engine-not-baselined"); + } else { if (expected.version !== version) baselineMismatches.push("browser-version"); if (expected.outcomesSha256 !== outcomeSha256) baselineMismatches.push("all-outcomes"); @@ -195,8 +203,17 @@ for (const [name, launcher] of engines) { if (expected.differencesSha256 !== differenceSha256) { baselineMismatches.push("difference-inventory"); } + const expectedCaseIds = [...expectedKnownDifferences.keys()].sort(); + const actualCaseIds = differenceRecords.map(({ id }) => id).sort(); + if (JSON.stringify(expectedCaseIds) !== JSON.stringify(actualCaseIds)) { + baselineMismatches.push("difference-case-identifiers"); + } } const baselineMatches = baselineMismatches.length === 0; + const classifiedDifferences = differenceRecords.map((difference) => ({ + ...difference, + ...expectedKnownDifferences?.get(difference.id) + })); results.push({ name, version, @@ -206,7 +223,7 @@ for (const [name, launcher] of engines) { knownDifferences: { count: failures.length, sha256: differenceSha256, - reason: baseline.reason + cases: classifiedDifferences }, baselineMismatches, failures: baselineMatches ? [] : failures diff --git a/scripts/oracles/run-public-fragment-browser-diff.mjs b/scripts/oracles/run-public-fragment-browser-diff.mjs index 78a4559..9fd2c01 100644 --- a/scripts/oracles/run-public-fragment-browser-diff.mjs +++ b/scripts/oracles/run-public-fragment-browser-diff.mjs @@ -140,7 +140,7 @@ function normalizePublicNode(node) { } function normalizePublic(testCase) { - return parseFragment(testCase.html, testCase.context, testCase.options).children.map(normalizePublicNode); + return parseFragment(testCase.html, testCase.context, testCase.options).tree.children.map(normalizePublicNode); } async function normalizeBrowser(page, testCase) { diff --git a/scripts/qualification/run-public-fragments.mjs b/scripts/qualification/run-public-fragments.mjs index 5234ac9..c9c1c0d 100644 --- a/scripts/qualification/run-public-fragments.mjs +++ b/scripts/qualification/run-public-fragments.mjs @@ -26,7 +26,7 @@ for (const relativePath of corpus.fixtureFiles) { baseCases += fixtureCases.length; for (const fixtureCase of expandTreeDatCases(fixtureCases, { includeModeInId: true })) { executions += 1; - const fragment = parseFragment( + const { tree: fragment } = parseFragment( fixtureCase.data, { ...fixtureCase.fragmentContext, attributes: [] }, { diff --git a/scripts/qualification/run-public-serialization-roundtrip.mjs b/scripts/qualification/run-public-serialization-roundtrip.mjs index b337184..49d2687 100644 --- a/scripts/qualification/run-public-serialization-roundtrip.mjs +++ b/scripts/qualification/run-public-serialization-roundtrip.mjs @@ -72,9 +72,9 @@ function normalizeNodes(nodes) { const failures = []; const positive = []; for (const testCase of POSITIVE_CASES) { - const first = parseFragment(testCase.html, HTML_DIV_CONTEXT); + const first = parseFragment(testCase.html, HTML_DIV_CONTEXT).tree; const serialized = serialize(first); - const second = parseFragment(serialized, HTML_DIV_CONTEXT); + const second = parseFragment(serialized, HTML_DIV_CONTEXT).tree; const before = normalizeNodes(first.children); const after = normalizeNodes(second.children); const stable = JSON.stringify(before) === JSON.stringify(after); @@ -87,16 +87,16 @@ for (const testCase of POSITIVE_CASES) { if (!stable) failures.push({ id: testCase.id, reason: "unexpected-roundtrip-difference" }); } -const plaintextFirst = parseFragment("a<b>", HTML_DIV_CONTEXT); +const plaintextFirst = parseFragment("<plaintext>a<b>", HTML_DIV_CONTEXT).tree; const plaintextSerialized = serialize(plaintextFirst); -const plaintextSecond = parseFragment(plaintextSerialized, HTML_DIV_CONTEXT); +const plaintextSecond = parseFragment(plaintextSerialized, HTML_DIV_CONTEXT).tree; const rawCase = PUBLIC_SERIALIZATION_QUALIFICATION_CASES.find( (testCase) => testCase.id === "classified/raw-effective-end-tag" ); if (rawCase === undefined) throw new Error("missing raw effective-end-tag qualification case"); const rawNode = createPublicSerializationNode(rawCase.descriptor); const rawSerialized = serialize(rawNode); -const rawSecond = parseFragment(rawSerialized, HTML_DIV_CONTEXT); +const rawSecond = parseFragment(rawSerialized, HTML_DIV_CONTEXT).tree; const classified = [ { diff --git a/scripts/qualification/run-public-serialization.mjs b/scripts/qualification/run-public-serialization.mjs index 0b393fc..8158aed 100644 --- a/scripts/qualification/run-public-serialization.mjs +++ b/scripts/qualification/run-public-serialization.mjs @@ -80,7 +80,7 @@ for (const testCase of PUBLIC_SERIALIZATION_QUALIFICATION_CASES) { const parsedWptOutcomes = []; for (let index = 0; index < WPT_SERIALIZING_OUTER_EXPECTATIONS.length; index += 1) { const expected = WPT_SERIALIZING_OUTER_EXPECTATIONS[index]; - const fragment = parseFragment(expected, HTML_DIV_CONTEXT); + const { tree: fragment } = parseFragment(expected, HTML_DIV_CONTEXT); const node = fragment.children[0]; const actual = node === undefined ? "" : serialize(node); const id = `serializing.html#outerHTML-${String(index)}`; diff --git a/scripts/qualification/run-resource-limits.mjs b/scripts/qualification/run-resource-limits.mjs index d29e43a..14065be 100644 --- a/scripts/qualification/run-resource-limits.mjs +++ b/scripts/qualification/run-resource-limits.mjs @@ -61,6 +61,16 @@ function fixture(name) { expectedLimit: 65_536 }; } + if (name === "byte-source-none" || name === "byte-source-text") { + const sourceRetention = name === "byte-source-none" ? "none" : "text"; + return { + input: new Uint8Array(8 * 1024 * 1024).fill(0x20), + collectAfterRun: true, + run(mod, input) { + return mod.parseBytes(input, { sourceRetention }); + } + }; + } if (name === "trace") { return { input: "\0".repeat(2_000), @@ -96,6 +106,7 @@ if (workerCase) { } const wallMs = performance.now() - startedAt; const cpu = process.cpuUsage(cpuBefore); + if (selected.collectAfterRun === true) globalThis.gc(); const after = process.memoryUsage(); const peakRssBytes = process.resourceUsage().maxRSS * 1024; @@ -135,7 +146,8 @@ if (workerCase) { actual: failure.actual } : { code: "OK" }, - retainedResultKind: retainedResult?.tree?.kind ?? retainedResult?.kind ?? null + retainedResultKind: retainedResult?.tree?.kind ?? retainedResult?.kind ?? null, + retainedSourceCodeUnits: retainedResult?.sourceText?.length ?? 0 }; process.stdout.write(`${JSON.stringify(report)}\n`); process.exit(0); @@ -144,19 +156,28 @@ if (workerCase) { const fixtureNames = ["nodes", "attributes", "errors", "decoded-output", "trace"]; const comparisons = []; -for (const name of fixtureNames) { - const runs = {}; - for (const mode of ["unbounded", "bounded"]) { - const result = spawnSync( - process.execPath, - ["--expose-gc", fileURLToPath(import.meta.url), `--worker=${name}`, ...(mode === "bounded" ? ["--bounded"] : [])], - { encoding: "utf8", maxBuffer: 1024 * 1024 } - ); - if (result.status !== 0) { - throw new Error(`${name}/${mode} worker failed:\n${result.stderr || result.stdout}`); - } - runs[mode] = JSON.parse(result.stdout.trim()); +function runWorker(name, isBounded = false) { + const result = spawnSync( + process.execPath, + [ + "--expose-gc", + fileURLToPath(import.meta.url), + `--worker=${name}`, + ...(isBounded ? ["--bounded"] : []) + ], + { encoding: "utf8", maxBuffer: 1024 * 1024 } + ); + if (result.status !== 0) { + throw new Error(`${name}/${isBounded ? "bounded" : "unbounded"} worker failed:\n${result.stderr || result.stdout}`); } + return JSON.parse(result.stdout.trim()); +} + +for (const name of fixtureNames) { + const runs = { + unbounded: runWorker(name), + bounded: runWorker(name, true) + }; if (runs.bounded.retainedHeapDeltaBytes >= runs.unbounded.retainedHeapDeltaBytes) { throw new Error(`${name}: bounded retained heap was not lower than the unbounded run`); } @@ -169,6 +190,19 @@ for (const name of fixtureNames) { }); } +const byteSourceRetention = { + none: runWorker("byte-source-none"), + text: runWorker("byte-source-text") +}; +if (byteSourceRetention.none.retainedSourceCodeUnits !== 0 || + byteSourceRetention.text.retainedSourceCodeUnits !== byteSourceRetention.text.inputBytes) { + throw new Error("byte source-retention workers returned the wrong public source shape"); +} +if (byteSourceRetention.none.retainedHeapDeltaBytes >= + byteSourceRetention.text.retainedHeapDeltaBytes) { + throw new Error("byte parsing without source retention did not retain less heap"); +} + const report = { schemaVersion: 1, suite: "html-parser-resource-limits", @@ -183,10 +217,17 @@ const report = { isolation: "one fresh process for each bounded or unbounded fixture", baseline: "explicit GC after fixture construction and before the parse", retainedHeap: "heapUsed sampled immediately after synchronous return or throw", + sourceRetentionHeap: "heapUsed sampled after explicit post-parse GC for byte source-retention fixtures", peakRss: "process.resourceUsage().maxRSS high-water mark", - assertion: "every bounded run reports limit + 1 and retains less heap than its paired unbounded run" + assertion: "hard budgets fail at limit + 1 and byte parsing retains decoded source only when requested" }, - comparisons + comparisons, + byteSourceRetention: { + ...byteSourceRetention, + retainedHeapReductionBytes: + byteSourceRetention.text.retainedHeapDeltaBytes - + byteSourceRetention.none.retainedHeapDeltaBytes + } }; await writeJson(options.report, report); diff --git a/scripts/smoke/browser-smoke.mjs b/scripts/smoke/browser-smoke.mjs index 7d3e085..2996ee2 100644 --- a/scripts/smoke/browser-smoke.mjs +++ b/scripts/smoke/browser-smoke.mjs @@ -84,10 +84,11 @@ async function runBrowserSmoke(baseUrl) { const serialized = mod.serialize(parsed); const parsedBytes = mod.parseBytes(new TextEncoder().encode(sampleHtml)).tree; const bytesSerialized = mod.serialize(parsedBytes); - const fragment = mod.parseFragment("child", { + const fragmentResult = mod.parseFragment("child", { namespaceUri: mod.HTML_NAMESPACE_URI, localName: "section" - }); + }, { budgets: { maxSteps: 10_000 } }); + const fragment = fragmentResult.tree; const streamParsed = (await mod.parseStream(streamFromText(sampleHtml))).tree; const streamSerialized = mod.serialize(streamParsed); @@ -108,15 +109,19 @@ async function runBrowserSmoke(baseUrl) { abortError = error; } - const tokenKinds = (await mod.tokenizeByteStreamEager( - streamFromText(sampleHtml) - )).map((token) => token.kind); + const tokenization = await mod.tokenizeByteStreamEager( + streamFromText(sampleHtml), + { budgets: { maxSteps: 10_000 } } + ); + const tokenKinds = tokenization.tokens.map((token) => token.kind); const stablePayload = { serialized, bytesSerialized, streamSerialized, fragmentContext: fragment.context, + fragmentResources: fragmentResult.metadata.resourceUsage, + tokenizationResources: tokenization.metadata.resourceUsage, tokenKinds }; const hashBuffer = await globalThis.crypto.subtle.digest( @@ -135,13 +140,16 @@ async function runBrowserSmoke(baseUrl) { parseBytes: bytesSerialized.includes("<p>smoke</p>"), parseStream: streamSerialized.includes("<p>smoke</p>"), parseFragment: fragment.context.localName === "section" && - fragment.context.namespaceUri === mod.HTML_NAMESPACE_URI, + fragment.context.namespaceUri === mod.HTML_NAMESPACE_URI && + Number.isSafeInteger(fragmentResult.metadata.resourceUsage.steps), traceSummary: traceSummary?.mode === "summary" && !("events" in traceSummary) && traceSummary.summary.eventCount > 0 && traceSummary.summary.eventKinds.includes("token") && JSON.stringify(traceSummary) === JSON.stringify(secondTraceSummary), - tokenizeByteStreamEager: tokenKinds.includes("startTag") && tokenKinds.includes("endTag"), + tokenizeByteStreamEager: tokenKinds.includes("startTag") && + tokenKinds.includes("endTag") && + Number.isSafeInteger(tokenization.metadata.resourceUsage.steps), errorGuards: typeof mod.isHtmlBudgetExceededError === "function" && mod.isHtmlBudgetExceededError(budgetError) && budgetError.code === "BUDGET_EXCEEDED" && diff --git a/scripts/smoke/control.mjs b/scripts/smoke/control.mjs index 1a9b539..946b2d7 100644 --- a/scripts/smoke/control.mjs +++ b/scripts/smoke/control.mjs @@ -149,7 +149,15 @@ async function computeDeterminismHash() { const canonicalPayload = { node: normalizeNode(parsed), - parseErrors: Array.isArray(parsed.errors) ? parsed.errors.map((entry) => normalizeParseError(entry)) : [] + parseErrors: Array.isArray(parsed.errors) ? parsed.errors.map((entry) => normalizeParseError(entry)) : [], + fragment: parseFragment("<p a=1>x</p>", { + namespaceUri: HTML_NAMESPACE_URI, + localName: "section" + }, { budgets: { maxSteps: 10_000 } }), + tokenization: await tokenizeByteStreamEager( + createByteStream([new TextEncoder().encode("<p a=1>x</p>")]), + { budgets: { maxSteps: 10_000 } } + ) }; return sha256Hex(JSON.stringify(canonicalPayload)); @@ -198,11 +206,16 @@ async function runSmokeAssertions() { const second = parse("deterministic"); ensure(JSON.stringify(first) === JSON.stringify(second), "deterministic output mismatch"); - const fragment = parseFragment("child", { + const fragmentResult = parseFragment("child", { namespaceUri: HTML_NAMESPACE_URI, localName: "section" - }); + }, { budgets: { maxSteps: 10_000 } }); + const { tree: fragment } = fragmentResult; ensure(fragment.context.localName === "section", "fragment context mismatch"); + ensure( + Number.isSafeInteger(fragmentResult.metadata.resourceUsage.steps), + "fragment resource metadata mismatch" + ); const sampleBytes = new Uint8Array([ 0x3c, 0x6d, 0x65, 0x74, 0x61, 0x20, 0x63, 0x68, 0x61, 0x72, 0x73, 0x65, 0x74, 0x3d, 0x77, 0x69, 0x6e, 0x64, @@ -218,13 +231,19 @@ async function runSmokeAssertions() { "parseStream output mismatch vs parseBytes" ); - const tokenKinds = (await tokenizeByteStreamEager( - createByteStream([new TextEncoder().encode("<p>smoke</p>")]) - )).map((token) => token.kind); + const tokenization = await tokenizeByteStreamEager( + createByteStream([new TextEncoder().encode("<p>smoke</p>")]), + { budgets: { maxSteps: 10_000 } } + ); + const tokenKinds = tokenization.tokens.map((token) => token.kind); ensure( JSON.stringify(tokenKinds) === JSON.stringify(["startTag", "chars", "endTag", "eof"]), "tokenizeByteStreamEager mismatch" ); + ensure( + Number.isSafeInteger(tokenization.metadata.resourceUsage.steps), + "tokenization resource metadata mismatch" + ); const outlineResult = outline(parsed); ensure(outlineResult.entries.length === 0, "outline generation mismatch"); diff --git a/src/internal/encoding/sniff.ts b/src/internal/encoding/sniff.ts index 6109846..ea8cae1 100644 --- a/src/internal/encoding/sniff.ts +++ b/src/internal/encoding/sniff.ts @@ -1,17 +1,17 @@ -interface EncodingSniffOptions { +export interface EncodingSniffOptions { readonly transportEncodingLabel?: string; readonly maxPrescanBytes?: number; readonly defaultEncoding?: string; } -interface EncodingSniffResult { +export interface EncodingSniffResult { readonly encoding: string; readonly source: "bom" | "transport" | "meta" | "default"; } -interface HtmlByteDecodeOptions extends EncodingSniffOptions { - readonly onDecodedChunk?: (chunk: string) => void; -} +export type EncodingSniffDecision = + | { readonly status: "pending" } + | { readonly status: "decided"; readonly result: EncodingSniffResult }; const WINDOWS_1252_ALIASES = new Set([ "iso-8859-1", @@ -37,6 +37,33 @@ function detectBom(bytes: Uint8Array): string | null { return null; } +function bomPrefixState( + bytes: Uint8Array, + endOfStream: boolean +): "pending" | "absent" | "utf-8" | "utf-16be" | "utf-16le" { + const first = bytes[0]; + if (first === undefined) return endOfStream ? "absent" : "pending"; + if (first === 0xef) { + const second = bytes[1]; + if (second === undefined) return endOfStream ? "absent" : "pending"; + if (second !== 0xbb) return "absent"; + const third = bytes[2]; + if (third === undefined) return endOfStream ? "absent" : "pending"; + return third === 0xbf ? "utf-8" : "absent"; + } + if (first === 0xfe) { + const second = bytes[1]; + if (second === undefined) return endOfStream ? "absent" : "pending"; + return second === 0xff ? "utf-16be" : "absent"; + } + if (first === 0xff) { + const second = bytes[1]; + if (second === undefined) return endOfStream ? "absent" : "pending"; + return second === 0xfe ? "utf-16le" : "absent"; + } + return "absent"; +} + function stripQuotes(value: string): string { const trimmed = value.trim(); if ( @@ -291,31 +318,46 @@ export function sniffHtmlEncoding(bytes: Uint8Array, options: EncodingSniffOptio return { encoding: defaultEncoding, source: "default" }; } -export function decodeHtmlBytes(bytes: Uint8Array, options: HtmlByteDecodeOptions = {}): { text: string; sniff: EncodingSniffResult } { - const sniff = sniffHtmlEncoding(bytes, { - ...(options.transportEncodingLabel !== undefined - ? { transportEncodingLabel: options.transportEncodingLabel } - : {}), - ...(options.maxPrescanBytes !== undefined ? { maxPrescanBytes: options.maxPrescanBytes } : {}), - ...(options.defaultEncoding !== undefined ? { defaultEncoding: options.defaultEncoding } : {}) - }); - const decoder = new TextDecoder(sniff.encoding); - const parts: string[] = []; - const decodeChunkBytes = 16_384; - for (let offset = 0; offset < bytes.byteLength; offset += decodeChunkBytes) { - const decoded = decoder.decode(bytes.subarray(offset, offset + decodeChunkBytes), { stream: true }); - if (decoded.length > 0) { - options.onDecodedChunk?.(decoded); - parts.push(decoded); +/** Determines whether the available stream prefix is sufficient to select an encoding. */ +export function decideHtmlEncoding( + bytes: Uint8Array, + options: EncodingSniffOptions & { readonly endOfStream: boolean } +): EncodingSniffDecision { + const bomState = bomPrefixState(bytes, options.endOfStream); + if (bomState === "pending") return Object.freeze({ status: "pending" }); + if (bomState !== "absent") { + return Object.freeze({ + status: "decided", + result: Object.freeze({ encoding: bomState, source: "bom" }) + }); + } + + if (options.transportEncodingLabel !== undefined) { + const transport = canonicalizeLabel(options.transportEncodingLabel, "transport"); + if (transport !== null) { + return Object.freeze({ + status: "decided", + result: Object.freeze({ encoding: transport, source: "transport" }) + }); } } - const final = decoder.decode(); - if (final.length > 0) { - options.onDecodedChunk?.(final); - parts.push(final); + + const maxPrescanBytes = options.maxPrescanBytes ?? 16_384; + const meta = sniffMetaCharset(bytes, maxPrescanBytes); + if (meta !== null) { + return Object.freeze({ + status: "decided", + result: Object.freeze({ encoding: meta, source: "meta" }) + }); } - return { - text: parts.join(""), - sniff - }; + if (!options.endOfStream && bytes.byteLength < maxPrescanBytes) { + return Object.freeze({ status: "pending" }); + } + + const fallback = canonicalizeLabel(options.defaultEncoding ?? "windows-1252", "default") ?? + "windows-1252"; + return Object.freeze({ + status: "decided", + result: Object.freeze({ encoding: fallback, source: "default" }) + }); } diff --git a/src/internal/foundation/internal-state-error.ts b/src/internal/foundation/internal-state-error.ts index b8afa62..e206847 100644 --- a/src/internal/foundation/internal-state-error.ts +++ b/src/internal/foundation/internal-state-error.ts @@ -1,15 +1,17 @@ const INTERNAL_STATE_COMPONENT_BY_REASON = Object.freeze({ CHARACTER_REFERENCE_ASCII_CONSUMPTION_MISMATCH: "character-reference-consumer", + ENCODING_SNIFF_BOUNDED_PREFIX_UNDECIDED: "encoding-sniff", + ENCODING_SNIFF_EOF_UNDECIDED: "encoding-sniff", FOUNDATION_CURSOR_REQUESTED_MORE_AFTER_CLOSE: "foundation-driver", GENERATED_CHARACTER_REFERENCE_ENTRY_MISSING: "named-character-references", GENERATED_CHARACTER_REFERENCE_LENGTH_MISMATCH: "named-character-references", INPUT_CURSOR_BUFFER_UNDERRUN: "input-cursor", INPUT_CURSOR_RECONSUME_CHARACTER_MISSING: "input-cursor", PUBLIC_PARSER_CHILD_CONVERSION_MISSING: "public-parser", + PUBLIC_PARSER_BYTES_ENGINE_NOT_INITIALIZED: "public-parser", PUBLIC_PARSER_ROOT_CONVERSION_MISSING: "public-parser", PUBLIC_PARSER_STREAM_ENGINE_NOT_INITIALIZED: "public-parser", PUBLIC_PARSER_TEMPLATE_CONTENT_KIND_MISMATCH: "public-parser", - PUBLIC_TOKENIZER_RETAINED_TEXT_MISSING: "public-tokenizer", TOKENIZER_BUFFERED_CHARACTER_MISSING: "tokenizer", TOKENIZER_CDATA_BRACKET_POSITION_MISSING: "tokenizer", TOKENIZER_CHARACTER_REFERENCE_MISSING: "tokenizer", diff --git a/src/internal/foundation/name-validation.ts b/src/internal/foundation/name-validation.ts index 9dc0bd3..95fb490 100644 --- a/src/internal/foundation/name-validation.ts +++ b/src/internal/foundation/name-validation.ts @@ -7,6 +7,13 @@ const FORBIDDEN_HTML_ATTRIBUTE_SYNTAX = new Set([ 0x3e // > ]); +const FORBIDDEN_HTML_ELEMENT_SYNTAX = new Set([ + 0x20, // ASCII space + 0x2f, // / + 0x3c, // < + 0x3e // > +]); + function isControl(codePoint: number): boolean { return codePoint <= 0x1f || (codePoint >= 0x7f && codePoint <= 0x9f); } @@ -67,3 +74,17 @@ export function isHtmlAttributeName(value: string): boolean { } return true; } + +/** Whether a value can be emitted as one HTML tag name without changing tokenization. */ +export function isHtmlElementName(value: string): boolean { + if (value.length === 0) return false; + for (const character of value) { + const codePoint = character.codePointAt(0); + if (codePoint === undefined || isControl(codePoint) || + FORBIDDEN_HTML_ELEMENT_SYNTAX.has(codePoint) || + (codePoint >= 0xd800 && codePoint <= 0xdfff)) { + return false; + } + } + return true; +} diff --git a/src/public/chunking.ts b/src/public/chunking.ts index 0cf8e8d..ecad099 100644 --- a/src/public/chunking.ts +++ b/src/public/chunking.ts @@ -4,6 +4,7 @@ import { normalizeChunkOptions } from "./operation.ts"; import { serializeNodes } from "./serialization.ts"; +import { validateSerializableInput } from "./tree-validation.ts"; import type { Chunk, @@ -23,6 +24,7 @@ export function chunk(tree: DocumentTree | FragmentTree, options: ChunkOptions = startedAt ); operation.checkpoint(); + validateSerializableInput(tree, operation); const maxChars = normalizedOptions.maxChars ?? 8192; const maxNodes = normalizedOptions.maxNodes ?? 256; const maxBytes = normalizedOptions.maxBytes ?? Number.POSITIVE_INFINITY; diff --git a/src/public/html-input.ts b/src/public/html-input.ts index 7c03348..984351f 100644 --- a/src/public/html-input.ts +++ b/src/public/html-input.ts @@ -1,4 +1,8 @@ -import { sniffHtmlEncoding } from "../internal/encoding/sniff.ts"; +import { + decideHtmlEncoding, + sniffHtmlEncoding +} from "../internal/encoding/sniff.ts"; +import { failInternalState } from "../internal/foundation/internal-state-error.ts"; import { enforceBudget } from "./budgets.ts"; import { @@ -34,6 +38,12 @@ interface StreamDecoderState { readonly sniff: StreamEncodingSniff; } +interface DecodeCallbacks { + readonly retainText: boolean; + readonly onEncodingSniff?: (sniff: StreamEncodingSniff) => void; + readonly onDecodedChunk?: (chunk: string) => void; +} + export function requireString(value: unknown, option: string): asserts value is string { if (typeof value !== "string") { throw new HtmlConfigurationError(option, "INVALID_VALUE", "must be a string"); @@ -106,6 +116,96 @@ export class DecodedUtf8BudgetCounter { } } +class DecodedOutputCollector { + readonly #operation: OperationContext; + readonly #budget: DecodedUtf8BudgetCounter; + readonly #parts: string[] | null; + readonly #onDecodedChunk: ((chunk: string) => void) | undefined; + #codeUnits = 0; + + constructor( + limit: number | undefined, + callbacks: DecodeCallbacks, + operation: OperationContext + ) { + this.#operation = operation; + this.#budget = new DecodedUtf8BudgetCounter(limit, operation); + this.#parts = callbacks.retainText ? [] : null; + this.#onDecodedChunk = callbacks.onDecodedChunk; + } + + append(value: string): void { + if (value.length === 0) return; + this.#budget.append(value); + const nextCodeUnits = this.#codeUnits + value.length; + if (!Number.isSafeInteger(nextCodeUnits)) { + throw new HtmlConfigurationError( + "input", + "INVALID_VALUE", + "decoded input must fit in a safe UTF-16 code-unit count" + ); + } + this.#codeUnits = nextCodeUnits; + this.#onDecodedChunk?.(value); + this.#parts?.push(value); + this.#operation.checkpoint(); + } + + get bytes(): number { + return this.#budget.bytes; + } + + get codeUnits(): number { + return this.#codeUnits; + } + + text(): string | null { + return this.#parts?.join("") ?? null; + } +} + +function decodeTransportBytes( + decoder: TextDecoder, + bytes: Uint8Array, + output: DecodedOutputCollector, + operation: OperationContext +): void { + const decodeChunkBytes = 16_384; + for (let offset = 0; offset < bytes.byteLength; offset += decodeChunkBytes) { + operation.checkpoint(); + output.append(decoder.decode(bytes.subarray(offset, offset + decodeChunkBytes), { stream: true })); + } +} + +/** Decodes an in-memory byte input without retaining decoded text unless requested. */ +export function decodeByteArray( + bytes: Uint8Array, + options: DecodeCallbacks & { + readonly transportEncodingLabel?: string; + readonly maxDecodedUtf8Bytes?: number; + }, + operation: OperationContext +): StreamDecodeResult { + operation.checkpoint(); + const sniff = sniffHtmlEncoding(bytes, options.transportEncodingLabel === undefined + ? {} + : { transportEncodingLabel: options.transportEncodingLabel }); + options.onEncodingSniff?.(sniff); + const output = new DecodedOutputCollector(options.maxDecodedUtf8Bytes, options, operation); + const decoder = new TextDecoder(sniff.encoding); + decodeTransportBytes(decoder, bytes, output, operation); + output.append(decoder.decode()); + return Object.freeze({ + text: output.text(), + sniff, + totalBytes: bytes.byteLength, + decodedUtf8Bytes: output.bytes, + decodedCodeUnits: output.codeUnits, + encodingPrescanBytes: 0, + encodingPrescanLimitBytes: 0 + }); +} + async function readStreamChunk( reader: ReadableStreamDefaultReader<Uint8Array>, operation: OperationContext @@ -186,55 +286,32 @@ export async function decodeByteStream( DEFAULT_STREAM_ENCODING_PRESCAN_BYTES, budgets?.maxEncodingPrescanBytes ?? DEFAULT_STREAM_ENCODING_PRESCAN_BYTES ); - const pendingBytesBuffer = new Uint8Array(prescanLimit); + const pendingBytesBuffer = new Uint8Array(Math.max(3, prescanLimit)); let pendingBytes = 0; let encodingPrescanBytes = 0; let decoderState: StreamDecoderState | undefined; - const decodedParts: string[] | null = options.retainText ? [] : null; - let decodedCodeUnits = 0; - const decodedBudget = new DecodedUtf8BudgetCounter( + const output = new DecodedOutputCollector( budgets?.maxDecodedUtf8Bytes, + options, operation ); const sniffOptions = options.transportEncodingLabel === undefined ? { maxPrescanBytes: prescanLimit } : { transportEncodingLabel: options.transportEncodingLabel, maxPrescanBytes: prescanLimit }; - const appendDecoded = (value: string): void => { - if (value.length > 0) { - decodedBudget.append(value); - const nextCodeUnits = decodedCodeUnits + value.length; - if (!Number.isSafeInteger(nextCodeUnits)) { - throw new HtmlConfigurationError( - "input", - "INVALID_VALUE", - "decoded input must fit in a safe UTF-16 code-unit count" - ); - } - decodedCodeUnits = nextCodeUnits; - options.onDecodedChunk?.(value); - decodedParts?.push(value); - } - }; - const decodeBytes = (decoder: TextDecoder, bytes: Uint8Array): void => { - const decodeChunkBytes = 16_384; - for (let offset = 0; offset < bytes.byteLength; offset += decodeChunkBytes) { - operation.checkpoint(); - appendDecoded(decoder.decode(bytes.subarray(offset, offset + decodeChunkBytes), { stream: true })); - } - }; - const initializeDecoder = (): StreamDecoderState => { + const initializeDecoder = (endOfStream: boolean): StreamDecoderState | undefined => { const bufferedBytes = pendingBytesBuffer.subarray(0, pendingBytes); - const sniff = sniffHtmlEncoding(bufferedBytes, sniffOptions); + const decision = decideHtmlEncoding(bufferedBytes, { ...sniffOptions, endOfStream }); + if (decision.status === "pending") return undefined; + const sniff = decision.result; const state = { decoder: new TextDecoder(sniff.encoding), sniff }; decoderState = state; options.onEncodingSniff?.(sniff); - decodeBytes(state.decoder, bufferedBytes); + decodeTransportBytes(state.decoder, bufferedBytes, output, operation); return state; }; try { - if (prescanLimit === 0) initializeDecoder(); for (;;) { const next = await readStreamChunk(reader, operation); if (next.done) break; @@ -249,31 +326,70 @@ export async function decodeByteStream( operation.checkpoint(); if (decoderState === undefined) { - const bytesToBuffer = Math.min(chunkValue.byteLength, prescanLimit - pendingBytes); - pendingBytesBuffer.set(chunkValue.subarray(0, bytesToBuffer), pendingBytes); - pendingBytes += bytesToBuffer; - encodingPrescanBytes = Math.max(encodingPrescanBytes, pendingBytes); - if (pendingBytes < prescanLimit) continue; - const activeDecoder = initializeDecoder().decoder; - if (bytesToBuffer < chunkValue.byteLength) { - decodeBytes(activeDecoder, chunkValue.subarray(bytesToBuffer)); + let offset = 0; + while (offset < chunkValue.byteLength) { + const capacity = pendingBytesBuffer.byteLength - pendingBytes; + if (capacity === 0) { + const initialized = initializeDecoder(false); + if (initialized === undefined) { + failInternalState("ENCODING_SNIFF_BOUNDED_PREFIX_UNDECIDED"); + } + break; + } + let bytesToBuffer = Math.min(chunkValue.byteLength - offset, capacity); + if (pendingBytes < 3) { + bytesToBuffer = 1; + } else { + const nextTagEnd = chunkValue.indexOf(0x3e, offset); + if (nextTagEnd !== -1) { + bytesToBuffer = Math.min(bytesToBuffer, nextTagEnd - offset + 1); + } + } + pendingBytesBuffer.set( + chunkValue.subarray(offset, offset + bytesToBuffer), + pendingBytes + ); + pendingBytes += bytesToBuffer; + offset += bytesToBuffer; + encodingPrescanBytes = Math.max( + encodingPrescanBytes, + Math.min(pendingBytes, prescanLimit) + ); + const lastBufferedByte = pendingBytesBuffer[pendingBytes - 1]; + const initialized = pendingBytes <= 3 || lastBufferedByte === 0x3e || + pendingBytes === pendingBytesBuffer.byteLength + ? initializeDecoder(false) + : undefined; + if (initialized !== undefined) break; + } + const initializedDecoder = decoderState as StreamDecoderState | undefined; + if (initializedDecoder !== undefined && offset < chunkValue.byteLength) { + decodeTransportBytes( + initializedDecoder.decoder, + chunkValue.subarray(offset), + output, + operation + ); } continue; } - decodeBytes(decoderState.decoder, chunkValue); + decodeTransportBytes(decoderState.decoder, chunkValue, output, operation); } - const finalState = decoderState ?? initializeDecoder(); - appendDecoded(finalState.decoder.decode()); - return { - text: decodedParts?.join("") ?? null, + const finalState = decoderState ?? initializeDecoder(true); + if (finalState === undefined) { + failInternalState("ENCODING_SNIFF_EOF_UNDECIDED"); + } + output.append(finalState.decoder.decode()); + return Object.freeze({ + text: output.text(), sniff: finalState.sniff, totalBytes: total, - decodedUtf8Bytes: decodedBudget.bytes, - decodedCodeUnits, + decodedUtf8Bytes: output.bytes, + decodedCodeUnits: output.codeUnits, encodingPrescanBytes, encodingPrescanLimitBytes: prescanLimit - }; + }); } catch (error) { try { const cancellation = reader.cancel(error); diff --git a/src/public/model.ts b/src/public/model.ts index 7c85892..2ad9720 100644 --- a/src/public/model.ts +++ b/src/public/model.ts @@ -1,5 +1,83 @@ +import { HtmlConfigurationError } from "./errors.ts"; + import type { OperationContext } from "./operation.ts"; -import type { HtmlNode } from "./types.ts"; +import type { ElementNode, HtmlNode } from "./types.ts"; + +type UnknownRecord = Readonly<Record<PropertyKey, unknown>>; + +function invalidModel(option: string, expected: string): never { + throw new HtmlConfigurationError(option, "INVALID_VALUE", expected); +} + +function modelRecord(value: unknown, option: string): UnknownRecord { + if (typeof value !== "object" || value === null || Array.isArray(value)) { + invalidModel(option, "must be a node record"); + } + return value as UnknownRecord; +} + +function modelArray(value: unknown, option: string): readonly unknown[] { + if (!Array.isArray(value)) invalidModel(option, "must be a dense array"); + for (let index = 0; index < value.length; index += 1) { + if (!Object.hasOwn(value, index)) invalidModel(option, "must be a dense array"); + } + return value; +} + +function modelString(value: unknown, option: string): string { + if (typeof value !== "string") invalidModel(option, "must be a string"); + return value; +} + +function validateAttributeShape(value: unknown, option: string): void { + const attribute = modelRecord(value, option); + const namespaceUri = attribute["namespaceUri"]; + if (namespaceUri !== null && typeof namespaceUri !== "string") { + invalidModel(option, "must contain a string or null namespaceUri"); + } + modelString(attribute["localName"], option); + modelString(attribute["value"], option); + if (attribute["prefix"] !== undefined) modelString(attribute["prefix"], option); +} + +/** Validates the complete shape read by element query helpers. @internal */ +export function requireElementNode(value: unknown, option: string): ElementNode { + const element = modelRecord(value, option); + if (element["kind"] !== "element") invalidModel(option, "must be an element node"); + if (!Number.isSafeInteger(element["id"]) || (element["id"] as number) < 1) { + invalidModel(option, "must have a positive safe-integer id"); + } + modelString(element["namespaceUri"], option); + modelString(element["localName"], option); + if (element["prefix"] !== undefined) modelString(element["prefix"], option); + const attributes = modelArray(element["attributes"], option); + for (const attribute of attributes) validateAttributeShape(attribute, option); + modelArray(element["children"], option); + return value as ElementNode; +} + +function requireTraversableNode(value: unknown): HtmlNode { + const node = modelRecord(value, "tree"); + const kind = node["kind"]; + if (!Number.isSafeInteger(node["id"]) || (node["id"] as number) < 1) { + invalidModel("tree", "must contain nodes with positive safe-integer ids"); + } + if (kind === "element") return requireElementNode(value, "tree"); + if (kind === "templateContent") { + modelArray(node["children"], "tree"); + } else if (kind === "text" || kind === "comment") { + modelString(node["value"], "tree"); + } else if (kind === "processingInstruction") { + modelString(node["target"], "tree"); + modelString(node["data"], "tree"); + } else if (kind === "doctype") { + modelString(node["name"], "tree"); + modelRecord(node["externalId"], "tree"); + } else { + invalidModel("tree", "must contain only supported HTML node kinds"); + } + return value as HtmlNode; +} /** HTML namespace URI assigned by the tree builder. */ export const HTML_NAMESPACE_URI = "http://www.w3.org/1999/xhtml"; @@ -33,6 +111,24 @@ export function ownedChildNodes(node: HtmlNode): readonly HtmlNode[] { return node.templateContent === undefined ? node.children : [node.templateContent]; } +/** Rejects cyclic and multiply-owned caller graphs during public traversal. @internal */ +export class OwnedNodeTracker { + readonly #seen = new WeakSet<object>(); + + observe(value: unknown): HtmlNode { + const node = requireTraversableNode(value); + if (this.#seen.has(node)) { + throw new HtmlConfigurationError( + "tree", + "INVALID_VALUE", + "must be acyclic and give every node exactly one owner" + ); + } + this.#seen.add(node); + return node; + } +} + /** @internal */ export function* iterateNodes( nodes: readonly HtmlNode[], @@ -40,6 +136,7 @@ export function* iterateNodes( operation: OperationContext ): IterableIterator<{ readonly node: HtmlNode; readonly depth: number }> { const stack: { readonly node: HtmlNode; readonly depth: number }[] = []; + const ownership = new OwnedNodeTracker(); for (let index = nodes.length - 1; index >= 0; index -= 1) { const node = nodes[index]; if (node !== undefined) stack.push({ node, depth }); @@ -48,8 +145,9 @@ export function* iterateNodes( const entry = stack.pop(); if (entry === undefined) continue; operation.checkpoint(); - yield entry; - const descendants = ownedChildNodes(entry.node); + const node = ownership.observe(entry.node); + yield { node, depth: entry.depth }; + const descendants = ownedChildNodes(node); for (let index = descendants.length - 1; index >= 0; index -= 1) { const child = descendants[index]; if (child !== undefined) stack.push({ node: child, depth: entry.depth + 1 }); @@ -61,12 +159,14 @@ export function* iterateNodes( export function countNodes(node: HtmlNode, operation: OperationContext): number { let count = 0; const stack: HtmlNode[] = [node]; + const ownership = new OwnedNodeTracker(); while (stack.length > 0) { const current = stack.pop(); if (current === undefined) continue; operation.checkpoint(); + const node = ownership.observe(current); count += 1; - const descendants = ownedChildNodes(current); + const descendants = ownedChildNodes(node); for (let index = descendants.length - 1; index >= 0; index -= 1) { const child = descendants[index]; if (child !== undefined) stack.push(child); diff --git a/src/public/operation.ts b/src/public/operation.ts index 10deaa5..c803f75 100644 --- a/src/public/operation.ts +++ b/src/public/operation.ts @@ -41,6 +41,7 @@ const TOKENIZE_BYTE_STREAM_EAGER_BUDGET_KEYS = Object.freeze([ "maxInputBytes", "maxEncodingPrescanBytes", "maxDecodedUtf8Bytes", + "maxSteps", "maxParseErrors", "maxAttributesPerElement", "maxAttributeBytes", diff --git a/src/public/parsed-document-registry.ts b/src/public/parsed-output-registry.ts similarity index 65% rename from src/public/parsed-document-registry.ts rename to src/public/parsed-output-registry.ts index e8f3e9a..f3f5d1d 100644 --- a/src/public/parsed-document-registry.ts +++ b/src/public/parsed-output-registry.ts @@ -1,4 +1,9 @@ -import type { ParsedDocument, PatchPlan } from "./types.ts"; +import type { + DocumentTree, + FragmentTree, + ParsedDocument, + PatchPlan +} from "./types.ts"; interface ParsedDocumentRegistration { readonly sourceText: string | null; @@ -6,6 +11,7 @@ interface ParsedDocumentRegistration { } const parsedDocuments = new WeakMap<ParsedDocument, ParsedDocumentRegistration>(); +const parserOwnedTrees = new WeakSet<DocumentTree | FragmentTree>(); const patchPlans = new WeakMap<PatchPlan, ParsedDocument>(); /** Registers identity-bound state which must not be forgeable through public object shapes. */ @@ -15,6 +21,12 @@ export function registerParsedDocument( spansCaptured: boolean ): void { parsedDocuments.set(document, Object.freeze({ sourceText, spansCaptured })); + parserOwnedTrees.add(document.tree); +} + +/** Registers an immutable parser-owned fragment tree. */ +export function registerParsedFragmentTree(tree: FragmentTree): void { + parserOwnedTrees.add(tree); } /** Returns identity-bound parse state, or undefined for an unrecognized object. */ @@ -24,6 +36,11 @@ export function parsedDocumentRegistration( return parsedDocuments.get(document); } +/** Whether a root is a deeply frozen tree produced by this module instance. */ +export function isParserOwnedTree(value: unknown): value is DocumentTree | FragmentTree { + return typeof value === "object" && value !== null && parserOwnedTrees.has(value as DocumentTree); +} + /** Binds a frozen patch plan to the exact parsed document which produced it. */ export function registerPatchPlan(plan: PatchPlan, document: ParsedDocument): void { patchPlans.set(plan, document); diff --git a/src/public/parsing.ts b/src/public/parsing.ts index 3081042..8344c89 100644 --- a/src/public/parsing.ts +++ b/src/public/parsing.ts @@ -1,4 +1,3 @@ -import { decodeHtmlBytes } from "../internal/encoding/sniff.ts"; import { failInternalState, requireInternalValue @@ -19,7 +18,7 @@ import { enforceBudget } from "./budgets.ts"; import { HtmlAbortError, HtmlBudgetExceededError } from "./errors.ts"; import { normalizeFragmentContext, toEngineFragmentContext } from "./fragment-context.ts"; import { - DecodedUtf8BudgetCounter, + decodeByteArray, decodeByteStream, requireByteArray, requireReadableByteStream, @@ -36,7 +35,10 @@ import { normalizeTokenizeByteStreamEagerOptions } from "./operation.ts"; import { normalizeParseErrorId, TraceSink } from "./parse-trace.ts"; -import { registerParsedDocument } from "./parsed-document-registry.ts"; +import { + registerParsedDocument, + registerParsedFragmentTree +} from "./parsed-output-registry.ts"; import type { OperationContext } from "./operation.ts"; import type { @@ -49,15 +51,17 @@ import type { NodeId, ParseBytesOptions, ParseFragmentOptions, + ParsedFragment, ParseOptions, ParsedDocument, - ParsedDocumentMetadata, + ParseMetadata, ParseResourceUsage, ParseStreamOptions, Span, SpanProvenance, TemplateContentNode, Token, + TokenizeByteStreamEagerResult, TokenizeByteStreamEagerOptions } from "./types.ts"; import type { @@ -84,12 +88,12 @@ import type { } from "../internal/html-engine/tree-model.ts"; interface ParseInputContext { - readonly inputKind: ParsedDocumentMetadata["inputKind"]; + readonly inputKind: ParseMetadata["inputKind"]; readonly byteLength: number; readonly decodedUtf8ByteLength: number; readonly decodedCodeUnits: number; readonly transportByteLength: number | null; - readonly metadataEncoding: ParsedDocumentMetadata["encoding"]; + readonly metadataEncoding: ParseMetadata["encoding"]; readonly encodingPrescanBytes: number; readonly operation: OperationContext; readonly startedAt: number; @@ -780,6 +784,29 @@ function finishDocumentOperation( return parsed; } +function fragmentResourceUsage( + inputBytes: number, + decodedCodeUnits: number, + budgets: ParseFragmentOptions["budgets"], + result: HtmlEngineProductResult, + trace: TraceSink +): ParseResourceUsage { + return Object.freeze({ + inputBytes, + decodedUtf8Bytes: inputBytes, + decodedCodeUnits, + steps: budgets?.maxSteps === undefined ? null : result.resources.steps, + nodes: result.resources.nodes, + maxDepth: result.resources.maxDepth, + parseErrors: result.resources.parseErrors, + attributes: result.resources.attributes, + attributeUtf8Bytes: result.resources.attributeUtf8Bytes, + encodingPrescanBytes: 0, + traceEvents: trace.eventCount, + traceUtf8Bytes: trace.eventUtf8Bytes + }); +} + function parseDocumentOperation( html: string, options: ParseOptions | ParseBytesOptions | ParseStreamOptions, @@ -844,28 +871,65 @@ export function parseBytes( const operation = createOperationContext(normalized.budgets?.maxTimeMs, normalized.signal, startedAt); operation.checkpoint(); enforceBudget("maxInputBytes", normalized.budgets?.maxInputBytes, bytes.byteLength); - const decodedBudget = new DecodedUtf8BudgetCounter( - normalized.budgets?.maxDecodedUtf8Bytes, + const trace = new TraceSink( + normalized.trace ?? "none", + normalized.onTraceEvent, + normalized.budgets, operation ); - const decoded = decodeHtmlBytes(bytes, { - ...(normalized.transportEncodingLabel === undefined - ? {} - : { transportEncodingLabel: normalized.transportEncodingLabel }), - onDecodedChunk(chunk): void { decodedBudget.append(chunk); } - }); - return parseDocumentOperation(decoded.text, normalized, { + let engine: ReturnType<typeof createDocumentEngine> | undefined; + let decoded; + try { + decoded = decodeByteArray(bytes, { + retainText: normalized.sourceRetention === "text", + ...(normalized.transportEncodingLabel === undefined + ? {} + : { transportEncodingLabel: normalized.transportEncodingLabel }), + ...(normalized.budgets?.maxDecodedUtf8Bytes === undefined + ? {} + : { maxDecodedUtf8Bytes: normalized.budgets.maxDecodedUtf8Bytes }), + onEncodingSniff(sniff): void { + trace.emit({ + kind: "decode", + source: "sniff", + encoding: sniff.encoding, + sniffSource: sniff.source + }); + engine = createDocumentEngine(normalized, startedAt, trace); + }, + onDecodedChunk(chunk): void { + requireInternalValue( + engine, + "PUBLIC_PARSER_BYTES_ENGINE_NOT_INITIALIZED" + ).session.write(chunk); + } + }, operation); + } catch (error) { + return mapEngineFailure(error); + } + const activeEngine = requireInternalValue( + engine, + "PUBLIC_PARSER_BYTES_ENGINE_NOT_INITIALIZED" + ); + let result: HtmlEngineProductResult; + try { + result = activeEngine.session.finishForPublicConversion(); + } catch (error) { + return mapEngineFailure(error); + } + trace.emitBudget("maxInputBytes", normalized.budgets?.maxInputBytes, bytes.byteLength); + return finishDocumentOperation(decoded.text, normalized, { inputKind: "bytes", byteLength: bytes.byteLength, - decodedUtf8ByteLength: decodedBudget.bytes, - decodedCodeUnits: decoded.text.length, + decodedUtf8ByteLength: decoded.decodedUtf8Bytes, + decodedCodeUnits: decoded.decodedCodeUnits, transportByteLength: bytes.byteLength, metadataEncoding: { name: decoded.sniff.encoding, source: decoded.sniff.source }, encodingPrescanBytes: 0, operation, startedAt, decode: { source: "sniff", encoding: decoded.sniff.encoding, sniffSource: decoded.sniff.source } - }); + }, trace, result, activeEngine.state.tokenCount); } /** @@ -879,14 +943,14 @@ export function parseBytes( * namespaceUri: HTML_NAMESPACE_URI, * localName: "tr" * }); - * console.log(fragment.children.length); + * console.log(fragment.tree.children.length); * ``` */ export function parseFragment( html: string, contextInput: HtmlFragmentContextInput, options: ParseFragmentOptions = {} -): FragmentTree { +): ParsedFragment { const startedAt = performance.now(); const normalized = normalizeParseFragmentOptions(options); requireString(html, "input"); @@ -989,7 +1053,7 @@ export function parseFragment( encodingPrescanBytes: null, encodingPrescanLimitBytes: null }); - return Object.freeze({ + const tree: FragmentTree = Object.freeze({ id: fragmentId, kind: "fragment", context, @@ -1000,6 +1064,22 @@ export function parseFragment( errors, ...(traceResult === undefined ? {} : { trace: traceResult }) }); + registerParsedFragmentTree(tree); + return Object.freeze({ + tree, + metadata: Object.freeze({ + inputKind: "text", + transportByteLength: null, + encoding: Object.freeze({ name: null, source: "already-decoded" }), + resourceUsage: fragmentResourceUsage( + inputBytes, + html.length, + normalized.budgets, + result, + trace + ) + }) + }); } function publicToken(token: HtmlToken, operation: OperationContext): Token { @@ -1039,16 +1119,16 @@ function publicToken(token: HtmlToken, operation: OperationContext): Token { export async function tokenizeByteStreamEager( stream: ReadableStream<Uint8Array>, options: TokenizeByteStreamEagerOptions = {} -): Promise<readonly Token[]> { +): Promise<TokenizeByteStreamEagerResult> { const startedAt = performance.now(); const normalized = normalizeTokenizeByteStreamEagerOptions(options); requireReadableByteStream(stream, "input"); const operation = createOperationContext(normalized.budgets?.maxTimeMs, normalized.signal, startedAt); - const decoded = await decodeByteStream(stream, { - ...normalized, - retainText: true - }, operation); + operation.checkpoint(); const limits: EngineResourceLimits = Object.freeze({ + ...(normalized.budgets?.maxSteps === undefined + ? {} + : { maxSteps: normalized.budgets.maxSteps }), ...(normalized.budgets?.maxParseErrors === undefined ? {} : { maxParseErrors: normalized.budgets.maxParseErrors }), @@ -1065,7 +1145,8 @@ export async function tokenizeByteStreamEager( const resources = createEngineResourceGuard({ limits, ...(normalized.signal === undefined ? {} : { signal: normalized.signal }), - startedAt + startedAt, + trackSteps: normalized.budgets?.maxSteps !== undefined }); const tokens: Token[] = []; const tokenizer = new HtmlTokenizer(resources, { @@ -1075,14 +1156,37 @@ export async function tokenizeByteStreamEager( return token.kind !== "start-tag" || token.selfClosing; } }, { reuseInputCharacters: true }); + let decoded; try { - tokenizer.write(requireInternalValue(decoded.text, "PUBLIC_TOKENIZER_RETAINED_TEXT_MISSING")); + decoded = await decodeByteStream(stream, { + ...normalized, + retainText: false, + onDecodedChunk(chunk): void { tokenizer.write(chunk); } + }, operation); tokenizer.close(); } catch (error) { return mapEngineFailure(error); } operation.checkpoint(); - return Object.freeze(tokens); + const usage = resources.snapshot(); + return Object.freeze({ + tokens: Object.freeze(tokens), + metadata: Object.freeze({ + inputKind: "stream", + transportByteLength: decoded.totalBytes, + encoding: Object.freeze({ name: decoded.sniff.encoding, source: decoded.sniff.source }), + resourceUsage: Object.freeze({ + inputBytes: decoded.totalBytes, + decodedUtf8Bytes: decoded.decodedUtf8Bytes, + decodedCodeUnits: decoded.decodedCodeUnits, + steps: normalized.budgets?.maxSteps === undefined ? null : usage.steps, + parseErrors: usage.parseErrors, + attributes: usage.attributes, + attributeUtf8Bytes: usage.attributeUtf8Bytes, + encodingPrescanBytes: decoded.encodingPrescanBytes + }) + }) + }); } /** diff --git a/src/public/patching.ts b/src/public/patching.ts index 4ca3508..57a16ed 100644 --- a/src/public/patching.ts +++ b/src/public/patching.ts @@ -15,7 +15,7 @@ import { parsedDocumentRegistration, patchPlanBelongsTo, registerPatchPlan -} from "./parsed-document-registry.ts"; +} from "./parsed-output-registry.ts"; import type { Edit, diff --git a/src/public/querying.ts b/src/public/querying.ts index d755a29..0cc68ad 100644 --- a/src/public/querying.ts +++ b/src/public/querying.ts @@ -1,9 +1,11 @@ +import { HtmlConfigurationError } from "./errors.ts"; import { requireString } from "./html-input.ts"; import { HTML_NAMESPACE_URI, asciiLowercase, isHtmlElement, - iterateNodes + iterateNodes, + requireElementNode } from "./model.ts"; import { createOperationContext, @@ -21,6 +23,42 @@ import type { OperationOptions } from "./types.ts"; +function invalid(option: string, expected: string): never { + throw new HtmlConfigurationError(option, "INVALID_VALUE", expected); +} + +function requireTree( + value: unknown, + option: string +): asserts value is DocumentTree | FragmentTree { + if (typeof value !== "object" || value === null || Array.isArray(value)) { + invalid(option, "must be a document or fragment tree"); + } + const source = value as Readonly<Record<PropertyKey, unknown>>; + if ((source["kind"] !== "document" && source["kind"] !== "fragment") || + !Array.isArray(source["children"])) { + invalid(option, "must be a document or fragment tree"); + } +} + +function requireCallback(value: unknown, option: string): asserts value is (...args: never[]) => unknown { + if (typeof value !== "function") invalid(option, "must be a function"); +} + +function requireNodeId(value: unknown, option: string): asserts value is NodeId { + if (!Number.isSafeInteger(value) || (value as number) < 1) { + invalid(option, "must be a positive safe integer"); + } +} + +function requireNullableString(value: unknown, option: string): asserts value is string | null { + if (value !== null) requireString(value, option); +} + +function requireOptionalString(value: unknown, option: string): asserts value is string | undefined { + if (value !== undefined) requireString(value, option); +} + /** Visits every node in depth-first tree order. */ export function walk( tree: DocumentTree | FragmentTree, @@ -29,6 +67,8 @@ export function walk( ): void { const startedAt = performance.now(); const normalizedOptions = normalizeOperationOptions(options); + requireTree(tree, "tree"); + requireCallback(visitor, "visitor"); const operation = createOperationContext( normalizedOptions.maxTimeMs, normalizedOptions.signal, @@ -49,6 +89,8 @@ export function walkElements( ): void { const startedAt = performance.now(); const normalizedOptions = normalizeOperationOptions(options); + requireTree(tree, "tree"); + requireCallback(visitor, "visitor"); const operation = createOperationContext( normalizedOptions.maxTimeMs, normalizedOptions.signal, @@ -71,6 +113,8 @@ export function findById( ): HtmlNode | null { const startedAt = performance.now(); const normalizedOptions = normalizeOperationOptions(options); + requireTree(tree, "tree"); + requireNodeId(id, "id"); const operation = createOperationContext( normalizedOptions.maxTimeMs, normalizedOptions.signal, @@ -91,11 +135,13 @@ export function getAttributeValue( node: Extract<HtmlNode, { kind: "element" }>, name: string ): string | undefined { - if (node.namespaceUri !== HTML_NAMESPACE_URI) { + const element = requireElementNode(node, "node"); + requireString(name, "name"); + if (element.namespaceUri !== HTML_NAMESPACE_URI) { return undefined; } const target = asciiLowercase(name); - for (const attribute of node.attributes) { + for (const attribute of element.attributes) { if (attribute.namespaceUri === null && asciiLowercase(attribute.localName) === target) { return attribute.value; } @@ -117,7 +163,10 @@ export function getAttributeValueNS( namespaceUri: string | null, localName: string ): string | undefined { - return node.attributes.find((attribute) => + const element = requireElementNode(node, "node"); + requireNullableString(namespaceUri, "namespaceUri"); + requireString(localName, "localName"); + return element.attributes.find((attribute) => attribute.namespaceUri === namespaceUri && attribute.localName === localName )?.value; } @@ -151,6 +200,8 @@ export function findAllByTagName( options: OperationOptions = {} ): IterableIterator<Extract<HtmlNode, { kind: "element" }>> { const startedAt = performance.now(); + requireTree(tree, "tree"); + requireString(tagName, "tagName"); const normalizedOptions = normalizeOperationOptions(options); const operation = createOperationContext( normalizedOptions.maxTimeMs, @@ -186,6 +237,7 @@ export function findAllByTagNameNS( options: OperationOptions = {} ): IterableIterator<Extract<HtmlNode, { kind: "element" }>> { const startedAt = performance.now(); + requireTree(tree, "tree"); requireString(namespaceUri, "namespaceUri"); requireString(localName, "localName"); const normalizedOptions = normalizeOperationOptions(options); @@ -231,6 +283,9 @@ export function findAllByAttr( options: OperationOptions = {} ): IterableIterator<Extract<HtmlNode, { kind: "element" }>> { const startedAt = performance.now(); + requireTree(tree, "tree"); + requireString(name, "name"); + requireOptionalString(value, "value"); const normalizedOptions = normalizeOperationOptions(options); const operation = createOperationContext( normalizedOptions.maxTimeMs, @@ -272,10 +327,10 @@ export function findAllByAttrNS( options: OperationOptions = {} ): IterableIterator<Extract<HtmlNode, { kind: "element" }>> { const startedAt = performance.now(); - if (namespaceUri !== null) { - requireString(namespaceUri, "namespaceUri"); - } + requireTree(tree, "tree"); + requireNullableString(namespaceUri, "namespaceUri"); requireString(localName, "localName"); + requireOptionalString(value, "value"); const normalizedOptions = normalizeOperationOptions(options); const operation = createOperationContext( normalizedOptions.maxTimeMs, diff --git a/src/public/serialization.ts b/src/public/serialization.ts index ed35c55..500be11 100644 --- a/src/public/serialization.ts +++ b/src/public/serialization.ts @@ -11,6 +11,7 @@ import { createOperationContext, normalizeSerializeOptions } from "./operation.ts"; +import { validateSerializableInput } from "./tree-validation.ts"; import type { OperationContext } from "./operation.ts"; import type { @@ -19,6 +20,7 @@ import type { FragmentTree, HtmlNode, HtmlScriptingMode, + SerializableNode, SerializeOptions } from "./types.ts"; @@ -128,7 +130,7 @@ export function serializeNodes( /** Serializes a parsed tree or one complete node representation as HTML. */ export function serialize( - tree: DocumentTree | FragmentTree | HtmlNode, + tree: DocumentTree | FragmentTree | SerializableNode, options: SerializeOptions = {} ): string { const startedAt = performance.now(); @@ -139,6 +141,7 @@ export function serialize( startedAt ); operation.checkpoint(); + validateSerializableInput(tree, operation); const scriptingMode = normalizedOptions.scriptingMode ?? (tree.kind === "document" || tree.kind === "fragment" ? tree.scriptingMode : "inert"); if (tree.kind === "document" || tree.kind === "fragment") { diff --git a/src/public/text-extraction.ts b/src/public/text-extraction.ts index 898a053..5693f5b 100644 --- a/src/public/text-extraction.ts +++ b/src/public/text-extraction.ts @@ -6,6 +6,7 @@ import { import { utf8ByteLength } from "./html-input.ts"; import { HTML_NAMESPACE_URI, + OwnedNodeTracker, asciiLowercase, isHtmlElement } from "./model.ts"; @@ -795,6 +796,7 @@ function* iterateVisibleExtractionChunks( } type Action = VisitAction | AppendAction; const stack: Action[] = []; + const ownership = new OwnedNodeTracker(); const pushVisits = ( nodes: readonly HtmlNode[], preserveWhitespace: boolean, @@ -846,6 +848,7 @@ function* iterateVisibleExtractionChunks( } const { node, preserveWhitespace, sourceOverride } = action; + ownership.observe(node); if (node.kind === "text") { pushAppend( node, @@ -1016,6 +1019,7 @@ function* iterateRawExtractionChunks( ? nodeOrTree.children : [nodeOrTree]; const stack: HtmlNode[] = []; + const ownership = new OwnedNodeTracker(); for (let index = roots.length - 1; index >= 0; index -= 1) { const root = roots[index]; if (root !== undefined) { @@ -1028,6 +1032,7 @@ function* iterateRawExtractionChunks( if (node === undefined) { continue; } + ownership.observe(node); if (node.kind === "text") { if (node.value.length > 0) { yield { @@ -1201,7 +1206,8 @@ function createTextOperations( }); } -const textOperations = createTextOperations(parseFragment); +const textOperations = createTextOperations((html, context, options) => + parseFragment(html, context, options).tree); /** * Iterates bounded policy tokens and returns the final result when fully drained. diff --git a/src/public/tree-validation.ts b/src/public/tree-validation.ts new file mode 100644 index 0000000..8f7229f --- /dev/null +++ b/src/public/tree-validation.ts @@ -0,0 +1,478 @@ +import { + isHtmlAttributeName, + isHtmlElementName, + isXmlLocalName +} from "../internal/foundation/name-validation.ts"; + +import { HtmlConfigurationError } from "./errors.ts"; +import { + HTML_NAMESPACE_URI, + MATHML_NAMESPACE_URI, + SVG_NAMESPACE_URI, + XLINK_NAMESPACE_URI, + XML_NAMESPACE_URI, + XMLNS_NAMESPACE_URI, + asciiLowercase +} from "./model.ts"; +import { isParserOwnedTree } from "./parsed-output-registry.ts"; + +import type { OperationContext } from "./operation.ts"; +import type { + Attribute, + DocumentTree, + FragmentTree, + SerializableNode, + Span +} from "./types.ts"; + +type UnknownRecord = Readonly<Record<PropertyKey, unknown>>; +type ChildPlacement = "document" | "fragment" | "element" | "template" | "standalone"; + +interface PendingNode { + readonly value: unknown; + readonly path: string; + readonly placement: ChildPlacement; +} + +function invalid(path: string, expected: string): never { + throw new HtmlConfigurationError(path, "INVALID_VALUE", expected); +} + +function record(value: unknown, path: string): UnknownRecord { + if (typeof value !== "object" || value === null || Array.isArray(value)) { + invalid(path, "must be a node record"); + } + return value as UnknownRecord; +} + +function read(value: UnknownRecord, key: PropertyKey, path: string): unknown { + try { + return value[key]; + } catch { + invalid(path, "must be a readable data property"); + } +} + +function string(value: unknown, path: string): string { + if (typeof value !== "string") invalid(path, "must be a string"); + return value; +} + +function optionalString(value: unknown, path: string): string | undefined { + if (value !== undefined && typeof value !== "string") { + invalid(path, "must be a string when provided"); + } + return value; +} + +function array(value: unknown, path: string): readonly unknown[] { + if (!Array.isArray(value)) invalid(path, "must be a dense array"); + for (let index = 0; index < value.length; index += 1) { + if (!Object.hasOwn(value, index)) invalid(`${path}[${String(index)}]`, "must be present"); + } + return value; +} + +function nodeId(value: unknown, path: string, ids: Set<number>): void { + if (!Number.isSafeInteger(value) || (value as number) < 1) { + invalid(path, "must be a positive safe integer"); + } + const id = value as number; + if (ids.has(id)) invalid(path, "must be unique within the serialized tree"); + ids.add(id); +} + +function span(value: unknown, path: string): Span | undefined { + if (value === undefined) return undefined; + const source = record(value, path); + const start = read(source, "start", `${path}.start`); + const end = read(source, "end", `${path}.end`); + if (!Number.isSafeInteger(start) || (start as number) < 0) { + invalid(`${path}.start`, "must be a non-negative safe integer"); + } + if (!Number.isSafeInteger(end) || (end as number) < (start as number)) { + invalid(`${path}.end`, "must be a safe integer no smaller than span.start"); + } + return value as Span; +} + +function nodeSourceFields(value: UnknownRecord, path: string): void { + const sourceSpan = span(read(value, "span", `${path}.span`), `${path}.span`); + const provenance = read(value, "spanProvenance", `${path}.spanProvenance`); + if (provenance !== undefined && provenance !== "input" && provenance !== "inferred") { + invalid(`${path}.spanProvenance`, 'must be "input" or "inferred" when provided'); + } + if (sourceSpan !== undefined && provenance !== "input") { + invalid(`${path}.spanProvenance`, 'must be "input" when span is present'); + } + if (provenance === "input" && sourceSpan === undefined) { + invalid(`${path}.span`, 'must be present when spanProvenance is "input"'); + } + if (provenance === "inferred" && sourceSpan !== undefined) { + invalid(`${path}.span`, 'must be absent when spanProvenance is "inferred"'); + } +} + +function expectedAttributePrefix( + namespaceUri: string, + localName: string +): string | undefined | null { + if (namespaceUri === XLINK_NAMESPACE_URI) return "xlink"; + if (namespaceUri === XML_NAMESPACE_URI) return "xml"; + if (namespaceUri === XMLNS_NAMESPACE_URI) return localName === "xmlns" ? undefined : "xmlns"; + return null; +} + +function validateAttribute( + value: unknown, + path: string, + htmlElement: boolean +): { readonly expandedName: string; readonly serializedName: string } { + const source = record(value, path); + const namespaceValue = read(source, "namespaceUri", `${path}.namespaceUri`); + if (namespaceValue !== null && (typeof namespaceValue !== "string" || namespaceValue.length === 0)) { + invalid(`${path}.namespaceUri`, "must be null or a non-empty namespace URI"); + } + const namespaceUri = namespaceValue; + const localName = string(read(source, "localName", `${path}.localName`), `${path}.localName`); + const prefix = optionalString(read(source, "prefix", `${path}.prefix`), `${path}.prefix`); + string(read(source, "value", `${path}.value`), `${path}.value`); + span(read(source, "span", `${path}.span`), `${path}.span`); + + if (namespaceUri === null) { + if (prefix !== undefined) invalid(`${path}.prefix`, "must be absent for an unnamespaced attribute"); + if (!isHtmlAttributeName(localName)) { + invalid(`${path}.localName`, "must be safely representable as one HTML attribute name"); + } + if (htmlElement && localName !== asciiLowercase(localName)) { + invalid(`${path}.localName`, "must use normalized ASCII lowercase on an HTML element"); + } + } else { + if (!isXmlLocalName(localName)) { + invalid(`${path}.localName`, "must be an XML local name for a namespaced attribute"); + } + if (prefix !== undefined && !isXmlLocalName(prefix)) { + invalid(`${path}.prefix`, "must be an XML local name when provided"); + } + const expectedPrefix = expectedAttributePrefix(namespaceUri, localName); + if (expectedPrefix !== null && prefix !== undefined && prefix !== expectedPrefix) { + invalid(`${path}.prefix`, "must match the reserved attribute namespace"); + } + if (expectedPrefix === null && (prefix === "xml" || prefix === "xmlns")) { + invalid(`${path}.prefix`, "must not use a reserved prefix for another namespace"); + } + } + + const serializedName = namespaceUri === null + ? localName + : namespaceUri === XML_NAMESPACE_URI + ? `xml:${localName}` + : namespaceUri === XMLNS_NAMESPACE_URI + ? localName === "xmlns" ? "xmlns" : `xmlns:${localName}` + : namespaceUri === XLINK_NAMESPACE_URI + ? `xlink:${localName}` + : prefix === undefined ? localName : `${prefix}:${localName}`; + return { + expandedName: `${namespaceUri ?? ""}\u0000${localName}`, + serializedName: asciiLowercase(serializedName) + }; +} + +function validateAttributes( + value: unknown, + path: string, + htmlElement: boolean +): readonly Attribute[] { + const attributes = array(value, path); + const expandedNames = new Set<string>(); + const serializedNames = new Set<string>(); + for (let index = 0; index < attributes.length; index += 1) { + const attributePath = `${path}[${String(index)}]`; + const identity = validateAttribute(attributes[index], attributePath, htmlElement); + if (expandedNames.has(identity.expandedName)) { + invalid(attributePath, "must have a unique namespace and local name"); + } + if (serializedNames.has(identity.serializedName)) { + invalid(attributePath, "must have a unique serialized HTML name"); + } + expandedNames.add(identity.expandedName); + serializedNames.add(identity.serializedName); + } + return value as readonly Attribute[]; +} + +function validateElementIdentity(value: UnknownRecord, path: string): { + readonly namespaceUri: string; + readonly localName: string; +} { + const namespaceUri = string( + read(value, "namespaceUri", `${path}.namespaceUri`), + `${path}.namespaceUri` + ); + if (namespaceUri.length === 0) invalid(`${path}.namespaceUri`, "must be a non-empty namespace URI"); + const localName = string(read(value, "localName", `${path}.localName`), `${path}.localName`); + const prefix = optionalString(read(value, "prefix", `${path}.prefix`), `${path}.prefix`); + const parserNamespace = namespaceUri === HTML_NAMESPACE_URI || + namespaceUri === SVG_NAMESPACE_URI || namespaceUri === MATHML_NAMESPACE_URI; + if (parserNamespace) { + if (prefix !== undefined) { + invalid(`${path}.prefix`, "must be absent for HTML, SVG, and MathML parser elements"); + } + if (!isHtmlElementName(localName)) { + invalid(`${path}.localName`, "must be safely representable as one HTML element name"); + } + if (namespaceUri === HTML_NAMESPACE_URI && localName !== asciiLowercase(localName)) { + invalid(`${path}.localName`, "must use normalized ASCII lowercase in the HTML namespace"); + } + } else { + if (!isXmlLocalName(localName)) { + invalid(`${path}.localName`, "must be an XML local name in a custom namespace"); + } + if (prefix !== undefined && !isXmlLocalName(prefix)) { + invalid(`${path}.prefix`, "must be an XML local name when provided"); + } + if (prefix === "xml" && namespaceUri !== XML_NAMESPACE_URI) { + invalid(`${path}.prefix`, "must bind the xml prefix to the XML namespace"); + } + if (prefix === "xmlns" || namespaceUri === XMLNS_NAMESPACE_URI) { + invalid(`${path}.namespaceUri`, "must not use the XMLNS namespace for an element"); + } + } + return { namespaceUri, localName }; +} + +function validateParseErrors(value: unknown, path: string): void { + const errors = array(value, path); + for (let index = 0; index < errors.length; index += 1) { + const errorPath = `${path}[${String(index)}]`; + const source = record(errors[index], errorPath); + if (read(source, "code", `${errorPath}.code`) !== "PARSER_ERROR") { + invalid(`${errorPath}.code`, 'must be "PARSER_ERROR"'); + } + string(read(source, "parseErrorId", `${errorPath}.parseErrorId`), `${errorPath}.parseErrorId`); + string(read(source, "message", `${errorPath}.message`), `${errorPath}.message`); + span(read(source, "span", `${errorPath}.span`), `${errorPath}.span`); + } +} + +function validateFragmentContextAttributes( + value: unknown, + path: string, + htmlContext: boolean +): void { + const attributes = array(value, path); + const expandedNames = new Set<string>(); + for (let index = 0; index < attributes.length; index += 1) { + const attributePath = `${path}[${String(index)}]`; + const source = record(attributes[index], attributePath); + const namespaceUri = read(source, "namespaceUri", `${attributePath}.namespaceUri`); + if (namespaceUri !== null && namespaceUri !== XLINK_NAMESPACE_URI && + namespaceUri !== XML_NAMESPACE_URI && namespaceUri !== XMLNS_NAMESPACE_URI) { + invalid( + `${attributePath}.namespaceUri`, + "must be null or the XLink, XML, or XMLNS namespace URI" + ); + } + const localName = string( + read(source, "localName", `${attributePath}.localName`), + `${attributePath}.localName` + ); + if (!isXmlLocalName(localName) || + (htmlContext && namespaceUri === null && localName !== asciiLowercase(localName))) { + invalid(`${attributePath}.localName`, "must be a normalized XML local name"); + } + string(read(source, "value", `${attributePath}.value`), `${attributePath}.value`); + const expandedName = `${namespaceUri ?? ""}\u0000${localName}`; + if (expandedNames.has(expandedName)) { + invalid(attributePath, "must have a unique namespace and local name"); + } + expandedNames.add(expandedName); + } +} + +function pushChildren( + children: readonly unknown[], + path: string, + placement: ChildPlacement, + pending: PendingNode[] +): void { + for (let index = children.length - 1; index >= 0; index -= 1) { + pending.push({ + value: children[index], + path: `${path}[${String(index)}]`, + placement + }); + } +} + +function validateRoot( + source: UnknownRecord, + kind: "document" | "fragment", + ids: Set<number>, + pending: PendingNode[] +): void { + nodeId(read(source, "id", "tree.id"), "tree.id", ids); + const scriptingMode = read(source, "scriptingMode", "tree.scriptingMode"); + if (scriptingMode !== "inert" && scriptingMode !== "disabled") { + invalid("tree.scriptingMode", 'must be "inert" or "disabled"'); + } + validateParseErrors(read(source, "errors", "tree.errors"), "tree.errors"); + const children = array(read(source, "children", "tree.children"), "tree.children"); + if (kind === "fragment") { + const documentMode = read(source, "documentMode", "tree.documentMode"); + if (documentMode !== "no-quirks" && documentMode !== "limited-quirks" && documentMode !== "quirks") { + invalid("tree.documentMode", "must be a supported document mode"); + } + if (typeof read(source, "hasFormInContextChain", "tree.hasFormInContextChain") !== "boolean") { + invalid("tree.hasFormInContextChain", "must be a boolean"); + } + const context = record(read(source, "context", "tree.context"), "tree.context"); + const namespaceUri = read(context, "namespaceUri", "tree.context.namespaceUri"); + if (namespaceUri !== HTML_NAMESPACE_URI && namespaceUri !== SVG_NAMESPACE_URI && + namespaceUri !== MATHML_NAMESPACE_URI) { + invalid("tree.context.namespaceUri", "must be the HTML, SVG, or MathML namespace URI"); + } + const localName = string( + read(context, "localName", "tree.context.localName"), + "tree.context.localName" + ); + if (!isXmlLocalName(localName) || + (namespaceUri === HTML_NAMESPACE_URI && localName !== asciiLowercase(localName))) { + invalid("tree.context.localName", "must be a normalized XML local name"); + } + validateFragmentContextAttributes( + read(context, "attributes", "tree.context.attributes"), + "tree.context.attributes", + namespaceUri === HTML_NAMESPACE_URI + ); + } + pushChildren(children, "tree.children", kind, pending); +} + +/** Validates every serializer-visible invariant before any output is emitted. */ +export function validateSerializableInput( + value: unknown, + operation: OperationContext +): asserts value is DocumentTree | FragmentTree | SerializableNode { + if (isParserOwnedTree(value)) return; + const root = record(value, "tree"); + const rootKind = read(root, "kind", "tree.kind"); + const ids = new Set<number>(); + const pending: PendingNode[] = []; + if (rootKind === "document" || rootKind === "fragment") { + validateRoot(root, rootKind, ids, pending); + } else { + pending.push({ value, path: "tree", placement: "standalone" }); + } + + const seen = new WeakMap<object, string>(); + const documentKinds = { doctypes: 0, elements: 0, elementSeen: false }; + while (pending.length > 0) { + operation.checkpoint(); + const entry = pending.pop(); + if (entry === undefined) continue; + const node = record(entry.value, entry.path); + const priorPath = seen.get(node); + if (priorPath !== undefined) { + invalid(entry.path, `must be uniquely owned; first encountered at ${priorPath}`); + } + seen.set(node, entry.path); + nodeId(read(node, "id", `${entry.path}.id`), `${entry.path}.id`, ids); + const kind = read(node, "kind", `${entry.path}.kind`); + if (kind === "templateContent") { + if (entry.placement !== "template") { + invalid(entry.path, "template content must be owned by exactly one HTML template element"); + } + const provenance = read(node, "spanProvenance", `${entry.path}.spanProvenance`); + if (provenance !== undefined && provenance !== "inferred") { + invalid(`${entry.path}.spanProvenance`, 'must be "inferred" when provided'); + } + const children = array(read(node, "children", `${entry.path}.children`), `${entry.path}.children`); + pushChildren(children, `${entry.path}.children`, "element", pending); + continue; + } + if (kind !== "element" && kind !== "text" && kind !== "comment" && + kind !== "processingInstruction" && kind !== "doctype") { + invalid(`${entry.path}.kind`, "must be a supported serializable node kind"); + } + if (entry.placement !== "standalone") { + if (kind === "doctype" && entry.placement !== "document") { + invalid(entry.path, "a doctype can only be a direct document child or standalone node"); + } + if (kind === "text" && entry.placement === "document") { + invalid(entry.path, "a text node cannot be a direct document child"); + } + } + if (entry.placement === "document") { + if (kind === "doctype") { + documentKinds.doctypes += 1; + if (documentKinds.doctypes > 1 || documentKinds.elementSeen) { + invalid(entry.path, "a document has at most one doctype before its element"); + } + } else if (kind === "element") { + documentKinds.elements += 1; + documentKinds.elementSeen = true; + if (documentKinds.elements > 1) invalid(entry.path, "a document has at most one element child"); + } + } + nodeSourceFields(node, entry.path); + if (kind === "text" || kind === "comment") { + string(read(node, "value", `${entry.path}.value`), `${entry.path}.value`); + continue; + } + if (kind === "processingInstruction") { + string(read(node, "target", `${entry.path}.target`), `${entry.path}.target`); + string(read(node, "data", `${entry.path}.data`), `${entry.path}.data`); + continue; + } + if (kind === "doctype") { + const name = string(read(node, "name", `${entry.path}.name`), `${entry.path}.name`); + if (!isHtmlElementName(name)) { + invalid(`${entry.path}.name`, "must be safely representable as one doctype name"); + } + const externalId = record( + read(node, "externalId", `${entry.path}.externalId`), + `${entry.path}.externalId` + ); + const externalKind = read(externalId, "kind", `${entry.path}.externalId.kind`); + if (externalKind === "public") { + string(read(externalId, "publicId", `${entry.path}.externalId.publicId`), `${entry.path}.externalId.publicId`); + const systemId = read(externalId, "systemId", `${entry.path}.externalId.systemId`); + if (systemId !== null) string(systemId, `${entry.path}.externalId.systemId`); + } else if (externalKind === "system") { + string(read(externalId, "systemId", `${entry.path}.externalId.systemId`), `${entry.path}.externalId.systemId`); + } else if (externalKind !== "none") { + invalid(`${entry.path}.externalId.kind`, 'must be "none", "public", or "system"'); + } + continue; + } + + const identity = validateElementIdentity(node, entry.path); + validateAttributes( + read(node, "attributes", `${entry.path}.attributes`), + `${entry.path}.attributes`, + identity.namespaceUri === HTML_NAMESPACE_URI + ); + const children = array(read(node, "children", `${entry.path}.children`), `${entry.path}.children`); + const templateContent = read(node, "templateContent", `${entry.path}.templateContent`); + const isTemplate = identity.namespaceUri === HTML_NAMESPACE_URI && identity.localName === "template"; + if (isTemplate) { + if (children.length !== 0) { + invalid(`${entry.path}.children`, "must be empty because an HTML template owns templateContent"); + } + if (templateContent === undefined) { + invalid(`${entry.path}.templateContent`, "must be present on an HTML template element"); + } + pending.push({ + value: templateContent, + path: `${entry.path}.templateContent`, + placement: "template" + }); + } else { + if (templateContent !== undefined) { + invalid(`${entry.path}.templateContent`, "is only valid on an HTML template element"); + } + pushChildren(children, `${entry.path}.children`, "element", pending); + } + } +} diff --git a/src/public/types.ts b/src/public/types.ts index 83edac8..ed4c532 100644 --- a/src/public/types.ts +++ b/src/public/types.ts @@ -152,15 +152,15 @@ export interface ParseFragmentOptions extends Omit<ParseOptions, "sourceRetentio readonly hasFormAncestor?: boolean; } -/** Parse limits for byte streams, including bounded encoding-prescan retention. */ +/** Parse limits for byte streams, including bounded optional meta-prescan retention. */ export interface ParseStreamBudgetOptions extends ParseBudgetOptions { - /** Transport bytes retained for encoding prescan; zero disables prescan retention. */ + /** Bytes retained for optional meta-encoding prescan; mandatory BOM detection is independent. */ readonly maxEncodingPrescanBytes?: number; } /** Options accepted by full-document byte-stream parsing. */ export interface ParseStreamOptions extends Omit<ParseBytesOptions, "budgets"> { - /** Parse and stream-prescan resource limits. */ + /** Parse and optional stream meta-prescan resource limits. */ readonly budgets?: ParseStreamBudgetOptions; } @@ -170,6 +170,7 @@ export type TokenizeByteStreamEagerBudgetOptions = Pick< | "maxInputBytes" | "maxEncodingPrescanBytes" | "maxDecodedUtf8Bytes" + | "maxSteps" | "maxParseErrors" | "maxAttributesPerElement" | "maxAttributeBytes" @@ -376,7 +377,7 @@ export interface TraceBudgetEvent { readonly status: "ok" | "exceeded"; } -/** Byte-stream consumption and encoding-prescan retention metrics. */ +/** Byte-stream consumption and optional meta-prescan retention metrics. */ export interface TraceStreamEvent { /** One-based event order within this parse. */ readonly seq: number; @@ -384,9 +385,9 @@ export interface TraceStreamEvent { readonly kind: "stream"; /** Total transport bytes read through EOF. */ readonly bytesRead: number; - /** Maximum transport bytes retained simultaneously for encoding prescan. */ + /** Maximum transport bytes retained simultaneously for optional meta prescan. */ readonly encodingPrescanBytes: number; - /** Effective prescan retention cap after the implementation maximum is applied. */ + /** Effective optional meta-prescan cap after the implementation maximum is applied. */ readonly encodingPrescanLimitBytes: number; } @@ -429,9 +430,9 @@ export interface TraceSummary { readonly decodedUtf8Bytes: number; /** Total stream transport bytes, or null for non-stream input. */ readonly bytesRead: number | null; - /** Stream encoding-prescan high-water bytes, or null for non-stream input. */ + /** Optional stream meta-prescan high-water bytes, or null for non-stream input. */ readonly encodingPrescanBytes: number | null; - /** Effective stream encoding-prescan cap, or null for non-stream input. */ + /** Effective optional stream meta-prescan cap, or null for non-stream input. */ readonly encodingPrescanLimitBytes: number | null; /** Total observed event count, whether or not events were retained. */ readonly eventCount: number; @@ -577,6 +578,9 @@ export type HtmlNode = | ProcessingInstructionNode | DoctypeNode; +/** Complete node accepted as a standalone serialization input. */ +export type SerializableNode = Exclude<HtmlNode, TemplateContentNode>; + /** Callback invoked for a node and its zero-based traversal depth. */ export type NodeVisitor = (node: HtmlNode, depth: number) => void; /** Callback invoked for an element and its zero-based traversal depth. */ @@ -598,7 +602,7 @@ export interface DocumentTree { readonly trace?: TraceResult; } -/** Successful full-document parser resource observations. */ +/** Successful document or fragment parser resource observations. */ export interface ParseResourceUsage { /** UTF-8 bytes for text input; supplied transport bytes for byte/stream input. */ readonly inputBytes: number; @@ -618,7 +622,7 @@ export interface ParseResourceUsage { readonly attributes: number; /** UTF-8 bytes in attempted attribute names and decoded values. */ readonly attributeUtf8Bytes: number; - /** Stream encoding-prescan retained-byte high-water mark; zero otherwise. */ + /** Optional stream meta-prescan retained-byte high-water mark; zero otherwise. */ readonly encodingPrescanBytes: number; /** Observable trace events emitted; zero when tracing and observation are disabled. */ readonly traceEvents: number; @@ -634,8 +638,8 @@ export interface ParseEncodingMetadata { readonly source: "already-decoded" | "bom" | "transport" | "meta" | "default"; } -/** Deterministic metadata for one full-document parse. */ -export interface ParsedDocumentMetadata { +/** Deterministic metadata for one document or fragment parse. */ +export interface ParseMetadata { /** Public input variant used for this parse. */ readonly inputKind: "text" | "bytes" | "stream"; /** Bytes supplied to a byte/stream API, or null for already-decoded text. */ @@ -653,7 +657,7 @@ export interface ParsedDocument { /** Exact decoded input when `sourceRetention: "text"`; otherwise null. */ readonly sourceText: string | null; /** Input, encoding, and resource evidence from this parse. */ - readonly metadata: ParsedDocumentMetadata; + readonly metadata: ParseMetadata; } /** Immutable root returned by fragment parsing. */ @@ -678,6 +682,54 @@ export interface FragmentTree { readonly trace?: TraceResult; } +/** Canonical result returned by fragment parsing. */ +export interface ParsedFragment { + /** Parsed fragment tree. */ + readonly tree: FragmentTree; + /** Input and successful resource evidence from this parse. */ + readonly metadata: ParseMetadata; +} + +/** Successful resource observations from eager byte-stream tokenization. */ +export interface TokenizationResourceUsage { + /** Transport bytes read through EOF. */ + readonly inputBytes: number; + /** UTF-8 bytes produced by decoding. */ + readonly decodedUtf8Bytes: number; + /** UTF-16 code units produced by decoding. */ + readonly decodedCodeUnits: number; + /** Deterministic tokenizer checkpoints, or null when no step limit enabled counting. */ + readonly steps: number | null; + /** Tokenizer diagnostics observed while producing tokens. */ + readonly parseErrors: number; + /** Attempted start-tag attributes, including duplicates later discarded. */ + readonly attributes: number; + /** UTF-8 bytes in attempted attribute names and decoded values. */ + readonly attributeUtf8Bytes: number; + /** Optional meta-encoding prefix retained at its high-water mark. */ + readonly encodingPrescanBytes: number; +} + +/** Metadata for one eager byte-stream tokenization operation. */ +export interface TokenizeByteStreamEagerMetadata { + /** Public input variant used by this operation. */ + readonly inputKind: "stream"; + /** Transport bytes read through EOF. */ + readonly transportByteLength: number; + /** Encoding decision from the owning decode pipeline. */ + readonly encoding: ParseEncodingMetadata; + /** Successful deterministic resource observations. */ + readonly resourceUsage: TokenizationResourceUsage; +} + +/** Immutable eager byte-stream tokenization result. */ +export interface TokenizeByteStreamEagerResult { + /** Logical tokens in emission order, including EOF. */ + readonly tokens: readonly Token[]; + /** Decode and tokenizer resource evidence. */ + readonly metadata: TokenizeByteStreamEagerMetadata; +} + /** One structural outline entry derived from a heading or sectioning element. */ export interface OutlineEntry { /** Identity of the source element. */ diff --git a/test/behavior/fragment-context.test.js b/test/behavior/fragment-context.test.js index 73f6d7d..0b71bae 100644 --- a/test/behavior/fragment-context.test.js +++ b/test/behavior/fragment-context.test.js @@ -20,7 +20,7 @@ test("fragment context is normalized, retained, and deeply immutable", () => { localName: "TeXtArEa", attributes: [{ namespaceUri: XLINK_NAMESPACE_URI, localName: "href", value: "#x" }] }; - const fragment = parseFragment("a<b>&amp;", input); + const { tree: fragment } = parseFragment("a<b>&amp;", input); assert.deepEqual(fragment.context, { namespaceUri: HTML_NAMESPACE_URI, @@ -39,7 +39,7 @@ test("fragment context is normalized, retained, and deeply immutable", () => { }); test("foreign fragment contexts and semantic attributes drive integration rules", () => { - const svg = parseFragment("<lineargradient/><foreignObject><p>x</p></foreignObject>", { + const { tree: svg } = parseFragment("<lineargradient/><foreignObject><p>x</p></foreignObject>", { namespaceUri: SVG_NAMESPACE_URI, localName: "svg" }); @@ -55,7 +55,7 @@ test("foreign fragment contexts and semantic attributes drive integration rules" assert.equal(foreignObject.children[0]?.kind, "element"); assert.equal(foreignObject.children[0]?.namespaceUri, HTML_NAMESPACE_URI); - const annotation = parseFragment("<p>x</p>", { + const { tree: annotation } = parseFragment("<p>x</p>", { namespaceUri: MATHML_NAMESPACE_URI, localName: "annotation-xml", attributes: [{ namespaceUri: null, localName: "encoding", value: "text/html" }] @@ -66,14 +66,14 @@ test("foreign fragment contexts and semantic attributes drive integration rules" test("fragment environment controls tree construction and is retained on the result", () => { const context = htmlContext("div"); - const standards = parseFragment("<p>x<table><tr><td>y", context); - const quirks = parseFragment("<p>x<table><tr><td>y", context, { documentMode: "quirks" }); + const { tree: standards } = parseFragment("<p>x<table><tr><td>y", context); + const { tree: quirks } = parseFragment("<p>x<table><tr><td>y", context, { documentMode: "quirks" }); assert.notDeepEqual(quirks.children, standards.children); assert.equal(standards.documentMode, "no-quirks"); assert.equal(quirks.documentMode, "quirks"); - const withoutForm = parseFragment("<form><input>", context); - const withForm = parseFragment("<form><input>", context, { hasFormAncestor: true }); + const { tree: withoutForm } = parseFragment("<form><input>", context); + const { tree: withForm } = parseFragment("<form><input>", context, { hasFormAncestor: true }); assert.equal(withoutForm.children[0]?.kind, "element"); assert.equal(withoutForm.children[0]?.localName, "form"); assert.equal(withForm.children[0]?.kind, "element"); @@ -81,14 +81,14 @@ test("fragment environment controls tree construction and is retained on the res assert.equal(withForm.hasFormInContextChain, true); assert.equal(withoutForm.hasFormInContextChain, false); - const formContext = parseFragment("<form><input>", htmlContext("form")); + const { tree: formContext } = parseFragment("<form><input>", htmlContext("form")); assert.equal(formContext.children[0]?.kind, "element"); assert.equal(formContext.children[0]?.localName, "input"); assert.equal(formContext.hasFormInContextChain, true); }); test("serialization and chunking inherit a fragment's scripting mode", () => { - const fragment = parseFragment( + const { tree: fragment } = parseFragment( "<noscript>&lt;b&gt;</noscript>", htmlContext("div"), { scriptingMode: "disabled" } @@ -100,7 +100,7 @@ test("serialization and chunking inherit a fragment's scripting mode", () => { }); test("frameset fragments serialize frame as an HTML void element", () => { - const fragment = parseFragment("</frameset><frame>", htmlContext("frameset")); + const { tree: fragment } = parseFragment("</frameset><frame>", htmlContext("frameset")); assert.equal(serialize(fragment), "<frame>"); }); @@ -152,7 +152,7 @@ test("fragment context and environment validation reject ambiguous configuration error.reason === "INVALID_VALUE" ); } - assert.equal(parseFragment("x", htmlContext("Élément")).context.localName, "Élément"); + assert.equal(parseFragment("x", htmlContext("Élément")).tree.context.localName, "Élément"); assert.throws( () => parseFragment("x", { ...htmlContext("div"), diff --git a/test/behavior/namespaces-stack.test.js b/test/behavior/namespaces-stack.test.js index ea73a04..d281cb3 100644 --- a/test/behavior/namespaces-stack.test.js +++ b/test/behavior/namespaces-stack.test.js @@ -210,7 +210,13 @@ test("deep parsed and caller-built trees remain stack-safe across the public sur children: [child] }; } - const manual = { id: depth + 2, kind: "document", children: [child], errors: [] }; + const manual = { + id: depth + 2, + kind: "document", + scriptingMode: "inert", + children: [child], + errors: [] + }; assert.equal(rawText(manual), "leaf"); assert.equal(visibleTextValue(manual), "leaf"); assert.equal(serialize(manual).includes("leaf"), true); diff --git a/test/behavior/parse-result-metadata.test.js b/test/behavior/parse-result-metadata.test.js index 5056583..9fd57f9 100644 --- a/test/behavior/parse-result-metadata.test.js +++ b/test/behavior/parse-result-metadata.test.js @@ -150,7 +150,7 @@ test("byte and stream results own decoding evidence and exact retained source", assert.equal(fromBytes.metadata.inputKind, "bytes"); assert.equal(fromBytes.metadata.resourceUsage.encodingPrescanBytes, 0); assert.equal(fromStream.metadata.inputKind, "stream"); - assert.equal(fromStream.metadata.resourceUsage.encodingPrescanBytes, bytes.byteLength); + assert.equal(fromStream.metadata.resourceUsage.encodingPrescanBytes, 1); }); test("byte and stream metadata preserve BOM, meta, and default encoding evidence", async () => { @@ -241,6 +241,37 @@ test("step observations are available exactly when deterministic counting is ena assert.equal(stepBudget?.actual, tracked.metadata.resourceUsage.steps); }); +test("fragment results expose the same immutable resource evidence as document results", () => { + const source = "<b a=1>x&amp;y</b>"; + const result = parseFragment( + source, + { namespaceUri: HTML_NAMESPACE_URI, localName: "section" }, + { budgets: { maxSteps: 1_000 } } + ); + assert.deepEqual(Object.keys(result), ["tree", "metadata"]); + assert.equal(result.tree.kind, "fragment"); + assert.deepEqual(result.metadata.encoding, { name: null, source: "already-decoded" }); + assert.equal(result.metadata.inputKind, "text"); + assert.equal(result.metadata.transportByteLength, null); + assert.equal(result.metadata.resourceUsage.inputBytes, new TextEncoder().encode(source).byteLength); + assert.equal(result.metadata.resourceUsage.decodedUtf8Bytes, new TextEncoder().encode(source).byteLength); + assert.equal(result.metadata.resourceUsage.decodedCodeUnits, source.length); + assert.ok(result.metadata.resourceUsage.steps > 0); + assert.equal(result.metadata.resourceUsage.nodes, 5); + assert.equal(result.metadata.resourceUsage.maxDepth, 3); + assert.equal(result.metadata.resourceUsage.attributes, 1); + assert.equal(Object.isFrozen(result), true); + assert.equal(Object.isFrozen(result.metadata), true); + assert.equal(Object.isFrozen(result.metadata.resourceUsage), true); + + const exactSteps = result.metadata.resourceUsage.steps; + assert.doesNotThrow(() => parseFragment( + source, + { namespaceUri: HTML_NAMESPACE_URI, localName: "section" }, + { budgets: { maxSteps: exactSteps } } + )); +}); + test("patch planning and application require exact registered parse identity", () => { const source = "<p>alpha</p>"; const document = parse(source, { captureSpans: true, sourceRetention: "text" }); diff --git a/test/behavior/public-integration.test.js b/test/behavior/public-integration.test.js index e6e4ae2..c379098 100644 --- a/test/behavior/public-integration.test.js +++ b/test/behavior/public-integration.test.js @@ -195,9 +195,9 @@ test("text, byte, stream, and fragment entrypoints preserve their input contract assert.deepEqual(fromStream.tree, text.tree); assert.equal(fromBytes.metadata.inputKind, "bytes"); assert.equal(fromStream.metadata.inputKind, "stream"); - assert.equal(fromStream.metadata.resourceUsage.encodingPrescanBytes, bytes.byteLength); + assert.equal(fromStream.metadata.resourceUsage.encodingPrescanBytes, 1); - const fragment = parseFragment( + const { tree: fragment } = parseFragment( "<td>x", { namespaceUri: HTML_NAMESPACE_URI, localName: "table" }, { captureSpans: true } @@ -242,7 +242,7 @@ test("visible-text nested markup remains on the production parser route", () => test("production tokenization retains the current-standard token vocabulary", async () => { const source = "<?build release?><p a=1>a&amp;b</p>"; - const tokens = await tokenizeByteStreamEager( + const { tokens } = await tokenizeByteStreamEager( byteStream([new TextEncoder().encode(source)]) ); assert.deepEqual(tokens, [ diff --git a/test/behavior/roundtrip.test.js b/test/behavior/roundtrip.test.js index 26792e4..ba71242 100644 --- a/test/behavior/roundtrip.test.js +++ b/test/behavior/roundtrip.test.js @@ -23,7 +23,7 @@ test("tokenizer handles long numeric character references without overflow failu const leadingZeroHex = await tokenizeByteStreamEager( byteStream(`&#x${"0".repeat(256)}41;`) ); - assert.deepEqual(leadingZeroHex, [ + assert.deepEqual(leadingZeroHex.tokens, [ { kind: "chars", value: "A" }, { kind: "eof" } ]); @@ -31,7 +31,7 @@ test("tokenizer handles long numeric character references without overflow failu const outOfRangeDecimal = await tokenizeByteStreamEager( byteStream(`&#${"9".repeat(400)};`) ); - assert.deepEqual(outOfRangeDecimal, [ + assert.deepEqual(outOfRangeDecimal.tokens, [ { kind: "chars", value: "\uFFFD" }, { kind: "eof" } ]); diff --git a/test/behavior/serialization.test.js b/test/behavior/serialization.test.js index 604a43f..e1f0533 100644 --- a/test/behavior/serialization.test.js +++ b/test/behavior/serialization.test.js @@ -2,7 +2,9 @@ import assert from "node:assert/strict"; import test from "node:test"; import { + HTML_NAMESPACE_URI, HtmlConfigurationError, + SVG_NAMESPACE_URI, serialize } from "../../dist/mod.js"; import { @@ -23,3 +25,64 @@ test("public serializer validates the scripting environment", () => { error.option === "options.scriptingMode" && error.reason === "INVALID_VALUE" ); }); + +function htmlElement(overrides = {}) { + return { + id: 1, + kind: "element", + namespaceUri: HTML_NAMESPACE_URI, + localName: "div", + attributes: [], + children: [], + ...overrides + }; +} + +function assertInvalidTree(value, option) { + assert.throws( + () => serialize(value), + (error) => error instanceof HtmlConfigurationError && + error.reason === "INVALID_VALUE" && error.option === option + ); +} + +test("public serializer rejects unsafe names, namespaces, prefixes, and duplicate attributes", () => { + assertInvalidTree(htmlElement({ localName: "div><script" }), "tree.localName"); + assertInvalidTree(htmlElement({ + attributes: [{ namespaceUri: null, localName: "bad name", value: "x" }] + }), "tree.attributes[0].localName"); + assertInvalidTree(htmlElement({ namespaceUri: "" }), "tree.namespaceUri"); + assertInvalidTree(htmlElement({ prefix: "html" }), "tree.prefix"); + assertInvalidTree({ + ...htmlElement({ namespaceUri: SVG_NAMESPACE_URI }), + attributes: [ + { namespaceUri: null, localName: "CLASS", value: "a" }, + { namespaceUri: null, localName: "class", value: "b" } + ] + }, "tree.attributes[1]"); + assertInvalidTree(htmlElement({ + attributes: [ + { namespaceUri: "urn:one", prefix: "xml", localName: "lang", value: "en" } + ] + }), "tree.attributes[0].prefix"); +}); + +test("public serializer requires unique acyclic ownership and exact template structure", () => { + const cyclic = htmlElement({ children: [] }); + cyclic.children.push(cyclic); + assertInvalidTree(cyclic, "tree.children[0]"); + + const shared = { id: 2, kind: "text", value: "x" }; + assertInvalidTree(htmlElement({ children: [shared, shared] }), "tree.children[1]"); + assertInvalidTree(htmlElement({ children: [{ id: 1, kind: "text", value: "x" }] }), "tree.children[0].id"); + assertInvalidTree({ id: 1, kind: "templateContent", children: [] }, "tree"); + assertInvalidTree(htmlElement({ localName: "template" }), "tree.templateContent"); + assertInvalidTree(htmlElement({ + templateContent: { id: 2, kind: "templateContent", children: [] } + }), "tree.templateContent"); + assertInvalidTree(htmlElement({ + localName: "template", + children: [{ id: 2, kind: "text", value: "wrong owner" }], + templateContent: { id: 3, kind: "templateContent", children: [] } + }), "tree.children"); +}); diff --git a/test/behavior/streaming.test.js b/test/behavior/streaming.test.js index 723834c..99a44ca 100644 --- a/test/behavior/streaming.test.js +++ b/test/behavior/streaming.test.js @@ -85,6 +85,75 @@ test("stream encoding prescan uses its documented implementation maximum", async } }); +test("mandatory BOM detection is independent from the optional meta-prescan cap", async () => { + const fixtures = [ + { + bytes: new Uint8Array([0xef, 0xbb, 0xbf, 0x3c, 0x70, 0x3e, 0x78]), + encoding: "utf-8" + }, + { + bytes: new Uint8Array([0xfe, 0xff, 0x00, 0x3c, 0x00, 0x70, 0x00, 0x3e, 0x00, 0x78]), + encoding: "utf-16be" + }, + { + bytes: new Uint8Array([0xff, 0xfe, 0x3c, 0x00, 0x70, 0x00, 0x3e, 0x00, 0x78, 0x00]), + encoding: "utf-16le" + } + ]; + for (const fixture of fixtures) { + for (const maxEncodingPrescanBytes of [0, 1, 2]) { + const result = await parseStream(createByteStream([ + fixture.bytes.subarray(0, 1), + fixture.bytes.subarray(1, 2), + fixture.bytes.subarray(2) + ]), { + sourceRetention: "text", + budgets: { maxEncodingPrescanBytes } + }); + assert.deepEqual(result.metadata.encoding, { + name: fixture.encoding, + source: "bom" + }); + assert.equal(result.sourceText, "<p>x"); + assert.ok( + result.metadata.resourceUsage.encodingPrescanBytes <= maxEncodingPrescanBytes + ); + } + } +}); + +test("stream decoding commits as soon as BOM precedence and encoding evidence are final", async () => { + const fixtures = [ + { + id: "transport", + bytes: new TextEncoder().encode("<p>transport tail</p>"), + options: { transportEncodingLabel: "utf-8" }, + expectedDecisionBytes: 1 + }, + { + id: "meta", + bytes: new TextEncoder().encode("<meta charset=utf-8><p>meta tail</p>"), + options: {}, + expectedDecisionBytes: new TextEncoder().encode("<meta charset=utf-8>").byteLength + } + ]; + for (const fixture of fixtures) { + const pullCounter = { count: 0 }; + let decisionPulls = null; + const chunks = [...fixture.bytes].map((value) => new Uint8Array([value])); + const result = await parseStream(createPullCountStream(chunks, pullCounter), { + ...fixture.options, + budgets: { maxEncodingPrescanBytes: 100 }, + onTraceEvent(event) { + if (event.kind === "decode") decisionPulls = pullCounter.count; + } + }); + assert.equal(decisionPulls, fixture.expectedDecisionBytes, fixture.id); + assert.ok(decisionPulls < fixture.bytes.byteLength, fixture.id); + assert.equal(result.metadata.resourceUsage.encodingPrescanBytes, decisionPulls); + } +}); + test("parseStream budget outcome is independent of upstream chunk boundaries", async () => { const bytes = new TextEncoder().encode(`<p>${"x".repeat(20_000)}</p>`); const chunks = []; @@ -230,7 +299,7 @@ test("parseStream and tokenizeByteStreamEager release their readers after succes assert.equal(parseInput.locked, false); const tokenizeInput = createByteStream([new TextEncoder().encode("<p>tokenized</p>")]); - const tokens = await tokenizeByteStreamEager(tokenizeInput); + const { tokens } = await tokenizeByteStreamEager(tokenizeInput); assert.ok(tokens.length > 0); assert.equal(tokenizeInput.locked, false); }); @@ -289,11 +358,31 @@ test("tokenizeByteStreamEager returns a deterministic token sequence", async () const second = await collect(); assert.deepEqual(first, second); assert.deepEqual( - first.map((entry) => entry.kind), + first.tokens.map((entry) => entry.kind), ["startTag", "chars", "endTag", "eof"] ); }); +test("eager tokenization exposes exact deterministic work and enforces maxSteps", async () => { + const input = () => createByteStream([new TextEncoder().encode("<p a=1>x&amp;y</p>")]); + const untracked = await tokenizeByteStreamEager(input()); + assert.equal(untracked.metadata.resourceUsage.steps, null); + assert.equal(Object.isFrozen(untracked), true); + assert.equal(Object.isFrozen(untracked.tokens), true); + assert.equal(Object.isFrozen(untracked.metadata.resourceUsage), true); + + const measured = await tokenizeByteStreamEager(input(), { budgets: { maxSteps: 1_000 } }); + const steps = measured.metadata.resourceUsage.steps; + assert.ok(Number.isSafeInteger(steps)); + assert.ok(steps > 0); + await tokenizeByteStreamEager(input(), { budgets: { maxSteps: steps } }); + await assert.rejects( + tokenizeByteStreamEager(input(), { budgets: { maxSteps: steps - 1 } }), + (error) => error instanceof HtmlBudgetExceededError && + error.budget === "maxSteps" && error.limit === steps - 1 && error.actual === steps + ); +}); + test("eager tokenization is invariant across chunk patterns and single-byte encoding", async () => { const bytes = new Uint8Array([ ...asciiBytes("<meta charset=windows-1252><p>"), @@ -317,7 +406,7 @@ test("eager tokenization is invariant across chunk patterns and single-byte enco } assert.deepEqual(results[0], results[1]); assert.deepEqual(results[1], results[2]); - assert.equal(results[0].find((token) => token.kind === "chars")?.value, "é"); + assert.equal(results[0].tokens.find((token) => token.kind === "chars")?.value, "é"); }); test("outline and chunk stay deterministic", () => { @@ -332,7 +421,7 @@ test("outline and chunk stay deterministic", () => { }); test("chunk enforces maxBytes when configured", () => { - const fragment = parseFragment( + const { tree: fragment } = parseFragment( "<p>a</p><p>bb</p><p>ccc</p>", { namespaceUri: HTML_NAMESPACE_URI, localName: "section" } ); diff --git a/test/behavior/trace-schema.test.js b/test/behavior/trace-schema.test.js index 4845525..04808b9 100644 --- a/test/behavior/trace-schema.test.js +++ b/test/behavior/trace-schema.test.js @@ -124,7 +124,7 @@ test("trace counts logical parser tokens once, including EOF", () => { "a<b>&amp;</b>", htmlContext("textarea"), { trace: "events" } - )), 2); + ).tree), 2); }); test("trace includes parseError events for malformed input", () => { diff --git a/test/behavior/traversal-extraction.test.js b/test/behavior/traversal-extraction.test.js index f538266..cb0c8d8 100644 --- a/test/behavior/traversal-extraction.test.js +++ b/test/behavior/traversal-extraction.test.js @@ -2,11 +2,19 @@ import assert from "node:assert/strict"; import test from "node:test"; import { + HtmlConfigurationError, TEXT_CONTENT_POLICY, + chunk, extractText, findAllByAttr, + findAllByAttrNS, findAllByTagName, + findAllByTagNameNS, findById, + getAttributeValue, + getAttributeValueNS, + hasAttribute, + hasAttributeNS, outline, parse, walk, @@ -41,6 +49,82 @@ test("walk and walkElements are deterministic", () => { assert.ok(firstElements.length >= 3); }); +test("query helpers reject invalid JavaScript arguments with one public error category", () => { + const { tree } = parse("<p class=x></p>"); + const element = [...findAllByTagName(tree, "p")][0]; + assert.ok(element); + const invalidCalls = [ + () => getAttributeValue(element, 1), + () => getAttributeValue(null, "class"), + () => getAttributeValue({ ...element, attributes: [null] }, "class"), + () => hasAttribute(element, 1), + () => getAttributeValueNS(element, 1, "class"), + () => getAttributeValueNS(element, null, 1), + () => hasAttributeNS(element, null, 1), + () => findAllByTagName(tree, 1), + () => findAllByTagNameNS(tree, 1, "p"), + () => findAllByTagNameNS(tree, "urn:test", 1), + () => findAllByAttr(tree, 1), + () => findAllByAttr(tree, "class", 1), + () => findAllByAttrNS(tree, 1, "class"), + () => findAllByAttrNS(tree, null, 1), + () => findAllByAttrNS(tree, null, "class", 1), + () => findById(tree, 0), + () => findById(null, 1), + () => walk(null, () => undefined), + () => walk(tree, null), + () => walkElements(tree, null), + () => [...findAllByAttr({ + ...tree, + children: [{ ...element, attributes: null }] + }, "class")] + ]; + for (const invoke of invalidCalls) { + assert.throws(invoke, HtmlConfigurationError); + } +}); + +test("all public tree traversals reject cyclic caller graphs deterministically", () => { + const element = { + id: 2, + kind: "element", + namespaceUri: "http://www.w3.org/1999/xhtml", + localName: "div", + attributes: [], + children: [] + }; + element.children.push(element); + const tree = { + id: 1, + kind: "fragment", + context: { + namespaceUri: "http://www.w3.org/1999/xhtml", + localName: "div", + attributes: [] + }, + scriptingMode: "inert", + documentMode: "no-quirks", + hasFormInContextChain: false, + children: [element], + errors: [] + }; + + const operations = [ + () => walk(tree, () => undefined), + () => [...findAllByTagName(tree, "div")], + () => outline(tree), + () => extractText(tree, { + policy: TEXT_CONTENT_POLICY, + maxOutputBytes: 100, + maxTokens: 100 + }), + () => chunk(tree) + ]; + for (const operation of operations) { + assert.throws(operation, HtmlConfigurationError); + } +}); + test("outline entry text uses a scalar-safe 200-byte UTF-8 prefix", () => { const { tree } = parse(`<h1>${"😀".repeat(100)}Z</h1>`); const entry = outline(tree).entries[0]; diff --git a/test/contracts/consumers/npm/runtime.mjs b/test/contracts/consumers/npm/runtime.mjs index 92d9686..505fc3b 100644 --- a/test/contracts/consumers/npm/runtime.mjs +++ b/test/contracts/consumers/npm/runtime.mjs @@ -17,7 +17,7 @@ if (!serialize(bytes.tree).includes("<p>bytes</p>")) { throw new Error("installed package byte parsing failed"); } -const fragment = parseFragment("<b>fragment</b>", { +const { tree: fragment } = parseFragment("<b>fragment</b>", { namespaceUri: HTML_NAMESPACE_URI, localName: "section" }); diff --git a/test/contracts/html-product-contract.type-test.ts b/test/contracts/html-product-contract.type-test.ts index e3a24df..d1bcee9 100644 --- a/test/contracts/html-product-contract.type-test.ts +++ b/test/contracts/html-product-contract.type-test.ts @@ -43,8 +43,8 @@ const contextualFragment = parseFragment("<p>x", mathContext, { documentMode: "limited-quirks", hasFormAncestor: true }); -const normalizedContext: HtmlFragmentContext = contextualFragment.context; -const documentMode: HtmlDocumentMode = contextualFragment.documentMode; +const normalizedContext: HtmlFragmentContext = contextualFragment.tree.context; +const documentMode: HtmlDocumentMode = contextualFragment.tree.documentMode; const attributeNamespace: HtmlAttributeNamespaceUri = normalizedContext.attributes[0]?.namespaceUri ?? null; void documentMode; void attributeNamespace; @@ -92,13 +92,16 @@ if (element?.kind === "element") { void attribute.name; } } -const tokenPromise: Promise<readonly ( - ProcessingInstructionToken | Exclude<Awaited<ReturnType<typeof tokenizeByteStreamEager>>[number], ProcessingInstructionToken> -)[]> = tokenizeByteStreamEager(new ReadableStream()); +const tokenPromise = tokenizeByteStreamEager(new ReadableStream()); +const tokenKinds: readonly ( + ProcessingInstructionToken | + Exclude<Awaited<typeof tokenPromise>["tokens"][number], ProcessingInstructionToken> +)[] = (await tokenPromise).tokens; void fragment; void templateContent; void instruction; void tokenPromise; +void tokenKinds; const scriptingMode: HtmlScriptingMode = "inert"; const serializeOptions: SerializeOptions = { scriptingMode, maxTimeMs: 10 }; void serializeOptions; diff --git a/test/fixtures/qualification/document-browser-baseline.json b/test/fixtures/qualification/document-browser-baseline.json index a96aa07..475cd5a 100644 --- a/test/fixtures/qualification/document-browser-baseline.json +++ b/test/fixtures/qualification/document-browser-baseline.json @@ -7,7 +7,100 @@ "wptScriptingInvariant": 1698, "supplemental": 26 }, - "reason": "Exact reviewed differences between the pinned current-standard WPT expectations and the browser versions bundled by Playwright 1.61.1.", + "knownDifferenceGroups": [ + { + "classification": "processing-instruction-implementation-lag", + "explanation": "The pinned WPT snapshot requires current HTML processing-instruction nodes; these browser DOMParser versions still expose different node shapes for the enumerated inputs.", + "engines": ["chromium", "firefox", "webkit"], + "caseIds": [ + "wpt:test/fixtures/upstream/wpt-tree-construction/resources/html5test-com.dat#12", + "wpt:test/fixtures/upstream/wpt-tree-construction/resources/tests1.dat#40", + "wpt:test/fixtures/upstream/wpt-tree-construction/resources/tests1.dat#44", + "wpt:test/fixtures/upstream/wpt-tree-construction/resources/tests1.dat#47" + ], + "caseIdPrefix": "wpt:test/fixtures/upstream/wpt-tree-construction/resources/processing-instructions.dat#", + "caseNumbers": [ + 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, + 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, + 39, 40, 41, 42, 43, 44, 45, 46, 47, 48, 49, 50, 51, 52, 53, 54, 55, 56, + 57, 58, 59, 60, 61, 62, 63, 64, 65, 101, 102, 103, 104, 105, 106, 108, 109, + 110, 111, 114, 115, 116, 117, 118, 119, 120, 124 + ] + }, + { + "classification": "foreign-cdata-implementation-lag", + "explanation": "Chromium 149 and WebKit 26.5 differ from the pinned WPT tree for CDATA sections at the enumerated SVG and MathML integration points.", + "engines": ["chromium", "webkit"], + "caseIds": [ + "wpt:test/fixtures/upstream/wpt-tree-construction/resources/html5test-com.dat#14", + "wpt:test/fixtures/upstream/wpt-tree-construction/resources/html5test-com.dat#15", + "wpt:test/fixtures/upstream/wpt-tree-construction/resources/html5test-com.dat#18" + ] + }, + { + "classification": "character-reference-implementation-difference", + "explanation": "Chromium 149 produces a different attribute character-reference result for this focused ambiguous-ampersand case while the parser result matches the pinned tokenizer and WPT expectations.", + "engines": ["chromium"], + "caseIds": ["supplemental:entities"] + }, + { + "classification": "adoption-recovery-implementation-difference", + "explanation": "Firefox 151 differs from the pinned WPT active-formatting recovery tree for this nested nobr/table case.", + "engines": ["firefox"], + "caseIds": [ + "wpt:test/fixtures/upstream/wpt-tree-construction/resources/adoption02.dat#3" + ] + }, + { + "classification": "obsolete-select-element-implementation-difference", + "explanation": "Firefox 151 differs from the pinned WPT treatment of the enumerated obsolete menuitem and keygen elements in select mode.", + "engines": ["firefox"], + "caseIds": [ + "wpt:test/fixtures/upstream/wpt-tree-construction/resources/menuitem-element.dat#14", + "wpt:test/fixtures/upstream/wpt-tree-construction/resources/tests7.dat#34" + ] + }, + { + "classification": "customizable-select-implementation-lag", + "explanation": "Firefox 151 does not yet produce the pinned current customizable-select and relaxed-select trees for these exact HTML, table, foreign-content, and plaintext interactions.", + "engines": ["firefox"], + "caseIds": [ + "wpt:test/fixtures/upstream/wpt-tree-construction/resources/tables01.dat#18", + "wpt:test/fixtures/upstream/wpt-tree-construction/resources/tests1.dat#30", + "wpt:test/fixtures/upstream/wpt-tree-construction/resources/tests1.dat#100", + "wpt:test/fixtures/upstream/wpt-tree-construction/resources/tests10.dat#4", + "wpt:test/fixtures/upstream/wpt-tree-construction/resources/tests10.dat#5", + "wpt:test/fixtures/upstream/wpt-tree-construction/resources/tests10.dat#17", + "wpt:test/fixtures/upstream/wpt-tree-construction/resources/tests10.dat#18", + "wpt:test/fixtures/upstream/wpt-tree-construction/resources/tests18.dat#14", + "wpt:test/fixtures/upstream/wpt-tree-construction/resources/tests18.dat#15", + "wpt:test/fixtures/upstream/wpt-tree-construction/resources/tests9.dat#5", + "wpt:test/fixtures/upstream/wpt-tree-construction/resources/tests9.dat#6", + "wpt:test/fixtures/upstream/wpt-tree-construction/resources/tests9.dat#18", + "wpt:test/fixtures/upstream/wpt-tree-construction/resources/tests9.dat#19", + "wpt:test/fixtures/upstream/wpt-tree-construction/resources/webkit02.dat#36", + "wpt:test/fixtures/upstream/wpt-tree-construction/resources/webkit02.dat#38", + "wpt:test/fixtures/upstream/wpt-tree-construction/resources/webkit02.dat#39", + "wpt:test/fixtures/upstream/wpt-tree-construction/resources/webkit02.dat#40", + "wpt:test/fixtures/upstream/wpt-tree-construction/resources/webkit02.dat#41", + "wpt:test/fixtures/upstream/wpt-tree-construction/resources/webkit02.dat#42", + "wpt:test/fixtures/upstream/wpt-tree-construction/resources/webkit02.dat#43", + "supplemental:relaxed-select" + ] + }, + { + "classification": "selectedcontent-implementation-lag", + "explanation": "Firefox 151 differs from the pinned current selectedcontent option-cloning and selectedness-update behavior for these exact cases.", + "engines": ["firefox"], + "caseIds": [ + "wpt:test/fixtures/upstream/wpt-tree-construction/resources/webkit02.dat#45", + "wpt:test/fixtures/upstream/wpt-tree-construction/resources/webkit02.dat#46", + "wpt:test/fixtures/upstream/wpt-tree-construction/resources/webkit02.dat#47", + "wpt:test/fixtures/upstream/wpt-tree-construction/resources/webkit02.dat#48", + "supplemental:selectedcontent-option-clone" + ] + } + ], "engines": { "chromium": { "version": "149.0.7827.55", diff --git a/test/support/public-serialization-qualification-cases.mjs b/test/support/public-serialization-qualification-cases.mjs index 17b9deb..28270be 100644 --- a/test/support/public-serialization-qualification-cases.mjs +++ b/test/support/public-serialization-qualification-cases.mjs @@ -100,11 +100,12 @@ function publicNode(descriptor, ids) { })), children: descriptor.children.map((child) => publicNode(child, ids)) }; - if (descriptor.templateChildren !== undefined) { + if (descriptor.templateChildren !== undefined || + (descriptor.namespaceUri === HTML_NAMESPACE_URI && descriptor.localName === "template")) { node.templateContent = { id: ids.next++, kind: "templateContent", - children: descriptor.templateChildren.map((child) => publicNode(child, ids)), + children: (descriptor.templateChildren ?? []).map((child) => publicNode(child, ids)), spanProvenance: "inferred" }; } diff --git a/test/tooling/document-browser-baseline.test.js b/test/tooling/document-browser-baseline.test.js new file mode 100644 index 0000000..05bec8a --- /dev/null +++ b/test/tooling/document-browser-baseline.test.js @@ -0,0 +1,50 @@ +import assert from "node:assert/strict"; +import test from "node:test"; + +import { expandKnownDifferenceGroups } from + "../../scripts/oracles/document-browser-baseline.mjs"; + +test("document browser baselines expand to reviewed per-case evidence", () => { + const inventory = expandKnownDifferenceGroups([ + { + classification: "implementation-lag", + explanation: "Pinned behavior differs in these exact cases.", + engines: ["chromium", "firefox"], + caseIds: ["case:one"], + caseIdPrefix: "case:", + caseNumbers: [2, 3] + } + ], ["chromium", "firefox", "webkit"]); + + assert.deepEqual([...inventory.get("chromium").keys()], ["case:one", "case:2", "case:3"]); + assert.deepEqual(inventory.get("firefox").get("case:2"), { + classification: "implementation-lag", + explanation: "Pinned behavior differs in these exact cases." + }); + assert.equal(inventory.get("webkit").size, 0); +}); + +test("document browser baselines reject unclassified and ambiguous evidence", () => { + assert.throws( + () => expandKnownDifferenceGroups(undefined, ["chromium"]), + /reviewed known-difference groups/ + ); + assert.throws( + () => expandKnownDifferenceGroups([{ + classification: "lag", + explanation: "reason", + engines: ["unknown"], + caseIds: ["case"] + }], ["chromium"]), + /unsupported engine unknown/ + ); + assert.throws( + () => expandKnownDifferenceGroups([{ + classification: "lag", + explanation: "reason", + engines: ["chromium"], + caseIds: ["case", "case"] + }], ["chromium"]), + /classifies case twice/ + ); +}); diff --git a/test/tooling/registry-integrity.test.js b/test/tooling/registry-integrity.test.js index bbc41be..b44788f 100644 --- a/test/tooling/registry-integrity.test.js +++ b/test/tooling/registry-integrity.test.js @@ -219,4 +219,6 @@ test("published provenance is bound to the version tag rather than current HEAD" ); assert.match(auditWorkflow, /fetch-depth: 0/); assert.match(auditWorkflow, /fetch-tags: true/); + assert.match(auditWorkflow, /--registry=npm --require-present/); + assert.match(auditWorkflow, /--registry=jsr --require-present/); }); From 44e85fea49a56b4c10fffe6c544a384d940a8015 Mon Sep 17 00:00:00 2001 From: Ismail El Korchi <ismail.elkorchi@gmail.com> Date: Tue, 21 Jul 2026 23:42:41 +0100 Subject: [PATCH 2/3] ci: align surface checks with runtime ownership --- .github/workflows/ci.yml | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 7159452..b731fc3 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -44,10 +44,6 @@ jobs: - name: Build run: npm run build - - name: Public surface parity - if: matrix.node-version == 20 - run: npm run test:public-surface - - name: Test run: npm test @@ -87,3 +83,7 @@ jobs: - name: Runtime agreement run: npm run qualification:runtime + + - name: Public surface parity + if: matrix.name == 'linux' + run: npm run test:public-surface From f7df10de098457b99399fb4db32d4170e90c0eb2 Mon Sep 17 00:00:00 2001 From: Ismail El Korchi <ismail.elkorchi@gmail.com> Date: Tue, 21 Jul 2026 23:51:31 +0100 Subject: [PATCH 3/3] docs: align CI runtime ownership --- docs/maintainers/testing.md | 8 +++++--- 1 file changed, 5 insertions(+), 3 deletions(-) diff --git a/docs/maintainers/testing.md b/docs/maintainers/testing.md index 71a38ab..a984571 100644 --- a/docs/maintainers/testing.md +++ b/docs/maintainers/testing.md @@ -136,6 +136,8 @@ evidence, and cross-revision performance. Both profiles fail at the first failed command and retain diagnostic reports under `reports/`; they do not assign an artificial quality score. -The blocking CI runtime contract runs on Linux, macOS, and Windows. Linux also -executes the separate Node, Deno, Bun, and browser jobs so cross-runtime output -agreement and platform coverage remain distinct signals. +The blocking CI runtime contract verifies exact Node, Deno, and Bun output +agreement on Linux, macOS, and Windows. The separate Linux Node-version matrix +owns linting, types, documentation, builds, and tests; npm/JSR public-surface +parity belongs to the Linux cross-runtime job because it requires both Node and +Deno. Browser engines are exercised by the qualification workflow.