diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml
index 20c3863..b731fc3 100644
--- a/.github/workflows/ci.yml
+++ b/.github/workflows/ci.yml
@@ -29,12 +29,6 @@ jobs:
node-version: ${{ matrix.node-version }}
cache: npm
- - name: Setup Deno
- if: matrix.node-version == 20
- uses: denoland/setup-deno@22d081ff2d3a40755e97629de92e3bcbfa7cf2ed # v2.0.5
- with:
- deno-version: v2.x
-
- name: Install
run: npm ci
@@ -50,15 +44,22 @@ jobs:
- name: Build
run: npm run build
- - name: Public surface parity
- if: matrix.node-version == 20
- run: npm run test:public-surface
-
- name: Test
run: npm test
- deno:
- runs-on: ubuntu-latest
+ platform-runtime:
+ strategy:
+ fail-fast: false
+ matrix:
+ include:
+ - os: ubuntu-latest
+ name: linux
+ - os: macos-latest
+ name: macos
+ - os: windows-latest
+ name: windows
+ runs-on: ${{ matrix.os }}
+ name: runtime-${{ matrix.name }}
steps:
- name: Checkout
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
@@ -74,35 +75,15 @@ jobs:
with:
deno-version: v2.x
- - name: Install
- run: npm ci
-
- - name: Build
- run: npm run build
-
- - name: Control smoke (Deno)
- run: npm run smoke:deno
-
- bun:
- runs-on: ubuntu-latest
- steps:
- - name: Checkout
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
-
- - name: Setup Node
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
- with:
- node-version: 20
- cache: npm
-
- name: Setup Bun
uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2.2.0
- name: Install
run: npm ci
- - name: Build
- run: npm run build
+ - name: Runtime agreement
+ run: npm run qualification:runtime
- - name: Control smoke (Bun)
- run: npm run smoke:bun
+ - name: Public surface parity
+ if: matrix.name == 'linux'
+ run: npm run test:public-surface
diff --git a/.github/workflows/release-audit.yml b/.github/workflows/release-audit.yml
index 7e4b6e4..39694d1 100644
--- a/.github/workflows/release-audit.yml
+++ b/.github/workflows/release-audit.yml
@@ -57,10 +57,10 @@ jobs:
run: deno publish --dry-run
- name: Inspect npm publication state
- run: node scripts/release/check-registry-version.mjs --registry=npm
+ run: node scripts/release/check-registry-version.mjs --registry=npm --require-present
- name: Inspect JSR publication state
- run: node scripts/release/check-registry-version.mjs --registry=jsr
+ run: node scripts/release/check-registry-version.mjs --registry=jsr --require-present
- name: Report outdated dependencies
continue-on-error: true
diff --git a/.github/workflows/runtime-latest.yml b/.github/workflows/runtime-latest.yml
index fdde7a9..f9e8b4b 100644
--- a/.github/workflows/runtime-latest.yml
+++ b/.github/workflows/runtime-latest.yml
@@ -14,7 +14,12 @@ concurrency:
jobs:
latest-runtime-smoke:
- runs-on: ubuntu-latest
+ strategy:
+ fail-fast: false
+ matrix:
+ os: [ubuntu-latest, macos-latest, windows-latest]
+ runs-on: ${{ matrix.os }}
+ name: latest-${{ matrix.os }}
timeout-minutes: 60
continue-on-error: true
steps:
@@ -41,17 +46,21 @@ jobs:
run: npm ci
- name: Lint
+ if: matrix.os == 'ubuntu-latest'
run: npm run lint
- name: Typecheck
+ if: matrix.os == 'ubuntu-latest'
run: npm run typecheck
- name: Test
+ if: matrix.os == 'ubuntu-latest'
run: npm run test
- name: Runtime smoke matrix
run: npm run smoke:runtimes
- name: Report outdated dependencies
+ if: matrix.os == 'ubuntu-latest'
continue-on-error: true
run: npm run deps:outdated
diff --git a/CHANGELOG.md b/CHANGELOG.md
index d972495..820fe32 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -4,6 +4,23 @@ All notable changes are documented in this file.
## Unreleased
+- Validate complete caller-constructed serialization graphs before emitting
+ markup, including names, namespaces, prefixes, attributes, document
+ placement, template ownership, unique ids, unique ownership, and acyclicity;
+ apply the same ownership safety to chunking and public tree traversals.
+- Separate mandatory BOM detection from optional meta-encoding prescan,
+ initialize streaming decode as soon as encoding evidence is final, and feed
+ byte-array decoding directly into the parser without retaining decoded source
+ when source retention is disabled.
+- Return resource metadata from fragment parsing and eager stream tokenization,
+ and expose and enforce deterministic `maxSteps` for eager tokenization.
+- Validate all public query-helper arguments consistently with
+ `HtmlConfigurationError`.
+- Require scheduled registry audits to find the released version on both npm
+ and JSR, and record every accepted browser difference with its exact case id,
+ classification, explanation, and result hashes.
+- Run the blocking runtime contract on Linux, macOS, and Windows.
+
## [0.2.0] - 2026-07-21
- Validate the complete runtime edit shape in source patching, normalize HTML
diff --git a/README.md b/README.md
index 8fa75d7..5c4b57d 100644
--- a/README.md
+++ b/README.md
@@ -60,16 +60,21 @@ ParsedDocument
└── metadata input, encoding, and observed resource usage
```
-`parseFragment()` returns a `FragmentTree` directly and requires an explicit
-namespace-aware context:
+`parseFragment()` returns a `ParsedFragment` with the immutable tree and the
+same successful resource evidence available for document parsing. It requires
+an explicit namespace-aware context:
```ts
import { HTML_NAMESPACE_URI, parseFragment } from "@ismail-elkorchi/html-parser";
-const rows = parseFragment("
A B", {
namespaceUri: HTML_NAMESPACE_URI,
localName: "tbody"
}, {
budgets: { maxNodes: 32, maxDepth: 8 }
});
-console.log(fragment.kind); // "fragment"
+console.log(fragment.kind, metadata.resourceUsage.nodes); // "fragment", observed nodes
console.log(serialize(fragment));
```
@@ -92,7 +92,7 @@ parsing environment:
descriptor is an HTML `form`. A context that is itself an HTML `form` is
recognized directly and does not require a redundant option.
-These values are retained on the returned fragment. `serialize(fragment)` and
+These values are retained on the returned fragment tree. `serialize(fragment)` and
`chunk(fragment)` inherit its scripting mode; an explicit serialization option
still overrides it. Supply environment values from the real context document
when parity with browser `innerHTML` parsing matters.
@@ -100,7 +100,7 @@ when parity with browser `innerHTML` parsing matters.
## Parse diagnostics
Malformed HTML is normally recovered according to HTML parsing rules. The
-result keeps non-fatal diagnostics in `tree.errors` or `fragment.errors`; each
+result keeps non-fatal diagnostics in its tree's `errors` array; each
entry has a stable `parseErrorId`, location data when available, and a short
message. `getParseErrorSpecRef()` returns the dedicated HTML Standard anchor
for a named tokenizer or input-stream error. Unnamed tree-construction errors
diff --git a/docs/streams-and-encoding.md b/docs/streams-and-encoding.md
index 97f4ad7..98ccd06 100644
--- a/docs/streams-and-encoding.md
+++ b/docs/streams-and-encoding.md
@@ -48,8 +48,10 @@ During conversion to the immutable public result, already-converted internal
children and attributes are released so the mutable and immutable trees are
not both retained in full at the peak.
-`tokenizeByteStreamEager()` has the same eager boundary: it returns all tokens
-after EOF and does not expose tokens progressively.
+`tokenizeByteStreamEager()` has the same eager boundary: it returns a frozen
+`{ tokens, metadata }` result after EOF and does not expose tokens
+progressively. Set `budgets.maxSteps` to impose deterministic tokenizer work;
+the observed count is then returned in `metadata.resourceUsage.steps`.
## Encoding selection
@@ -57,11 +59,19 @@ Byte and stream entry points use the same HTML sniffing and decoding pipeline.
`metadata.encoding.source` reports `"bom"`, `"transport"`, `"meta"`, or
`"default"`; `metadata.encoding.name` reports the selected WHATWG encoding.
-`maxEncodingPrescanBytes` limits how much transport prefix is retained for
-encoding prescan. It is not a total-input budget and later bytes do not make it
-throw. The implementation never retains more than 16,384 bytes for prescan,
-even when a larger value is configured. Use `maxInputBytes` for total transport
-bytes and `maxDecodedUtf8Bytes` for decoded UTF-8 size.
+Mandatory BOM detection always examines the required prefix of up to three
+bytes and is independent of `maxEncodingPrescanBytes`. That option limits only
+the prefix retained for optional ` ` prescanning; zero disables
+meta prescanning without disabling UTF-8 or UTF-16 BOM recognition. It is not a
+total-input budget and later bytes do not make it throw. The implementation
+never retains more than 16,384 bytes for optional meta prescan, even when a
+larger value is configured.
+
+Decoding starts as soon as the higher-priority evidence is final: after a BOM,
+after BOM absence plus a recognized transport label, or after a complete early
+meta declaration. Otherwise the default is selected at the configured cap or
+EOF. Use `maxInputBytes` for total transport bytes and
+`maxDecodedUtf8Bytes` for decoded UTF-8 size.
## Cancellation and stream failures
diff --git a/scripts/mutation/config.json b/scripts/mutation/config.json
index 2828128..2849bec 100644
--- a/scripts/mutation/config.json
+++ b/scripts/mutation/config.json
@@ -31,7 +31,12 @@
},
{
"targetFile": "dist/public/serialization.js",
- "testCommand": ["node", "--test", "test/behavior/fragment-context.test.js"],
+ "testCommand": [
+ "node",
+ "--test",
+ "test/behavior/fragment-context.test.js",
+ "test/behavior/serialization.test.js"
+ ],
"mutants": [
{
"id": "parsed-tree-serializer-scripting-inheritance",
@@ -41,6 +46,54 @@
}
]
},
+ {
+ "targetFile": "dist/public/tree-validation.js",
+ "testCommand": ["node", "--test", "test/behavior/serialization.test.js"],
+ "mutants": [
+ {
+ "id": "serializer-element-name-preflight",
+ "description": "Allow an unsafe caller-constructed element name through serializer preflight.",
+ "search": "if (!isHtmlElementName(localName)) {",
+ "replace": "if (false && !isHtmlElementName(localName)) {"
+ },
+ {
+ "id": "serializer-serialized-attribute-uniqueness",
+ "description": "Allow two caller attributes to emit the same serialized HTML name.",
+ "search": "if (serializedNames.has(identity.serializedName)) {",
+ "replace": "if (false && serializedNames.has(identity.serializedName)) {"
+ },
+ {
+ "id": "serializer-template-content-ownership",
+ "description": "Allow an HTML template element without its owned template-content node.",
+ "search": "if (templateContent === undefined) {\n invalid(`${entry.path}.templateContent`, \"must be present on an HTML template element\");\n }\n pending.push({\n value: templateContent,\n path: `${entry.path}.templateContent`,\n placement: \"template\"\n });",
+ "replace": "if (templateContent !== undefined) {\n pending.push({\n value: templateContent,\n path: `${entry.path}.templateContent`,\n placement: \"template\"\n });\n }"
+ }
+ ]
+ },
+ {
+ "targetFile": "dist/public/html-input.js",
+ "testCommand": ["node", "--test", "test/behavior/streaming.test.js"],
+ "mutants": [
+ {
+ "id": "stream-early-meta-decision",
+ "description": "Delay a complete early meta-encoding decision until the prescan cap or EOF.",
+ "search": "pendingBytes <= 3 || lastBufferedByte === 0x3e ||",
+ "replace": "pendingBytes <= 3 || false ||"
+ }
+ ]
+ },
+ {
+ "targetFile": "dist/public/model.js",
+ "testCommand": ["node", "--test", "test/behavior/traversal-extraction.test.js"],
+ "mutants": [
+ {
+ "id": "query-element-attribute-shape",
+ "description": "Skip attribute-record validation at the shared query and traversal boundary.",
+ "search": "for (const attribute of attributes)\n validateAttributeShape(attribute, option);",
+ "replace": "for (const attribute of [])\n validateAttributeShape(attribute, option);"
+ }
+ ]
+ },
{
"targetFile": "dist/public/chunking.js",
"testCommand": [
@@ -185,7 +238,12 @@
},
{
"targetFile": "dist/internal/encoding/sniff.js",
- "testCommand": ["node", "--test", "test/behavior/encoding-sniff.test.js"],
+ "testCommand": [
+ "node",
+ "--test",
+ "test/behavior/encoding-sniff.test.js",
+ "test/behavior/streaming.test.js"
+ ],
"mutants": [
{
"id": "encoding-alias-windows1252",
@@ -210,6 +268,12 @@
"description": "Disable UTF-8 BOM detection.",
"search": "if (bytes.length >= 3 && bytes[0] === 0xef && bytes[1] === 0xbb && bytes[2] === 0xbf) {",
"replace": "if (false && bytes.length >= 3 && bytes[0] === 0xef && bytes[1] === 0xbb && bytes[2] === 0xbf) {"
+ },
+ {
+ "id": "stream-bom-prefix-wait",
+ "description": "Stop waiting for the complete mandatory BOM prefix before lower-priority encoding evidence.",
+ "search": "if (bomState === \"pending\")\n return Object.freeze({ status: \"pending\" });",
+ "replace": "if (false && bomState === \"pending\")\n return Object.freeze({ status: \"pending\" });"
}
]
},
@@ -768,8 +832,8 @@
{
"id": "product-trace-byte-accounting",
"description": "Discard retained trace byte accounting from public resource observations.",
- "search": "traceUtf8Bytes: trace.eventUtf8Bytes",
- "replace": "traceUtf8Bytes: 0"
+ "search": "encodingPrescanBytes: input.encodingPrescanBytes,\n traceEvents: trace.eventCount,\n traceUtf8Bytes: trace.eventUtf8Bytes",
+ "replace": "encodingPrescanBytes: input.encodingPrescanBytes,\n traceEvents: trace.eventCount,\n traceUtf8Bytes: 0"
},
{
"id": "product-step-budget-forwarding",
@@ -800,6 +864,24 @@
"description": "Discard decoded chunks instead of feeding tree construction before EOF.",
"search": "requireInternalValue(engine, \"PUBLIC_PARSER_STREAM_ENGINE_NOT_INITIALIZED\").session.write(chunk);",
"replace": "void chunk;"
+ },
+ {
+ "id": "eager-tokenization-step-limit",
+ "description": "Drop the public eager-tokenization step limit before engine construction.",
+ "search": ": { maxSteps: normalized.budgets.maxSteps }),",
+ "replace": ": {}),"
+ },
+ {
+ "id": "eager-tokenization-step-evidence",
+ "description": "Hide the observed eager-tokenization step count after successful work.",
+ "search": "steps: normalized.budgets?.maxSteps === undefined ? null : usage.steps,",
+ "replace": "steps: null,"
+ },
+ {
+ "id": "fragment-step-evidence",
+ "description": "Hide the observed fragment parser step count after successful work.",
+ "search": "function fragmentResourceUsage(inputBytes, decodedCodeUnits, budgets, result, trace) {\n return Object.freeze({\n inputBytes,\n decodedUtf8Bytes: inputBytes,\n decodedCodeUnits,\n steps: budgets?.maxSteps === undefined ? null : result.resources.steps,",
+ "replace": "function fragmentResourceUsage(inputBytes, decodedCodeUnits, budgets, result, trace) {\n return Object.freeze({\n inputBytes,\n decodedUtf8Bytes: inputBytes,\n decodedCodeUnits,\n steps: null,"
}
]
}
diff --git a/scripts/oracles/document-browser-baseline.mjs b/scripts/oracles/document-browser-baseline.mjs
new file mode 100644
index 0000000..8812ec9
--- /dev/null
+++ b/scripts/oracles/document-browser-baseline.mjs
@@ -0,0 +1,51 @@
+/** Expands reviewed known-difference groups into an exact per-engine case inventory. */
+export function expandKnownDifferenceGroups(groups, engineNames) {
+ if (!Array.isArray(groups)) {
+ throw new Error("Document browser baseline requires reviewed known-difference groups");
+ }
+ const byEngine = new Map(engineNames.map((name) => [name, new Map()]));
+ for (const group of groups) {
+ if (typeof group?.classification !== "string" || group.classification.length === 0 ||
+ typeof group.explanation !== "string" || group.explanation.length === 0 ||
+ !Array.isArray(group.engines) || group.engines.length === 0) {
+ throw new Error("Document browser baseline has an invalid known-difference classification");
+ }
+ if (group.caseIds !== undefined && !Array.isArray(group.caseIds)) {
+ throw new Error("Document browser baseline has an invalid case-id inventory");
+ }
+ const caseIds = [...(group.caseIds ?? [])];
+ if (group.caseIdPrefix !== undefined || group.caseNumbers !== undefined) {
+ if (typeof group.caseIdPrefix !== "string" || !Array.isArray(group.caseNumbers)) {
+ throw new Error("Document browser baseline has an invalid case-id expansion");
+ }
+ for (const caseNumber of group.caseNumbers) {
+ if (!Number.isSafeInteger(caseNumber) || caseNumber < 1) {
+ throw new Error("Document browser baseline case numbers must be positive integers");
+ }
+ caseIds.push(`${group.caseIdPrefix}${String(caseNumber)}`);
+ }
+ }
+ if (caseIds.length === 0 || caseIds.some((id) => typeof id !== "string" || id.length === 0)) {
+ throw new Error("Document browser baseline classification must identify every case");
+ }
+ if (new Set(group.engines).size !== group.engines.length) {
+ throw new Error("Document browser baseline repeats an engine in one classification");
+ }
+ for (const engine of group.engines) {
+ const inventory = byEngine.get(engine);
+ if (inventory === undefined) {
+ throw new Error(`Document browser baseline names unsupported engine ${String(engine)}`);
+ }
+ for (const id of caseIds) {
+ if (inventory.has(id)) {
+ throw new Error(`Document browser baseline classifies ${id} twice for ${engine}`);
+ }
+ inventory.set(id, Object.freeze({
+ classification: group.classification,
+ explanation: group.explanation
+ }));
+ }
+ }
+ }
+ return byEngine;
+}
diff --git a/scripts/oracles/run-document-browser-diff.mjs b/scripts/oracles/run-document-browser-diff.mjs
index 86f9e94..192c655 100644
--- a/scripts/oracles/run-document-browser-diff.mjs
+++ b/scripts/oracles/run-document-browser-diff.mjs
@@ -8,6 +8,7 @@ import { parse } from "../../dist/mod.js";
import { parseTreeDatFixtures } from "../../test/support/tree-dat.mjs";
import { verifyWptTreeCorpus } from "../../test/support/wpt-tree-corpus.mjs";
import { writeJson } from "../lib/report.mjs";
+import { expandKnownDifferenceGroups } from "./document-browser-baseline.mjs";
const BASELINE_PATH = "test/fixtures/qualification/document-browser-baseline.json";
@@ -150,6 +151,10 @@ if (baseline.schemaVersion !== 1 ||
typeof baseline.engines !== "object" || baseline.engines === null) {
throw new Error("Document browser baseline does not match the pinned qualification inputs");
}
+const knownDifferencesByEngine = expandKnownDifferenceGroups(
+ baseline.knownDifferenceGroups,
+ Object.keys(AVAILABLE_ENGINES)
+);
const results = [];
for (const [name, launcher] of engines) {
@@ -186,8 +191,11 @@ for (const [name, launcher] of engines) {
const differenceSha256 = sha256(differenceRecords);
const version = browser.version();
const expected = baseline.engines[name];
+ const expectedKnownDifferences = knownDifferencesByEngine.get(name);
const baselineMismatches = [];
- if (expected === undefined) baselineMismatches.push("engine-not-baselined");
+ if (expected === undefined || expectedKnownDifferences === undefined) {
+ baselineMismatches.push("engine-not-baselined");
+ }
else {
if (expected.version !== version) baselineMismatches.push("browser-version");
if (expected.outcomesSha256 !== outcomeSha256) baselineMismatches.push("all-outcomes");
@@ -195,8 +203,17 @@ for (const [name, launcher] of engines) {
if (expected.differencesSha256 !== differenceSha256) {
baselineMismatches.push("difference-inventory");
}
+ const expectedCaseIds = [...expectedKnownDifferences.keys()].sort();
+ const actualCaseIds = differenceRecords.map(({ id }) => id).sort();
+ if (JSON.stringify(expectedCaseIds) !== JSON.stringify(actualCaseIds)) {
+ baselineMismatches.push("difference-case-identifiers");
+ }
}
const baselineMatches = baselineMismatches.length === 0;
+ const classifiedDifferences = differenceRecords.map((difference) => ({
+ ...difference,
+ ...expectedKnownDifferences?.get(difference.id)
+ }));
results.push({
name,
version,
@@ -206,7 +223,7 @@ for (const [name, launcher] of engines) {
knownDifferences: {
count: failures.length,
sha256: differenceSha256,
- reason: baseline.reason
+ cases: classifiedDifferences
},
baselineMismatches,
failures: baselineMatches ? [] : failures
diff --git a/scripts/oracles/run-public-fragment-browser-diff.mjs b/scripts/oracles/run-public-fragment-browser-diff.mjs
index 78a4559..9fd2c01 100644
--- a/scripts/oracles/run-public-fragment-browser-diff.mjs
+++ b/scripts/oracles/run-public-fragment-browser-diff.mjs
@@ -140,7 +140,7 @@ function normalizePublicNode(node) {
}
function normalizePublic(testCase) {
- return parseFragment(testCase.html, testCase.context, testCase.options).children.map(normalizePublicNode);
+ return parseFragment(testCase.html, testCase.context, testCase.options).tree.children.map(normalizePublicNode);
}
async function normalizeBrowser(page, testCase) {
diff --git a/scripts/qualification/run-public-fragments.mjs b/scripts/qualification/run-public-fragments.mjs
index 5234ac9..c9c1c0d 100644
--- a/scripts/qualification/run-public-fragments.mjs
+++ b/scripts/qualification/run-public-fragments.mjs
@@ -26,7 +26,7 @@ for (const relativePath of corpus.fixtureFiles) {
baseCases += fixtureCases.length;
for (const fixtureCase of expandTreeDatCases(fixtureCases, { includeModeInId: true })) {
executions += 1;
- const fragment = parseFragment(
+ const { tree: fragment } = parseFragment(
fixtureCase.data,
{ ...fixtureCase.fragmentContext, attributes: [] },
{
diff --git a/scripts/qualification/run-public-serialization-roundtrip.mjs b/scripts/qualification/run-public-serialization-roundtrip.mjs
index b337184..49d2687 100644
--- a/scripts/qualification/run-public-serialization-roundtrip.mjs
+++ b/scripts/qualification/run-public-serialization-roundtrip.mjs
@@ -72,9 +72,9 @@ function normalizeNodes(nodes) {
const failures = [];
const positive = [];
for (const testCase of POSITIVE_CASES) {
- const first = parseFragment(testCase.html, HTML_DIV_CONTEXT);
+ const first = parseFragment(testCase.html, HTML_DIV_CONTEXT).tree;
const serialized = serialize(first);
- const second = parseFragment(serialized, HTML_DIV_CONTEXT);
+ const second = parseFragment(serialized, HTML_DIV_CONTEXT).tree;
const before = normalizeNodes(first.children);
const after = normalizeNodes(second.children);
const stable = JSON.stringify(before) === JSON.stringify(after);
@@ -87,16 +87,16 @@ for (const testCase of POSITIVE_CASES) {
if (!stable) failures.push({ id: testCase.id, reason: "unexpected-roundtrip-difference" });
}
-const plaintextFirst = parseFragment("a", HTML_DIV_CONTEXT);
+const plaintextFirst = parseFragment("a", HTML_DIV_CONTEXT).tree;
const plaintextSerialized = serialize(plaintextFirst);
-const plaintextSecond = parseFragment(plaintextSerialized, HTML_DIV_CONTEXT);
+const plaintextSecond = parseFragment(plaintextSerialized, HTML_DIV_CONTEXT).tree;
const rawCase = PUBLIC_SERIALIZATION_QUALIFICATION_CASES.find(
(testCase) => testCase.id === "classified/raw-effective-end-tag"
);
if (rawCase === undefined) throw new Error("missing raw effective-end-tag qualification case");
const rawNode = createPublicSerializationNode(rawCase.descriptor);
const rawSerialized = serialize(rawNode);
-const rawSecond = parseFragment(rawSerialized, HTML_DIV_CONTEXT);
+const rawSecond = parseFragment(rawSerialized, HTML_DIV_CONTEXT).tree;
const classified = [
{
diff --git a/scripts/qualification/run-public-serialization.mjs b/scripts/qualification/run-public-serialization.mjs
index 0b393fc..8158aed 100644
--- a/scripts/qualification/run-public-serialization.mjs
+++ b/scripts/qualification/run-public-serialization.mjs
@@ -80,7 +80,7 @@ for (const testCase of PUBLIC_SERIALIZATION_QUALIFICATION_CASES) {
const parsedWptOutcomes = [];
for (let index = 0; index < WPT_SERIALIZING_OUTER_EXPECTATIONS.length; index += 1) {
const expected = WPT_SERIALIZING_OUTER_EXPECTATIONS[index];
- const fragment = parseFragment(expected, HTML_DIV_CONTEXT);
+ const { tree: fragment } = parseFragment(expected, HTML_DIV_CONTEXT);
const node = fragment.children[0];
const actual = node === undefined ? "" : serialize(node);
const id = `serializing.html#outerHTML-${String(index)}`;
diff --git a/scripts/qualification/run-resource-limits.mjs b/scripts/qualification/run-resource-limits.mjs
index d29e43a..14065be 100644
--- a/scripts/qualification/run-resource-limits.mjs
+++ b/scripts/qualification/run-resource-limits.mjs
@@ -61,6 +61,16 @@ function fixture(name) {
expectedLimit: 65_536
};
}
+ if (name === "byte-source-none" || name === "byte-source-text") {
+ const sourceRetention = name === "byte-source-none" ? "none" : "text";
+ return {
+ input: new Uint8Array(8 * 1024 * 1024).fill(0x20),
+ collectAfterRun: true,
+ run(mod, input) {
+ return mod.parseBytes(input, { sourceRetention });
+ }
+ };
+ }
if (name === "trace") {
return {
input: "\0".repeat(2_000),
@@ -96,6 +106,7 @@ if (workerCase) {
}
const wallMs = performance.now() - startedAt;
const cpu = process.cpuUsage(cpuBefore);
+ if (selected.collectAfterRun === true) globalThis.gc();
const after = process.memoryUsage();
const peakRssBytes = process.resourceUsage().maxRSS * 1024;
@@ -135,7 +146,8 @@ if (workerCase) {
actual: failure.actual
}
: { code: "OK" },
- retainedResultKind: retainedResult?.tree?.kind ?? retainedResult?.kind ?? null
+ retainedResultKind: retainedResult?.tree?.kind ?? retainedResult?.kind ?? null,
+ retainedSourceCodeUnits: retainedResult?.sourceText?.length ?? 0
};
process.stdout.write(`${JSON.stringify(report)}\n`);
process.exit(0);
@@ -144,19 +156,28 @@ if (workerCase) {
const fixtureNames = ["nodes", "attributes", "errors", "decoded-output", "trace"];
const comparisons = [];
-for (const name of fixtureNames) {
- const runs = {};
- for (const mode of ["unbounded", "bounded"]) {
- const result = spawnSync(
- process.execPath,
- ["--expose-gc", fileURLToPath(import.meta.url), `--worker=${name}`, ...(mode === "bounded" ? ["--bounded"] : [])],
- { encoding: "utf8", maxBuffer: 1024 * 1024 }
- );
- if (result.status !== 0) {
- throw new Error(`${name}/${mode} worker failed:\n${result.stderr || result.stdout}`);
- }
- runs[mode] = JSON.parse(result.stdout.trim());
+function runWorker(name, isBounded = false) {
+ const result = spawnSync(
+ process.execPath,
+ [
+ "--expose-gc",
+ fileURLToPath(import.meta.url),
+ `--worker=${name}`,
+ ...(isBounded ? ["--bounded"] : [])
+ ],
+ { encoding: "utf8", maxBuffer: 1024 * 1024 }
+ );
+ if (result.status !== 0) {
+ throw new Error(`${name}/${isBounded ? "bounded" : "unbounded"} worker failed:\n${result.stderr || result.stdout}`);
}
+ return JSON.parse(result.stdout.trim());
+}
+
+for (const name of fixtureNames) {
+ const runs = {
+ unbounded: runWorker(name),
+ bounded: runWorker(name, true)
+ };
if (runs.bounded.retainedHeapDeltaBytes >= runs.unbounded.retainedHeapDeltaBytes) {
throw new Error(`${name}: bounded retained heap was not lower than the unbounded run`);
}
@@ -169,6 +190,19 @@ for (const name of fixtureNames) {
});
}
+const byteSourceRetention = {
+ none: runWorker("byte-source-none"),
+ text: runWorker("byte-source-text")
+};
+if (byteSourceRetention.none.retainedSourceCodeUnits !== 0 ||
+ byteSourceRetention.text.retainedSourceCodeUnits !== byteSourceRetention.text.inputBytes) {
+ throw new Error("byte source-retention workers returned the wrong public source shape");
+}
+if (byteSourceRetention.none.retainedHeapDeltaBytes >=
+ byteSourceRetention.text.retainedHeapDeltaBytes) {
+ throw new Error("byte parsing without source retention did not retain less heap");
+}
+
const report = {
schemaVersion: 1,
suite: "html-parser-resource-limits",
@@ -183,10 +217,17 @@ const report = {
isolation: "one fresh process for each bounded or unbounded fixture",
baseline: "explicit GC after fixture construction and before the parse",
retainedHeap: "heapUsed sampled immediately after synchronous return or throw",
+ sourceRetentionHeap: "heapUsed sampled after explicit post-parse GC for byte source-retention fixtures",
peakRss: "process.resourceUsage().maxRSS high-water mark",
- assertion: "every bounded run reports limit + 1 and retains less heap than its paired unbounded run"
+ assertion: "hard budgets fail at limit + 1 and byte parsing retains decoded source only when requested"
},
- comparisons
+ comparisons,
+ byteSourceRetention: {
+ ...byteSourceRetention,
+ retainedHeapReductionBytes:
+ byteSourceRetention.text.retainedHeapDeltaBytes -
+ byteSourceRetention.none.retainedHeapDeltaBytes
+ }
};
await writeJson(options.report, report);
diff --git a/scripts/smoke/browser-smoke.mjs b/scripts/smoke/browser-smoke.mjs
index 7d3e085..2996ee2 100644
--- a/scripts/smoke/browser-smoke.mjs
+++ b/scripts/smoke/browser-smoke.mjs
@@ -84,10 +84,11 @@ async function runBrowserSmoke(baseUrl) {
const serialized = mod.serialize(parsed);
const parsedBytes = mod.parseBytes(new TextEncoder().encode(sampleHtml)).tree;
const bytesSerialized = mod.serialize(parsedBytes);
- const fragment = mod.parseFragment("child", {
+ const fragmentResult = mod.parseFragment("child", {
namespaceUri: mod.HTML_NAMESPACE_URI,
localName: "section"
- });
+ }, { budgets: { maxSteps: 10_000 } });
+ const fragment = fragmentResult.tree;
const streamParsed = (await mod.parseStream(streamFromText(sampleHtml))).tree;
const streamSerialized = mod.serialize(streamParsed);
@@ -108,15 +109,19 @@ async function runBrowserSmoke(baseUrl) {
abortError = error;
}
- const tokenKinds = (await mod.tokenizeByteStreamEager(
- streamFromText(sampleHtml)
- )).map((token) => token.kind);
+ const tokenization = await mod.tokenizeByteStreamEager(
+ streamFromText(sampleHtml),
+ { budgets: { maxSteps: 10_000 } }
+ );
+ const tokenKinds = tokenization.tokens.map((token) => token.kind);
const stablePayload = {
serialized,
bytesSerialized,
streamSerialized,
fragmentContext: fragment.context,
+ fragmentResources: fragmentResult.metadata.resourceUsage,
+ tokenizationResources: tokenization.metadata.resourceUsage,
tokenKinds
};
const hashBuffer = await globalThis.crypto.subtle.digest(
@@ -135,13 +140,16 @@ async function runBrowserSmoke(baseUrl) {
parseBytes: bytesSerialized.includes(" smoke
"),
parseStream: streamSerialized.includes("smoke
"),
parseFragment: fragment.context.localName === "section" &&
- fragment.context.namespaceUri === mod.HTML_NAMESPACE_URI,
+ fragment.context.namespaceUri === mod.HTML_NAMESPACE_URI &&
+ Number.isSafeInteger(fragmentResult.metadata.resourceUsage.steps),
traceSummary: traceSummary?.mode === "summary" &&
!("events" in traceSummary) &&
traceSummary.summary.eventCount > 0 &&
traceSummary.summary.eventKinds.includes("token") &&
JSON.stringify(traceSummary) === JSON.stringify(secondTraceSummary),
- tokenizeByteStreamEager: tokenKinds.includes("startTag") && tokenKinds.includes("endTag"),
+ tokenizeByteStreamEager: tokenKinds.includes("startTag") &&
+ tokenKinds.includes("endTag") &&
+ Number.isSafeInteger(tokenization.metadata.resourceUsage.steps),
errorGuards: typeof mod.isHtmlBudgetExceededError === "function" &&
mod.isHtmlBudgetExceededError(budgetError) &&
budgetError.code === "BUDGET_EXCEEDED" &&
diff --git a/scripts/smoke/control.mjs b/scripts/smoke/control.mjs
index 1a9b539..946b2d7 100644
--- a/scripts/smoke/control.mjs
+++ b/scripts/smoke/control.mjs
@@ -149,7 +149,15 @@ async function computeDeterminismHash() {
const canonicalPayload = {
node: normalizeNode(parsed),
- parseErrors: Array.isArray(parsed.errors) ? parsed.errors.map((entry) => normalizeParseError(entry)) : []
+ parseErrors: Array.isArray(parsed.errors) ? parsed.errors.map((entry) => normalizeParseError(entry)) : [],
+ fragment: parseFragment("x
", {
+ namespaceUri: HTML_NAMESPACE_URI,
+ localName: "section"
+ }, { budgets: { maxSteps: 10_000 } }),
+ tokenization: await tokenizeByteStreamEager(
+ createByteStream([new TextEncoder().encode("x
")]),
+ { budgets: { maxSteps: 10_000 } }
+ )
};
return sha256Hex(JSON.stringify(canonicalPayload));
@@ -198,11 +206,16 @@ async function runSmokeAssertions() {
const second = parse("deterministic");
ensure(JSON.stringify(first) === JSON.stringify(second), "deterministic output mismatch");
- const fragment = parseFragment("child", {
+ const fragmentResult = parseFragment("child", {
namespaceUri: HTML_NAMESPACE_URI,
localName: "section"
- });
+ }, { budgets: { maxSteps: 10_000 } });
+ const { tree: fragment } = fragmentResult;
ensure(fragment.context.localName === "section", "fragment context mismatch");
+ ensure(
+ Number.isSafeInteger(fragmentResult.metadata.resourceUsage.steps),
+ "fragment resource metadata mismatch"
+ );
const sampleBytes = new Uint8Array([
0x3c, 0x6d, 0x65, 0x74, 0x61, 0x20, 0x63, 0x68, 0x61, 0x72, 0x73, 0x65, 0x74, 0x3d, 0x77, 0x69, 0x6e, 0x64,
@@ -218,13 +231,19 @@ async function runSmokeAssertions() {
"parseStream output mismatch vs parseBytes"
);
- const tokenKinds = (await tokenizeByteStreamEager(
- createByteStream([new TextEncoder().encode("smoke
")])
- )).map((token) => token.kind);
+ const tokenization = await tokenizeByteStreamEager(
+ createByteStream([new TextEncoder().encode("smoke
")]),
+ { budgets: { maxSteps: 10_000 } }
+ );
+ const tokenKinds = tokenization.tokens.map((token) => token.kind);
ensure(
JSON.stringify(tokenKinds) === JSON.stringify(["startTag", "chars", "endTag", "eof"]),
"tokenizeByteStreamEager mismatch"
);
+ ensure(
+ Number.isSafeInteger(tokenization.metadata.resourceUsage.steps),
+ "tokenization resource metadata mismatch"
+ );
const outlineResult = outline(parsed);
ensure(outlineResult.entries.length === 0, "outline generation mismatch");
diff --git a/src/internal/encoding/sniff.ts b/src/internal/encoding/sniff.ts
index 6109846..ea8cae1 100644
--- a/src/internal/encoding/sniff.ts
+++ b/src/internal/encoding/sniff.ts
@@ -1,17 +1,17 @@
-interface EncodingSniffOptions {
+export interface EncodingSniffOptions {
readonly transportEncodingLabel?: string;
readonly maxPrescanBytes?: number;
readonly defaultEncoding?: string;
}
-interface EncodingSniffResult {
+export interface EncodingSniffResult {
readonly encoding: string;
readonly source: "bom" | "transport" | "meta" | "default";
}
-interface HtmlByteDecodeOptions extends EncodingSniffOptions {
- readonly onDecodedChunk?: (chunk: string) => void;
-}
+export type EncodingSniffDecision =
+ | { readonly status: "pending" }
+ | { readonly status: "decided"; readonly result: EncodingSniffResult };
const WINDOWS_1252_ALIASES = new Set([
"iso-8859-1",
@@ -37,6 +37,33 @@ function detectBom(bytes: Uint8Array): string | null {
return null;
}
+function bomPrefixState(
+ bytes: Uint8Array,
+ endOfStream: boolean
+): "pending" | "absent" | "utf-8" | "utf-16be" | "utf-16le" {
+ const first = bytes[0];
+ if (first === undefined) return endOfStream ? "absent" : "pending";
+ if (first === 0xef) {
+ const second = bytes[1];
+ if (second === undefined) return endOfStream ? "absent" : "pending";
+ if (second !== 0xbb) return "absent";
+ const third = bytes[2];
+ if (third === undefined) return endOfStream ? "absent" : "pending";
+ return third === 0xbf ? "utf-8" : "absent";
+ }
+ if (first === 0xfe) {
+ const second = bytes[1];
+ if (second === undefined) return endOfStream ? "absent" : "pending";
+ return second === 0xff ? "utf-16be" : "absent";
+ }
+ if (first === 0xff) {
+ const second = bytes[1];
+ if (second === undefined) return endOfStream ? "absent" : "pending";
+ return second === 0xfe ? "utf-16le" : "absent";
+ }
+ return "absent";
+}
+
function stripQuotes(value: string): string {
const trimmed = value.trim();
if (
@@ -291,31 +318,46 @@ export function sniffHtmlEncoding(bytes: Uint8Array, options: EncodingSniffOptio
return { encoding: defaultEncoding, source: "default" };
}
-export function decodeHtmlBytes(bytes: Uint8Array, options: HtmlByteDecodeOptions = {}): { text: string; sniff: EncodingSniffResult } {
- const sniff = sniffHtmlEncoding(bytes, {
- ...(options.transportEncodingLabel !== undefined
- ? { transportEncodingLabel: options.transportEncodingLabel }
- : {}),
- ...(options.maxPrescanBytes !== undefined ? { maxPrescanBytes: options.maxPrescanBytes } : {}),
- ...(options.defaultEncoding !== undefined ? { defaultEncoding: options.defaultEncoding } : {})
- });
- const decoder = new TextDecoder(sniff.encoding);
- const parts: string[] = [];
- const decodeChunkBytes = 16_384;
- for (let offset = 0; offset < bytes.byteLength; offset += decodeChunkBytes) {
- const decoded = decoder.decode(bytes.subarray(offset, offset + decodeChunkBytes), { stream: true });
- if (decoded.length > 0) {
- options.onDecodedChunk?.(decoded);
- parts.push(decoded);
+/** Determines whether the available stream prefix is sufficient to select an encoding. */
+export function decideHtmlEncoding(
+ bytes: Uint8Array,
+ options: EncodingSniffOptions & { readonly endOfStream: boolean }
+): EncodingSniffDecision {
+ const bomState = bomPrefixState(bytes, options.endOfStream);
+ if (bomState === "pending") return Object.freeze({ status: "pending" });
+ if (bomState !== "absent") {
+ return Object.freeze({
+ status: "decided",
+ result: Object.freeze({ encoding: bomState, source: "bom" })
+ });
+ }
+
+ if (options.transportEncodingLabel !== undefined) {
+ const transport = canonicalizeLabel(options.transportEncodingLabel, "transport");
+ if (transport !== null) {
+ return Object.freeze({
+ status: "decided",
+ result: Object.freeze({ encoding: transport, source: "transport" })
+ });
}
}
- const final = decoder.decode();
- if (final.length > 0) {
- options.onDecodedChunk?.(final);
- parts.push(final);
+
+ const maxPrescanBytes = options.maxPrescanBytes ?? 16_384;
+ const meta = sniffMetaCharset(bytes, maxPrescanBytes);
+ if (meta !== null) {
+ return Object.freeze({
+ status: "decided",
+ result: Object.freeze({ encoding: meta, source: "meta" })
+ });
}
- return {
- text: parts.join(""),
- sniff
- };
+ if (!options.endOfStream && bytes.byteLength < maxPrescanBytes) {
+ return Object.freeze({ status: "pending" });
+ }
+
+ const fallback = canonicalizeLabel(options.defaultEncoding ?? "windows-1252", "default") ??
+ "windows-1252";
+ return Object.freeze({
+ status: "decided",
+ result: Object.freeze({ encoding: fallback, source: "default" })
+ });
}
diff --git a/src/internal/foundation/internal-state-error.ts b/src/internal/foundation/internal-state-error.ts
index b8afa62..e206847 100644
--- a/src/internal/foundation/internal-state-error.ts
+++ b/src/internal/foundation/internal-state-error.ts
@@ -1,15 +1,17 @@
const INTERNAL_STATE_COMPONENT_BY_REASON = Object.freeze({
CHARACTER_REFERENCE_ASCII_CONSUMPTION_MISMATCH: "character-reference-consumer",
+ ENCODING_SNIFF_BOUNDED_PREFIX_UNDECIDED: "encoding-sniff",
+ ENCODING_SNIFF_EOF_UNDECIDED: "encoding-sniff",
FOUNDATION_CURSOR_REQUESTED_MORE_AFTER_CLOSE: "foundation-driver",
GENERATED_CHARACTER_REFERENCE_ENTRY_MISSING: "named-character-references",
GENERATED_CHARACTER_REFERENCE_LENGTH_MISMATCH: "named-character-references",
INPUT_CURSOR_BUFFER_UNDERRUN: "input-cursor",
INPUT_CURSOR_RECONSUME_CHARACTER_MISSING: "input-cursor",
PUBLIC_PARSER_CHILD_CONVERSION_MISSING: "public-parser",
+ PUBLIC_PARSER_BYTES_ENGINE_NOT_INITIALIZED: "public-parser",
PUBLIC_PARSER_ROOT_CONVERSION_MISSING: "public-parser",
PUBLIC_PARSER_STREAM_ENGINE_NOT_INITIALIZED: "public-parser",
PUBLIC_PARSER_TEMPLATE_CONTENT_KIND_MISMATCH: "public-parser",
- PUBLIC_TOKENIZER_RETAINED_TEXT_MISSING: "public-tokenizer",
TOKENIZER_BUFFERED_CHARACTER_MISSING: "tokenizer",
TOKENIZER_CDATA_BRACKET_POSITION_MISSING: "tokenizer",
TOKENIZER_CHARACTER_REFERENCE_MISSING: "tokenizer",
diff --git a/src/internal/foundation/name-validation.ts b/src/internal/foundation/name-validation.ts
index 9dc0bd3..95fb490 100644
--- a/src/internal/foundation/name-validation.ts
+++ b/src/internal/foundation/name-validation.ts
@@ -7,6 +7,13 @@ const FORBIDDEN_HTML_ATTRIBUTE_SYNTAX = new Set([
0x3e // >
]);
+const FORBIDDEN_HTML_ELEMENT_SYNTAX = new Set([
+ 0x20, // ASCII space
+ 0x2f, // /
+ 0x3c, // <
+ 0x3e // >
+]);
+
function isControl(codePoint: number): boolean {
return codePoint <= 0x1f || (codePoint >= 0x7f && codePoint <= 0x9f);
}
@@ -67,3 +74,17 @@ export function isHtmlAttributeName(value: string): boolean {
}
return true;
}
+
+/** Whether a value can be emitted as one HTML tag name without changing tokenization. */
+export function isHtmlElementName(value: string): boolean {
+ if (value.length === 0) return false;
+ for (const character of value) {
+ const codePoint = character.codePointAt(0);
+ if (codePoint === undefined || isControl(codePoint) ||
+ FORBIDDEN_HTML_ELEMENT_SYNTAX.has(codePoint) ||
+ (codePoint >= 0xd800 && codePoint <= 0xdfff)) {
+ return false;
+ }
+ }
+ return true;
+}
diff --git a/src/public/chunking.ts b/src/public/chunking.ts
index 0cf8e8d..ecad099 100644
--- a/src/public/chunking.ts
+++ b/src/public/chunking.ts
@@ -4,6 +4,7 @@ import {
normalizeChunkOptions
} from "./operation.ts";
import { serializeNodes } from "./serialization.ts";
+import { validateSerializableInput } from "./tree-validation.ts";
import type {
Chunk,
@@ -23,6 +24,7 @@ export function chunk(tree: DocumentTree | FragmentTree, options: ChunkOptions =
startedAt
);
operation.checkpoint();
+ validateSerializableInput(tree, operation);
const maxChars = normalizedOptions.maxChars ?? 8192;
const maxNodes = normalizedOptions.maxNodes ?? 256;
const maxBytes = normalizedOptions.maxBytes ?? Number.POSITIVE_INFINITY;
diff --git a/src/public/html-input.ts b/src/public/html-input.ts
index 7c03348..984351f 100644
--- a/src/public/html-input.ts
+++ b/src/public/html-input.ts
@@ -1,4 +1,8 @@
-import { sniffHtmlEncoding } from "../internal/encoding/sniff.ts";
+import {
+ decideHtmlEncoding,
+ sniffHtmlEncoding
+} from "../internal/encoding/sniff.ts";
+import { failInternalState } from "../internal/foundation/internal-state-error.ts";
import { enforceBudget } from "./budgets.ts";
import {
@@ -34,6 +38,12 @@ interface StreamDecoderState {
readonly sniff: StreamEncodingSniff;
}
+interface DecodeCallbacks {
+ readonly retainText: boolean;
+ readonly onEncodingSniff?: (sniff: StreamEncodingSniff) => void;
+ readonly onDecodedChunk?: (chunk: string) => void;
+}
+
export function requireString(value: unknown, option: string): asserts value is string {
if (typeof value !== "string") {
throw new HtmlConfigurationError(option, "INVALID_VALUE", "must be a string");
@@ -106,6 +116,96 @@ export class DecodedUtf8BudgetCounter {
}
}
+class DecodedOutputCollector {
+ readonly #operation: OperationContext;
+ readonly #budget: DecodedUtf8BudgetCounter;
+ readonly #parts: string[] | null;
+ readonly #onDecodedChunk: ((chunk: string) => void) | undefined;
+ #codeUnits = 0;
+
+ constructor(
+ limit: number | undefined,
+ callbacks: DecodeCallbacks,
+ operation: OperationContext
+ ) {
+ this.#operation = operation;
+ this.#budget = new DecodedUtf8BudgetCounter(limit, operation);
+ this.#parts = callbacks.retainText ? [] : null;
+ this.#onDecodedChunk = callbacks.onDecodedChunk;
+ }
+
+ append(value: string): void {
+ if (value.length === 0) return;
+ this.#budget.append(value);
+ const nextCodeUnits = this.#codeUnits + value.length;
+ if (!Number.isSafeInteger(nextCodeUnits)) {
+ throw new HtmlConfigurationError(
+ "input",
+ "INVALID_VALUE",
+ "decoded input must fit in a safe UTF-16 code-unit count"
+ );
+ }
+ this.#codeUnits = nextCodeUnits;
+ this.#onDecodedChunk?.(value);
+ this.#parts?.push(value);
+ this.#operation.checkpoint();
+ }
+
+ get bytes(): number {
+ return this.#budget.bytes;
+ }
+
+ get codeUnits(): number {
+ return this.#codeUnits;
+ }
+
+ text(): string | null {
+ return this.#parts?.join("") ?? null;
+ }
+}
+
+function decodeTransportBytes(
+ decoder: TextDecoder,
+ bytes: Uint8Array,
+ output: DecodedOutputCollector,
+ operation: OperationContext
+): void {
+ const decodeChunkBytes = 16_384;
+ for (let offset = 0; offset < bytes.byteLength; offset += decodeChunkBytes) {
+ operation.checkpoint();
+ output.append(decoder.decode(bytes.subarray(offset, offset + decodeChunkBytes), { stream: true }));
+ }
+}
+
+/** Decodes an in-memory byte input without retaining decoded text unless requested. */
+export function decodeByteArray(
+ bytes: Uint8Array,
+ options: DecodeCallbacks & {
+ readonly transportEncodingLabel?: string;
+ readonly maxDecodedUtf8Bytes?: number;
+ },
+ operation: OperationContext
+): StreamDecodeResult {
+ operation.checkpoint();
+ const sniff = sniffHtmlEncoding(bytes, options.transportEncodingLabel === undefined
+ ? {}
+ : { transportEncodingLabel: options.transportEncodingLabel });
+ options.onEncodingSniff?.(sniff);
+ const output = new DecodedOutputCollector(options.maxDecodedUtf8Bytes, options, operation);
+ const decoder = new TextDecoder(sniff.encoding);
+ decodeTransportBytes(decoder, bytes, output, operation);
+ output.append(decoder.decode());
+ return Object.freeze({
+ text: output.text(),
+ sniff,
+ totalBytes: bytes.byteLength,
+ decodedUtf8Bytes: output.bytes,
+ decodedCodeUnits: output.codeUnits,
+ encodingPrescanBytes: 0,
+ encodingPrescanLimitBytes: 0
+ });
+}
+
async function readStreamChunk(
reader: ReadableStreamDefaultReader,
operation: OperationContext
@@ -186,55 +286,32 @@ export async function decodeByteStream(
DEFAULT_STREAM_ENCODING_PRESCAN_BYTES,
budgets?.maxEncodingPrescanBytes ?? DEFAULT_STREAM_ENCODING_PRESCAN_BYTES
);
- const pendingBytesBuffer = new Uint8Array(prescanLimit);
+ const pendingBytesBuffer = new Uint8Array(Math.max(3, prescanLimit));
let pendingBytes = 0;
let encodingPrescanBytes = 0;
let decoderState: StreamDecoderState | undefined;
- const decodedParts: string[] | null = options.retainText ? [] : null;
- let decodedCodeUnits = 0;
- const decodedBudget = new DecodedUtf8BudgetCounter(
+ const output = new DecodedOutputCollector(
budgets?.maxDecodedUtf8Bytes,
+ options,
operation
);
const sniffOptions = options.transportEncodingLabel === undefined
? { maxPrescanBytes: prescanLimit }
: { transportEncodingLabel: options.transportEncodingLabel, maxPrescanBytes: prescanLimit };
- const appendDecoded = (value: string): void => {
- if (value.length > 0) {
- decodedBudget.append(value);
- const nextCodeUnits = decodedCodeUnits + value.length;
- if (!Number.isSafeInteger(nextCodeUnits)) {
- throw new HtmlConfigurationError(
- "input",
- "INVALID_VALUE",
- "decoded input must fit in a safe UTF-16 code-unit count"
- );
- }
- decodedCodeUnits = nextCodeUnits;
- options.onDecodedChunk?.(value);
- decodedParts?.push(value);
- }
- };
- const decodeBytes = (decoder: TextDecoder, bytes: Uint8Array): void => {
- const decodeChunkBytes = 16_384;
- for (let offset = 0; offset < bytes.byteLength; offset += decodeChunkBytes) {
- operation.checkpoint();
- appendDecoded(decoder.decode(bytes.subarray(offset, offset + decodeChunkBytes), { stream: true }));
- }
- };
- const initializeDecoder = (): StreamDecoderState => {
+ const initializeDecoder = (endOfStream: boolean): StreamDecoderState | undefined => {
const bufferedBytes = pendingBytesBuffer.subarray(0, pendingBytes);
- const sniff = sniffHtmlEncoding(bufferedBytes, sniffOptions);
+ const decision = decideHtmlEncoding(bufferedBytes, { ...sniffOptions, endOfStream });
+ if (decision.status === "pending") return undefined;
+ const sniff = decision.result;
const state = { decoder: new TextDecoder(sniff.encoding), sniff };
decoderState = state;
options.onEncodingSniff?.(sniff);
- decodeBytes(state.decoder, bufferedBytes);
+ decodeTransportBytes(state.decoder, bufferedBytes, output, operation);
return state;
};
try {
- if (prescanLimit === 0) initializeDecoder();
for (;;) {
const next = await readStreamChunk(reader, operation);
if (next.done) break;
@@ -249,31 +326,70 @@ export async function decodeByteStream(
operation.checkpoint();
if (decoderState === undefined) {
- const bytesToBuffer = Math.min(chunkValue.byteLength, prescanLimit - pendingBytes);
- pendingBytesBuffer.set(chunkValue.subarray(0, bytesToBuffer), pendingBytes);
- pendingBytes += bytesToBuffer;
- encodingPrescanBytes = Math.max(encodingPrescanBytes, pendingBytes);
- if (pendingBytes < prescanLimit) continue;
- const activeDecoder = initializeDecoder().decoder;
- if (bytesToBuffer < chunkValue.byteLength) {
- decodeBytes(activeDecoder, chunkValue.subarray(bytesToBuffer));
+ let offset = 0;
+ while (offset < chunkValue.byteLength) {
+ const capacity = pendingBytesBuffer.byteLength - pendingBytes;
+ if (capacity === 0) {
+ const initialized = initializeDecoder(false);
+ if (initialized === undefined) {
+ failInternalState("ENCODING_SNIFF_BOUNDED_PREFIX_UNDECIDED");
+ }
+ break;
+ }
+ let bytesToBuffer = Math.min(chunkValue.byteLength - offset, capacity);
+ if (pendingBytes < 3) {
+ bytesToBuffer = 1;
+ } else {
+ const nextTagEnd = chunkValue.indexOf(0x3e, offset);
+ if (nextTagEnd !== -1) {
+ bytesToBuffer = Math.min(bytesToBuffer, nextTagEnd - offset + 1);
+ }
+ }
+ pendingBytesBuffer.set(
+ chunkValue.subarray(offset, offset + bytesToBuffer),
+ pendingBytes
+ );
+ pendingBytes += bytesToBuffer;
+ offset += bytesToBuffer;
+ encodingPrescanBytes = Math.max(
+ encodingPrescanBytes,
+ Math.min(pendingBytes, prescanLimit)
+ );
+ const lastBufferedByte = pendingBytesBuffer[pendingBytes - 1];
+ const initialized = pendingBytes <= 3 || lastBufferedByte === 0x3e ||
+ pendingBytes === pendingBytesBuffer.byteLength
+ ? initializeDecoder(false)
+ : undefined;
+ if (initialized !== undefined) break;
+ }
+ const initializedDecoder = decoderState as StreamDecoderState | undefined;
+ if (initializedDecoder !== undefined && offset < chunkValue.byteLength) {
+ decodeTransportBytes(
+ initializedDecoder.decoder,
+ chunkValue.subarray(offset),
+ output,
+ operation
+ );
}
continue;
}
- decodeBytes(decoderState.decoder, chunkValue);
+ decodeTransportBytes(decoderState.decoder, chunkValue, output, operation);
}
- const finalState = decoderState ?? initializeDecoder();
- appendDecoded(finalState.decoder.decode());
- return {
- text: decodedParts?.join("") ?? null,
+ const finalState = decoderState ?? initializeDecoder(true);
+ if (finalState === undefined) {
+ failInternalState("ENCODING_SNIFF_EOF_UNDECIDED");
+ }
+ output.append(finalState.decoder.decode());
+ return Object.freeze({
+ text: output.text(),
sniff: finalState.sniff,
totalBytes: total,
- decodedUtf8Bytes: decodedBudget.bytes,
- decodedCodeUnits,
+ decodedUtf8Bytes: output.bytes,
+ decodedCodeUnits: output.codeUnits,
encodingPrescanBytes,
encodingPrescanLimitBytes: prescanLimit
- };
+ });
} catch (error) {
try {
const cancellation = reader.cancel(error);
diff --git a/src/public/model.ts b/src/public/model.ts
index 7c85892..2ad9720 100644
--- a/src/public/model.ts
+++ b/src/public/model.ts
@@ -1,5 +1,83 @@
+import { HtmlConfigurationError } from "./errors.ts";
+
import type { OperationContext } from "./operation.ts";
-import type { HtmlNode } from "./types.ts";
+import type { ElementNode, HtmlNode } from "./types.ts";
+
+type UnknownRecord = Readonly>;
+
+function invalidModel(option: string, expected: string): never {
+ throw new HtmlConfigurationError(option, "INVALID_VALUE", expected);
+}
+
+function modelRecord(value: unknown, option: string): UnknownRecord {
+ if (typeof value !== "object" || value === null || Array.isArray(value)) {
+ invalidModel(option, "must be a node record");
+ }
+ return value as UnknownRecord;
+}
+
+function modelArray(value: unknown, option: string): readonly unknown[] {
+ if (!Array.isArray(value)) invalidModel(option, "must be a dense array");
+ for (let index = 0; index < value.length; index += 1) {
+ if (!Object.hasOwn(value, index)) invalidModel(option, "must be a dense array");
+ }
+ return value;
+}
+
+function modelString(value: unknown, option: string): string {
+ if (typeof value !== "string") invalidModel(option, "must be a string");
+ return value;
+}
+
+function validateAttributeShape(value: unknown, option: string): void {
+ const attribute = modelRecord(value, option);
+ const namespaceUri = attribute["namespaceUri"];
+ if (namespaceUri !== null && typeof namespaceUri !== "string") {
+ invalidModel(option, "must contain a string or null namespaceUri");
+ }
+ modelString(attribute["localName"], option);
+ modelString(attribute["value"], option);
+ if (attribute["prefix"] !== undefined) modelString(attribute["prefix"], option);
+}
+
+/** Validates the complete shape read by element query helpers. @internal */
+export function requireElementNode(value: unknown, option: string): ElementNode {
+ const element = modelRecord(value, option);
+ if (element["kind"] !== "element") invalidModel(option, "must be an element node");
+ if (!Number.isSafeInteger(element["id"]) || (element["id"] as number) < 1) {
+ invalidModel(option, "must have a positive safe-integer id");
+ }
+ modelString(element["namespaceUri"], option);
+ modelString(element["localName"], option);
+ if (element["prefix"] !== undefined) modelString(element["prefix"], option);
+ const attributes = modelArray(element["attributes"], option);
+ for (const attribute of attributes) validateAttributeShape(attribute, option);
+ modelArray(element["children"], option);
+ return value as ElementNode;
+}
+
+function requireTraversableNode(value: unknown): HtmlNode {
+ const node = modelRecord(value, "tree");
+ const kind = node["kind"];
+ if (!Number.isSafeInteger(node["id"]) || (node["id"] as number) < 1) {
+ invalidModel("tree", "must contain nodes with positive safe-integer ids");
+ }
+ if (kind === "element") return requireElementNode(value, "tree");
+ if (kind === "templateContent") {
+ modelArray(node["children"], "tree");
+ } else if (kind === "text" || kind === "comment") {
+ modelString(node["value"], "tree");
+ } else if (kind === "processingInstruction") {
+ modelString(node["target"], "tree");
+ modelString(node["data"], "tree");
+ } else if (kind === "doctype") {
+ modelString(node["name"], "tree");
+ modelRecord(node["externalId"], "tree");
+ } else {
+ invalidModel("tree", "must contain only supported HTML node kinds");
+ }
+ return value as HtmlNode;
+}
/** HTML namespace URI assigned by the tree builder. */
export const HTML_NAMESPACE_URI = "http://www.w3.org/1999/xhtml";
@@ -33,6 +111,24 @@ export function ownedChildNodes(node: HtmlNode): readonly HtmlNode[] {
return node.templateContent === undefined ? node.children : [node.templateContent];
}
+/** Rejects cyclic and multiply-owned caller graphs during public traversal. @internal */
+export class OwnedNodeTracker {
+ readonly #seen = new WeakSet();
+
+ observe(value: unknown): HtmlNode {
+ const node = requireTraversableNode(value);
+ if (this.#seen.has(node)) {
+ throw new HtmlConfigurationError(
+ "tree",
+ "INVALID_VALUE",
+ "must be acyclic and give every node exactly one owner"
+ );
+ }
+ this.#seen.add(node);
+ return node;
+ }
+}
+
/** @internal */
export function* iterateNodes(
nodes: readonly HtmlNode[],
@@ -40,6 +136,7 @@ export function* iterateNodes(
operation: OperationContext
): IterableIterator<{ readonly node: HtmlNode; readonly depth: number }> {
const stack: { readonly node: HtmlNode; readonly depth: number }[] = [];
+ const ownership = new OwnedNodeTracker();
for (let index = nodes.length - 1; index >= 0; index -= 1) {
const node = nodes[index];
if (node !== undefined) stack.push({ node, depth });
@@ -48,8 +145,9 @@ export function* iterateNodes(
const entry = stack.pop();
if (entry === undefined) continue;
operation.checkpoint();
- yield entry;
- const descendants = ownedChildNodes(entry.node);
+ const node = ownership.observe(entry.node);
+ yield { node, depth: entry.depth };
+ const descendants = ownedChildNodes(node);
for (let index = descendants.length - 1; index >= 0; index -= 1) {
const child = descendants[index];
if (child !== undefined) stack.push({ node: child, depth: entry.depth + 1 });
@@ -61,12 +159,14 @@ export function* iterateNodes(
export function countNodes(node: HtmlNode, operation: OperationContext): number {
let count = 0;
const stack: HtmlNode[] = [node];
+ const ownership = new OwnedNodeTracker();
while (stack.length > 0) {
const current = stack.pop();
if (current === undefined) continue;
operation.checkpoint();
+ const node = ownership.observe(current);
count += 1;
- const descendants = ownedChildNodes(current);
+ const descendants = ownedChildNodes(node);
for (let index = descendants.length - 1; index >= 0; index -= 1) {
const child = descendants[index];
if (child !== undefined) stack.push(child);
diff --git a/src/public/operation.ts b/src/public/operation.ts
index 10deaa5..c803f75 100644
--- a/src/public/operation.ts
+++ b/src/public/operation.ts
@@ -41,6 +41,7 @@ const TOKENIZE_BYTE_STREAM_EAGER_BUDGET_KEYS = Object.freeze([
"maxInputBytes",
"maxEncodingPrescanBytes",
"maxDecodedUtf8Bytes",
+ "maxSteps",
"maxParseErrors",
"maxAttributesPerElement",
"maxAttributeBytes",
diff --git a/src/public/parsed-document-registry.ts b/src/public/parsed-output-registry.ts
similarity index 65%
rename from src/public/parsed-document-registry.ts
rename to src/public/parsed-output-registry.ts
index e8f3e9a..f3f5d1d 100644
--- a/src/public/parsed-document-registry.ts
+++ b/src/public/parsed-output-registry.ts
@@ -1,4 +1,9 @@
-import type { ParsedDocument, PatchPlan } from "./types.ts";
+import type {
+ DocumentTree,
+ FragmentTree,
+ ParsedDocument,
+ PatchPlan
+} from "./types.ts";
interface ParsedDocumentRegistration {
readonly sourceText: string | null;
@@ -6,6 +11,7 @@ interface ParsedDocumentRegistration {
}
const parsedDocuments = new WeakMap();
+const parserOwnedTrees = new WeakSet();
const patchPlans = new WeakMap();
/** Registers identity-bound state which must not be forgeable through public object shapes. */
@@ -15,6 +21,12 @@ export function registerParsedDocument(
spansCaptured: boolean
): void {
parsedDocuments.set(document, Object.freeze({ sourceText, spansCaptured }));
+ parserOwnedTrees.add(document.tree);
+}
+
+/** Registers an immutable parser-owned fragment tree. */
+export function registerParsedFragmentTree(tree: FragmentTree): void {
+ parserOwnedTrees.add(tree);
}
/** Returns identity-bound parse state, or undefined for an unrecognized object. */
@@ -24,6 +36,11 @@ export function parsedDocumentRegistration(
return parsedDocuments.get(document);
}
+/** Whether a root is a deeply frozen tree produced by this module instance. */
+export function isParserOwnedTree(value: unknown): value is DocumentTree | FragmentTree {
+ return typeof value === "object" && value !== null && parserOwnedTrees.has(value as DocumentTree);
+}
+
/** Binds a frozen patch plan to the exact parsed document which produced it. */
export function registerPatchPlan(plan: PatchPlan, document: ParsedDocument): void {
patchPlans.set(plan, document);
diff --git a/src/public/parsing.ts b/src/public/parsing.ts
index 3081042..8344c89 100644
--- a/src/public/parsing.ts
+++ b/src/public/parsing.ts
@@ -1,4 +1,3 @@
-import { decodeHtmlBytes } from "../internal/encoding/sniff.ts";
import {
failInternalState,
requireInternalValue
@@ -19,7 +18,7 @@ import { enforceBudget } from "./budgets.ts";
import { HtmlAbortError, HtmlBudgetExceededError } from "./errors.ts";
import { normalizeFragmentContext, toEngineFragmentContext } from "./fragment-context.ts";
import {
- DecodedUtf8BudgetCounter,
+ decodeByteArray,
decodeByteStream,
requireByteArray,
requireReadableByteStream,
@@ -36,7 +35,10 @@ import {
normalizeTokenizeByteStreamEagerOptions
} from "./operation.ts";
import { normalizeParseErrorId, TraceSink } from "./parse-trace.ts";
-import { registerParsedDocument } from "./parsed-document-registry.ts";
+import {
+ registerParsedDocument,
+ registerParsedFragmentTree
+} from "./parsed-output-registry.ts";
import type { OperationContext } from "./operation.ts";
import type {
@@ -49,15 +51,17 @@ import type {
NodeId,
ParseBytesOptions,
ParseFragmentOptions,
+ ParsedFragment,
ParseOptions,
ParsedDocument,
- ParsedDocumentMetadata,
+ ParseMetadata,
ParseResourceUsage,
ParseStreamOptions,
Span,
SpanProvenance,
TemplateContentNode,
Token,
+ TokenizeByteStreamEagerResult,
TokenizeByteStreamEagerOptions
} from "./types.ts";
import type {
@@ -84,12 +88,12 @@ import type {
} from "../internal/html-engine/tree-model.ts";
interface ParseInputContext {
- readonly inputKind: ParsedDocumentMetadata["inputKind"];
+ readonly inputKind: ParseMetadata["inputKind"];
readonly byteLength: number;
readonly decodedUtf8ByteLength: number;
readonly decodedCodeUnits: number;
readonly transportByteLength: number | null;
- readonly metadataEncoding: ParsedDocumentMetadata["encoding"];
+ readonly metadataEncoding: ParseMetadata["encoding"];
readonly encodingPrescanBytes: number;
readonly operation: OperationContext;
readonly startedAt: number;
@@ -780,6 +784,29 @@ function finishDocumentOperation(
return parsed;
}
+function fragmentResourceUsage(
+ inputBytes: number,
+ decodedCodeUnits: number,
+ budgets: ParseFragmentOptions["budgets"],
+ result: HtmlEngineProductResult,
+ trace: TraceSink
+): ParseResourceUsage {
+ return Object.freeze({
+ inputBytes,
+ decodedUtf8Bytes: inputBytes,
+ decodedCodeUnits,
+ steps: budgets?.maxSteps === undefined ? null : result.resources.steps,
+ nodes: result.resources.nodes,
+ maxDepth: result.resources.maxDepth,
+ parseErrors: result.resources.parseErrors,
+ attributes: result.resources.attributes,
+ attributeUtf8Bytes: result.resources.attributeUtf8Bytes,
+ encodingPrescanBytes: 0,
+ traceEvents: trace.eventCount,
+ traceUtf8Bytes: trace.eventUtf8Bytes
+ });
+}
+
function parseDocumentOperation(
html: string,
options: ParseOptions | ParseBytesOptions | ParseStreamOptions,
@@ -844,28 +871,65 @@ export function parseBytes(
const operation = createOperationContext(normalized.budgets?.maxTimeMs, normalized.signal, startedAt);
operation.checkpoint();
enforceBudget("maxInputBytes", normalized.budgets?.maxInputBytes, bytes.byteLength);
- const decodedBudget = new DecodedUtf8BudgetCounter(
- normalized.budgets?.maxDecodedUtf8Bytes,
+ const trace = new TraceSink(
+ normalized.trace ?? "none",
+ normalized.onTraceEvent,
+ normalized.budgets,
operation
);
- const decoded = decodeHtmlBytes(bytes, {
- ...(normalized.transportEncodingLabel === undefined
- ? {}
- : { transportEncodingLabel: normalized.transportEncodingLabel }),
- onDecodedChunk(chunk): void { decodedBudget.append(chunk); }
- });
- return parseDocumentOperation(decoded.text, normalized, {
+ let engine: ReturnType | undefined;
+ let decoded;
+ try {
+ decoded = decodeByteArray(bytes, {
+ retainText: normalized.sourceRetention === "text",
+ ...(normalized.transportEncodingLabel === undefined
+ ? {}
+ : { transportEncodingLabel: normalized.transportEncodingLabel }),
+ ...(normalized.budgets?.maxDecodedUtf8Bytes === undefined
+ ? {}
+ : { maxDecodedUtf8Bytes: normalized.budgets.maxDecodedUtf8Bytes }),
+ onEncodingSniff(sniff): void {
+ trace.emit({
+ kind: "decode",
+ source: "sniff",
+ encoding: sniff.encoding,
+ sniffSource: sniff.source
+ });
+ engine = createDocumentEngine(normalized, startedAt, trace);
+ },
+ onDecodedChunk(chunk): void {
+ requireInternalValue(
+ engine,
+ "PUBLIC_PARSER_BYTES_ENGINE_NOT_INITIALIZED"
+ ).session.write(chunk);
+ }
+ }, operation);
+ } catch (error) {
+ return mapEngineFailure(error);
+ }
+ const activeEngine = requireInternalValue(
+ engine,
+ "PUBLIC_PARSER_BYTES_ENGINE_NOT_INITIALIZED"
+ );
+ let result: HtmlEngineProductResult;
+ try {
+ result = activeEngine.session.finishForPublicConversion();
+ } catch (error) {
+ return mapEngineFailure(error);
+ }
+ trace.emitBudget("maxInputBytes", normalized.budgets?.maxInputBytes, bytes.byteLength);
+ return finishDocumentOperation(decoded.text, normalized, {
inputKind: "bytes",
byteLength: bytes.byteLength,
- decodedUtf8ByteLength: decodedBudget.bytes,
- decodedCodeUnits: decoded.text.length,
+ decodedUtf8ByteLength: decoded.decodedUtf8Bytes,
+ decodedCodeUnits: decoded.decodedCodeUnits,
transportByteLength: bytes.byteLength,
metadataEncoding: { name: decoded.sniff.encoding, source: decoded.sniff.source },
encodingPrescanBytes: 0,
operation,
startedAt,
decode: { source: "sniff", encoding: decoded.sniff.encoding, sniffSource: decoded.sniff.source }
- });
+ }, trace, result, activeEngine.state.tokenCount);
}
/**
@@ -879,14 +943,14 @@ export function parseBytes(
* namespaceUri: HTML_NAMESPACE_URI,
* localName: "tr"
* });
- * console.log(fragment.children.length);
+ * console.log(fragment.tree.children.length);
* ```
*/
export function parseFragment(
html: string,
contextInput: HtmlFragmentContextInput,
options: ParseFragmentOptions = {}
-): FragmentTree {
+): ParsedFragment {
const startedAt = performance.now();
const normalized = normalizeParseFragmentOptions(options);
requireString(html, "input");
@@ -989,7 +1053,7 @@ export function parseFragment(
encodingPrescanBytes: null,
encodingPrescanLimitBytes: null
});
- return Object.freeze({
+ const tree: FragmentTree = Object.freeze({
id: fragmentId,
kind: "fragment",
context,
@@ -1000,6 +1064,22 @@ export function parseFragment(
errors,
...(traceResult === undefined ? {} : { trace: traceResult })
});
+ registerParsedFragmentTree(tree);
+ return Object.freeze({
+ tree,
+ metadata: Object.freeze({
+ inputKind: "text",
+ transportByteLength: null,
+ encoding: Object.freeze({ name: null, source: "already-decoded" }),
+ resourceUsage: fragmentResourceUsage(
+ inputBytes,
+ html.length,
+ normalized.budgets,
+ result,
+ trace
+ )
+ })
+ });
}
function publicToken(token: HtmlToken, operation: OperationContext): Token {
@@ -1039,16 +1119,16 @@ function publicToken(token: HtmlToken, operation: OperationContext): Token {
export async function tokenizeByteStreamEager(
stream: ReadableStream,
options: TokenizeByteStreamEagerOptions = {}
-): Promise {
+): Promise {
const startedAt = performance.now();
const normalized = normalizeTokenizeByteStreamEagerOptions(options);
requireReadableByteStream(stream, "input");
const operation = createOperationContext(normalized.budgets?.maxTimeMs, normalized.signal, startedAt);
- const decoded = await decodeByteStream(stream, {
- ...normalized,
- retainText: true
- }, operation);
+ operation.checkpoint();
const limits: EngineResourceLimits = Object.freeze({
+ ...(normalized.budgets?.maxSteps === undefined
+ ? {}
+ : { maxSteps: normalized.budgets.maxSteps }),
...(normalized.budgets?.maxParseErrors === undefined
? {}
: { maxParseErrors: normalized.budgets.maxParseErrors }),
@@ -1065,7 +1145,8 @@ export async function tokenizeByteStreamEager(
const resources = createEngineResourceGuard({
limits,
...(normalized.signal === undefined ? {} : { signal: normalized.signal }),
- startedAt
+ startedAt,
+ trackSteps: normalized.budgets?.maxSteps !== undefined
});
const tokens: Token[] = [];
const tokenizer = new HtmlTokenizer(resources, {
@@ -1075,14 +1156,37 @@ export async function tokenizeByteStreamEager(
return token.kind !== "start-tag" || token.selfClosing;
}
}, { reuseInputCharacters: true });
+ let decoded;
try {
- tokenizer.write(requireInternalValue(decoded.text, "PUBLIC_TOKENIZER_RETAINED_TEXT_MISSING"));
+ decoded = await decodeByteStream(stream, {
+ ...normalized,
+ retainText: false,
+ onDecodedChunk(chunk): void { tokenizer.write(chunk); }
+ }, operation);
tokenizer.close();
} catch (error) {
return mapEngineFailure(error);
}
operation.checkpoint();
- return Object.freeze(tokens);
+ const usage = resources.snapshot();
+ return Object.freeze({
+ tokens: Object.freeze(tokens),
+ metadata: Object.freeze({
+ inputKind: "stream",
+ transportByteLength: decoded.totalBytes,
+ encoding: Object.freeze({ name: decoded.sniff.encoding, source: decoded.sniff.source }),
+ resourceUsage: Object.freeze({
+ inputBytes: decoded.totalBytes,
+ decodedUtf8Bytes: decoded.decodedUtf8Bytes,
+ decodedCodeUnits: decoded.decodedCodeUnits,
+ steps: normalized.budgets?.maxSteps === undefined ? null : usage.steps,
+ parseErrors: usage.parseErrors,
+ attributes: usage.attributes,
+ attributeUtf8Bytes: usage.attributeUtf8Bytes,
+ encodingPrescanBytes: decoded.encodingPrescanBytes
+ })
+ })
+ });
}
/**
diff --git a/src/public/patching.ts b/src/public/patching.ts
index 4ca3508..57a16ed 100644
--- a/src/public/patching.ts
+++ b/src/public/patching.ts
@@ -15,7 +15,7 @@ import {
parsedDocumentRegistration,
patchPlanBelongsTo,
registerPatchPlan
-} from "./parsed-document-registry.ts";
+} from "./parsed-output-registry.ts";
import type {
Edit,
diff --git a/src/public/querying.ts b/src/public/querying.ts
index d755a29..0cc68ad 100644
--- a/src/public/querying.ts
+++ b/src/public/querying.ts
@@ -1,9 +1,11 @@
+import { HtmlConfigurationError } from "./errors.ts";
import { requireString } from "./html-input.ts";
import {
HTML_NAMESPACE_URI,
asciiLowercase,
isHtmlElement,
- iterateNodes
+ iterateNodes,
+ requireElementNode
} from "./model.ts";
import {
createOperationContext,
@@ -21,6 +23,42 @@ import type {
OperationOptions
} from "./types.ts";
+function invalid(option: string, expected: string): never {
+ throw new HtmlConfigurationError(option, "INVALID_VALUE", expected);
+}
+
+function requireTree(
+ value: unknown,
+ option: string
+): asserts value is DocumentTree | FragmentTree {
+ if (typeof value !== "object" || value === null || Array.isArray(value)) {
+ invalid(option, "must be a document or fragment tree");
+ }
+ const source = value as Readonly>;
+ if ((source["kind"] !== "document" && source["kind"] !== "fragment") ||
+ !Array.isArray(source["children"])) {
+ invalid(option, "must be a document or fragment tree");
+ }
+}
+
+function requireCallback(value: unknown, option: string): asserts value is (...args: never[]) => unknown {
+ if (typeof value !== "function") invalid(option, "must be a function");
+}
+
+function requireNodeId(value: unknown, option: string): asserts value is NodeId {
+ if (!Number.isSafeInteger(value) || (value as number) < 1) {
+ invalid(option, "must be a positive safe integer");
+ }
+}
+
+function requireNullableString(value: unknown, option: string): asserts value is string | null {
+ if (value !== null) requireString(value, option);
+}
+
+function requireOptionalString(value: unknown, option: string): asserts value is string | undefined {
+ if (value !== undefined) requireString(value, option);
+}
+
/** Visits every node in depth-first tree order. */
export function walk(
tree: DocumentTree | FragmentTree,
@@ -29,6 +67,8 @@ export function walk(
): void {
const startedAt = performance.now();
const normalizedOptions = normalizeOperationOptions(options);
+ requireTree(tree, "tree");
+ requireCallback(visitor, "visitor");
const operation = createOperationContext(
normalizedOptions.maxTimeMs,
normalizedOptions.signal,
@@ -49,6 +89,8 @@ export function walkElements(
): void {
const startedAt = performance.now();
const normalizedOptions = normalizeOperationOptions(options);
+ requireTree(tree, "tree");
+ requireCallback(visitor, "visitor");
const operation = createOperationContext(
normalizedOptions.maxTimeMs,
normalizedOptions.signal,
@@ -71,6 +113,8 @@ export function findById(
): HtmlNode | null {
const startedAt = performance.now();
const normalizedOptions = normalizeOperationOptions(options);
+ requireTree(tree, "tree");
+ requireNodeId(id, "id");
const operation = createOperationContext(
normalizedOptions.maxTimeMs,
normalizedOptions.signal,
@@ -91,11 +135,13 @@ export function getAttributeValue(
node: Extract,
name: string
): string | undefined {
- if (node.namespaceUri !== HTML_NAMESPACE_URI) {
+ const element = requireElementNode(node, "node");
+ requireString(name, "name");
+ if (element.namespaceUri !== HTML_NAMESPACE_URI) {
return undefined;
}
const target = asciiLowercase(name);
- for (const attribute of node.attributes) {
+ for (const attribute of element.attributes) {
if (attribute.namespaceUri === null && asciiLowercase(attribute.localName) === target) {
return attribute.value;
}
@@ -117,7 +163,10 @@ export function getAttributeValueNS(
namespaceUri: string | null,
localName: string
): string | undefined {
- return node.attributes.find((attribute) =>
+ const element = requireElementNode(node, "node");
+ requireNullableString(namespaceUri, "namespaceUri");
+ requireString(localName, "localName");
+ return element.attributes.find((attribute) =>
attribute.namespaceUri === namespaceUri && attribute.localName === localName
)?.value;
}
@@ -151,6 +200,8 @@ export function findAllByTagName(
options: OperationOptions = {}
): IterableIterator> {
const startedAt = performance.now();
+ requireTree(tree, "tree");
+ requireString(tagName, "tagName");
const normalizedOptions = normalizeOperationOptions(options);
const operation = createOperationContext(
normalizedOptions.maxTimeMs,
@@ -186,6 +237,7 @@ export function findAllByTagNameNS(
options: OperationOptions = {}
): IterableIterator> {
const startedAt = performance.now();
+ requireTree(tree, "tree");
requireString(namespaceUri, "namespaceUri");
requireString(localName, "localName");
const normalizedOptions = normalizeOperationOptions(options);
@@ -231,6 +283,9 @@ export function findAllByAttr(
options: OperationOptions = {}
): IterableIterator> {
const startedAt = performance.now();
+ requireTree(tree, "tree");
+ requireString(name, "name");
+ requireOptionalString(value, "value");
const normalizedOptions = normalizeOperationOptions(options);
const operation = createOperationContext(
normalizedOptions.maxTimeMs,
@@ -272,10 +327,10 @@ export function findAllByAttrNS(
options: OperationOptions = {}
): IterableIterator> {
const startedAt = performance.now();
- if (namespaceUri !== null) {
- requireString(namespaceUri, "namespaceUri");
- }
+ requireTree(tree, "tree");
+ requireNullableString(namespaceUri, "namespaceUri");
requireString(localName, "localName");
+ requireOptionalString(value, "value");
const normalizedOptions = normalizeOperationOptions(options);
const operation = createOperationContext(
normalizedOptions.maxTimeMs,
diff --git a/src/public/serialization.ts b/src/public/serialization.ts
index ed35c55..500be11 100644
--- a/src/public/serialization.ts
+++ b/src/public/serialization.ts
@@ -11,6 +11,7 @@ import {
createOperationContext,
normalizeSerializeOptions
} from "./operation.ts";
+import { validateSerializableInput } from "./tree-validation.ts";
import type { OperationContext } from "./operation.ts";
import type {
@@ -19,6 +20,7 @@ import type {
FragmentTree,
HtmlNode,
HtmlScriptingMode,
+ SerializableNode,
SerializeOptions
} from "./types.ts";
@@ -128,7 +130,7 @@ export function serializeNodes(
/** Serializes a parsed tree or one complete node representation as HTML. */
export function serialize(
- tree: DocumentTree | FragmentTree | HtmlNode,
+ tree: DocumentTree | FragmentTree | SerializableNode,
options: SerializeOptions = {}
): string {
const startedAt = performance.now();
@@ -139,6 +141,7 @@ export function serialize(
startedAt
);
operation.checkpoint();
+ validateSerializableInput(tree, operation);
const scriptingMode = normalizedOptions.scriptingMode ??
(tree.kind === "document" || tree.kind === "fragment" ? tree.scriptingMode : "inert");
if (tree.kind === "document" || tree.kind === "fragment") {
diff --git a/src/public/text-extraction.ts b/src/public/text-extraction.ts
index 898a053..5693f5b 100644
--- a/src/public/text-extraction.ts
+++ b/src/public/text-extraction.ts
@@ -6,6 +6,7 @@ import {
import { utf8ByteLength } from "./html-input.ts";
import {
HTML_NAMESPACE_URI,
+ OwnedNodeTracker,
asciiLowercase,
isHtmlElement
} from "./model.ts";
@@ -795,6 +796,7 @@ function* iterateVisibleExtractionChunks(
}
type Action = VisitAction | AppendAction;
const stack: Action[] = [];
+ const ownership = new OwnedNodeTracker();
const pushVisits = (
nodes: readonly HtmlNode[],
preserveWhitespace: boolean,
@@ -846,6 +848,7 @@ function* iterateVisibleExtractionChunks(
}
const { node, preserveWhitespace, sourceOverride } = action;
+ ownership.observe(node);
if (node.kind === "text") {
pushAppend(
node,
@@ -1016,6 +1019,7 @@ function* iterateRawExtractionChunks(
? nodeOrTree.children
: [nodeOrTree];
const stack: HtmlNode[] = [];
+ const ownership = new OwnedNodeTracker();
for (let index = roots.length - 1; index >= 0; index -= 1) {
const root = roots[index];
if (root !== undefined) {
@@ -1028,6 +1032,7 @@ function* iterateRawExtractionChunks(
if (node === undefined) {
continue;
}
+ ownership.observe(node);
if (node.kind === "text") {
if (node.value.length > 0) {
yield {
@@ -1201,7 +1206,8 @@ function createTextOperations(
});
}
-const textOperations = createTextOperations(parseFragment);
+const textOperations = createTextOperations((html, context, options) =>
+ parseFragment(html, context, options).tree);
/**
* Iterates bounded policy tokens and returns the final result when fully drained.
diff --git a/src/public/tree-validation.ts b/src/public/tree-validation.ts
new file mode 100644
index 0000000..8f7229f
--- /dev/null
+++ b/src/public/tree-validation.ts
@@ -0,0 +1,478 @@
+import {
+ isHtmlAttributeName,
+ isHtmlElementName,
+ isXmlLocalName
+} from "../internal/foundation/name-validation.ts";
+
+import { HtmlConfigurationError } from "./errors.ts";
+import {
+ HTML_NAMESPACE_URI,
+ MATHML_NAMESPACE_URI,
+ SVG_NAMESPACE_URI,
+ XLINK_NAMESPACE_URI,
+ XML_NAMESPACE_URI,
+ XMLNS_NAMESPACE_URI,
+ asciiLowercase
+} from "./model.ts";
+import { isParserOwnedTree } from "./parsed-output-registry.ts";
+
+import type { OperationContext } from "./operation.ts";
+import type {
+ Attribute,
+ DocumentTree,
+ FragmentTree,
+ SerializableNode,
+ Span
+} from "./types.ts";
+
+type UnknownRecord = Readonly>;
+type ChildPlacement = "document" | "fragment" | "element" | "template" | "standalone";
+
+interface PendingNode {
+ readonly value: unknown;
+ readonly path: string;
+ readonly placement: ChildPlacement;
+}
+
+function invalid(path: string, expected: string): never {
+ throw new HtmlConfigurationError(path, "INVALID_VALUE", expected);
+}
+
+function record(value: unknown, path: string): UnknownRecord {
+ if (typeof value !== "object" || value === null || Array.isArray(value)) {
+ invalid(path, "must be a node record");
+ }
+ return value as UnknownRecord;
+}
+
+function read(value: UnknownRecord, key: PropertyKey, path: string): unknown {
+ try {
+ return value[key];
+ } catch {
+ invalid(path, "must be a readable data property");
+ }
+}
+
+function string(value: unknown, path: string): string {
+ if (typeof value !== "string") invalid(path, "must be a string");
+ return value;
+}
+
+function optionalString(value: unknown, path: string): string | undefined {
+ if (value !== undefined && typeof value !== "string") {
+ invalid(path, "must be a string when provided");
+ }
+ return value;
+}
+
+function array(value: unknown, path: string): readonly unknown[] {
+ if (!Array.isArray(value)) invalid(path, "must be a dense array");
+ for (let index = 0; index < value.length; index += 1) {
+ if (!Object.hasOwn(value, index)) invalid(`${path}[${String(index)}]`, "must be present");
+ }
+ return value;
+}
+
+function nodeId(value: unknown, path: string, ids: Set): void {
+ if (!Number.isSafeInteger(value) || (value as number) < 1) {
+ invalid(path, "must be a positive safe integer");
+ }
+ const id = value as number;
+ if (ids.has(id)) invalid(path, "must be unique within the serialized tree");
+ ids.add(id);
+}
+
+function span(value: unknown, path: string): Span | undefined {
+ if (value === undefined) return undefined;
+ const source = record(value, path);
+ const start = read(source, "start", `${path}.start`);
+ const end = read(source, "end", `${path}.end`);
+ if (!Number.isSafeInteger(start) || (start as number) < 0) {
+ invalid(`${path}.start`, "must be a non-negative safe integer");
+ }
+ if (!Number.isSafeInteger(end) || (end as number) < (start as number)) {
+ invalid(`${path}.end`, "must be a safe integer no smaller than span.start");
+ }
+ return value as Span;
+}
+
+function nodeSourceFields(value: UnknownRecord, path: string): void {
+ const sourceSpan = span(read(value, "span", `${path}.span`), `${path}.span`);
+ const provenance = read(value, "spanProvenance", `${path}.spanProvenance`);
+ if (provenance !== undefined && provenance !== "input" && provenance !== "inferred") {
+ invalid(`${path}.spanProvenance`, 'must be "input" or "inferred" when provided');
+ }
+ if (sourceSpan !== undefined && provenance !== "input") {
+ invalid(`${path}.spanProvenance`, 'must be "input" when span is present');
+ }
+ if (provenance === "input" && sourceSpan === undefined) {
+ invalid(`${path}.span`, 'must be present when spanProvenance is "input"');
+ }
+ if (provenance === "inferred" && sourceSpan !== undefined) {
+ invalid(`${path}.span`, 'must be absent when spanProvenance is "inferred"');
+ }
+}
+
+function expectedAttributePrefix(
+ namespaceUri: string,
+ localName: string
+): string | undefined | null {
+ if (namespaceUri === XLINK_NAMESPACE_URI) return "xlink";
+ if (namespaceUri === XML_NAMESPACE_URI) return "xml";
+ if (namespaceUri === XMLNS_NAMESPACE_URI) return localName === "xmlns" ? undefined : "xmlns";
+ return null;
+}
+
+function validateAttribute(
+ value: unknown,
+ path: string,
+ htmlElement: boolean
+): { readonly expandedName: string; readonly serializedName: string } {
+ const source = record(value, path);
+ const namespaceValue = read(source, "namespaceUri", `${path}.namespaceUri`);
+ if (namespaceValue !== null && (typeof namespaceValue !== "string" || namespaceValue.length === 0)) {
+ invalid(`${path}.namespaceUri`, "must be null or a non-empty namespace URI");
+ }
+ const namespaceUri = namespaceValue;
+ const localName = string(read(source, "localName", `${path}.localName`), `${path}.localName`);
+ const prefix = optionalString(read(source, "prefix", `${path}.prefix`), `${path}.prefix`);
+ string(read(source, "value", `${path}.value`), `${path}.value`);
+ span(read(source, "span", `${path}.span`), `${path}.span`);
+
+ if (namespaceUri === null) {
+ if (prefix !== undefined) invalid(`${path}.prefix`, "must be absent for an unnamespaced attribute");
+ if (!isHtmlAttributeName(localName)) {
+ invalid(`${path}.localName`, "must be safely representable as one HTML attribute name");
+ }
+ if (htmlElement && localName !== asciiLowercase(localName)) {
+ invalid(`${path}.localName`, "must use normalized ASCII lowercase on an HTML element");
+ }
+ } else {
+ if (!isXmlLocalName(localName)) {
+ invalid(`${path}.localName`, "must be an XML local name for a namespaced attribute");
+ }
+ if (prefix !== undefined && !isXmlLocalName(prefix)) {
+ invalid(`${path}.prefix`, "must be an XML local name when provided");
+ }
+ const expectedPrefix = expectedAttributePrefix(namespaceUri, localName);
+ if (expectedPrefix !== null && prefix !== undefined && prefix !== expectedPrefix) {
+ invalid(`${path}.prefix`, "must match the reserved attribute namespace");
+ }
+ if (expectedPrefix === null && (prefix === "xml" || prefix === "xmlns")) {
+ invalid(`${path}.prefix`, "must not use a reserved prefix for another namespace");
+ }
+ }
+
+ const serializedName = namespaceUri === null
+ ? localName
+ : namespaceUri === XML_NAMESPACE_URI
+ ? `xml:${localName}`
+ : namespaceUri === XMLNS_NAMESPACE_URI
+ ? localName === "xmlns" ? "xmlns" : `xmlns:${localName}`
+ : namespaceUri === XLINK_NAMESPACE_URI
+ ? `xlink:${localName}`
+ : prefix === undefined ? localName : `${prefix}:${localName}`;
+ return {
+ expandedName: `${namespaceUri ?? ""}\u0000${localName}`,
+ serializedName: asciiLowercase(serializedName)
+ };
+}
+
+function validateAttributes(
+ value: unknown,
+ path: string,
+ htmlElement: boolean
+): readonly Attribute[] {
+ const attributes = array(value, path);
+ const expandedNames = new Set();
+ const serializedNames = new Set();
+ for (let index = 0; index < attributes.length; index += 1) {
+ const attributePath = `${path}[${String(index)}]`;
+ const identity = validateAttribute(attributes[index], attributePath, htmlElement);
+ if (expandedNames.has(identity.expandedName)) {
+ invalid(attributePath, "must have a unique namespace and local name");
+ }
+ if (serializedNames.has(identity.serializedName)) {
+ invalid(attributePath, "must have a unique serialized HTML name");
+ }
+ expandedNames.add(identity.expandedName);
+ serializedNames.add(identity.serializedName);
+ }
+ return value as readonly Attribute[];
+}
+
+function validateElementIdentity(value: UnknownRecord, path: string): {
+ readonly namespaceUri: string;
+ readonly localName: string;
+} {
+ const namespaceUri = string(
+ read(value, "namespaceUri", `${path}.namespaceUri`),
+ `${path}.namespaceUri`
+ );
+ if (namespaceUri.length === 0) invalid(`${path}.namespaceUri`, "must be a non-empty namespace URI");
+ const localName = string(read(value, "localName", `${path}.localName`), `${path}.localName`);
+ const prefix = optionalString(read(value, "prefix", `${path}.prefix`), `${path}.prefix`);
+ const parserNamespace = namespaceUri === HTML_NAMESPACE_URI ||
+ namespaceUri === SVG_NAMESPACE_URI || namespaceUri === MATHML_NAMESPACE_URI;
+ if (parserNamespace) {
+ if (prefix !== undefined) {
+ invalid(`${path}.prefix`, "must be absent for HTML, SVG, and MathML parser elements");
+ }
+ if (!isHtmlElementName(localName)) {
+ invalid(`${path}.localName`, "must be safely representable as one HTML element name");
+ }
+ if (namespaceUri === HTML_NAMESPACE_URI && localName !== asciiLowercase(localName)) {
+ invalid(`${path}.localName`, "must use normalized ASCII lowercase in the HTML namespace");
+ }
+ } else {
+ if (!isXmlLocalName(localName)) {
+ invalid(`${path}.localName`, "must be an XML local name in a custom namespace");
+ }
+ if (prefix !== undefined && !isXmlLocalName(prefix)) {
+ invalid(`${path}.prefix`, "must be an XML local name when provided");
+ }
+ if (prefix === "xml" && namespaceUri !== XML_NAMESPACE_URI) {
+ invalid(`${path}.prefix`, "must bind the xml prefix to the XML namespace");
+ }
+ if (prefix === "xmlns" || namespaceUri === XMLNS_NAMESPACE_URI) {
+ invalid(`${path}.namespaceUri`, "must not use the XMLNS namespace for an element");
+ }
+ }
+ return { namespaceUri, localName };
+}
+
+function validateParseErrors(value: unknown, path: string): void {
+ const errors = array(value, path);
+ for (let index = 0; index < errors.length; index += 1) {
+ const errorPath = `${path}[${String(index)}]`;
+ const source = record(errors[index], errorPath);
+ if (read(source, "code", `${errorPath}.code`) !== "PARSER_ERROR") {
+ invalid(`${errorPath}.code`, 'must be "PARSER_ERROR"');
+ }
+ string(read(source, "parseErrorId", `${errorPath}.parseErrorId`), `${errorPath}.parseErrorId`);
+ string(read(source, "message", `${errorPath}.message`), `${errorPath}.message`);
+ span(read(source, "span", `${errorPath}.span`), `${errorPath}.span`);
+ }
+}
+
+function validateFragmentContextAttributes(
+ value: unknown,
+ path: string,
+ htmlContext: boolean
+): void {
+ const attributes = array(value, path);
+ const expandedNames = new Set();
+ for (let index = 0; index < attributes.length; index += 1) {
+ const attributePath = `${path}[${String(index)}]`;
+ const source = record(attributes[index], attributePath);
+ const namespaceUri = read(source, "namespaceUri", `${attributePath}.namespaceUri`);
+ if (namespaceUri !== null && namespaceUri !== XLINK_NAMESPACE_URI &&
+ namespaceUri !== XML_NAMESPACE_URI && namespaceUri !== XMLNS_NAMESPACE_URI) {
+ invalid(
+ `${attributePath}.namespaceUri`,
+ "must be null or the XLink, XML, or XMLNS namespace URI"
+ );
+ }
+ const localName = string(
+ read(source, "localName", `${attributePath}.localName`),
+ `${attributePath}.localName`
+ );
+ if (!isXmlLocalName(localName) ||
+ (htmlContext && namespaceUri === null && localName !== asciiLowercase(localName))) {
+ invalid(`${attributePath}.localName`, "must be a normalized XML local name");
+ }
+ string(read(source, "value", `${attributePath}.value`), `${attributePath}.value`);
+ const expandedName = `${namespaceUri ?? ""}\u0000${localName}`;
+ if (expandedNames.has(expandedName)) {
+ invalid(attributePath, "must have a unique namespace and local name");
+ }
+ expandedNames.add(expandedName);
+ }
+}
+
+function pushChildren(
+ children: readonly unknown[],
+ path: string,
+ placement: ChildPlacement,
+ pending: PendingNode[]
+): void {
+ for (let index = children.length - 1; index >= 0; index -= 1) {
+ pending.push({
+ value: children[index],
+ path: `${path}[${String(index)}]`,
+ placement
+ });
+ }
+}
+
+function validateRoot(
+ source: UnknownRecord,
+ kind: "document" | "fragment",
+ ids: Set,
+ pending: PendingNode[]
+): void {
+ nodeId(read(source, "id", "tree.id"), "tree.id", ids);
+ const scriptingMode = read(source, "scriptingMode", "tree.scriptingMode");
+ if (scriptingMode !== "inert" && scriptingMode !== "disabled") {
+ invalid("tree.scriptingMode", 'must be "inert" or "disabled"');
+ }
+ validateParseErrors(read(source, "errors", "tree.errors"), "tree.errors");
+ const children = array(read(source, "children", "tree.children"), "tree.children");
+ if (kind === "fragment") {
+ const documentMode = read(source, "documentMode", "tree.documentMode");
+ if (documentMode !== "no-quirks" && documentMode !== "limited-quirks" && documentMode !== "quirks") {
+ invalid("tree.documentMode", "must be a supported document mode");
+ }
+ if (typeof read(source, "hasFormInContextChain", "tree.hasFormInContextChain") !== "boolean") {
+ invalid("tree.hasFormInContextChain", "must be a boolean");
+ }
+ const context = record(read(source, "context", "tree.context"), "tree.context");
+ const namespaceUri = read(context, "namespaceUri", "tree.context.namespaceUri");
+ if (namespaceUri !== HTML_NAMESPACE_URI && namespaceUri !== SVG_NAMESPACE_URI &&
+ namespaceUri !== MATHML_NAMESPACE_URI) {
+ invalid("tree.context.namespaceUri", "must be the HTML, SVG, or MathML namespace URI");
+ }
+ const localName = string(
+ read(context, "localName", "tree.context.localName"),
+ "tree.context.localName"
+ );
+ if (!isXmlLocalName(localName) ||
+ (namespaceUri === HTML_NAMESPACE_URI && localName !== asciiLowercase(localName))) {
+ invalid("tree.context.localName", "must be a normalized XML local name");
+ }
+ validateFragmentContextAttributes(
+ read(context, "attributes", "tree.context.attributes"),
+ "tree.context.attributes",
+ namespaceUri === HTML_NAMESPACE_URI
+ );
+ }
+ pushChildren(children, "tree.children", kind, pending);
+}
+
+/** Validates every serializer-visible invariant before any output is emitted. */
+export function validateSerializableInput(
+ value: unknown,
+ operation: OperationContext
+): asserts value is DocumentTree | FragmentTree | SerializableNode {
+ if (isParserOwnedTree(value)) return;
+ const root = record(value, "tree");
+ const rootKind = read(root, "kind", "tree.kind");
+ const ids = new Set();
+ const pending: PendingNode[] = [];
+ if (rootKind === "document" || rootKind === "fragment") {
+ validateRoot(root, rootKind, ids, pending);
+ } else {
+ pending.push({ value, path: "tree", placement: "standalone" });
+ }
+
+ const seen = new WeakMap();
+ const documentKinds = { doctypes: 0, elements: 0, elementSeen: false };
+ while (pending.length > 0) {
+ operation.checkpoint();
+ const entry = pending.pop();
+ if (entry === undefined) continue;
+ const node = record(entry.value, entry.path);
+ const priorPath = seen.get(node);
+ if (priorPath !== undefined) {
+ invalid(entry.path, `must be uniquely owned; first encountered at ${priorPath}`);
+ }
+ seen.set(node, entry.path);
+ nodeId(read(node, "id", `${entry.path}.id`), `${entry.path}.id`, ids);
+ const kind = read(node, "kind", `${entry.path}.kind`);
+ if (kind === "templateContent") {
+ if (entry.placement !== "template") {
+ invalid(entry.path, "template content must be owned by exactly one HTML template element");
+ }
+ const provenance = read(node, "spanProvenance", `${entry.path}.spanProvenance`);
+ if (provenance !== undefined && provenance !== "inferred") {
+ invalid(`${entry.path}.spanProvenance`, 'must be "inferred" when provided');
+ }
+ const children = array(read(node, "children", `${entry.path}.children`), `${entry.path}.children`);
+ pushChildren(children, `${entry.path}.children`, "element", pending);
+ continue;
+ }
+ if (kind !== "element" && kind !== "text" && kind !== "comment" &&
+ kind !== "processingInstruction" && kind !== "doctype") {
+ invalid(`${entry.path}.kind`, "must be a supported serializable node kind");
+ }
+ if (entry.placement !== "standalone") {
+ if (kind === "doctype" && entry.placement !== "document") {
+ invalid(entry.path, "a doctype can only be a direct document child or standalone node");
+ }
+ if (kind === "text" && entry.placement === "document") {
+ invalid(entry.path, "a text node cannot be a direct document child");
+ }
+ }
+ if (entry.placement === "document") {
+ if (kind === "doctype") {
+ documentKinds.doctypes += 1;
+ if (documentKinds.doctypes > 1 || documentKinds.elementSeen) {
+ invalid(entry.path, "a document has at most one doctype before its element");
+ }
+ } else if (kind === "element") {
+ documentKinds.elements += 1;
+ documentKinds.elementSeen = true;
+ if (documentKinds.elements > 1) invalid(entry.path, "a document has at most one element child");
+ }
+ }
+ nodeSourceFields(node, entry.path);
+ if (kind === "text" || kind === "comment") {
+ string(read(node, "value", `${entry.path}.value`), `${entry.path}.value`);
+ continue;
+ }
+ if (kind === "processingInstruction") {
+ string(read(node, "target", `${entry.path}.target`), `${entry.path}.target`);
+ string(read(node, "data", `${entry.path}.data`), `${entry.path}.data`);
+ continue;
+ }
+ if (kind === "doctype") {
+ const name = string(read(node, "name", `${entry.path}.name`), `${entry.path}.name`);
+ if (!isHtmlElementName(name)) {
+ invalid(`${entry.path}.name`, "must be safely representable as one doctype name");
+ }
+ const externalId = record(
+ read(node, "externalId", `${entry.path}.externalId`),
+ `${entry.path}.externalId`
+ );
+ const externalKind = read(externalId, "kind", `${entry.path}.externalId.kind`);
+ if (externalKind === "public") {
+ string(read(externalId, "publicId", `${entry.path}.externalId.publicId`), `${entry.path}.externalId.publicId`);
+ const systemId = read(externalId, "systemId", `${entry.path}.externalId.systemId`);
+ if (systemId !== null) string(systemId, `${entry.path}.externalId.systemId`);
+ } else if (externalKind === "system") {
+ string(read(externalId, "systemId", `${entry.path}.externalId.systemId`), `${entry.path}.externalId.systemId`);
+ } else if (externalKind !== "none") {
+ invalid(`${entry.path}.externalId.kind`, 'must be "none", "public", or "system"');
+ }
+ continue;
+ }
+
+ const identity = validateElementIdentity(node, entry.path);
+ validateAttributes(
+ read(node, "attributes", `${entry.path}.attributes`),
+ `${entry.path}.attributes`,
+ identity.namespaceUri === HTML_NAMESPACE_URI
+ );
+ const children = array(read(node, "children", `${entry.path}.children`), `${entry.path}.children`);
+ const templateContent = read(node, "templateContent", `${entry.path}.templateContent`);
+ const isTemplate = identity.namespaceUri === HTML_NAMESPACE_URI && identity.localName === "template";
+ if (isTemplate) {
+ if (children.length !== 0) {
+ invalid(`${entry.path}.children`, "must be empty because an HTML template owns templateContent");
+ }
+ if (templateContent === undefined) {
+ invalid(`${entry.path}.templateContent`, "must be present on an HTML template element");
+ }
+ pending.push({
+ value: templateContent,
+ path: `${entry.path}.templateContent`,
+ placement: "template"
+ });
+ } else {
+ if (templateContent !== undefined) {
+ invalid(`${entry.path}.templateContent`, "is only valid on an HTML template element");
+ }
+ pushChildren(children, `${entry.path}.children`, "element", pending);
+ }
+ }
+}
diff --git a/src/public/types.ts b/src/public/types.ts
index 83edac8..ed4c532 100644
--- a/src/public/types.ts
+++ b/src/public/types.ts
@@ -152,15 +152,15 @@ export interface ParseFragmentOptions extends Omit {
- /** Parse and stream-prescan resource limits. */
+ /** Parse and optional stream meta-prescan resource limits. */
readonly budgets?: ParseStreamBudgetOptions;
}
@@ -170,6 +170,7 @@ export type TokenizeByteStreamEagerBudgetOptions = Pick<
| "maxInputBytes"
| "maxEncodingPrescanBytes"
| "maxDecodedUtf8Bytes"
+ | "maxSteps"
| "maxParseErrors"
| "maxAttributesPerElement"
| "maxAttributeBytes"
@@ -376,7 +377,7 @@ export interface TraceBudgetEvent {
readonly status: "ok" | "exceeded";
}
-/** Byte-stream consumption and encoding-prescan retention metrics. */
+/** Byte-stream consumption and optional meta-prescan retention metrics. */
export interface TraceStreamEvent {
/** One-based event order within this parse. */
readonly seq: number;
@@ -384,9 +385,9 @@ export interface TraceStreamEvent {
readonly kind: "stream";
/** Total transport bytes read through EOF. */
readonly bytesRead: number;
- /** Maximum transport bytes retained simultaneously for encoding prescan. */
+ /** Maximum transport bytes retained simultaneously for optional meta prescan. */
readonly encodingPrescanBytes: number;
- /** Effective prescan retention cap after the implementation maximum is applied. */
+ /** Effective optional meta-prescan cap after the implementation maximum is applied. */
readonly encodingPrescanLimitBytes: number;
}
@@ -429,9 +430,9 @@ export interface TraceSummary {
readonly decodedUtf8Bytes: number;
/** Total stream transport bytes, or null for non-stream input. */
readonly bytesRead: number | null;
- /** Stream encoding-prescan high-water bytes, or null for non-stream input. */
+ /** Optional stream meta-prescan high-water bytes, or null for non-stream input. */
readonly encodingPrescanBytes: number | null;
- /** Effective stream encoding-prescan cap, or null for non-stream input. */
+ /** Effective optional stream meta-prescan cap, or null for non-stream input. */
readonly encodingPrescanLimitBytes: number | null;
/** Total observed event count, whether or not events were retained. */
readonly eventCount: number;
@@ -577,6 +578,9 @@ export type HtmlNode =
| ProcessingInstructionNode
| DoctypeNode;
+/** Complete node accepted as a standalone serialization input. */
+export type SerializableNode = Exclude;
+
/** Callback invoked for a node and its zero-based traversal depth. */
export type NodeVisitor = (node: HtmlNode, depth: number) => void;
/** Callback invoked for an element and its zero-based traversal depth. */
@@ -598,7 +602,7 @@ export interface DocumentTree {
readonly trace?: TraceResult;
}
-/** Successful full-document parser resource observations. */
+/** Successful document or fragment parser resource observations. */
export interface ParseResourceUsage {
/** UTF-8 bytes for text input; supplied transport bytes for byte/stream input. */
readonly inputBytes: number;
@@ -618,7 +622,7 @@ export interface ParseResourceUsage {
readonly attributes: number;
/** UTF-8 bytes in attempted attribute names and decoded values. */
readonly attributeUtf8Bytes: number;
- /** Stream encoding-prescan retained-byte high-water mark; zero otherwise. */
+ /** Optional stream meta-prescan retained-byte high-water mark; zero otherwise. */
readonly encodingPrescanBytes: number;
/** Observable trace events emitted; zero when tracing and observation are disabled. */
readonly traceEvents: number;
@@ -634,8 +638,8 @@ export interface ParseEncodingMetadata {
readonly source: "already-decoded" | "bom" | "transport" | "meta" | "default";
}
-/** Deterministic metadata for one full-document parse. */
-export interface ParsedDocumentMetadata {
+/** Deterministic metadata for one document or fragment parse. */
+export interface ParseMetadata {
/** Public input variant used for this parse. */
readonly inputKind: "text" | "bytes" | "stream";
/** Bytes supplied to a byte/stream API, or null for already-decoded text. */
@@ -653,7 +657,7 @@ export interface ParsedDocument {
/** Exact decoded input when `sourceRetention: "text"`; otherwise null. */
readonly sourceText: string | null;
/** Input, encoding, and resource evidence from this parse. */
- readonly metadata: ParsedDocumentMetadata;
+ readonly metadata: ParseMetadata;
}
/** Immutable root returned by fragment parsing. */
@@ -678,6 +682,54 @@ export interface FragmentTree {
readonly trace?: TraceResult;
}
+/** Canonical result returned by fragment parsing. */
+export interface ParsedFragment {
+ /** Parsed fragment tree. */
+ readonly tree: FragmentTree;
+ /** Input and successful resource evidence from this parse. */
+ readonly metadata: ParseMetadata;
+}
+
+/** Successful resource observations from eager byte-stream tokenization. */
+export interface TokenizationResourceUsage {
+ /** Transport bytes read through EOF. */
+ readonly inputBytes: number;
+ /** UTF-8 bytes produced by decoding. */
+ readonly decodedUtf8Bytes: number;
+ /** UTF-16 code units produced by decoding. */
+ readonly decodedCodeUnits: number;
+ /** Deterministic tokenizer checkpoints, or null when no step limit enabled counting. */
+ readonly steps: number | null;
+ /** Tokenizer diagnostics observed while producing tokens. */
+ readonly parseErrors: number;
+ /** Attempted start-tag attributes, including duplicates later discarded. */
+ readonly attributes: number;
+ /** UTF-8 bytes in attempted attribute names and decoded values. */
+ readonly attributeUtf8Bytes: number;
+ /** Optional meta-encoding prefix retained at its high-water mark. */
+ readonly encodingPrescanBytes: number;
+}
+
+/** Metadata for one eager byte-stream tokenization operation. */
+export interface TokenizeByteStreamEagerMetadata {
+ /** Public input variant used by this operation. */
+ readonly inputKind: "stream";
+ /** Transport bytes read through EOF. */
+ readonly transportByteLength: number;
+ /** Encoding decision from the owning decode pipeline. */
+ readonly encoding: ParseEncodingMetadata;
+ /** Successful deterministic resource observations. */
+ readonly resourceUsage: TokenizationResourceUsage;
+}
+
+/** Immutable eager byte-stream tokenization result. */
+export interface TokenizeByteStreamEagerResult {
+ /** Logical tokens in emission order, including EOF. */
+ readonly tokens: readonly Token[];
+ /** Decode and tokenizer resource evidence. */
+ readonly metadata: TokenizeByteStreamEagerMetadata;
+}
+
/** One structural outline entry derived from a heading or sectioning element. */
export interface OutlineEntry {
/** Identity of the source element. */
diff --git a/test/behavior/fragment-context.test.js b/test/behavior/fragment-context.test.js
index 73f6d7d..0b71bae 100644
--- a/test/behavior/fragment-context.test.js
+++ b/test/behavior/fragment-context.test.js
@@ -20,7 +20,7 @@ test("fragment context is normalized, retained, and deeply immutable", () => {
localName: "TeXtArEa",
attributes: [{ namespaceUri: XLINK_NAMESPACE_URI, localName: "href", value: "#x" }]
};
- const fragment = parseFragment("a&", input);
+ const { tree: fragment } = parseFragment("a&", input);
assert.deepEqual(fragment.context, {
namespaceUri: HTML_NAMESPACE_URI,
@@ -39,7 +39,7 @@ test("fragment context is normalized, retained, and deeply immutable", () => {
});
test("foreign fragment contexts and semantic attributes drive integration rules", () => {
- const svg = parseFragment("x
", {
+ const { tree: svg } = parseFragment("x
", {
namespaceUri: SVG_NAMESPACE_URI,
localName: "svg"
});
@@ -55,7 +55,7 @@ test("foreign fragment contexts and semantic attributes drive integration rules"
assert.equal(foreignObject.children[0]?.kind, "element");
assert.equal(foreignObject.children[0]?.namespaceUri, HTML_NAMESPACE_URI);
- const annotation = parseFragment(" x
", {
+ const { tree: annotation } = parseFragment("x
", {
namespaceUri: MATHML_NAMESPACE_URI,
localName: "annotation-xml",
attributes: [{ namespaceUri: null, localName: "encoding", value: "text/html" }]
@@ -66,14 +66,14 @@ test("foreign fragment contexts and semantic attributes drive integration rules"
test("fragment environment controls tree construction and is retained on the result", () => {
const context = htmlContext("div");
- const standards = parseFragment("x
y", context);
- const quirks = parseFragment("x
y", context, { documentMode: "quirks" });
+ const { tree: standards } = parseFragment("x
y", context);
+ const { tree: quirks } = parseFragment("x
y", context, { documentMode: "quirks" });
assert.notDeepEqual(quirks.children, standards.children);
assert.equal(standards.documentMode, "no-quirks");
assert.equal(quirks.documentMode, "quirks");
- const withoutForm = parseFragment("