diff --git a/.gitignore b/.gitignore
index b88b9b1..5893d04 100644
--- a/.gitignore
+++ b/.gitignore
@@ -1,6 +1,7 @@
node_modules/
dist/
release/
+.engine-build/
coverage/
playwright-report/
test-results/
@@ -9,4 +10,3 @@ test-results/
.env
.env.*
!.env.example
-
diff --git a/.prettierignore b/.prettierignore
new file mode 100644
index 0000000..c805541
--- /dev/null
+++ b/.prettierignore
@@ -0,0 +1 @@
+public/engines/pcre2/pcre2.mjs
diff --git a/CHANGELOG.md b/CHANGELOG.md
index a468414..1761bbb 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -1,5 +1,58 @@
# Changelog
+## Unreleased
+
+## 0.2.0 — 2026-07-27
+
+- Open Quick reference and Capabilities as populated, accessible in-viewport
+ dialogs with keyboard dismissal, focus handling and explicit empty states.
+- Add flavour-neutral version, flag, option and execution-worker registries
+ and expose only the ECMAScript and fully wired PCRE2 runtimes.
+- Validate flavour/version/options consistently across workbench requests,
+ projects and test suites, including legacy schema-v1 identities.
+- Add a production local-first Corpus / Apply workspace with bounded UTF-8
+ multi-file and pasted-text input, sequential killable worker jobs, progress,
+ cancellation, explicit whole-document/independent-line semantics, preserved
+ line separators, bounded capture extraction and replacement previews,
+ per-document exact/partial summaries, content-free JSON/CSV/NDJSON reports
+ and privacy-gated exact-output downloads and bounded ZIP export.
+- Keep corpus content and results ephemeral: project schema v1 stores only the
+ selected workspace mode and rejects injected corpus payloads.
+- Add a deterministic, offline PCRE2 10.47 WebAssembly runtime with pinned
+ source/toolchain identities, provenance, licences and checksums.
+- Add bounded PCRE2 compile, match and native-substitution ABI calls with
+ match/depth/heap caps, copied caller-owned records, cleanup on every exit,
+ UTF-8-to-UTF-16 range normalization and lone-surrogate rejection.
+- Run PCRE2 in a dedicated killable/recoverable worker, expose its real
+ flavour-specific flags and options, and add lexical syntax/replacement
+ coverage, official quick reference, conformance and browser tests.
+- Add an ABI-v3 PCRE2 automatic-callout operation with caller-owned event and
+ mark buffers, dual event/byte caps, native truncation stop, exact UTF-8 and
+ UTF-16 positions, a separate killable worker and a bounded viewer that keeps
+ reported fields distinct from derived movement labels.
+- Add ECMAScript-scoped advisory static risk diagnostics plus cancellable
+ bounded growth probes and cold/warm native-engine benchmarks with median/p95,
+ separate timing/outcome plots, runtime identity and strict
+ input/output/wall limits.
+- Add an ECMAScript/PCRE2 comparison orchestrator with exact per-side requests,
+ independent worker timeout outcomes, normalized-range match/capture
+ alignment, replacement differences and explicit non-comparable states.
+- Add a reviewed PCRE2 10.47 8-bit C17 generator with exact UTF-8 byte arrays,
+ native/host limits, error handling, deterministic source identities and
+ compile/execute golden gates against the named toolchain.
+- Add deterministic, cancellable failing-subject minimization for saved-test
+ failures, capture ranges, exact engine timeouts and ECMAScript/PCRE2
+ mismatches, with real-worker verification, explicit budgets and honest
+ transform-local minimality.
+- Add deterministic, seed-addressed ECMAScript positive/negative case
+ generation from the normalized AST, with bounded synthesis, actual-engine
+ verification, explicit unsupported coverage, cancellable workers and
+ provenance-preserving unit-test integration.
+- Add an idempotent grammar-backed ECMAScript literal/control formatter with an
+ exact preview, separate source/candidate syntax and engine workers, capture
+ shape checks, applicable unit-test replay, explicit inconclusive states and a
+ mandatory confirmation before apply.
+
## 0.1.0 — 2026-07-24
- Added the open-source ECMAScript 2025 syntax provider using regexpp 4.12.2.
diff --git a/LICENSES/Emscripten-MIT.txt b/LICENSES/Emscripten-MIT.txt
new file mode 100644
index 0000000..18ad35a
--- /dev/null
+++ b/LICENSES/Emscripten-MIT.txt
@@ -0,0 +1,19 @@
+Copyright (c) 2018 Emscripten authors (see AUTHORS in Emscripten)
+
+Permission is hereby granted, free of charge, to any person obtaining a copy
+of this software and associated documentation files (the "Software"), to deal
+in the Software without restriction, including without limitation the rights
+to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
+copies of the Software, and to permit persons to whom the Software is
+furnished to do so, subject to the following conditions:
+
+The above copyright notice and this permission notice shall be included in all
+copies or substantial portions of the Software.
+
+THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
+IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
+FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
+AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
+LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
+OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
+SOFTWARE.
diff --git a/LICENSES/PCRE2.txt b/LICENSES/PCRE2.txt
new file mode 100644
index 0000000..f6fba35
--- /dev/null
+++ b/LICENSES/PCRE2.txt
@@ -0,0 +1,104 @@
+PCRE2 Licence
+=============
+
+| SPDX-License-Identifier: | BSD-3-Clause WITH PCRE2-exception |
+|---------|-------|
+
+PCRE2 is a library of functions to support regular expressions whose syntax
+and semantics are as close as possible to those of the Perl 5 language.
+
+Releases 10.00 and above of PCRE2 are distributed under the terms of the "BSD"
+licence, as specified below, with one exemption for certain binary
+redistributions. The documentation for PCRE2, supplied in the "doc" directory,
+is distributed under the same terms as the software itself. The data in the
+testdata directory is not copyrighted and is in the public domain.
+
+The basic library functions are written in C and are freestanding. Also
+included in the distribution is a just-in-time compiler that can be used to
+optimize pattern matching. This is an optional feature that can be omitted when
+the library is built. The just-in-time compiler is separately licensed under the
+"2-clause BSD" licence.
+
+
+COPYRIGHT
+---------
+
+### The basic library functions
+
+ Written by: Philip Hazel
+ Email local part: Philip.Hazel
+ Email domain: gmail.com
+
+ Retired from University of Cambridge Computing Service,
+ Cambridge, England.
+
+ Copyright (c) 1997-2007 University of Cambridge
+ Copyright (c) 2007-2024 Philip Hazel
+ All rights reserved.
+
+### PCRE2 Just-In-Time compilation support
+
+ Written by: Zoltan Herczeg
+ Email local part: hzmester
+ Email domain: freemail.hu
+
+ Copyright (c) 2010-2024 Zoltan Herczeg
+ All rights reserved.
+
+### Stack-less Just-In-Time compiler
+
+ Written by: Zoltan Herczeg
+ Email local part: hzmester
+ Email domain: freemail.hu
+
+ Copyright (c) 2009-2024 Zoltan Herczeg
+ All rights reserved.
+
+The code in the `deps/sljit` directory has its own LICENSE file.
+
+### All other contributions
+
+Many other contributors have participated in the authorship of PCRE2. As PCRE2
+has never required a Contributor Licensing Agreement, or other copyright
+assignment agreement, all contributions have copyright retained by each
+original contributor or their employer.
+
+
+THE "BSD" LICENCE
+-----------------
+
+Redistribution and use in source and binary forms, with or without
+modification, are permitted provided that the following conditions are met:
+
+* Redistributions of source code must retain the above copyright notices,
+ this list of conditions and the following disclaimer.
+
+* Redistributions in binary form must reproduce the above copyright
+ notices, this list of conditions and the following disclaimer in the
+ documentation and/or other materials provided with the distribution.
+
+* Neither the name of the University of Cambridge nor the names of any
+ contributors may be used to endorse or promote products derived from this
+ software without specific prior written permission.
+
+THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
+AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
+IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
+ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE
+LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
+CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
+SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
+INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
+CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
+ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
+POSSIBILITY OF SUCH DAMAGE.
+
+
+EXEMPTION FOR BINARY LIBRARY-LIKE PACKAGES
+------------------------------------------
+
+The second condition in the BSD licence (covering binary redistributions) does
+not apply all the way down a chain of software. If binary package A includes
+PCRE2, it must respect the condition, but if package B is software that
+includes package A, the condition is not imposed on package B unless it uses
+PCRE2 independently.
diff --git a/README.md b/README.md
index 0274a00..bf20e5b 100644
--- a/README.md
+++ b/README.md
@@ -8,21 +8,45 @@ Patterns, replacement templates, test text and project files stay in the
browser. There is no account, backend, upload, remote execution, telemetry,
runtime CDN, remote font, or cloud-save service.
-## Current milestone
+## Current release — 0.2.0
-Version 0.1.0 is the first production flavour slice: ECMAScript.
+The v0.2.0 release contains two real production engine slices: ECMAScript and
+PCRE2 10.47.
- Syntax acceptance: `@eslint-community/regexpp` 4.12.2, ECMAScript 2025
grammar; application-owned structural explanations are partial and
feature-matrix tested.
- Execution engine: the current browser's native ECMAScript `RegExp`.
- Native and editor offsets: UTF-16 code units.
+- PCRE2 execution and substitution: pinned official PCRE2 10.47 compiled to a
+ self-hosted 8-bit WebAssembly pack with UTF/UCP always enabled.
+- PCRE2 native UTF-8 byte ranges are retained and normalized to exact editor
+ UTF-16 ranges; inputs with lone surrogates are rejected instead of rewritten.
+- PCRE2 match-step, depth and heap limits are configurable inside hard caps.
+- PCRE2 automatic callouts are copied through a separate ABI operation and
+ displayed from a dedicated killable trace worker under 50,000-event and
+ 10 MiB serialized-data hard caps.
- Parsing and execution run in separate workers.
- Actual execution can be terminated by killing its worker.
- The worker is recreated after timeout, crash or cancellation.
- Capture ranges use the engine's `d` indices result.
- Named, numbered, optional, empty and repeated-final captures are represented.
-- Match, replacement, typed list/extraction and unit-test modes are available.
+- Match, replacement, typed list/extraction, corpus/apply and unit-test modes
+ are available.
+- ECMAScript-specific advisory static risk findings, bounded dynamic growth
+ probes and cold/warm benchmarks run locally; timing and growth execution is
+ isolated in a separately killable worker.
+- ECMAScript and PCRE2 can be compared through two exact, independently timed
+ requests without translating patterns, flags or replacements. Incomplete or
+ rejected sides are explicitly not comparable.
+- The reviewed PCRE2 C17 generator emits exact UTF-8 byte arrays, configured
+ limits and error handling for the PCRE2 10.47 8-bit API.
+- A cancellable, deterministic subject minimizer verifies saved-test failures,
+ capture-range failures, exact timeouts and ECMAScript/PCRE2 mismatches against
+ the selected real workers under explicit evaluation and wall budgets.
+- Deterministic ECMAScript test-case generation samples the normalized AST from
+ an explicit seed, reports structural coverage and unsupported constructs, and
+ labels cases only after verification by the selected real engine worker.
- Project JSON import/export and optional IndexedDB save are available.
- Interactive test text is not persisted by default.
@@ -40,39 +64,67 @@ hierarchy uses syntactic capture nesting while preserving actual engine ranges.
ECMAScript exposes only the final retained capture of a repeated group; Regex
Tools says so and never invents capture history.
-Browser ECMAScript engines do not expose their internal backtracking trace.
-Version 0.1.0 therefore presents no ECMAScript execution trace and no
-educational trace disguised as one.
+Browser ECMAScript exposes no actual engine trace. PCRE2 has a separate bounded
+automatic-callout viewer. Its event fields are reported by PCRE2, while
+“forward”, “same position” and “apparent backtrack” are visibly derived only
+from adjacent reported subject positions. The callout stream is not described
+as a complete record of every internal engine action.
## Workbench modes
- **Match** — live highlighting, extraction tree and bounded capture table.
-- **Replace** — native ECMAScript `RegExp` matches with bounded,
- specification-compatible string substitution; output completeness is
- reported separately from bounded per-match original/result, changed-range and
- token-contribution previews that synchronize the replacement, pattern,
- subject and output editors.
+- **Replace** — engine-native bounded substitution for ECMAScript and PCRE2;
+ output completeness is reported separately from bounded per-match
+ original/result, changed-range and token-contribution previews that
+ synchronize the replacement, pattern, subject and output editors.
- **List** — non-executable typed templates with text, CSV, JSON and NDJSON
export; export is disabled when engine collection or bounded rendering is
incomplete.
+- **Corpus / Apply** — bounded multi-file UTF-8 and pasted-text input processed
+ sequentially in killable workers. Whole-document and independent-line
+ semantics are explicit; line apply preserves original separators. Progress,
+ cancellation, bounded capture samples, replacement previews, per-document
+ exact/partial summaries, content-free JSON/CSV/NDJSON reports and
+ privacy-gated exact-output ZIP/downloads are available. Original files are
+ never modified.
- **Unit tests** — editable and cloneable versioned presence, count, full-match,
capture, replacement, timing and expected-timeout assertions. Capture a
complete current result, select only cases to rerun, filter failures, and
import/export a standalone validated JSON suite; execution stays in isolated
workers.
-Corpus, comparison, debugging, analysis, generation, minimization, formatting
-and code generation remain roadmap items. Controls for them are not exposed in
-this release.
+The auxiliary **PCRE trace**, ECMAScript **Analysis**, **Compare & code**,
+**Generate cases**, **Minimize** and **Format** workspaces are separate from
+normal matching, replacement, tests and corpus execution. Trace collection is
+never included in benchmark measurements.
+Comparison retains native ranges while aligning actual results by normalized
+editor UTF-16 ranges. Agreement is described only as evidence for the current
+subject, never as proof of pattern equivalence.
+Case generation is bounded sampling, not a proof of language coverage. Version
+1 supports only a complete accepted ECMAScript normalized AST; the partial
+PCRE2 provider is reported as unavailable rather than passed through or
+relabeled. See
+[`docs/GENERATED_CASES.md`](docs/GENERATED_CASES.md).
+Minimization verifies every accepted candidate against the exact selected
+worker request and claims only local minimality under its documented
+Unicode-scalar transforms; see
+[`docs/MINIMIZATION.md`](docs/MINIMIZATION.md).
+Formatting is currently ECMAScript-only and changes only parser-reported
+literal slash/control representations. Apply requires independent reparse,
+actual-engine current-subject/replacement comparison, every applicable exact
+unit test and explicit confirmation; see
+[`docs/FORMATTING.md`](docs/FORMATTING.md).
## Flags and iteration
Pattern and flags are stored separately; delimiters are presentation only.
-Native `g` and `y` behavior is preserved. The optional **Scan all** action is
-explicit. When neither `g` nor `y` is present it adds `g` only to that execution
-request, visibly records the internal flag, and does not change the saved user
-flags. The `d` flag may likewise be added internally to obtain exact indices
-without changing matching semantics.
+ECMAScript preserves native `g` and `y`; PCRE2 offers the application-level
+`g` iteration flag plus `i`, `m`, `s`, `x`, `U` and `J`. PCRE2 UTF/UCP mode is
+mandatory because browser strings are Unicode and partial-byte ranges cannot be
+represented safely in the editor. The optional **Scan all** action adds only an
+internal iteration flag to that request and does not change saved flags.
+ECMAScript may add internal `d` to obtain exact indices without changing match
+semantics.
Zero-length global iteration advances using the selected Unicode semantics and
cannot loop forever.
@@ -81,29 +133,63 @@ cannot loop forever.
Important defaults:
-| Limit | Default |
-| ------------------------------ | -----------------: |
-| Live parse debounce | 120 ms |
-| Live execution debounce | 220 ms |
-| Live execution timeout | 250 ms |
-| Manual execution timeout | 2 s |
-| Advanced maximum timeout | 10 s |
-| Pattern hard limit | 64 Ki UTF-16 units |
-| Interactive subject hard limit | 16 MiB |
-| Replacement template | 64 Ki UTF-16 units |
-| List template | 16 Ki UTF-16 units |
-| Per-match replacement preview | 200 matches |
-| Replacement contributions | 2,000 |
-| Maximum matches | 10,000 |
-| Maximum capture groups | 1,000 |
-| Maximum capture rows | 100,000 |
-| Replacement preview | 64 MiB |
-| Project JSON document | 32 MiB UTF-8 |
-| Standalone test-suite JSON | 32 MiB UTF-8 |
-| In-memory unit-test suite | 32 MiB UTF-8 |
-| Tests per suite | 1,000 |
-| Retained message per test | 2 KiB UTF-8 |
-| Unit-test suite wall time | 60 s |
+| Limit | Default |
+| ------------------------------- | -----------------: |
+| Live parse debounce | 120 ms |
+| Live execution debounce | 220 ms |
+| Live execution timeout | 250 ms |
+| Manual execution timeout | 2 s |
+| Advanced maximum timeout | 10 s |
+| Pattern hard limit | 64 Ki UTF-16 units |
+| Interactive subject hard limit | 16 MiB |
+| Corpus documents | 256 |
+| Per corpus document | 16 MiB |
+| Aggregate corpus input | 256 MiB |
+| Independent corpus lines | 100,000 |
+| Aggregate corpus matches | 100,000 |
+| Capture summaries per document | 32 |
+| Samples per capture | 3 |
+| Retained applied output | 64 MiB |
+| Applied ZIP | 68 MiB |
+| Corpus batch wall time | 5 min |
+| Replacement template | 64 Ki UTF-16 units |
+| List template | 16 Ki UTF-16 units |
+| Per-match replacement preview | 200 matches |
+| Replacement contributions | 2,000 |
+| Maximum matches | 10,000 |
+| Maximum capture groups | 1,000 |
+| Maximum capture rows | 100,000 |
+| Replacement preview | 64 MiB |
+| Project JSON document | 32 MiB UTF-8 |
+| Standalone test-suite JSON | 32 MiB UTF-8 |
+| In-memory unit-test suite | 32 MiB UTF-8 |
+| Tests per suite | 1,000 |
+| Retained message per test | 2 KiB UTF-8 |
+| Unit-test suite wall time | 60 s |
+| PCRE2 trace events | 50,000 |
+| PCRE2 serialized trace | 10 MiB |
+| Comparison retained differences | 5,000 |
+| Comparison retained alignments | 2,000 |
+| Static analysis traversal | 20,000 nodes |
+| Static analysis findings | 100 |
+| Growth probe steps | 24 |
+| Analysis aggregate wall time | 30 s |
+| Generated verified cases | 48 |
+| Generated candidate attempts | 192 |
+| Generation aggregate wall time | 15 s |
+| Generation aggregate candidates | 1 MiB |
+| Minimizer subject | 1 MiB UTF-8 |
+| Minimizer candidate evaluations | 2,000 |
+| Minimizer aggregate wall time | 60 s |
+| Formatter validation tests | 1,000 |
+| Formatter test wall time | 60 s |
+| Retained minimizer history | 200 |
+
+Analysis defaults to three warm-up plus 15 measured samples (caps: 100 and
+1,000), with a 2-second per-sample worker limit that can be raised to 10
+seconds. Growth generation defaults to 1 MiB, preflights the central 16 MiB
+subject cap and rejects more than 10,000,000 repetitions. Replacement timing
+is skipped above an 8 Mi UTF-16-unit output estimate.
A timeout terminates actual execution. It is reported as **timed out**, never as
“no match”. The next request creates a fresh worker.
@@ -128,8 +214,12 @@ require separate explicit opt-ins for JSON export and IndexedDB save.
Standalone test-suite JSON likewise omits subjects by default, is schema- and
field-validated on import, and appends at most 1,000 total tests. Add, update,
clone and append-import also preflight the complete in-memory suite against a
-32 MiB aggregate bound. Corpus content is not supported or persisted in this
-release. Import, export and IndexedDB save enforce one 32 MiB aggregate UTF-8
+32 MiB aggregate bound. Added generated tests retain their validated generator
+version, seed, candidate ID and intended outcome; editing one removes that
+provenance. Schema v1 can remember that the corpus workspace was selected, but
+corpus documents, contents, outputs and results are always ephemeral and
+excluded from JSON and IndexedDB. Imports that try to inject those fields are
+rejected. Import, export and IndexedDB save enforce one 32 MiB aggregate UTF-8
JSON-document limit in addition to per-field limits. The latest local save is
read through an `updatedAt` index cursor; loading it does not materialize the
complete project store.
@@ -154,6 +244,9 @@ npm run test:conformance
npm run test:browser
npm run engines:build
npm run engines:verify
+npm run engines:pcre2:build -- --source-dir /path/to/pcre2 --emcc /path/to/emcc
+npm run engines:pcre2:verify
+npm run engines:pcre2:install
npm run build
npm run toolbox:check
npm run check
@@ -173,17 +266,34 @@ The deterministic release ZIP can be consumed by Toolbox Portal using an exact
artifact URL, manifest ID, version and SHA-256. Portal builds consume the
artifact; they do not build Regex Tools source.
-Version 0.1.0 has no WebAssembly. Its minimum CSP needs self-hosted scripts and
-workers, not `wasm-unsafe-eval`. See
-[`docs/PORTAL_REQUIREMENTS.md`](docs/PORTAL_REQUIREMENTS.md).
+The PCRE2 worker loads self-hosted JavaScript and WebAssembly only. Deployment
+must serve `.wasm` as `application/wasm` and allow WebAssembly compilation in
+`script-src`; see [`docs/PORTAL_REQUIREMENTS.md`](docs/PORTAL_REQUIREMENTS.md).
## Flavour roadmap
-The next flavour is official PCRE2 10.47 WebAssembly. A technical spike from the
-signed official tag has already demonstrated Unicode matching, named captures,
-substitution, automatic and explicit callouts, and match/depth/heap limits.
-PCRE2 is not shipped until the application-owned bridge, byte/UTF-16 offsets,
-trace caps, browser tests and source/licence material meet the acceptance gate.
+PCRE2 10.47 is the second implemented flavour. Its application-owned ABI
+provides bounded compilation, matching, native substitution and separate
+automatic-callout tracing, copied caller-owned records, deterministic
+allocation cleanup and explicit result completeness. The exact
+source/toolchain build is reproducible offline and the verified pack is bundled
+with its metadata, checksums and licence.
+
+Generated-case version 1 is ECMAScript-only. It traverses the complete
+application-owned ECMAScript AST under node, depth, attempt, byte, repetition
+and wall-time caps, then verifies every retained label through the actual
+browser `RegExp` adapter. PCRE2 generation remains unavailable until its syntax
+provider can expose a complete normalized tree.
+
+Pattern formatting is likewise ECMAScript-only. It consumes the complete
+regexpp token model and does not reinterpret the partial PCRE2 structure.
+PCRE2 formatting remains unavailable until a complete grammar-backed
+implementation passes equivalent idempotence, reparse, engine and test gates.
+
+The first advertised generated-code target is the PCRE2 10.47 8-bit C API.
+Its deterministic match and replacement fixtures compile and execute against
+the exact official release. See
+[`docs/COMPARISON_AND_CODEGEN.md`](docs/COMPARISON_AND_CODEGEN.md).
Python, Go, Rust, .NET, Java and legacy PCRE follow only through their actual
named runtimes. No flavour is emulated through JavaScript and relabelled.
@@ -198,7 +308,7 @@ notices; see [`THIRD_PARTY_NOTICES.md`](THIRD_PARTY_NOTICES.md) and
No regex101 application source, branding, generated explanation prose,
reference text or visual design is copied. RegexLib and RGXP.RU fixtures are not
redistributed because a suitable fixture licence was not established. No
-RegexHub fixture is shipped in v0.1.0.
+RegexHub fixture is shipped in v0.2.0.
## Known limitations
@@ -211,7 +321,17 @@ RegexHub fixture is shipped in v0.1.0.
error node, not a fabricated multi-character span or partial provider AST.
- Native JavaScript compile errors do not consistently expose a machine-readable
source offset; the provider diagnostic supplies the range.
-- Capture history and actual ECMAScript execution tracing are unavailable.
+- Capture history is unavailable. Browser ECMAScript exposes no engine trace;
+ PCRE2 exposes bounded automatic callouts, not a complete internal trace.
+ Movement classifications are derived and labelled as such.
+- Static risk heuristics and generated-input observations apply only to
+ ECMAScript 2025 and are advisory. They are never proof of safety or of a
+ pattern’s general time complexity; analysis settings and results are
+ ephemeral.
+- PCRE2 structural explanations are deliberately partial. The application
+ recognizes common capture forms and PCRE2-only constructs for reference, but
+ the actual PCRE2 compiler is authoritative. Branch-reset capture numbering is
+ omitted from the explanation rather than guessed.
- Capture-table rendering is paged at 200 rows while aggregate collection is
bounded separately.
- Explanation/extraction trees and editor decorations show bounded 2,000-node
@@ -235,8 +355,17 @@ RegexHub fixture is shipped in v0.1.0.
- Unit-test failure messages retain at most 2 KiB UTF-8 per case, so 1,000
results stay below the 2 MiB log budget. Oversized exact replacement results
offer a lightweight `should-match` draft when one is valid and fits the suite.
-- Benchmark, risk analysis, generated cases, minimization and corpus processing
- are not in this release.
+- Corpus input accepts recognized UTF-8 text only. It does not auto-detect
+ legacy encodings or recursively traverse directories.
+- Corpus capture extraction is a bounded per-document summary: full match plus
+ at most 31 capture groups, three 160-unit samples each. Content-free reports
+ include counts and clipping metadata, never the samples themselves.
+- Generated cases deliberately sample only ECMAScript and report
+ backreferences, lookaround, boundaries, complex Unicode sets and other
+ unsupported constraints instead of claiming exhaustive coverage.
+- Pattern formatting is intentionally narrow and ECMAScript-only. Its bounded
+ source/candidate and unit-test validation is evidence for the checked
+ snapshots, not a proof of equivalence over all subjects.
Architecture, security, performance, provenance and flavour details live in
[`docs/`](docs/).
diff --git a/SOURCE.md b/SOURCE.md
index 4be6470..b55c84b 100644
--- a/SOURCE.md
+++ b/SOURCE.md
@@ -1,9 +1,10 @@
# Source identity
- Project: Regex Tools
-- Version: 0.1.0
+- Release version: `0.2.0`
+- Release tag: `v0.2.0`
- Repository:
-- Release tag: `v0.1.0`
+- Previous release tag: `v0.1.0`
- Licence: GPL-3.0-or-later
- Copyright: © 2026 Albrecht Degering
@@ -19,5 +20,14 @@ runtime or build-time CDN fetch. Complete preferred-form source is the tagged
repository. The release archive contains compiled static assets plus legal,
source-identity and attribution documents.
-No PCRE2, Python, Go, Rust, .NET, Java or legacy-PCRE runtime is included in
-version 0.1.0.
+The explicit offline PCRE2 build is documented in `docs/ENGINE_BUILDS.md`.
+PCRE2 source is not vendored: the gate accepts only the pinned, clean official
+10.47 checkout and Emscripten 6.0.4 compiler. Generated staging is ignored; the
+verified runtime files, deterministic metadata, checksums and exact licence are
+installed under `public/engines/pcre2/` and included in the application source
+and build.
+
+The v0.2.0 release includes PCRE2. Python, Go, Rust, .NET, Java and legacy PCRE
+remain unimplemented. The v0.1.0 release remains available under its historical
+tag and artifact coordinates. A future public artifact must bump package,
+manifest, tag and source identity together; it must not reuse `v0.2.0`.
diff --git a/THIRD_PARTY_NOTICES.md b/THIRD_PARTY_NOTICES.md
index c8a8f39..ad8cee6 100644
--- a/THIRD_PARTY_NOTICES.md
+++ b/THIRD_PARTY_NOTICES.md
@@ -1,27 +1,29 @@
# Third-party notices
Regex Tools original source is GPL-3.0-or-later. The production browser bundle
-also contains the compatible components below. Exact dependency resolution is
-recorded in `package-lock.json`.
+contains the compatible runtime components below.
+Exact dependency resolution is recorded in `package-lock.json`.
-| Component | Version | Licence | Shipped | Role and source |
-| -------------------------------- | ------------------------------------------------------------------------ | ---------- | ----------------- | ------------------------------------------------------------------------ |
-| `@add-ideas/toolbox-contract` | 0.2.2 | Apache-2.0 | Runtime | Manifest contract; |
-| `@add-ideas/toolbox-shell-react` | 0.2.2 | Apache-2.0 | Runtime | Shared shell; same source |
-| `@eslint-community/regexpp` | 4.12.2 | MIT | Syntax worker | ECMAScript parser; |
-| CodeMirror packages | state 6.7.1; view 6.43.6; language 6.12.4; commands 6.10.4; search 6.7.1 | MIT | Runtime | Editors; |
-| `@lezer/highlight` | 1.2.3 | MIT | Runtime | Editor highlighting support; |
-| React / React DOM | 19.2.7 | MIT | Runtime | User interface; |
-| `fflate` | 0.8.3 | MIT | Packaging utility | Deterministic release ZIP; |
-| Vite | 8.1.5 | MIT | Generated helpers | Production build; |
-| Rolldown | 1.1.5 | MIT | Generated helpers | Production bundling; |
+| Component | Version | Licence | Shipped | Role and source |
+| -------------------------------- | ------------------------------------------------------------------------ | --------------------------------- | ---------------------------- | -------------------------------------------------------------------------------------------- |
+| `@add-ideas/toolbox-contract` | 0.2.2 | Apache-2.0 | Runtime | Manifest contract; |
+| `@add-ideas/toolbox-shell-react` | 0.2.2 | Apache-2.0 | Runtime | Shared shell; same source |
+| `@eslint-community/regexpp` | 4.12.2 | MIT | Syntax worker | ECMAScript parser; |
+| CodeMirror packages | state 6.7.1; view 6.43.6; language 6.12.4; commands 6.10.4; search 6.7.1 | MIT | Runtime | Editors; |
+| `@lezer/highlight` | 1.2.3 | MIT | Runtime | Editor highlighting support; |
+| React / React DOM | 19.2.7 | MIT | Runtime | User interface; |
+| `fflate` | 0.8.3 | MIT | Runtime and build dependency | Local corpus-output ZIP and deterministic release ZIP; |
+| Emscripten generated runtime | 6.0.4 | MIT | Generated WebAssembly glue | PCRE2 module loader/runtime; |
+| PCRE2 | 10.47 | BSD-3-Clause WITH PCRE2-exception | WebAssembly runtime | Pinned official 8-bit engine; |
+| Vite | 8.1.5 | MIT | Generated helpers | Production build; |
+| Rolldown | 1.1.5 | MIT | Generated helpers | Production bundling; |
`@add-ideas/toolbox-testkit` 0.2.2 and the other test/build dependencies are
development-only and are not part of the static runtime bundle.
`recheck` 4.5.0 and `regexp-ast-analysis` 0.7.1 were inspected but deliberately
-not installed or shipped. PCRE2 10.47 was used only in an external feasibility
-spike and is not present in the v0.1.0 artifact.
+not installed or shipped. PCRE2 source is not vendored; its generated
+WebAssembly pack is shipped with exact metadata, checksums and licence.
Corresponding licence texts and copyright notices are in `LICENSES/`. The
release package additionally carries the exact Vite and Rolldown legal files
diff --git a/docs/ANALYSIS.md b/docs/ANALYSIS.md
new file mode 100644
index 0000000..88bd95f
--- /dev/null
+++ b/docs/ANALYSIS.md
@@ -0,0 +1,121 @@
+# Performance and risk analysis
+
+Regex Tools provides one deliberately scoped analysis implementation:
+ECMAScript 2025 patterns parsed into the application-owned normalized AST and
+executed by the browser's native `RegExp` runtime. The analyser does not apply
+ECMAScript findings to PCRE2 or any other flavour.
+
+Analysis is advisory. The UI uses “potential risk”, “possibly” and “observed
+under selected limits”. It never turns an empty finding list into “safe”,
+“guaranteed linear” or a vulnerability proof.
+
+## Static analysis
+
+The application-owned `regex-tools-ecmascript-heuristics` analyser walks at
+most 20,000 normalized nodes and returns at most 100 findings. Each finding
+contains:
+
+- flavour and UTF-16 source range;
+- stable rule identifier;
+- static or dynamically observed provenance;
+- explanation and example risk;
+- low, medium or high confidence;
+- limitations;
+- a suggested investigation.
+
+The current rules review nested and nullable repetition, ambiguous repeated
+alternatives, overlapping alternative prefixes, wildcards and backreferences
+inside repetition, repeated lookaround bodies, unanchored amplification,
+capture count, nesting depth and replacement-output expansion. These are
+structural heuristics. They do not model the native engine's complete compiled
+program, optimizations or calling context.
+
+## Dynamic growth probes
+
+The user defines a bounded subject family:
+
+```text
+prefix + repeatedFragment.repeat(repetitions) + suffix
+```
+
+The byte and UTF-16 size are calculated before the generated string is
+allocated. The repeated fragment is limited to 4,096 UTF-16 units, each affix
+to 16,384 units, the run to 24 samples, and repetitions to 10,000,000. The UI
+defaults to a 1 MiB generated-subject cap even though the central interactive
+hard cap remains 16 MiB.
+
+Every probe executes in a dedicated worker. The supervisor terminates and
+discards that worker on per-sample timeout, cancellation or crash. Runs also
+stop at:
+
+- the selected repetition bound;
+- the selected generated-byte bound;
+- the selected step bound;
+- the 30-second aggregate wall bound;
+- a selected normalized-growth threshold.
+
+The growth threshold compares the time ratio with the input-size ratio for two
+adjacent samples. Sub-millisecond measurements below a noise floor do not
+trigger it. A threshold observation characterizes only those generated inputs
+and that browser. Timeout, crash, compile rejection and non-match remain
+distinct states.
+
+The UI renders time and outcome in separate plots and keeps the exact sample
+table alongside them.
+
+## Benchmark method
+
+A benchmark begins with a fresh analysis worker. Worker creation and its
+identity response form the separate cold-start measurement. The first native
+engine use is retained as a cold sample. Configured warm-up samples are
+discarded; configured measured samples produce the warm statistics.
+
+Each sample measures these phases separately:
+
+- `RegExp` construction;
+- first `exec`;
+- bounded all-match iteration;
+- native string replacement when its conservative output bound is acceptable;
+- all-match throughput and match count.
+
+All-match iteration retains only counts, not values or captures. It follows the
+same explicit scan-all, global, sticky and Unicode zero-length advancement
+semantics as the workbench. The application-added indices flag remains in the
+effective flags so the measured operation reflects interactive execution.
+
+Replacement timing is skipped when match collection reaches its bound or a
+conservative estimate can exceed 8 Mi UTF-16 units. This prevents a benchmark
+from using an unbounded native replacement merely to obtain a timing.
+It is also skipped for explicit scan-all with sticky `y`: native
+`String.replace` processes only one sticky match, while the workbench's
+explicit scan-all operation can process a contiguous sequence. Reporting that
+different operation as the same benchmark would be misleading.
+
+Warm metrics include sample count, minimum, conventional median, nearest-rank
+p95 and maximum. Milliseconds are shown to two decimal places, with values
+below 0.01 ms labelled accordingly; this does not imply nanosecond accuracy.
+The result retains subject size, match count, effective flags, browser runtime
+identity and engine identity.
+
+Defaults:
+
+| Setting | Default | Hard bound |
+| ----------------------- | ------: | ---------: |
+| Warm-up samples | 3 | 100 |
+| Measured samples | 15 | 1,000 |
+| Per-sample timeout | 2 s | 10 s |
+| Aggregate wall time | 30 s | 30 s |
+| Matches per sample | 10,000 | 10,000 |
+| Replacement output gate | 8 MiU | 8 MiU |
+
+One subject is not a complete performance characterization. Results from
+different algorithms, pattern ports, browser engines, WebAssembly runtimes or
+machines are not directly comparable. Trace-enabled execution is never used
+for benchmark results.
+
+## Privacy and persistence
+
+Patterns, subjects, generated inputs, findings and benchmark samples remain in
+the browser. No analysis request is uploaded. Analysis settings and results are
+currently ephemeral and are not added to project JSON or IndexedDB; this also
+keeps large benchmark subjects out of persisted projects by default.
diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md
index 3d49116..d198537 100644
--- a/docs/ARCHITECTURE.md
+++ b/docs/ARCHITECTURE.md
@@ -5,9 +5,32 @@ boundaries.
```text
React workbench
- ├─ Pattern SyntaxSupervisor → syntax.worker → Regexpp provider → normalized AST
+ ├─ Pattern SyntaxSupervisor → syntax.worker
+ │ ├─ Regexpp ECMAScript provider → normalized AST
+ │ └─ PCRE2 lexical provider → partial normalized structure
├─ Replacement SyntaxSupervisor → syntax.worker → typed replacement tokens
- ├─ EngineSupervisor → ecmascript.worker → native RegExp → match DTOs
+ ├─ EngineSupervisor
+ │ ├─ ecmascript.worker → native RegExp → match DTOs
+ │ └─ pcre2.worker → PCRE2 10.47 WASM ABI → match DTOs
+ ├─ Pcre2TraceSupervisor → pcre2-trace.worker
+ │ └─ separate ABI-v3 PCRE2_AUTO_CALLOUT operation → bounded trace DTOs
+ ├─ AnalysisPanel
+ │ ├─ normalized-AST ECMAScript static heuristics
+ │ └─ AnalysisSupervisor → analysis.worker → native RegExp benchmark/growth probes
+ ├─ ComparisonOrchestrator
+ │ ├─ two exact syntax snapshots + independently killable engine workers
+ │ └─ normalized UTF-16 match/capture alignment + explicit incomplete states
+ ├─ PCRE2 C17 generator → validated snapshot → exact UTF-8 byte-array source
+ ├─ SubjectMinimizer
+ │ ├─ fixed syntax snapshots + exact real-engine failure or comparison oracle
+ │ └─ deterministic Unicode-scalar reducer → bounded local-minimal result
+ ├─ CaseGenerationOrchestrator
+ │ ├─ generation.worker → deterministic bounded ECMAScript AST candidates
+ │ └─ EngineSupervisor → selected actual-engine label verification
+ ├─ ECMAScript literal/control formatter
+ │ ├─ exact regexpp literal-token ranges → idempotent candidate preview
+ │ └─ PatternFormatValidator → paired syntax + actual-engine workers and exact tests
+ ├─ Corpus EngineSupervisor → sequential bounded document jobs → summaries/outputs
├─ Test SyntaxSupervisor + EngineSupervisor → isolated cancellable test runs
└─ project validator / serializer / IndexedDB persistence
```
@@ -21,6 +44,59 @@ request ID and worker generation. A supervisor allows one active request,
rejects stale responses, terminates on timeout/crash/cancel, and lazily creates
a new generation. Timeout is a distinct error state.
+PCRE2 tracing is not an `EngineSupervisor` match or replacement operation. Its
+own worker compiles with `PCRE2_AUTO_CALLOUT`, copies only complete fixed
+records and bounded mark bytes, and stops natively at either trace cap.
+Reported callout fields retain byte positions; movement classifications are a
+separate derived UI layer. Normal execution, corpus, tests and benchmarks never
+run through the trace operation.
+
+Static analysis traverses the normalized ECMAScript tree on the UI side under
+node/finding caps. Timing and generated-input growth probes use the independent
+analysis worker. Timeout, cancellation or crash discards that worker; no trace
+mode is benchmarked.
+
+Comparison is a separate orchestration layer, not an `EngineSupervisor`
+operation. It runs explicit ECMAScript and PCRE2 syntax/execution snapshots,
+retains each engine identity and native range model, and aligns only normalized
+editor ranges. Timeout, cancellation, compile rejection and truncation remain
+distinct non-comparable outcomes. Shared patterns are sent unchanged;
+per-flavour ports are user-owned variants.
+
+The PCRE2 C generator is application-owned and never implements browser
+execution. It validates one exact PCRE2 snapshot, emits user strings as
+independent UTF-8 byte arrays, and states the host-level mappings that the
+PCRE2 API cannot express directly. Its advertised C17 target is protected by
+deterministic golden identities and an exact PCRE2 10.47 compile/execute gate.
+
+Subject minimization is a main-thread bounded reducer whose expensive predicate
+checks stay in independently terminable engine workers. Fixed pattern syntax
+is parsed once for capture metadata. The baseline and every accepted candidate
+retain the exact target and engine identity. Timeout, crash, cancellation,
+worker failure and incomplete output are separate outcomes. Deterministic
+ddmin deletion is followed by a fixed-point local sweep over Unicode-scalar
+deletion and lower-rank canonical replacement; only completion of that sweep
+permits a transform-local minimality claim.
+
+Generated-case synthesis and verification are separate trust boundaries. The
+dedicated generation worker receives only a validated, complete ECMAScript
+normalized tree plus explicit settings and seed. The main-thread orchestrator
+then sends each bounded candidate to `EngineSupervisor`; only the selected real
+adapter can establish a `should-match` or `should-not-match` label. Worker
+timeout, crash, cancellation, compile rejection, wrong outcome and ordinary
+failure stay distinct. Settings/results are ephemeral, while explicitly added
+unit tests retain field-validated generator provenance.
+
+Formatting is not an engine operation and never rewrites from heuristic text
+alone. The ECMAScript formatter consumes an exact accepted regexpp snapshot and
+changes only parser-reported literal/control token ranges. Its validator
+reparses source and candidate separately, preserves capture shape, runs both in
+separate actual-engine workers against the current replacement snapshot, then
+replays every enabled test tied to the exact active pattern/version/flags/options.
+Any difference, incomplete output, one-sided timeout, identity change or
+inconclusive worker outcome blocks apply. The preview and observations are
+ephemeral, and user confirmation is required even after all bounded gates pass.
+
Pattern and replacement syntax use independent supervisors. A pattern snapshot
is bound to its exact pattern, ordered flags and revision; execution cannot use
acceptance or capture metadata from an earlier input. Replacement output is
@@ -32,10 +108,27 @@ over those actual matches, the complete replacement-token model and the bounded
output. They execute no replacement callback or user code and have separate
presentation and token-evaluation limits.
-Native ranges are retained in the engine's declared unit. Browser editor ranges
+Corpus documents are ephemeral React session state, not project state. A
+dedicated engine supervisor processes one document at a time and is terminated
+on cancellation, timeout or workspace exit. Batch orchestration retains only
+per-document summaries and, for apply runs, bounded outputs. It enforces
+document-count, per-document byte, aggregate input byte, aggregate match,
+aggregate output and batch wall-time limits. Exact-output export is unavailable
+if any document is partial, failed, cancelled or not run.
+
+Whole-document mode sends each document to the engine once. Independent-line
+mode segments CRLF, CR, LF, U+2028 and U+2029 without discarding them, executes
+each logical line as its own subject, and reattaches the exact original
+separator during apply. This makes anchor and cross-line behavior deliberately
+different rather than silently rewriting the pattern. Match DTOs are reduced to
+bounded per-group participation counts and samples before the next document.
+
+Native ranges are retained in the engine's declared unit. PCRE2 always uses
+UTF/UCP and exposes UTF-8 byte ranges; its ABI copies scalar records and names
+out of WebAssembly before returning and retains no code pointer. Browser editor ranges
are half-open UTF-16 code-unit ranges. Reusable converters cover UTF-8 byte and
-Unicode code-point offsets for later flavours, including invalid-boundary
-rejection and lone-surrogate detection.
+Unicode code-point offsets, including invalid-boundary rejection and
+lone-surrogate detection.
The release boundary is `dist/` plus checked-in legal/source documents. The
Portal consumes the resulting immutable ZIP and never imports React source.
diff --git a/docs/COMPARISON_AND_CODEGEN.md b/docs/COMPARISON_AND_CODEGEN.md
new file mode 100644
index 0000000..137851c
--- /dev/null
+++ b/docs/COMPARISON_AND_CODEGEN.md
@@ -0,0 +1,118 @@
+# ECMAScript/PCRE2 comparison and PCRE2 C generation
+
+## Comparison contract
+
+The comparison panel sends two explicit requests: one to the native browser
+ECMAScript worker and one to the bundled PCRE2 10.47 worker. It does not
+translate patterns, flags, options or replacement templates.
+
+Two pattern models are available:
+
+- **Shared pattern** sends the same string unchanged to both engines.
+- **Per-flavour variants** retains two explicit ports supplied by the user.
+
+Flags are selected independently from each flavour's registered allowlist.
+PCRE2 match, depth and heap bounds are part of its exact request. Both sides
+share the selected subject, scan-all intent, host result bounds and wall-clock
+worker timeout.
+
+Each side retains:
+
+- its syntax-provider request, acceptance, diagnostics and coverage;
+- the exact execution or replacement request sent to its worker;
+- a deterministic display key plus the complete authoritative request;
+- completion, timeout, cancellation or worker-failure status;
+- engine, runtime and adapter identity when the engine returns one;
+- user and effective flags;
+- native ranges and normalized editor UTF-16 ranges;
+- compile acceptance, matches, captures and replacement output.
+
+The orchestrator owns two syntax supervisors and the multi-engine execution
+supervisor. The two engine calls run concurrently through independently
+killable flavour workers. A side timeout kills only that worker. The shared
+Cancel action terminates every active comparison worker. Comparison remains
+outside `EngineSupervisor`, whose responsibility is one engine operation and
+lifecycle boundary.
+
+Match pairs are aligned by equal normalized UTF-16 ranges first. Unpaired
+matches use an explicitly labelled ordinal fallback. Captures are aligned by
+normalized range, then capture identity, then an explicitly labelled ordinal
+fallback. Original native offsets remain attached to both results.
+
+No equivalence claim is made when syntax or compilation is rejected, a worker
+times out/fails/is cancelled, results or values are truncated, or replacement
+output is incomplete. A completed matching result is labelled only as
+agreement for the current subject; it is evidence, not a proof that two
+patterns are equivalent. Timeout is never interpreted as “no match.”
+
+The retained DTO is bounded to 5,000 displayed difference records and 2,000
+match alignments while still counting all differences in the bounded engine
+result. The panel renders at most 250 differences and 100 alignments at once.
+
+## PCRE2 C17 generator
+
+The first reviewed code-generation target is the PCRE2 10.47 8-bit C API.
+Other languages remain unavailable until their generators have equivalent
+escaping, semantic and toolchain gates.
+
+The generator consumes a validated PCRE2 snapshot and:
+
+- requires the registered PCRE2 10.47 version and flag/option allowlists;
+- refuses unpaired UTF-16 surrogates rather than emitting lossy UTF-8;
+- emits pattern, subject and replacement as independent explicit UTF-8 byte
+ arrays, so quotes, backslashes, newlines, NUL, Unicode, dollar signs and
+ braces cannot escape into C source syntax;
+- enables mandatory `PCRE2_UTF` and `PCRE2_UCP`;
+- maps `i`, `m`, `s`, `x`, `U` and `J` to named PCRE2 compile constants;
+- implements application-level `g`/scan-all behavior explicitly;
+- handles the PCRE2 empty-match anchored retry before advancing one UTF-8
+ character;
+- configures native match, depth and heap limits;
+- enforces match/capture row caps in the generated match loop;
+- bounds replacement output with a host buffer and withholds output when its
+ post-run result-row bound is exceeded;
+- reports compile offsets and native API errors and cleans up all PCRE2 state;
+- rejects compilation against a PCRE2 header other than 10.47 and verifies the
+ runtime version.
+
+The generated program states limitations that cannot be mapped honestly:
+
+- PCRE2 has no `g` compile flag;
+- match/capture rows and replacement-output size are host-application bounds,
+ not PCRE2 matching options;
+- replacement result-row limits can be checked only after
+ `pcre2_substitute()` has completed;
+- PCRE2 exposes no native wall-clock timeout, so hard elapsed-time
+ cancellation belongs to the containing process;
+- C API offsets are UTF-8 bytes, not browser editor UTF-16 units.
+
+Generated snippets are examples only and are never used to implement the
+browser runtime.
+
+## Golden and toolchain gate
+
+`scripts/pcre2-codegen-golden.test.ts` pins deterministic SHA-256 identities
+for match and replacement fixtures:
+
+```text
+match 22b91d578a449b0ed5c30e6eadad30f19a2bcb26ac9db355c8ec7a69328675da
+replacement ab90a0d9831ea7b7145aa0072ac352b42baf9a795319d81f531b0f341b04a636
+```
+
+The compile/execute gate was run against a host static build from official
+signed tag `pcre2-10.47`, commit
+`f454e231fe5006dd7ff8f4693fd2b8eb94333429`. Both generated fixtures compile
+under C17 with `-Wall -Wextra -Werror`; the match fixture verifies native UTF-8
+ranges and named captures, and the replacement fixture verifies exact Unicode
+output and substitution count.
+
+To repeat the gate with an exact local PCRE2 installation:
+
+```sh
+PCRE2_CODEGEN_PREFIX=/path/to/pcre2-10.47-prefix \
+ npm test -- scripts/pcre2-codegen-golden.test.ts
+```
+
+The prefix must provide `include/pcre2.h` and
+`lib64/libpcre2-8.a`. A different static-library location can be supplied
+through `PCRE2_CODEGEN_LIBRARY`.
diff --git a/docs/ENGINE_BUILDS.md b/docs/ENGINE_BUILDS.md
index ec068a7..072ce6f 100644
--- a/docs/ENGINE_BUILDS.md
+++ b/docs/ENGINE_BUILDS.md
@@ -1,28 +1,74 @@
# Engine builds
-Version 0.1.0 ships no WebAssembly engine.
+The production application uses the browser's native `RegExp` for ECMAScript
+and bundles a verified PCRE2 10.47 WebAssembly pack. Normal application builds
+never download or compile an engine: the reviewed files under
+`public/engines/pcre2/` are copied into `dist/` and verified there.
-`npm run engines:build` verifies that this release has no external pack to
-build. After the production build, `npm run engines:verify` rejects any
-undeclared engine asset and confirms the packaged ECMAScript-only notice.
-Neither command accesses the network.
+## Rebuilding PCRE2
-## PCRE2 feasibility record
+The reproducible build accepts only:
-The next flavour was spiked outside the application repository against official
-PCRE2 10.47:
+- PCRE2 tag `pcre2-10.47`, signed tag object
+ `cd007b4466798f66d479d1442a407099e7c40050`, peeled commit
+ `f454e231fe5006dd7ff8f4693fd2b8eb94333429` and tree
+ `81a83a3552bd68d0ea7b7004f8bb6e7892f583ba`;
+- Emscripten 6.0.4, compiler revision
+ `fe5be6afdff43ad58860d821fcc8572a23f92d19`, from emsdk commit
+ `224ec5f9f2f72f09f9ce0e26d66bae7dbd8b692f`;
+- CMake 4.3.4 and Ninja 1.13.2;
+- 8-bit PCRE2 with Unicode enabled and JIT, threads and filesystem disabled.
-- signed tag object `cd007b4466798f66d479d1442a407099e7c40050`;
-- peeled commit `f454e231fe5006dd7ff8f4693fd2b8eb94333429`;
-- licence `BSD-3-Clause WITH PCRE2-exception`;
-- Emscripten 6.0.4;
-- 8-bit library, Unicode enabled, JIT disabled.
+The source and compiler checkouts must already exist locally and be completely
+clean. No command clones, downloads or updates them:
-Node and Chromium tests demonstrated version/config reporting, Unicode,
-named groups, global substitution, automatic and explicit callouts, and
-match/depth/heap errors. This was a technical spike only. No spike source,
-binary or PCRE2 source is included in v0.1.0.
+```sh
+npm run engines:pcre2:build -- \
+ --source-dir /absolute/path/to/pcre2 \
+ --emcc /absolute/path/to/emscripten/emcc
+npm run engines:pcre2:verify
+npm run engines:pcre2:install
+```
-Production PCRE2 support remains gated on a reviewed bridge, copied DTOs,
-allocation cleanup, exact source build, UTF-8/UTF-16 mapping, trace/output caps,
-worker recovery, deterministic assets, licences and browser conformance.
+Use `--force` only to replace the exact ignored `.engine-build/pcre2` output.
+The explicit install command first re-verifies that pack and then atomically
+replaces only `public/engines/pcre2`.
+
+The builder verifies Git objects and critical source/compiler hashes,
+sanitizes build flags, sets `SOURCE_DATE_EPOCH=1760997684`, builds serially and
+emits exactly:
+
+- `pcre2.mjs` and `pcre2.wasm`;
+- deterministic `engine-metadata.json`;
+- the exact upstream `LICENSE.txt`;
+- `SHA256SUMS`.
+
+The verifier checks the closed file set, metadata, source bridge hashes and all
+checksums. It validates and instantiates WebAssembly, checks ABI/engine/config
+identity, then executes native Unicode compile, bounded match and bounded
+substitution smoke cases plus normal and cap-stopped automatic-callout traces.
+
+The immutable signed-tag object is pinned. OpenPGP trust validation remains a
+release-operator step because the offline builder does not install or trust a
+key.
+
+## Runtime boundary
+
+ABI version 3 has no retained native handles. Each call compiles, obtains
+capture names, matches/substitutes or traces and releases its code, match data
+and match context before returning. The caller supplies fixed-capacity
+result/name/output/event/mark buffers. Hard maxima are 1 MiB pattern, 16 MiB
+subject, 1,000 captures, 10,000 matches, 100,000 capture rows, 50,000 trace
+events, 10 MiB serialized trace data and 128 MiB PCRE2 heap limit; the
+workbench uses stricter pattern and replacement input limits where applicable.
+
+PCRE2 always runs with UTF and UCP. Native UTF-8 byte offsets are retained in
+DTOs and converted only at valid code-point boundaries to half-open editor
+UTF-16 ranges. Lone UTF-16 surrogates are rejected before encoding.
+
+Normal execution lives only in `pcre2.worker`. Instrumented
+`PCRE2_AUTO_CALLOUT` collection is a separate ABI operation loaded only by
+`pcre2-trace.worker`; normal match, replacement, tests, corpus and benchmarks
+never call it. Each supervisor terminates its worker on timeout, crash,
+cancellation or supersession and lazily creates a clean generation for the
+next request.
diff --git a/docs/EXECUTION_TRACES.md b/docs/EXECUTION_TRACES.md
index 236ac8f..e337ad7 100644
--- a/docs/EXECUTION_TRACES.md
+++ b/docs/EXECUTION_TRACES.md
@@ -1,13 +1,39 @@
# Execution traces
-Version 0.1.0 exposes no execution trace.
+Browser ECMAScript APIs expose compilation, match, capture and replacement
+results, but no V8, SpiderMonkey or JavaScriptCore execution trace. Regex Tools
+does not relabel structural explanations, static findings or timing
+observations as an ECMAScript engine trace.
-Browser ECMAScript APIs return compilation, match, capture and replacement
-results but not the runtime's internal backtracking events. Regex Tools does
-not label structural explanations, static hints or adjacent-result inference
-as an actual engine trace.
+PCRE2 10.47 has a separate actual automatic-callout vertical. ABI version 3
+compiles the requested pattern with `PCRE2_AUTO_CALLOUT`, registers an
+application callback, and copies only:
-The next PCRE2 slice will use actual automatic/explicit callout events from an
-application-owned bridge. Event count, serialized bytes, match/depth/heap
-limits and wall-clock termination must all be enforced. Classifications such
-as “apparent backtrack” will remain visibly derived from adjacent events.
+- callout number;
+- pattern byte position and next-item byte length;
+- subject byte position;
+- capture-top and capture-last numbers;
+- bounded current mark text.
+
+These fields have `reported` provenance. Pattern and subject byte boundaries
+are validated and normalized to exact editor UTF-16 positions while the native
+values remain visible.
+
+Collection runs in `pcre2-trace.worker`, not the normal execution worker. It
+uses the selected PCRE2 match, depth and heap limits plus the supervisor wall
+clock. The callback retains only complete 32-byte records and mark slices under
+both a 50,000-event and 10 MiB serialized-data hard cap. At a cap it returns
+PCRE2’s reserved `PCRE2_ERROR_CALLOUT`, abandoning the native match; the result
+retains that native status, the last complete event and an explicit truncation
+flag. Timeout, cancellation or crash discards the complete worker.
+
+The viewer derives “forward”, “same position” and “apparent backtrack” solely
+from the difference between adjacent reported subject positions. Each label
+has `derived` provenance. This is useful navigation, not a claim that the
+stream reports every internal action. PCRE2 start optimizations can also
+conclude some requests without any callout.
+
+Tracing represents one exact `pcre2_match()` invocation. The application-level
+`g` iteration flag is not applied, and that fact is reported. Normal match,
+replacement, tests, corpus, comparisons and benchmarks never use the
+instrumented trace path.
diff --git a/docs/EXPLANATION_MODEL.md b/docs/EXPLANATION_MODEL.md
index 37f2e16..f5ed969 100644
--- a/docs/EXPLANATION_MODEL.md
+++ b/docs/EXPLANATION_MODEL.md
@@ -1,9 +1,9 @@
# Explanation model
-The community provider converts regexpp nodes into a stable application AST
-inside the syntax worker. Nodes carry deterministic IDs, UTF-16 ranges, raw
-source, capture metadata, quantifier bounds, assertion kind, support status,
-provider identity and provenance.
+The ECMAScript community provider converts regexpp nodes into a stable
+application AST inside the syntax worker. Nodes carry deterministic IDs,
+UTF-16 ranges, raw source, capture metadata, quantifier bounds, assertion kind,
+support status, provider identity and provenance.
Explanations are original deterministic text selected by node type. Nullable
and minimum/maximum consumed-length properties are derived recursively where
@@ -15,6 +15,13 @@ regexpp 4.12.2 does not expose tolerant recovery. A malformed pattern therefore
gets the provider's exact diagnostic plus an application-owned error node. No
partial regexpp AST is claimed.
+The PCRE2 provider currently emits a deliberately partial lexical tree for
+common capture forms. It does not infer branch-reset numbering or claim full
+nesting; authoritative compilation, capture counts, names and result ranges
+come from PCRE2 itself.
+
The explanation tree is structural. It is never called an actual execution
trace. Selecting a node selects and scrolls its exact editor range; selecting
-pattern text chooses the smallest enclosing node.
+pattern text chooses the smallest enclosing node. PCRE2’s separate trace viewer
+contains actual bounded automatic callouts; its movement labels remain a
+distinct derived layer and are never copied into the structural explanation.
diff --git a/docs/FLAVOUR_SUPPORT.md b/docs/FLAVOUR_SUPPORT.md
index fe88c9e..6590c4c 100644
--- a/docs/FLAVOUR_SUPPORT.md
+++ b/docs/FLAVOUR_SUPPORT.md
@@ -1,20 +1,55 @@
# Flavour support
-| Flavour | Syntax | Execution | Replacement | Captures | Trace | Native offsets |
-| ------------ | ------------------------------------------------------------------ | ----------------------------------------------- | ---------------------------------------------------- | -------------------------------------------------- | ----------- | ------------------- |
-| ECMAScript | regexpp 4.12.2 grammar; explanations partial/feature-matrix tested | Current browser `RegExp`; feature-matrix tested | Native matches; bounded ECMAScript `GetSubstitution` | Named/numbered; final repeated capture; no history | Unavailable | UTF-16 |
-| PCRE2 | Unavailable in release | Technical spike only; not shipped | Unavailable | Unavailable | Unavailable | Planned UTF-8 bytes |
-| PCRE | Unavailable | Unavailable | Unavailable | Unavailable | Unavailable | Unavailable |
-| Python `re` | Unavailable | Unavailable | Unavailable | Unavailable | Unavailable | Planned code points |
-| Go `regexp` | Unavailable | Unavailable | Unavailable | Unavailable | Unavailable | Planned UTF-8 bytes |
-| Rust `regex` | Unavailable | Unavailable | Unavailable | Unavailable | Unavailable | Planned UTF-8 bytes |
-| .NET | Unavailable | Unavailable | Unavailable | Planned capture history | Unavailable | Planned UTF-16 |
-| Java | Unavailable | Unavailable | Unavailable | Unavailable | Unavailable | Planned UTF-16 |
+| Flavour | Syntax | Execution | Replacement | Captures | Trace | Analysis | Generated cases | Native offsets |
+| ------------ | ------------------------------------------------------------ | ------------------------------------------------------ | ---------------------------------------- | -------------------------------------------------- | ------------------------------------------------------------ | ------------------------------------------------------ | ------------------------------------------------------- | ------------------- |
+| ECMAScript | regexpp 4.12.2; explanations partial/feature-matrix tested | Current browser `RegExp`; conformance tested | Bounded ECMAScript `GetSubstitution` | Named/numbered; final repeated capture; no history | Unavailable | Experimental static heuristics + bounded native probes | Bounded normalized-AST sampling; actual-engine verified | UTF-16 |
+| PCRE2 10.47 | Application lexical provider partial; compiler authoritative | Pinned 8-bit WASM; bounded/killable/conformance tested | Native bounded `pcre2_substitute()` loop | Native named/numbered records; no capture history | Bounded reported automatic callouts; derived movement labels | Unavailable; ECMAScript heuristics are never applied | Unavailable; structural syntax provider is partial | UTF-8 bytes |
+| PCRE | Unavailable | Unavailable | Unavailable | Unavailable | Unavailable | Unavailable | Unavailable | Unavailable |
+| Python `re` | Unavailable | Unavailable | Unavailable | Unavailable | Unavailable | Unavailable | Unavailable | Planned code points |
+| Go `regexp` | Unavailable | Unavailable | Unavailable | Unavailable | Unavailable | Unavailable | Unavailable | Planned UTF-8 bytes |
+| Rust `regex` | Unavailable | Unavailable | Unavailable | Unavailable | Unavailable | Unavailable | Unavailable | Planned UTF-8 bytes |
+| .NET | Unavailable | Unavailable | Unavailable | Planned capture history | Unavailable | Unavailable | Unavailable | Planned UTF-16 |
+| Java | Unavailable | Unavailable | Unavailable | Unavailable | Unavailable | Unavailable | Unavailable | Planned UTF-16 |
-ECMAScript tests cover `g`, `y`, `u`, internal `d`, zero-length iteration,
-named groups, unmatched/empty/repeated groups, lookbehind, backreferences,
-Unicode properties, astral input, replacement and result truncation.
+ECMAScript tests cover its complete exposed flag matrix, internal indices,
+zero-length iteration, captures, lookbehind, backreferences, Unicode properties
+and bounded replacement.
-Actual browser identity is displayed because ECMAScript behavior can evolve
-with the runtime. Syntax acceptance and engine compilation are separate;
-neither silently rewrites the pattern.
+PCRE2 tests cover `g`, `i`, `m`, `s`, `x`, `U`, `J`, mandatory UTF/UCP,
+branch-reset groups, recursion, `\K`, duplicate names, native substitution,
+astral byte/UTF-16 ranges, zero-length iteration, compile errors, match limits
+and result/output truncation. Trace tests cover reported automatic callouts,
+bounded mark copying, native stop at the complete-event cap, compile ranges,
+worker loading and browser rendering.
+
+ECMAScript static findings and generated-input timing observations are
+advisory. They describe only the selected ECMAScript 2025 pattern, browser
+runtime and bounded inputs; they are not proofs of safety or general
+complexity.
+
+ECMAScript generated-case version 1 uses only a complete accepted normalized
+AST, an explicit deterministic seed and bounded sampling strategies. Every
+retained positive or negative label is confirmed by the actual selected
+browser engine. PCRE2 generation is unavailable because its current lexical
+provider is intentionally partial; ECMAScript behavior is never relabelled as
+PCRE2.
+
+The PCRE2 explanation provider deliberately does not invent full structural
+coverage. In particular, branch-reset numbering is omitted from the syntax tree
+while actual capture numbers and names still come from PCRE2 results.
+
+ECMAScript and PCRE2 support exact-request comparison for syntax acceptance,
+engine compilation, matches, captures and engine-native replacement. Results
+align by normalized UTF-16 ranges while retaining each native offset model.
+Timeout, cancellation, rejection and truncation are never presented as
+equivalent results.
+
+Pattern formatting is available only for an exact accepted ECMAScript regexpp
+snapshot. It canonicalizes literal slashes and raw control/line-separator
+representations, then requires paired actual-engine and applicable exact-test
+validation before apply. PCRE2 formatting is unavailable because the current
+provider is deliberately partial; its syntax is never treated as ECMAScript.
+
+PCRE2 is the only current code-generation target. The reviewed C17 generator is
+pinned to the PCRE2 10.47 8-bit API; other language generators remain
+unavailable until they pass equivalent escaping and named-toolchain gates.
diff --git a/docs/FORMATTING.md b/docs/FORMATTING.md
new file mode 100644
index 0000000..9ed2a83
--- /dev/null
+++ b/docs/FORMATTING.md
@@ -0,0 +1,57 @@
+# Pattern formatting
+
+Regex Tools includes a deliberately narrow, local ECMAScript formatter. It
+canonicalizes representations that can be changed without inventing a visual
+layout:
+
+- an unescaped literal `/` becomes `\/`, making the pattern safe to copy into
+ an ECMAScript regex literal;
+- raw NUL and control code units become fixed-width `\xNN` or their canonical
+ short escape (`\t`, `\n`, `\v`, `\f`, `\r`);
+- raw U+2028 and U+2029 line separators become `\u2028` and `\u2029`;
+- legacy escaped raw controls are replaced as one parser-reported literal
+ token, so the formatter does not introduce an extra backslash.
+
+Existing explicit escape spellings, groups, alternatives, quantifiers, inline
+flags and literal whitespace are otherwise preserved byte for byte. ECMAScript
+regex whitespace is normally significant, so the tool does not insert
+indentation, line breaks or comments and does not claim to be a structural
+pretty-printer. The transform is idempotent.
+
+## Eligibility
+
+Formatting requires the exact current, accepted, non-recovered syntax snapshot
+from the bundled `@eslint-community/regexpp` provider. Transformations are
+derived from the provider's literal-token ranges rather than from a heuristic
+text scan. A stale parse, a malformed pattern, a partial provider, PCRE2 or any
+other flavour is reported as unavailable and is never silently rewritten.
+
+## Mandatory validation before apply
+
+The preview cannot be applied directly. Regex Tools first:
+
+1. reparses source and candidate independently and checks capture numbering,
+ names, repetition and parent-capture shape;
+2. compiles and runs the source and candidate in separate actual ECMAScript
+ engine workers against the current subject and replacement;
+3. compares exact normalized match, capture and replacement output, including
+ the engine identity;
+4. reruns every enabled unit test whose flavour, version, pattern, flags and
+ engine options exactly match the active source snapshot, and compares both
+ semantic output and assertion outcome.
+
+A compile rejection, one-sided timeout, crash, worker error, cancellation,
+truncated result, engine-identity change, incomplete test suite or semantic
+difference blocks application. Matching fixed timeouts are comparable only for
+an exact unit-test pair; two timeouts on the current interactive snapshot are
+inconclusive. If no unit test is applicable, the UI says so and still requires
+the current subject/replacement gate.
+
+Validation is bounded to 1,000 tests, the normal worker/input limits and a
+60-second aggregate test wall budget. A test is not started unless its complete
+configured timeout still fits that budget. The user must explicitly confirm
+the exact validated candidate before the Apply button is enabled.
+
+Passing these gates is strong, reproducible evidence for the current snapshot
+and exact tests. It is not a mathematical proof of equivalence for every
+possible subject.
diff --git a/docs/GENERATED_CASES.md b/docs/GENERATED_CASES.md
new file mode 100644
index 0000000..0935037
--- /dev/null
+++ b/docs/GENERATED_CASES.md
@@ -0,0 +1,135 @@
+# Generated test cases
+
+Regex Tools generator version 1 creates deterministic candidate subjects from
+the application-owned normalized AST and then verifies every candidate through
+the selected actual engine adapter. Generation is local-only and bounded. It is
+sampling, not proof of language coverage, equivalence or safety.
+
+## Supported scope
+
+Version 1 is scoped to an accepted ECMAScript 2025 normalized AST from the
+bundled regexpp-based provider. It requests:
+
+- a shortest structural candidate using the shortest visible alternative and
+ each quantifier minimum;
+- each structurally visible alternative;
+- quantifier minimum, adjacent and bounded upper values;
+- optional-absent and optional-present values;
+- bounded character-class, dot and Unicode-property representatives;
+- newline and non-boundary context around anchors;
+- deletion, replacement and boundary mutations as likely near misses;
+- deterministic random alternative, quantifier and class choices.
+
+The coverage report says `covered`, `partial`, `unsupported` or
+`not applicable` for each category. “Covered” means that the documented bounded
+sampling strategy was requested. It does not mean every string in the regular
+language was enumerated.
+
+PCRE2 generation is unavailable in version 1. Its actual engine is complete for
+the advertised execution operations, but its current structural syntax
+provider is intentionally partial. Regex Tools does not feed PCRE2 syntax to
+the ECMAScript generator or relabel ECMAScript behavior.
+
+## Unsupported and partial constructs
+
+The report includes exact normalized source ranges and a reason for every
+recognized gap, capped at 250 rendered entries. Version 1 does not solve:
+
+- capture-dependent backreferences;
+- lookaround or word-boundary constraints symbolically;
+- recursion, subroutine calls or conditional execution;
+- branch-reset and atomic grouping;
+- engine control verbs or callouts;
+- complex Unicode-set algebra and properties of strings;
+- recovered or provider-unsupported syntax.
+
+Surrounding candidates may still happen to satisfy a lookaround or boundary.
+They are retained only when the actual selected engine confirms the requested
+match outcome. That confirmation does not upgrade the generator’s structural
+coverage claim.
+
+## Determinism and seed
+
+The seed is explicit text of 1–128 UTF-16 units. Generator version 1 hashes its
+UTF-16 code units with a documented stable 32-bit FNV-1a step and drives an
+application-owned 32-bit deterministic sequence. It never calls
+`Math.random()`.
+
+The same generator version, normalized AST, flags, settings and seed produce
+the same candidate order and candidate identifiers. Verification timing is
+environment-dependent, and strict wall-time or engine-timeout bounds can
+therefore stop two runs at different points. The UI records both the original
+seed and its eight-digit hash. Generated unit tests retain:
+
+- generator id and version;
+- the original seed;
+- candidate id;
+- intended positive or negative outcome.
+
+That provenance survives project and test-suite JSON export when subjects are
+included. Editing a generated test turns it into a user-authored test and
+removes the generation provenance.
+
+## Actual-engine verification
+
+Candidate synthesis runs in a dedicated killable worker. The verifier then
+sends each exact candidate, pattern, flavour version, flags, engine options and
+scan mode through `EngineSupervisor`, which selects the registered real engine
+worker.
+
+- An intended positive is labelled `should-match` only after the engine returns
+ at least one match.
+- An intended negative is labelled `should-not-match` only after the engine
+ returns no match.
+- A candidate with the opposite result is discarded and retained only as a
+ bounded advanced diagnostic.
+- Engine compile rejection, timeout, worker crash, cancellation and ordinary
+ execution error remain distinct outcomes.
+- Cancellation terminates both candidate-synthesis and active engine workers.
+- Configuration changes invalidate results immediately; old cases are never
+ displayed against a new pattern or engine configuration.
+
+The engine name, version, adapter, runtime and native offset unit shown with the
+result come from the actual execution result, not static marketing metadata.
+
+## Bounds
+
+Defaults:
+
+| Bound | Default |
+| -------------------------------------- | ------------------------: |
+| Retained verified cases | 48 |
+| Candidate attempts | 192 |
+| Seeded random variants | 16 |
+| Per-case engine timeout | 500 ms |
+| Aggregate wall time | 15 s |
+| Bytes per candidate subject | 64 KiB |
+| Aggregate candidate subject bytes | 1 MiB |
+| Synthesized repetitions per quantifier | 32 |
+| Normalized AST traversal | 20,000 nodes / 512 levels |
+
+Hard caps:
+
+| Bound | Hard cap |
+| -------------------------------------- | -------: |
+| Retained verified cases | 1,000 |
+| Candidate attempts | 4,000 |
+| Per-case engine timeout | 10 s |
+| Aggregate wall time | 30 s |
+| Bytes per candidate subject | 16 MiB |
+| Aggregate candidate subject bytes | 16 MiB |
+| Synthesized repetitions per quantifier | 1,024 |
+| Advanced discard diagnostics | 250 |
+
+UTF-8 sizes are checked before a candidate is retained for verification.
+Quantifier expansion is preflighted before `String.prototype.repeat` allocates
+the output. Reaching a case, attempt, byte, AST, repetition or wall-time bound
+is visible in the result; it is never reported as complete coverage.
+
+## Privacy and persistence
+
+No pattern, candidate, verified subject, diagnostic or seed is uploaded.
+Results and settings are ephemeral until the user explicitly adds cases to the
+unit-test suite or downloads the verified JSON artifact. Test-suite export
+omits subjects by default because generated subjects may still encode literal
+pattern content. Enabling subject export is an explicit user choice.
diff --git a/docs/MINIMIZATION.md b/docs/MINIMIZATION.md
new file mode 100644
index 0000000..656dc5d
--- /dev/null
+++ b/docs/MINIMIZATION.md
@@ -0,0 +1,71 @@
+# Subject minimization
+
+The **Minimize** workspace reduces one failing subject locally against the
+selected real ECMAScript or PCRE2 worker. It can preserve:
+
+- the exact failure class of any supported saved unit-test assertion;
+- an identified capture's wrong participation, value or UTF-16 range;
+- an engine worker timeout at one fixed deadline;
+- an ECMAScript/PCRE2 semantic mismatch with the same mismatch-kind set; or
+- the same one-sided comparison timeout while the other engine completes with
+ authoritative, untruncated output.
+
+The fixed pattern, ordered flags, options, replacement and engine versions form
+the target identity. Syntax is parsed once to obtain capture metadata. Every
+accepted subject candidate is then verified through the selected execution
+worker; the reducer does not contain a substitute regex implementation.
+
+## Deterministic transforms
+
+Subjects must be well-formed Unicode. The reducer never splits a surrogate
+pair and rejects input containing a lone surrogate.
+
+Reduction order is stable:
+
+1. contiguous ddmin chunk deletion, left to right;
+2. single Unicode-scalar deletion, left to right; then
+3. lower-rank replacement by `a`, `0`, space, newline and `!`, in that order.
+
+The last two passes restart after every accepted change and continue to a fixed
+point. A complete pass with no accepted candidate establishes only **local
+minimality under those transforms**. It is not a proof of the shortest,
+globally minimal or semantically simplest reproducer. Evaluation or wall-budget
+exhaustion, an inconclusive candidate, cancellation and worker failure always
+return a result without a minimality claim.
+
+## Exact predicates
+
+A baseline run must reproduce the requested failure before reduction starts.
+Unit-test failures retain their assertion class; a wrong capture value cannot
+turn into a missing capture, and a semantic comparison retains the exact sorted
+set of baseline mismatch kinds. Rejected patterns and truncated match,
+capture or replacement data are not accepted as semantic evidence.
+
+Timeout is a worker deadline outcome, never a no-match. An explicit timeout
+predicate accepts only `WorkerRequestError("timeout")` at the configured fixed
+deadline. Comparison timeout reduction additionally requires the same flavour
+to time out and the peer flavour to complete with the same engine identity and
+complete data. Crash, unreadable worker response, ordinary worker error and
+cancellation have separate statuses. Crash minimization is unsupported and a
+crash never satisfies a timeout predicate.
+
+For a `must-time-out` unit-test assertion, an ordinary completion is the test
+failure being minimized; a timeout makes that assertion pass. The separate
+**Exact engine timeout** target is used to minimize an observed timeout.
+
+## Bounds and progress
+
+- Subject: 1 MiB UTF-8, separate from the larger interactive editor cap.
+- Candidate evaluations: at most 2,000, including the baseline.
+- Aggregate syntax-setup and reduction wall time: at most 60 seconds.
+- Candidate worker deadline: fixed for the complete run, at most 10 seconds.
+- Retained accepted-transform history: 200 entries; totals remain exact.
+
+A candidate starts only when its full fixed deadline fits inside the remaining
+aggregate wall budget. This prevents the aggregate deadline from being
+misreported as an engine timeout. Progress exposes phase, evaluation and wall
+budgets, accepted reductions, current scalar/byte sizes and the latest
+observation. Cancellation terminates syntax and execution workers.
+
+The minimized subject can be copied or deliberately applied to the main editor.
+Inputs and results stay in memory and are neither uploaded nor persisted.
diff --git a/docs/PARSER_PROFILES.md b/docs/PARSER_PROFILES.md
index b92f89c..a00cfde 100644
--- a/docs/PARSER_PROFILES.md
+++ b/docs/PARSER_PROFILES.md
@@ -7,10 +7,13 @@ community
```
It uses only reviewed open-source dependencies and builds without private
-registry credentials. Version 0.1.0 parses ECMAScript with regexpp 4.12.2.
+registry credentials. ECMAScript uses regexpp 4.12.2. PCRE2 uses an
+application-owned lexical provider for partial explanations and the actual
+bundled PCRE2 compiler for authoritative acceptance.
The product deliberately does not implement or reference an `@r101/parser`
profile. No commercial package alias, import, tarball, credential, licence key
or private CI path exists. Additional flavour syntax will be implemented
incrementally as open-source providers behind the same application-owned
-interface.
+interface. Partial coverage is exposed as metadata and never presented as a
+complete parser.
diff --git a/docs/PCRE2_NEXT_SLICES.md b/docs/PCRE2_NEXT_SLICES.md
new file mode 100644
index 0000000..3258385
--- /dev/null
+++ b/docs/PCRE2_NEXT_SLICES.md
@@ -0,0 +1,66 @@
+# PCRE2 advanced slices
+
+This document records the reviewed boundaries for advanced PCRE2 work. A
+feature is exposed only after its complete runtime and UI path passes the
+stated gates.
+
+## Automatic-callout trace — implemented
+
+ABI version 3 and the dedicated trace worker now provide:
+
+- a new ABI operation separate from normal matching, compiled with
+ `PCRE2_AUTO_CALLOUT`;
+- a fixed copied event record containing only reported callout number, pattern
+ byte position, next-item length, subject byte position, capture top/last and
+ bounded mark text;
+- caller-supplied event and text buffers capped by both
+ `maximumTraceEvents` and `maximumTraceBytes`;
+- explicit `eventsTruncated`, native status and last-complete-event fields;
+- the existing match/depth/heap limits and supervisor timeout;
+- cleanup identical to normal ABI calls.
+
+Tracing runs in its own worker and never replaces the normal result used by
+Match, Replace, Tests, Corpus or Analysis. “Forward”, “same position” and
+“apparent backtrack” are derived from adjacent reported positions only and
+carry derived provenance. The viewer explicitly states that a callout stream
+is not a complete account of every internal engine action.
+
+## ECMAScript versus PCRE2 comparison — implemented
+
+The comparison orchestrator issues two explicit requests with
+`Promise.allSettled` through independently killable flavour workers. It does
+not translate flags, patterns or replacement syntax.
+
+The comparison DTO retains:
+
+- each exact request identity, engine identity and independently complete or
+ timed-out result;
+- match/capture differences aligned by editor UTF-16 ranges, while retaining
+ each native offset model;
+- explicit “not comparable” states for syntax/compile failure, truncation,
+ timeout, cancellation, worker failure or flavour-only constructs;
+- a shared wall-clock cancellation action that terminates both workers.
+
+The orchestrator remains outside `EngineSupervisor`, preserving the
+single-engine lifecycle boundary. Shared patterns are sent unchanged and
+explicit per-flavour variants remain user-owned. Agreement is labelled only as
+evidence for the current subject.
+
+## PCRE2 C code generation — implemented
+
+The first reviewed target consumes a validated request snapshot and uses an
+application-owned C17 template. It:
+
+- identifies exact PCRE2 10.47, 8-bit, UTF/UCP semantics and selected flags;
+- emits pattern, subject and replacement independently as exact UTF-8 byte
+ arrays;
+- includes match/depth/heap limits and error handling in generated C examples;
+- states when a target binding cannot express an application-level `g` flag or
+ a Regex Tools output/result cap directly;
+- ships deterministic golden match/replacement fixtures that compile and
+ execute against official PCRE2 10.47 before the target is advertised.
+
+No generated snippet may be used as the implementation of the browser runtime.
+Other target languages remain unavailable until their own escaping and
+toolchain gates pass. See
+[`COMPARISON_AND_CODEGEN.md`](COMPARISON_AND_CODEGEN.md).
diff --git a/docs/PERFORMANCE.md b/docs/PERFORMANCE.md
index cbf1776..a8a56e6 100644
--- a/docs/PERFORMANCE.md
+++ b/docs/PERFORMANCE.md
@@ -10,6 +10,18 @@ browser, engine version, pattern and subject. The UI therefore reports observed
elapsed time and exact runtime identity but makes no universal throughput or
safety claim.
+PCRE2 cold execution includes loading and instantiating the self-hosted
+WebAssembly module inside its worker; warm requests reuse that instance.
+Every request still compiles and frees its pattern so no unbounded compiled-code
+cache can accumulate. Native match-step, depth and heap limits complement the
+supervisor wall clock, and cancellation discards the complete worker heap.
+
+Automatic-callout tracing is deliberately excluded from every benchmark. It
+compiles an independently instrumented PCRE2 pattern in a separate worker and
+can materially change runtime work. The trace path retains at most 50,000
+complete 32-byte event records and 10 MiB total serialized event/mark bytes;
+the viewer renders at most 2,000 retained events.
+
The production bundle separates syntax and execution workers. Interactive
syntax/extraction trees and editor decoration sets render at most 2,000 items
each and report both the rendered and actual totals. These are presentation
@@ -62,5 +74,57 @@ representation against a byte budget before materializing or writing it.
IndexedDB schema version 2 indexes `updatedAt`; latest-project loading opens one
descending cursor and validates only that record instead of calling `getAll()`.
-Formal cold/warm benchmarks, p95 statistics and growth charts are deferred to
-the analysis milestone and will run in killable workers.
+Corpus runs accept at most 256 documents, 16 MiB per document and 256 MiB of
+aggregate UTF-8 input. Jobs run sequentially so only one engine request is
+active at a time. A batch retains at most 100,000 matches and 64 MiB of applied
+output, and stops after five minutes of aggregate wall time. Each document also
+uses the selected manual timeout. Progress is reported after every document;
+cancellation kills the current worker and labels all remaining documents.
+Result tables contain summaries, not match/capture objects. ZIP export is
+available only for an all-exact apply batch and is created asynchronously.
+
+Independent-line mode preflights at most 100,000 logical lines and processes
+each as an isolated engine subject while preserving line separators for exact
+apply output. Per document, corpus extraction retains at most 32 group summaries
+(including the full match), three 160-UTF-16-unit samples per group and 100
+unique diagnostics. Output previews show 2,000 units. Content-free summary
+exports are capped at 4 MiB; the applied archive is capped at 68 MiB in addition
+to the 64 MiB uncompressed-output budget.
+
+ECMAScript analysis reports cold compilation separately from warm native
+execution. The default benchmark uses three warm-up and 15 measured samples;
+user caps are 100 warm-ups and 1,000 samples. It reports median and
+nearest-rank p95 values plus separate timing and outcome/growth plots with the
+exact runtime identity. Each sample defaults to a 2-second killable-worker
+limit (10-second cap), and the aggregate analysis wall time is 30 seconds.
+
+Growth probes retain at most 24 steps, generate at most 1 MiB by default under
+the central 16 MiB subject cap, and reject more than 10,000,000 repetitions
+before string allocation. Replacement timing is skipped when match collection
+is truncated or the estimated output exceeds 8 Mi UTF-16 units. Results are
+observations for the chosen runtime and generated inputs, never a general
+complexity proof. See [`ANALYSIS.md`](ANALYSIS.md).
+
+Generated-case synthesis defaults to 48 retained cases from at most 192
+candidates, 16 seeded-random variants, 64 KiB per candidate, 1 MiB aggregate
+candidate bytes and 15 seconds of aggregate wall time. Every candidate then
+uses a separate selected-engine request with a 500 ms default timeout.
+Generation preflights UTF-8 size and repetition bounds, caps normalized-AST
+traversal at 20,000 nodes / 512 levels and retains only bounded diagnostics.
+Hard caps are documented in [`GENERATED_CASES.md`](GENERATED_CASES.md).
+
+Subject minimization accepts at most 1 MiB UTF-8, performs at most 2,000 real
+engine candidate evaluations and stops after at most 60 seconds including
+syntax setup. One fixed worker timeout of at most 10 seconds applies to every
+candidate. A candidate starts only when that complete deadline fits in the
+remaining aggregate budget. Accepted-transform history retains 200 entries;
+evaluation and accepted totals remain exact. See
+[`MINIMIZATION.md`](MINIMIZATION.md).
+
+Formatter validation uses two syntax supervisors and two actual-engine
+supervisors so source/candidate outcomes cannot share compiled state. The
+current subject/replacement snapshot always runs; at most 1,000 enabled tests
+with the exact active identity are replayed. Test validation has a 60-second
+aggregate wall budget, and a test starts only when its complete configured
+worker timeout still fits. Preview details retain at most 10,000 transforms and
+the UI renders the first 200. See [`FORMATTING.md`](FORMATTING.md).
diff --git a/docs/PORTAL_REQUIREMENTS.md b/docs/PORTAL_REQUIREMENTS.md
index c6f2530..c519ad1 100644
--- a/docs/PORTAL_REQUIREMENTS.md
+++ b/docs/PORTAL_REQUIREMENTS.md
@@ -1,6 +1,6 @@
# Portal requirements
-Regex Tools 0.1.0 is a static nested-path-safe application.
+Regex Tools is a static nested-path-safe application.
Required delivery behavior:
@@ -9,11 +9,17 @@ Required delivery behavior:
revalidation/no-cache resources.
- Hashed Vite assets may use immutable long-term caching.
- CSP must allow `worker-src 'self' blob:`.
-- No WebAssembly MIME rule or `wasm-unsafe-eval` is required by v0.1.0.
+- `engines/pcre2/pcre2.wasm` must use `application/wasm`.
+- `script-src` must permit self-hosted WebAssembly compilation, normally with
+ `'wasm-unsafe-eval'`; broad `'unsafe-eval'` is neither needed nor recommended.
+- `engines/pcre2/pcre2.mjs`, metadata, checksums, licence and WASM should use
+ revalidation or a release-scoped immutable cache. They are not
+ content-hashed filenames and must never drift under one release identity.
The Portal must consume the immutable release ZIP, verify its SHA-256, verify
-manifest ID `de.add-ideas.regex-tools` and version `0.1.0`, and mount it at a
+manifest ID `de.add-ideas.regex-tools` and version `0.2.0`, and mount it at a
relative target such as `apps/regex/`.
-When PCRE2 ships later, the smallest additional change will be
-`application/wasm` delivery and `script-src 'self' 'wasm-unsafe-eval'`.
+The package, manifest, tag, artifact URL and checksum must all use the v0.2.0
+identity. The historical v0.1.0 artifact coordinates remain immutable and must
+not be overwritten or reused.
diff --git a/docs/REFERENCE_IMPLEMENTATIONS.md b/docs/REFERENCE_IMPLEMENTATIONS.md
index 8883c03..ab5fba8 100644
--- a/docs/REFERENCE_IMPLEMENTATIONS.md
+++ b/docs/REFERENCE_IMPLEMENTATIONS.md
@@ -8,7 +8,7 @@ Inspection date: 2026-07-24.
| Toolbox SDK | `53c40a61ba1581246f65773fcbb1c1cfd31ac98e` / 0.2.2 | Apache-2.0 | Contract, shell and release conventions; package APIs used, no source adapted. |
| Toolbox Portal | `a9c31c8986c40a0097966318e925083302e91e13` / 0.5.0 | AGPL-3.0-only | Assembly/lock conventions inspected; no source copied into the app. |
| regexpp | 4.12.2; npm integrity `sha512-EriSTlt5OC9/7SXkRSCAhfSxxoSUgBm33OH+IkwbdpgoqsSsUg7y3uh+IICI/Qg4BBWr3U2i39RpmycbxMq4ew==` | MIT | Pinned ECMAScript syntax provider. Provider AST is normalized in a worker. |
-| PCRE2 | signed tag 10.47; commit `f454e231fe5006dd7ff8f4693fd2b8eb94333429` | BSD-3-Clause WITH PCRE2-exception | External technical spike only; not shipped or copied. |
+| PCRE2 | signed tag 10.47; commit `f454e231fe5006dd7ff8f4693fd2b8eb94333429` | BSD-3-Clause WITH PCRE2-exception | Pinned offline build; generated verified WebAssembly pack is shipped, source is not vendored. |
| ECMAScript specification | ; living specification inspected 2026-07-24 | ECMA terms | Semantics reference; no prose copied. |
| regex101 | and public parser documentation | Product/commercial terms | Product reference only. No source, branding, explanations, reference prose or visual design copied. |
| `recheck` | 4.5.0 | MIT | Metadata inspected; not installed or shipped. |
@@ -17,5 +17,5 @@ Inspection date: 2026-07-24.
| RegexLib | Website inspected | Redistribution licence not established | No fixture copied. |
| RGXP.RU | Website inspected | Redistribution licence not established | No fixture copied. |
-All v0.1.0 conformance cases are project-authored. See
+All v0.1.0 and v0.2.0 conformance cases are project-authored. See
`tests/fixtures/README.md`.
diff --git a/docs/RELEASE.md b/docs/RELEASE.md
index 7fa1c86..194ecba 100644
--- a/docs/RELEASE.md
+++ b/docs/RELEASE.md
@@ -18,6 +18,9 @@ on publication failure, replaces an existing exact version only with
`--force`, and produces:
```text
-release/regex-tools-0.1.0.zip
-release/regex-tools-0.1.0.zip.sha256
+release/regex-tools-0.2.0.zip
+release/regex-tools-0.2.0.zip.sha256
```
+
+The existing `release/regex-tools-0.1.0.zip` and matching sidecar are historical
+immutable artifacts. Creating v0.2.0 must not replace, rename or remove them.
diff --git a/docs/SECURITY.md b/docs/SECURITY.md
index 9b2efd0..77ec32b 100644
--- a/docs/SECURITY.md
+++ b/docs/SECURITY.md
@@ -9,6 +9,16 @@ untrusted.
- A terminated worker is never reused.
- Pattern, subject, replacement/list template, match, capture and output sizes
are bounded.
+- PCRE2 adds native match-step, depth and heap caps. Its C ABI returns only
+ caller-owned bounded records and releases code, match data and match context
+ on every exit path.
+- PCRE2 automatic-callout tracing is a separate ABI and worker path. It copies
+ only complete fixed records and bounded mark text, stops matching at the
+ 50,000-event or 10 MiB serialized trace cap, and is killed on timeout,
+ cancellation or crash.
+- PCRE2 runs only in mandatory UTF/UCP mode. Exact UTF-8 byte boundaries are
+ normalized to editor UTF-16 offsets; lone surrogates are rejected instead of
+ being silently replaced by browser encoding.
- Replacement templates are strings; executable callbacks are unavailable.
- Replacement/list token decorations, rows, chips, diagnostics and list
evaluation work have separate presentation/operation caps; incomplete list
@@ -25,14 +35,51 @@ untrusted.
- Unit-test assertion diagnostics never stringify complete large values and
retain at most 2 KiB UTF-8 per case. Oversized replacement output is not
copied into an exact current-result draft.
+- Corpus file input is local, recognized UTF-8 text only. It is bounded before
+ decoding, rejects NUL/binary input, runs sequentially in a dedicated
+ cancellable supervisor and never modifies source files.
+- Corpus JSON/CSV/NDJSON summaries never contain source, capture-sample or
+ applied-output content. CSV cells are formula-injection protected. Local UI
+ previews are explicitly bounded. Applied files require an explicit
+ content-export acknowledgement; names are normalized against archive path
+ traversal and collisions. Incomplete applied output cannot be downloaded or
+ included in the all-document ZIP, whose bytes have a separate hard limit.
+- Corpus documents, contents, outputs and results are never placed in project
+ JSON or IndexedDB. Schema-v1 imports that attempt to inject corpus payloads
+ are rejected.
+- ECMAScript growth analysis preflights repetition count and UTF-8 bytes before
+ allocating a generated string. The analysis worker is terminated on timeout,
+ cancellation or crash; replacement timing also preflights its output bound.
+ Analysis settings and results are neither persisted nor uploaded.
+- Generated-case synthesis runs in its own killable worker under AST, depth,
+ attempt, repetition and UTF-8 byte caps. Labels come only from bounded
+ requests to the selected real engine worker; a wrong outcome, timeout, crash,
+ cancellation or compile rejection is never converted into a verified case.
+ Settings, discarded candidates and unselected results are ephemeral and
+ never uploaded.
+- Imported generated-test provenance is accepted only for the exact supported
+ generator id/version and must agree with the test assertion outcome. Unknown
+ provenance fields and future versions are rejected; editing a case removes
+ its provenance.
+- ECMAScript formatting consumes only exact accepted regexpp literal-token
+ ranges and never treats PCRE2 as ECMAScript. Source/candidate syntax and
+ execution use separate killable workers; compile failure, truncation,
+ timeout asymmetry, engine-identity drift, incomplete tests or any semantic
+ difference fails closed. Preview/results are ephemeral and apply requires an
+ explicit confirmation.
+- Subject minimization rejects lone surrogates, caps input at 1 MiB UTF-8,
+ evaluation count at 2,000 and aggregate wall time at 60 seconds. Each
+ predicate check runs in the selected killable engine worker. Timeout, crash,
+ cancellation and worker error remain distinct; truncated results fail
+ closed. Subjects and results are ephemeral and never uploaded or persisted.
- No backend, uploads, telemetry, remote corpus URL, CDN, remote font, `eval`,
`Function`, arbitrary command line or unsafe HTML rendering exists.
-Version 0.1.0 needs a CSP equivalent to:
+The PCRE2-enabled build needs a CSP equivalent to:
```text
default-src 'self';
-script-src 'self';
+script-src 'self' 'wasm-unsafe-eval';
worker-src 'self' blob:;
connect-src 'self';
img-src 'self' data:;
@@ -44,8 +91,9 @@ base-uri 'none';
form-action 'none';
```
-The Toolbox shell currently requires inline styles. No `unsafe-eval` or
-`wasm-unsafe-eval` is required until an actual WebAssembly flavour ships.
+The Toolbox shell currently requires inline styles. Corpus ZIP generation and
+PCRE2 use self-hosted modules and local workers under the existing
+`worker-src` policy. Broad `unsafe-eval` is not required.
Report vulnerabilities privately to the repository owner before public issue
details are posted.
diff --git a/engines/pcre2/CMakeLists.txt b/engines/pcre2/CMakeLists.txt
new file mode 100644
index 0000000..03d7312
--- /dev/null
+++ b/engines/pcre2/CMakeLists.txt
@@ -0,0 +1,71 @@
+# SPDX-License-Identifier: GPL-3.0-or-later
+
+cmake_minimum_required(VERSION 3.25)
+project(regex_pcre2_bridge LANGUAGES C)
+
+if(NOT EMSCRIPTEN)
+ message(FATAL_ERROR "The staged PCRE2 pack must be built with Emscripten.")
+endif()
+
+if(NOT DEFINED PCRE2_SOURCE_DIR OR NOT IS_DIRECTORY "${PCRE2_SOURCE_DIR}")
+ message(FATAL_ERROR "PCRE2_SOURCE_DIR must identify the verified PCRE2 source checkout.")
+endif()
+
+if(NOT DEFINED REGEX_PCRE2_OUTPUT_DIR)
+ message(FATAL_ERROR "REGEX_PCRE2_OUTPUT_DIR must identify the private staging directory.")
+endif()
+
+set(BUILD_SHARED_LIBS OFF CACHE BOOL "" FORCE)
+set(BUILD_STATIC_LIBS ON CACHE BOOL "" FORCE)
+set(PCRE2_BUILD_PCRE2_8 ON CACHE BOOL "" FORCE)
+set(PCRE2_BUILD_PCRE2_16 OFF CACHE BOOL "" FORCE)
+set(PCRE2_BUILD_PCRE2_32 OFF CACHE BOOL "" FORCE)
+set(PCRE2_BUILD_PCRE2GREP OFF CACHE BOOL "" FORCE)
+set(PCRE2_BUILD_TESTS OFF CACHE BOOL "" FORCE)
+set(PCRE2_DEBUG OFF CACHE STRING "" FORCE)
+set(PCRE2_REBUILD_CHARTABLES OFF CACHE BOOL "" FORCE)
+set(PCRE2_SHOW_REPORT OFF CACHE BOOL "" FORCE)
+set(PCRE2_STATIC_PIC OFF CACHE BOOL "" FORCE)
+set(PCRE2_SUPPORT_JIT OFF CACHE BOOL "" FORCE)
+set(PCRE2_SUPPORT_UNICODE ON CACHE BOOL "" FORCE)
+
+add_subdirectory("${PCRE2_SOURCE_DIR}" pcre2-source EXCLUDE_FROM_ALL)
+
+add_executable(regex-pcre2-pack pcre2_bridge.c)
+target_compile_features(regex-pcre2-pack PRIVATE c_std_11)
+target_compile_definitions(regex-pcre2-pack PRIVATE PCRE2_CODE_UNIT_WIDTH=8)
+target_compile_options(
+ regex-pcre2-pack
+ PRIVATE
+ -O3
+ -fno-ident
+ -fvisibility=hidden
+)
+target_link_libraries(regex-pcre2-pack PRIVATE pcre2-8-static)
+target_link_options(
+ regex-pcre2-pack
+ PRIVATE
+ -O3
+ --no-entry
+ "SHELL:-s ALLOW_MEMORY_GROWTH=1"
+ "SHELL:-s ASSERTIONS=0"
+ "SHELL:-s ENVIRONMENT=web,worker,node"
+ "SHELL:-s EXPORT_ES6=1"
+ "SHELL:-s EXPORT_NAME=createRegexPcre2"
+ "SHELL:-s EXPORTED_FUNCTIONS=['_regex_pcre2_bridge_abi_version','_regex_pcre2_compile_probe','_regex_pcre2_config_flags','_regex_pcre2_error_message','_regex_pcre2_execute','_regex_pcre2_self_test','_regex_pcre2_substitute','_regex_pcre2_trace','_regex_pcre2_version','_malloc','_free']"
+ "SHELL:-s EXPORTED_RUNTIME_METHODS=['HEAPU8','UTF8ToString']"
+ "SHELL:-s FILESYSTEM=0"
+ "SHELL:-s INITIAL_MEMORY=16777216"
+ "SHELL:-s MALLOC=emmalloc"
+ "SHELL:-s MAXIMUM_MEMORY=268435456"
+ "SHELL:-s MODULARIZE=1"
+ "SHELL:-s NO_EXIT_RUNTIME=1"
+ "SHELL:-s STRICT=1"
+)
+set_target_properties(
+ regex-pcre2-pack
+ PROPERTIES
+ OUTPUT_NAME pcre2
+ RUNTIME_OUTPUT_DIRECTORY "${REGEX_PCRE2_OUTPUT_DIR}"
+ SUFFIX ".mjs"
+)
diff --git a/engines/pcre2/README.md b/engines/pcre2/README.md
new file mode 100644
index 0000000..78e0022
--- /dev/null
+++ b/engines/pcre2/README.md
@@ -0,0 +1,27 @@
+# PCRE2 engine bridge
+
+This directory contains the GPL-3.0-or-later application bridge and
+deterministic build description for the bundled PCRE2 flavour. It deliberately
+does not vendor the upstream PCRE2 source tree or generated binaries.
+
+ABI version 3 exposes:
+
+- exact engine identity and configuration;
+- compile diagnostics with UTF-8 byte offsets;
+- bounded non-overlapping matching with copied group-zero/capture records and
+ copied capture-name metadata;
+- bounded native `pcre2_substitute()` semantics over the same retained match
+ sequence;
+- bounded automatic-callout tracing from one native `pcre2_match()` invocation,
+ with copied event and mark data and an immediate native stop at either trace
+ cap;
+- match-step, depth, heap, match-count, capture-row and output limits;
+- no native pointer or compiled-code handle across a call boundary.
+
+All PCRE2 allocations are released before an ABI call returns. JavaScript owns
+and frees every input/result buffer in `finally`, and the containing worker is
+terminated on timeout, crash, cancellation or supersession.
+
+The explicit offline source/toolchain identity, build, verification and local
+asset-install workflow is documented in
+[`docs/ENGINE_BUILDS.md`](../../docs/ENGINE_BUILDS.md).
diff --git a/engines/pcre2/pcre2_bridge.c b/engines/pcre2/pcre2_bridge.c
new file mode 100644
index 0000000..f5595da
--- /dev/null
+++ b/engines/pcre2/pcre2_bridge.c
@@ -0,0 +1,910 @@
+/* SPDX-License-Identifier: GPL-3.0-or-later */
+
+#define PCRE2_CODE_UNIT_WIDTH 8
+
+#include "pcre2_bridge.h"
+
+#include
+#include
+#include
+#include
+
+#define REGEX_PCRE2_COMPILE_FLAGS \
+ (REGEX_PCRE2_CASELESS | REGEX_PCRE2_MULTILINE | REGEX_PCRE2_DOTALL | \
+ REGEX_PCRE2_EXTENDED | REGEX_PCRE2_UNGREEDY | REGEX_PCRE2_UTF | \
+ REGEX_PCRE2_UCP | REGEX_PCRE2_DUPNAMES)
+#define REGEX_PCRE2_ALL_FLAGS \
+ (REGEX_PCRE2_COMPILE_FLAGS | REGEX_PCRE2_GLOBAL)
+
+_Static_assert(sizeof(regex_pcre2_compile_result) == 20,
+ "Compile result is part of ABI version 3.");
+_Static_assert(sizeof(regex_pcre2_limits) == 20,
+ "Limits record is part of ABI version 3.");
+_Static_assert(sizeof(regex_pcre2_match_record) == 16,
+ "Match record is part of ABI version 3.");
+_Static_assert(sizeof(regex_pcre2_name_record) == 12,
+ "Name record is part of ABI version 3.");
+_Static_assert(sizeof(regex_pcre2_run_result) == 60,
+ "Run result is part of ABI version 3.");
+_Static_assert(sizeof(regex_pcre2_trace_limits) == 20,
+ "Trace limits are part of ABI version 3.");
+_Static_assert(sizeof(regex_pcre2_trace_event) == 32,
+ "Trace event is part of ABI version 3.");
+_Static_assert(sizeof(regex_pcre2_trace_result) == 52,
+ "Trace result is part of ABI version 3.");
+
+typedef struct regex_pcre2_run_buffers {
+ regex_pcre2_match_record *records;
+ uint32_t record_capacity;
+ regex_pcre2_name_record *name_records;
+ uint32_t name_record_capacity;
+ uint8_t *name_bytes;
+ uint32_t name_bytes_capacity;
+ uint8_t *output;
+ uint32_t output_capacity;
+} regex_pcre2_run_buffers;
+
+typedef struct regex_pcre2_trace_state {
+ regex_pcre2_trace_event *events;
+ uint32_t event_capacity;
+ uint8_t *mark_bytes;
+ uint32_t mark_bytes_capacity;
+ uint32_t maximum_events;
+ uint32_t maximum_trace_bytes;
+ uint32_t event_count;
+ uint32_t total_event_count;
+ uint32_t mark_bytes_length;
+ uint32_t events_truncated;
+ uint32_t marks_truncated;
+} regex_pcre2_trace_state;
+
+static const uint8_t empty_bytes[1] = {0};
+
+static uint32_t to_pcre2_options(uint32_t application_flags) {
+ uint32_t options = 0;
+ if ((application_flags & REGEX_PCRE2_CASELESS) != 0) {
+ options |= PCRE2_CASELESS;
+ }
+ if ((application_flags & REGEX_PCRE2_MULTILINE) != 0) {
+ options |= PCRE2_MULTILINE;
+ }
+ if ((application_flags & REGEX_PCRE2_DOTALL) != 0) {
+ options |= PCRE2_DOTALL;
+ }
+ if ((application_flags & REGEX_PCRE2_EXTENDED) != 0) {
+ options |= PCRE2_EXTENDED;
+ }
+ if ((application_flags & REGEX_PCRE2_UNGREEDY) != 0) {
+ options |= PCRE2_UNGREEDY;
+ }
+ if ((application_flags & REGEX_PCRE2_UTF) != 0) {
+ options |= PCRE2_UTF;
+ }
+ if ((application_flags & REGEX_PCRE2_UCP) != 0) {
+ options |= PCRE2_UCP;
+ }
+ if ((application_flags & REGEX_PCRE2_DUPNAMES) != 0) {
+ options |= PCRE2_DUPNAMES;
+ }
+ return options;
+}
+
+static int32_t fail_compile(regex_pcre2_compile_result *result,
+ int32_t status) {
+ if (result != NULL) {
+ memset(result, 0, sizeof(*result));
+ result->status = status;
+ }
+ return status;
+}
+
+static int32_t fail_run(regex_pcre2_run_result *result,
+ int32_t status,
+ uint32_t phase,
+ uint32_t offset) {
+ if (result != NULL) {
+ memset(result, 0, sizeof(*result));
+ result->status = status;
+ result->error_phase = phase;
+ result->error_offset = offset;
+ }
+ return status;
+}
+
+static int32_t fail_trace(regex_pcre2_trace_result *result,
+ int32_t status,
+ uint32_t phase,
+ uint32_t offset) {
+ if (result != NULL) {
+ memset(result, 0, sizeof(*result));
+ result->status = status;
+ result->error_phase = phase;
+ result->error_offset = offset;
+ result->last_complete_event = UINT32_MAX;
+ }
+ return status;
+}
+
+static int input_pointer_valid(const uint8_t *value, uint32_t length) {
+ return value != NULL || length == 0;
+}
+
+static const uint8_t *nonnull_input(const uint8_t *value) {
+ return value == NULL ? empty_bytes : value;
+}
+
+static int limits_valid(const regex_pcre2_limits *limits) {
+ return limits != NULL && limits->maximum_matches > 0 &&
+ limits->maximum_matches <= REGEX_PCRE2_MAX_MATCHES &&
+ limits->maximum_capture_rows > 0 &&
+ limits->maximum_capture_rows <= REGEX_PCRE2_MAX_CAPTURE_ROWS &&
+ limits->match_limit > 0 &&
+ limits->match_limit <= REGEX_PCRE2_MAX_MATCH_LIMIT &&
+ limits->depth_limit > 0 &&
+ limits->depth_limit <= REGEX_PCRE2_MAX_DEPTH_LIMIT &&
+ limits->heap_limit_kib > 0 &&
+ limits->heap_limit_kib <= REGEX_PCRE2_MAX_HEAP_LIMIT_KIB;
+}
+
+static int buffers_valid(const regex_pcre2_run_buffers *buffers,
+ int substitution) {
+ if (buffers == NULL ||
+ (buffers->records == NULL && buffers->record_capacity != 0) ||
+ (buffers->name_records == NULL &&
+ buffers->name_record_capacity != 0) ||
+ (buffers->name_bytes == NULL && buffers->name_bytes_capacity != 0)) {
+ return 0;
+ }
+ /*
+ * PCRE2 writes a trailing NUL even when the semantic output capacity is
+ * zero, so the bridge contract always requires the advertised +1 byte.
+ */
+ if (substitution && buffers->output == NULL) {
+ return 0;
+ }
+ return 1;
+}
+
+static uint32_t bounded_offset(PCRE2_SIZE value) {
+ return value > UINT32_MAX ? UINT32_MAX : (uint32_t)value;
+}
+
+static int32_t copy_names(const pcre2_code *code,
+ regex_pcre2_run_buffers *buffers,
+ regex_pcre2_run_result *result) {
+ uint32_t entry_size = 0;
+ PCRE2_SPTR table = NULL;
+ uint32_t index = 0;
+ uint32_t name_offset = 0;
+
+ if (result->name_count == 0) {
+ return 0;
+ }
+ if (pcre2_pattern_info(code, PCRE2_INFO_NAMEENTRYSIZE, &entry_size) != 0 ||
+ pcre2_pattern_info(code, PCRE2_INFO_NAMETABLE, &table) != 0 ||
+ entry_size < 3 || table == NULL) {
+ return REGEX_PCRE2_ERROR_CONFIG;
+ }
+ for (index = 0; index < result->name_count; index += 1) {
+ const uint8_t *entry = table + (index * entry_size);
+ uint32_t group_number =
+ ((uint32_t)entry[0] << 8) | (uint32_t)entry[1];
+ uint32_t maximum_name = entry_size - 2;
+ uint32_t name_length = 0;
+ while (name_length < maximum_name && entry[2 + name_length] != 0) {
+ name_length += 1;
+ }
+ if (result->name_record_count >= buffers->name_record_capacity ||
+ name_length > buffers->name_bytes_capacity - name_offset) {
+ result->names_truncated = 1;
+ break;
+ }
+ buffers->name_records[result->name_record_count].group_number =
+ group_number;
+ buffers->name_records[result->name_record_count].name_offset =
+ name_offset;
+ buffers->name_records[result->name_record_count].name_length =
+ name_length;
+ if (name_length > 0) {
+ memcpy(buffers->name_bytes + name_offset, entry + 2, name_length);
+ }
+ result->name_record_count += 1;
+ name_offset += name_length;
+ }
+ result->name_bytes_length = name_offset;
+ return 0;
+}
+
+static int append_utf8_prefix(regex_pcre2_run_buffers *buffers,
+ regex_pcre2_run_result *result,
+ const uint8_t *source,
+ uint32_t length) {
+ uint32_t remaining = buffers->output_capacity - result->output_length;
+ uint32_t copied = length < remaining ? length : remaining;
+
+ if (copied < length) {
+ while (copied > 0 && (source[copied] & 0xc0u) == 0x80u) {
+ copied -= 1;
+ }
+ result->output_truncated = 1;
+ }
+ if (copied > 0) {
+ memcpy(buffers->output + result->output_length, source, copied);
+ result->output_length += copied;
+ }
+ return copied == length;
+}
+
+static int32_t configure_match_context_values(pcre2_match_context *context,
+ uint32_t match_limit,
+ uint32_t depth_limit,
+ uint32_t heap_limit_kib) {
+ if (pcre2_set_match_limit(context, match_limit) != 0 ||
+ pcre2_set_depth_limit(context, depth_limit) != 0 ||
+ pcre2_set_heap_limit(context, heap_limit_kib) != 0) {
+ return REGEX_PCRE2_ERROR_CONFIG;
+ }
+ return 0;
+}
+
+static int32_t configure_match_context(pcre2_match_context *context,
+ const regex_pcre2_limits *limits) {
+ return configure_match_context_values(
+ context, limits->match_limit, limits->depth_limit,
+ limits->heap_limit_kib);
+}
+
+static int trace_limits_valid(const regex_pcre2_trace_limits *limits) {
+ return limits != NULL && limits->maximum_events > 0 &&
+ limits->maximum_events <= REGEX_PCRE2_MAX_TRACE_EVENTS &&
+ limits->maximum_trace_bytes >= sizeof(regex_pcre2_trace_event) &&
+ limits->maximum_trace_bytes <= REGEX_PCRE2_MAX_TRACE_BYTES &&
+ limits->match_limit > 0 &&
+ limits->match_limit <= REGEX_PCRE2_MAX_MATCH_LIMIT &&
+ limits->depth_limit > 0 &&
+ limits->depth_limit <= REGEX_PCRE2_MAX_DEPTH_LIMIT &&
+ limits->heap_limit_kib > 0 &&
+ limits->heap_limit_kib <= REGEX_PCRE2_MAX_HEAP_LIMIT_KIB;
+}
+
+static uint32_t bounded_mark_length(PCRE2_SPTR mark, uint32_t *truncated) {
+ uint32_t length = 0;
+ if (mark == NULL) {
+ return 0;
+ }
+ while (length < REGEX_PCRE2_MAX_TRACE_MARK_BYTES && mark[length] != 0) {
+ length += 1;
+ }
+ if (length == REGEX_PCRE2_MAX_TRACE_MARK_BYTES && mark[length] != 0) {
+ *truncated = 1;
+ while (length > 0 && (mark[length] & 0xc0u) == 0x80u) {
+ length -= 1;
+ }
+ }
+ return length;
+}
+
+static int trace_callout(pcre2_callout_block *block, void *data) {
+ regex_pcre2_trace_state *state = (regex_pcre2_trace_state *)data;
+ regex_pcre2_trace_event *event = NULL;
+ uint32_t mark_length = 0;
+ uint32_t mark_was_truncated = 0;
+ uint64_t required_trace_bytes = 0;
+
+ if (block == NULL || state == NULL) {
+ return PCRE2_ERROR_CALLOUT;
+ }
+ if (state->total_event_count < UINT32_MAX) {
+ state->total_event_count += 1;
+ }
+ mark_length = bounded_mark_length(block->mark, &mark_was_truncated);
+ required_trace_bytes =
+ ((uint64_t)state->event_count + 1u) *
+ sizeof(regex_pcre2_trace_event) +
+ state->mark_bytes_length + mark_length;
+ if (state->event_count >= state->maximum_events ||
+ state->event_count >= state->event_capacity ||
+ required_trace_bytes > state->maximum_trace_bytes ||
+ mark_length > state->mark_bytes_capacity - state->mark_bytes_length) {
+ state->events_truncated = 1;
+ return PCRE2_ERROR_CALLOUT;
+ }
+
+ event = &state->events[state->event_count];
+ event->callout_number = block->callout_number;
+ event->pattern_position_byte = bounded_offset(block->pattern_position);
+ event->next_item_length_byte = bounded_offset(block->next_item_length);
+ event->subject_position_byte = bounded_offset(block->current_position);
+ event->capture_top = block->capture_top;
+ event->capture_last = block->capture_last;
+ event->mark_offset = state->mark_bytes_length;
+ event->mark_length = mark_length;
+ if (mark_length > 0) {
+ memcpy(state->mark_bytes + state->mark_bytes_length, block->mark,
+ mark_length);
+ state->mark_bytes_length += mark_length;
+ }
+ state->marks_truncated |= mark_was_truncated;
+ state->event_count += 1;
+ return 0;
+}
+
+static int32_t run_bounded(const uint8_t *pattern,
+ uint32_t pattern_length,
+ const uint8_t *subject,
+ uint32_t subject_length,
+ const uint8_t *replacement,
+ uint32_t replacement_length,
+ uint32_t application_flags,
+ const regex_pcre2_limits *limits,
+ regex_pcre2_run_buffers *buffers,
+ int substitution,
+ regex_pcre2_run_result *result) {
+ pcre2_code *code = NULL;
+ pcre2_match_data *match_data = NULL;
+ pcre2_match_context *match_context = NULL;
+ int compile_error = 0;
+ PCRE2_SIZE compile_offset = 0;
+ uint32_t capture_count = 0;
+ uint32_t name_count = 0;
+ uint32_t options = 0;
+ uint32_t search_offset = 0;
+ uint32_t match_options = 0;
+ uint32_t copied_until = 0;
+ uint32_t capture_rows = 0;
+ int32_t status = 0;
+ const uint8_t *safe_pattern = nonnull_input(pattern);
+ const uint8_t *safe_subject = nonnull_input(subject);
+ const uint8_t *safe_replacement = nonnull_input(replacement);
+
+ if (result == NULL || !input_pointer_valid(pattern, pattern_length) ||
+ !input_pointer_valid(subject, subject_length) ||
+ (substitution &&
+ !input_pointer_valid(replacement, replacement_length)) ||
+ !buffers_valid(buffers, substitution)) {
+ return fail_run(result, REGEX_PCRE2_ERROR_INVALID_ARGUMENT,
+ REGEX_PCRE2_PHASE_BRIDGE, 0);
+ }
+ memset(result, 0, sizeof(*result));
+
+ if ((application_flags & ~REGEX_PCRE2_ALL_FLAGS) != 0) {
+ return fail_run(result, REGEX_PCRE2_ERROR_UNSUPPORTED_FLAGS,
+ REGEX_PCRE2_PHASE_BRIDGE, 0);
+ }
+ if (pattern_length > REGEX_PCRE2_MAX_PATTERN_BYTES) {
+ return fail_run(result, REGEX_PCRE2_ERROR_PATTERN_TOO_LARGE,
+ REGEX_PCRE2_PHASE_BRIDGE, 0);
+ }
+ if (subject_length > REGEX_PCRE2_MAX_SUBJECT_BYTES) {
+ return fail_run(result, REGEX_PCRE2_ERROR_SUBJECT_TOO_LARGE,
+ REGEX_PCRE2_PHASE_BRIDGE, 0);
+ }
+ if (substitution &&
+ replacement_length > REGEX_PCRE2_MAX_REPLACEMENT_BYTES) {
+ return fail_run(result, REGEX_PCRE2_ERROR_REPLACEMENT_TOO_LARGE,
+ REGEX_PCRE2_PHASE_BRIDGE, 0);
+ }
+ if (!limits_valid(limits)) {
+ return fail_run(result, REGEX_PCRE2_ERROR_LIMIT_OUT_OF_RANGE,
+ REGEX_PCRE2_PHASE_BRIDGE, 0);
+ }
+
+ options = to_pcre2_options(application_flags);
+ result->effective_options = options;
+ code = pcre2_compile(safe_pattern, pattern_length, options, &compile_error,
+ &compile_offset, NULL);
+ if (code == NULL) {
+ status = compile_error;
+ result->status = status;
+ result->error_phase = REGEX_PCRE2_PHASE_COMPILE;
+ result->error_offset = bounded_offset(compile_offset);
+ goto cleanup;
+ }
+ if (pcre2_pattern_info(code, PCRE2_INFO_CAPTURECOUNT, &capture_count) != 0 ||
+ pcre2_pattern_info(code, PCRE2_INFO_NAMECOUNT, &name_count) != 0) {
+ status = REGEX_PCRE2_ERROR_CONFIG;
+ result->status = status;
+ result->error_phase = REGEX_PCRE2_PHASE_BRIDGE;
+ goto cleanup;
+ }
+ result->capture_count = capture_count;
+ result->name_count = name_count;
+ if (capture_count > REGEX_PCRE2_MAX_CAPTURE_GROUPS) {
+ status = REGEX_PCRE2_ERROR_CAPTURE_LIMIT;
+ result->status = status;
+ result->error_phase = REGEX_PCRE2_PHASE_BRIDGE;
+ goto cleanup;
+ }
+ status = copy_names(code, buffers, result);
+ if (status != 0) {
+ result->status = status;
+ result->error_phase = REGEX_PCRE2_PHASE_BRIDGE;
+ goto cleanup;
+ }
+
+ match_data = pcre2_match_data_create_from_pattern(code, NULL);
+ match_context = pcre2_match_context_create(NULL);
+ if (match_data == NULL || match_context == NULL) {
+ status = PCRE2_ERROR_NOMEMORY;
+ result->status = status;
+ result->error_phase = REGEX_PCRE2_PHASE_BRIDGE;
+ goto cleanup;
+ }
+ status = configure_match_context(match_context, limits);
+ if (status != 0) {
+ result->status = status;
+ result->error_phase = REGEX_PCRE2_PHASE_BRIDGE;
+ goto cleanup;
+ }
+
+ while (search_offset <= subject_length) {
+ int match_status =
+ pcre2_match(code, safe_subject, subject_length, search_offset,
+ match_options, match_data, match_context);
+ PCRE2_SIZE *ovector = NULL;
+ uint32_t required_records = capture_count + 1;
+ uint32_t required_capture_rows = capture_count == 0 ? 1 : capture_count;
+ uint32_t group = 0;
+ uint32_t start = 0;
+ uint32_t end = 0;
+
+ if (match_status == PCRE2_ERROR_NOMATCH) {
+ break;
+ }
+ if (match_status < 0) {
+ status = match_status;
+ result->status = status;
+ result->error_phase = REGEX_PCRE2_PHASE_MATCH;
+ result->error_offset = search_offset;
+ goto cleanup;
+ }
+ if (match_status == 0) {
+ status = REGEX_PCRE2_ERROR_CONFIG;
+ result->status = status;
+ result->error_phase = REGEX_PCRE2_PHASE_BRIDGE;
+ goto cleanup;
+ }
+
+ ovector = pcre2_get_ovector_pointer(match_data);
+ start = bounded_offset(ovector[0]);
+ end = bounded_offset(ovector[1]);
+ if (start > end || end > subject_length) {
+ status = REGEX_PCRE2_ERROR_CONFIG;
+ result->status = status;
+ result->error_phase = REGEX_PCRE2_PHASE_BRIDGE;
+ goto cleanup;
+ }
+ if (result->match_count >= limits->maximum_matches ||
+ capture_rows + required_capture_rows >
+ limits->maximum_capture_rows ||
+ required_records > buffers->record_capacity - result->record_count) {
+ result->results_truncated = 1;
+ break;
+ }
+
+ for (group = 0; group <= capture_count; group += 1) {
+ regex_pcre2_match_record *record =
+ &buffers->records[result->record_count];
+ PCRE2_SIZE native_start = ovector[group * 2];
+ PCRE2_SIZE native_end = ovector[group * 2 + 1];
+ record->match_number = result->match_count + 1;
+ record->group_number = group;
+ if (native_start == PCRE2_UNSET || native_end == PCRE2_UNSET) {
+ record->start_byte = REGEX_PCRE2_UNSET_OFFSET;
+ record->end_byte = REGEX_PCRE2_UNSET_OFFSET;
+ } else {
+ record->start_byte = bounded_offset(native_start);
+ record->end_byte = bounded_offset(native_end);
+ }
+ result->record_count += 1;
+ }
+ result->match_count += 1;
+ capture_rows += required_capture_rows;
+
+ if (substitution && result->output_truncated == 0) {
+ PCRE2_SIZE replacement_output_length =
+ buffers->output_capacity - result->output_length + 1;
+ int substitute_status = 0;
+ if (!append_utf8_prefix(buffers, result, safe_subject + copied_until,
+ start - copied_until)) {
+ break;
+ }
+ replacement_output_length =
+ buffers->output_capacity - result->output_length + 1;
+ substitute_status = pcre2_substitute(
+ code, safe_subject, subject_length, search_offset,
+ match_options | PCRE2_SUBSTITUTE_MATCHED |
+ PCRE2_SUBSTITUTE_REPLACEMENT_ONLY |
+ PCRE2_SUBSTITUTE_UNSET_EMPTY,
+ match_data, match_context, safe_replacement, replacement_length,
+ buffers->output + result->output_length, &replacement_output_length);
+ if (substitute_status == PCRE2_ERROR_NOMEMORY) {
+ result->output_truncated = 1;
+ break;
+ }
+ if (substitute_status < 0) {
+ status = substitute_status;
+ result->status = status;
+ result->error_phase = REGEX_PCRE2_PHASE_SUBSTITUTE;
+ result->error_offset = bounded_offset(replacement_output_length);
+ goto cleanup;
+ }
+ result->output_length += bounded_offset(replacement_output_length);
+ result->substitution_count += (uint32_t)substitute_status;
+ copied_until = end;
+ }
+
+ if ((application_flags & REGEX_PCRE2_GLOBAL) == 0) {
+ break;
+ }
+ {
+ PCRE2_SIZE next_offset = search_offset;
+ uint32_t next_options = 0;
+ if (!pcre2_next_match(match_data, &next_offset, &next_options)) {
+ break;
+ }
+ search_offset = bounded_offset(next_offset);
+ match_options = next_options;
+ }
+ }
+
+ if (substitution && result->output_truncated == 0) {
+ (void)append_utf8_prefix(buffers, result, safe_subject + copied_until,
+ subject_length - copied_until);
+ }
+
+cleanup:
+ if (match_context != NULL) {
+ pcre2_match_context_free(match_context);
+ }
+ if (match_data != NULL) {
+ pcre2_match_data_free(match_data);
+ }
+ if (code != NULL) {
+ pcre2_code_free(code);
+ }
+ return status;
+}
+
+uint32_t regex_pcre2_bridge_abi_version(void) {
+ return REGEX_PCRE2_BRIDGE_ABI_VERSION;
+}
+
+uint32_t regex_pcre2_config_flags(void) {
+ uint32_t unicode = 0;
+ uint32_t jit = 0;
+ uint32_t flags = 0;
+
+ if (pcre2_config(PCRE2_CONFIG_UNICODE, &unicode) != 0 ||
+ pcre2_config(PCRE2_CONFIG_JIT, &jit) != 0) {
+ return 0;
+ }
+ if (unicode != 0) {
+ flags |= REGEX_PCRE2_CONFIG_UNICODE;
+ }
+ if (jit != 0) {
+ flags |= REGEX_PCRE2_CONFIG_JIT;
+ }
+ return flags;
+}
+
+const uint8_t *regex_pcre2_version(void) {
+ static uint8_t version[64];
+ if (pcre2_config(PCRE2_CONFIG_VERSION, version) <= 0) {
+ version[0] = '\0';
+ }
+ return version;
+}
+
+int32_t regex_pcre2_error_message(int32_t error_code,
+ uint8_t *output,
+ uint32_t output_capacity) {
+ int status = 0;
+ if (output == NULL || output_capacity == 0) {
+ return REGEX_PCRE2_ERROR_INVALID_ARGUMENT;
+ }
+ status = pcre2_get_error_message(error_code, output, output_capacity);
+ if (status < 0) {
+ output[0] = '\0';
+ }
+ return status;
+}
+
+int32_t regex_pcre2_compile_probe(const uint8_t *pattern,
+ uint32_t pattern_length,
+ uint32_t application_flags,
+ regex_pcre2_compile_result *result) {
+ pcre2_code *code = NULL;
+ int error_code = 0;
+ PCRE2_SIZE error_offset = 0;
+ uint32_t capture_count = 0;
+ uint32_t name_count = 0;
+ uint32_t options = 0;
+
+ if (result == NULL || !input_pointer_valid(pattern, pattern_length)) {
+ return fail_compile(result, REGEX_PCRE2_ERROR_INVALID_ARGUMENT);
+ }
+ memset(result, 0, sizeof(*result));
+
+ if ((application_flags & ~REGEX_PCRE2_COMPILE_FLAGS) != 0) {
+ return fail_compile(result, REGEX_PCRE2_ERROR_UNSUPPORTED_FLAGS);
+ }
+ if (pattern_length > REGEX_PCRE2_MAX_PATTERN_BYTES) {
+ return fail_compile(result, REGEX_PCRE2_ERROR_PATTERN_TOO_LARGE);
+ }
+
+ options = to_pcre2_options(application_flags);
+ result->effective_options = options;
+ code = pcre2_compile(nonnull_input(pattern), pattern_length, options,
+ &error_code, &error_offset, NULL);
+ if (code == NULL) {
+ result->status = error_code;
+ result->error_offset = bounded_offset(error_offset);
+ return error_code;
+ }
+
+ if (pcre2_pattern_info(code, PCRE2_INFO_CAPTURECOUNT, &capture_count) != 0 ||
+ pcre2_pattern_info(code, PCRE2_INFO_NAMECOUNT, &name_count) != 0) {
+ pcre2_code_free(code);
+ return fail_compile(result, REGEX_PCRE2_ERROR_CONFIG);
+ }
+
+ result->capture_count = capture_count;
+ result->name_count = name_count;
+ pcre2_code_free(code);
+ return 0;
+}
+
+int32_t regex_pcre2_execute(
+ const uint8_t *pattern,
+ uint32_t pattern_length,
+ const uint8_t *subject,
+ uint32_t subject_length,
+ uint32_t application_flags,
+ const regex_pcre2_limits *limits,
+ regex_pcre2_match_record *records,
+ uint32_t record_capacity,
+ regex_pcre2_name_record *name_records,
+ uint32_t name_record_capacity,
+ uint8_t *name_bytes,
+ uint32_t name_bytes_capacity,
+ regex_pcre2_run_result *result) {
+ regex_pcre2_run_buffers buffers = {
+ records, record_capacity, name_records, name_record_capacity,
+ name_bytes, name_bytes_capacity, NULL, 0};
+ return run_bounded(pattern, pattern_length, subject, subject_length, NULL, 0,
+ application_flags, limits, &buffers, 0, result);
+}
+
+int32_t regex_pcre2_substitute(
+ const uint8_t *pattern,
+ uint32_t pattern_length,
+ const uint8_t *subject,
+ uint32_t subject_length,
+ const uint8_t *replacement,
+ uint32_t replacement_length,
+ uint32_t application_flags,
+ const regex_pcre2_limits *limits,
+ regex_pcre2_match_record *records,
+ uint32_t record_capacity,
+ regex_pcre2_name_record *name_records,
+ uint32_t name_record_capacity,
+ uint8_t *name_bytes,
+ uint32_t name_bytes_capacity,
+ uint8_t *output,
+ uint32_t output_capacity,
+ regex_pcre2_run_result *result) {
+ regex_pcre2_run_buffers buffers = {
+ records, record_capacity, name_records, name_record_capacity,
+ name_bytes, name_bytes_capacity, output, output_capacity};
+ return run_bounded(pattern, pattern_length, subject, subject_length,
+ replacement, replacement_length, application_flags,
+ limits, &buffers, 1, result);
+}
+
+int32_t regex_pcre2_trace(
+ const uint8_t *pattern,
+ uint32_t pattern_length,
+ const uint8_t *subject,
+ uint32_t subject_length,
+ uint32_t application_flags,
+ const regex_pcre2_trace_limits *limits,
+ regex_pcre2_trace_event *events,
+ uint32_t event_capacity,
+ uint8_t *mark_bytes,
+ uint32_t mark_bytes_capacity,
+ regex_pcre2_trace_result *result) {
+ pcre2_code *code = NULL;
+ pcre2_match_data *match_data = NULL;
+ pcre2_match_context *match_context = NULL;
+ regex_pcre2_trace_state state;
+ int compile_error = 0;
+ PCRE2_SIZE compile_offset = 0;
+ uint32_t capture_count = 0;
+ uint32_t options = 0;
+ int32_t match_status = 0;
+ int32_t status = 0;
+
+ if (result == NULL || !input_pointer_valid(pattern, pattern_length) ||
+ !input_pointer_valid(subject, subject_length) || events == NULL ||
+ event_capacity == 0 ||
+ (mark_bytes == NULL && mark_bytes_capacity != 0)) {
+ return fail_trace(result, REGEX_PCRE2_ERROR_INVALID_ARGUMENT,
+ REGEX_PCRE2_PHASE_BRIDGE, 0);
+ }
+ memset(result, 0, sizeof(*result));
+ result->last_complete_event = UINT32_MAX;
+ if ((application_flags & ~REGEX_PCRE2_ALL_FLAGS) != 0) {
+ return fail_trace(result, REGEX_PCRE2_ERROR_UNSUPPORTED_FLAGS,
+ REGEX_PCRE2_PHASE_BRIDGE, 0);
+ }
+ if (pattern_length > REGEX_PCRE2_MAX_PATTERN_BYTES) {
+ return fail_trace(result, REGEX_PCRE2_ERROR_PATTERN_TOO_LARGE,
+ REGEX_PCRE2_PHASE_BRIDGE, 0);
+ }
+ if (subject_length > REGEX_PCRE2_MAX_SUBJECT_BYTES) {
+ return fail_trace(result, REGEX_PCRE2_ERROR_SUBJECT_TOO_LARGE,
+ REGEX_PCRE2_PHASE_BRIDGE, 0);
+ }
+ if (!trace_limits_valid(limits) ||
+ event_capacity < limits->maximum_events ||
+ event_capacity > REGEX_PCRE2_MAX_TRACE_EVENTS ||
+ mark_bytes_capacity > REGEX_PCRE2_MAX_TRACE_BYTES) {
+ return fail_trace(result, REGEX_PCRE2_ERROR_LIMIT_OUT_OF_RANGE,
+ REGEX_PCRE2_PHASE_BRIDGE, 0);
+ }
+
+ options = to_pcre2_options(application_flags) | PCRE2_AUTO_CALLOUT;
+ result->effective_options = options;
+ code = pcre2_compile(nonnull_input(pattern), pattern_length, options,
+ &compile_error, &compile_offset, NULL);
+ if (code == NULL) {
+ status = compile_error;
+ result->status = status;
+ result->error_phase = REGEX_PCRE2_PHASE_COMPILE;
+ result->error_offset = bounded_offset(compile_offset);
+ goto cleanup;
+ }
+ if (pcre2_pattern_info(code, PCRE2_INFO_CAPTURECOUNT, &capture_count) != 0) {
+ status = REGEX_PCRE2_ERROR_CONFIG;
+ result->status = status;
+ result->error_phase = REGEX_PCRE2_PHASE_BRIDGE;
+ goto cleanup;
+ }
+ if (capture_count > REGEX_PCRE2_MAX_CAPTURE_GROUPS) {
+ status = REGEX_PCRE2_ERROR_CAPTURE_LIMIT;
+ result->status = status;
+ result->error_phase = REGEX_PCRE2_PHASE_BRIDGE;
+ goto cleanup;
+ }
+
+ match_data = pcre2_match_data_create_from_pattern(code, NULL);
+ match_context = pcre2_match_context_create(NULL);
+ if (match_data == NULL || match_context == NULL) {
+ status = PCRE2_ERROR_NOMEMORY;
+ result->status = status;
+ result->error_phase = REGEX_PCRE2_PHASE_BRIDGE;
+ goto cleanup;
+ }
+ memset(&state, 0, sizeof(state));
+ state.events = events;
+ state.event_capacity = event_capacity;
+ state.mark_bytes = mark_bytes;
+ state.mark_bytes_capacity = mark_bytes_capacity;
+ state.maximum_events = limits->maximum_events;
+ state.maximum_trace_bytes = limits->maximum_trace_bytes;
+ status = configure_match_context_values(
+ match_context, limits->match_limit, limits->depth_limit,
+ limits->heap_limit_kib);
+ if (status != 0 ||
+ pcre2_set_callout(match_context, trace_callout, &state) != 0) {
+ status = REGEX_PCRE2_ERROR_CONFIG;
+ result->status = status;
+ result->error_phase = REGEX_PCRE2_PHASE_BRIDGE;
+ goto cleanup;
+ }
+ /*
+ * The global flag is intentionally application-level for normal execution.
+ * A trace represents one exact pcre2_match() invocation.
+ */
+ match_status = pcre2_match(code, nonnull_input(subject), subject_length, 0, 0,
+ match_data, match_context);
+ result->native_match_status = match_status;
+ result->event_count = state.event_count;
+ result->total_event_count = state.total_event_count;
+ result->mark_bytes_length = state.mark_bytes_length;
+ result->trace_bytes_length =
+ state.event_count * (uint32_t)sizeof(regex_pcre2_trace_event) +
+ state.mark_bytes_length;
+ result->events_truncated = state.events_truncated;
+ result->marks_truncated = state.marks_truncated;
+ result->last_complete_event =
+ state.event_count == 0 ? UINT32_MAX : state.event_count - 1;
+ if (match_status >= 0) {
+ result->matched = 1;
+ } else if (match_status == PCRE2_ERROR_NOMATCH ||
+ (match_status == PCRE2_ERROR_CALLOUT &&
+ state.events_truncated != 0)) {
+ status = 0;
+ } else {
+ status = match_status;
+ result->status = status;
+ result->error_phase = REGEX_PCRE2_PHASE_MATCH;
+ if (state.event_count > 0) {
+ result->error_offset =
+ events[state.event_count - 1].subject_position_byte;
+ }
+ }
+
+cleanup:
+ if (match_context != NULL) {
+ pcre2_match_context_free(match_context);
+ }
+ if (match_data != NULL) {
+ pcre2_match_data_free(match_data);
+ }
+ if (code != NULL) {
+ pcre2_code_free(code);
+ }
+ return status;
+}
+
+int32_t regex_pcre2_self_test(void) {
+ static const uint8_t pattern[] = "(?\\p{L}+)-(?\\d+)";
+ static const uint8_t subject[] = "x Grüße-42 y";
+ static const uint8_t replacement[] = "$:$";
+ static const uint8_t trace_pattern[] = "a+b";
+ static const uint8_t trace_subject[] = "aaab";
+ regex_pcre2_compile_result compile_result;
+ regex_pcre2_run_result run_result;
+ regex_pcre2_trace_result trace_result;
+ regex_pcre2_limits limits = {10, 100, 1000000, 1000, 32768};
+ regex_pcre2_trace_limits trace_limits = {100, 4096, 1000000, 1000,
+ 32768};
+ regex_pcre2_match_record records[3];
+ regex_pcre2_name_record names[2];
+ regex_pcre2_trace_event trace_events[100];
+ uint8_t name_bytes[64];
+ uint8_t trace_marks[64];
+ uint8_t output[65];
+ int32_t status = regex_pcre2_compile_probe(
+ pattern, (uint32_t)(sizeof(pattern) - 1),
+ REGEX_PCRE2_UTF | REGEX_PCRE2_UCP, &compile_result);
+
+ if (status != 0 || compile_result.capture_count != 2 ||
+ compile_result.name_count != 2) {
+ return 1;
+ }
+ status = regex_pcre2_substitute(
+ pattern, (uint32_t)(sizeof(pattern) - 1), subject,
+ (uint32_t)(sizeof(subject) - 1), replacement,
+ (uint32_t)(sizeof(replacement) - 1),
+ REGEX_PCRE2_UTF | REGEX_PCRE2_UCP, &limits, records, 3, names, 2,
+ name_bytes, sizeof(name_bytes), output, sizeof(output) - 1, &run_result);
+ if (status != 0 || run_result.match_count != 1 ||
+ run_result.record_count != 3 || run_result.substitution_count != 1 ||
+ run_result.output_truncated != 0 ||
+ run_result.output_length != sizeof("x 42:Grüße y") - 1 ||
+ memcmp(output, "x 42:Grüße y", sizeof("x 42:Grüße y") - 1) != 0) {
+ return 2;
+ }
+ status = regex_pcre2_trace(
+ trace_pattern, (uint32_t)(sizeof(trace_pattern) - 1), trace_subject,
+ (uint32_t)(sizeof(trace_subject) - 1),
+ REGEX_PCRE2_UTF | REGEX_PCRE2_UCP, &trace_limits, trace_events, 100,
+ trace_marks, sizeof(trace_marks), &trace_result);
+ if (status != 0 || trace_result.status != 0 ||
+ trace_result.native_match_status <= 0 || trace_result.matched != 1 ||
+ trace_result.event_count == 0 || trace_result.events_truncated != 0 ||
+ trace_result.last_complete_event != trace_result.event_count - 1) {
+ return 3;
+ }
+ if (regex_pcre2_bridge_abi_version() != REGEX_PCRE2_BRIDGE_ABI_VERSION ||
+ regex_pcre2_config_flags() != REGEX_PCRE2_CONFIG_UNICODE) {
+ return 4;
+ }
+ if (strcmp((const char *)regex_pcre2_version(), "10.47 2025-10-21") != 0) {
+ return 5;
+ }
+ return 0;
+}
diff --git a/engines/pcre2/pcre2_bridge.h b/engines/pcre2/pcre2_bridge.h
new file mode 100644
index 0000000..8f6ed64
--- /dev/null
+++ b/engines/pcre2/pcre2_bridge.h
@@ -0,0 +1,236 @@
+/* SPDX-License-Identifier: GPL-3.0-or-later */
+
+#ifndef REGEX_TOOLS_PCRE2_BRIDGE_H
+#define REGEX_TOOLS_PCRE2_BRIDGE_H
+
+#include
+
+#define REGEX_PCRE2_BRIDGE_ABI_VERSION 3u
+#define REGEX_PCRE2_MAX_PATTERN_BYTES 1048576u
+#define REGEX_PCRE2_MAX_SUBJECT_BYTES 16777216u
+#define REGEX_PCRE2_MAX_REPLACEMENT_BYTES 262144u
+#define REGEX_PCRE2_MAX_MATCHES 10000u
+#define REGEX_PCRE2_MAX_CAPTURE_ROWS 100000u
+#define REGEX_PCRE2_MAX_CAPTURE_GROUPS 1000u
+#define REGEX_PCRE2_MAX_MATCH_LIMIT 100000000u
+#define REGEX_PCRE2_MAX_DEPTH_LIMIT 100000u
+#define REGEX_PCRE2_MAX_HEAP_LIMIT_KIB 131072u
+#define REGEX_PCRE2_MAX_TRACE_EVENTS 50000u
+#define REGEX_PCRE2_MAX_TRACE_BYTES 10485760u
+#define REGEX_PCRE2_MAX_TRACE_MARK_BYTES 1024u
+#define REGEX_PCRE2_UNSET_OFFSET UINT32_MAX
+
+enum regex_pcre2_flag {
+ REGEX_PCRE2_CASELESS = 1u << 0,
+ REGEX_PCRE2_MULTILINE = 1u << 1,
+ REGEX_PCRE2_DOTALL = 1u << 2,
+ REGEX_PCRE2_EXTENDED = 1u << 3,
+ REGEX_PCRE2_UNGREEDY = 1u << 4,
+ REGEX_PCRE2_UTF = 1u << 5,
+ REGEX_PCRE2_UCP = 1u << 6,
+ REGEX_PCRE2_DUPNAMES = 1u << 7,
+ REGEX_PCRE2_GLOBAL = 1u << 8
+};
+
+enum regex_pcre2_config_flag {
+ REGEX_PCRE2_CONFIG_UNICODE = 1u << 0,
+ REGEX_PCRE2_CONFIG_JIT = 1u << 1
+};
+
+enum regex_pcre2_bridge_error {
+ REGEX_PCRE2_ERROR_INVALID_ARGUMENT = -10001,
+ REGEX_PCRE2_ERROR_UNSUPPORTED_FLAGS = -10002,
+ REGEX_PCRE2_ERROR_PATTERN_TOO_LARGE = -10003,
+ REGEX_PCRE2_ERROR_CONFIG = -10004,
+ REGEX_PCRE2_ERROR_SUBJECT_TOO_LARGE = -10005,
+ REGEX_PCRE2_ERROR_REPLACEMENT_TOO_LARGE = -10006,
+ REGEX_PCRE2_ERROR_LIMIT_OUT_OF_RANGE = -10007,
+ REGEX_PCRE2_ERROR_CAPTURE_LIMIT = -10008,
+ REGEX_PCRE2_ERROR_OUTPUT_BUFFER = -10009
+};
+
+enum regex_pcre2_error_phase {
+ REGEX_PCRE2_PHASE_NONE = 0,
+ REGEX_PCRE2_PHASE_BRIDGE = 1,
+ REGEX_PCRE2_PHASE_COMPILE = 2,
+ REGEX_PCRE2_PHASE_MATCH = 3,
+ REGEX_PCRE2_PHASE_SUBSTITUTE = 4
+};
+
+/*
+ * This 20-byte, five-u32 record is retained for deterministic compile-only
+ * build verification. Runtime callers should use the bounded APIs below.
+ */
+typedef struct regex_pcre2_compile_result {
+ int32_t status;
+ uint32_t error_offset;
+ uint32_t capture_count;
+ uint32_t name_count;
+ uint32_t effective_options;
+} regex_pcre2_compile_result;
+
+/* All values are required and validated against the bridge hard maxima. */
+typedef struct regex_pcre2_limits {
+ uint32_t maximum_matches;
+ uint32_t maximum_capture_rows;
+ uint32_t match_limit;
+ uint32_t depth_limit;
+ uint32_t heap_limit_kib;
+} regex_pcre2_limits;
+
+/*
+ * One record is emitted for group zero and every capture group of each
+ * retained match. Unset captures use REGEX_PCRE2_UNSET_OFFSET for both bounds.
+ */
+typedef struct regex_pcre2_match_record {
+ uint32_t match_number;
+ uint32_t group_number;
+ uint32_t start_byte;
+ uint32_t end_byte;
+} regex_pcre2_match_record;
+
+/* Names are UTF-8 slices into the caller-owned name byte buffer. */
+typedef struct regex_pcre2_name_record {
+ uint32_t group_number;
+ uint32_t name_offset;
+ uint32_t name_length;
+} regex_pcre2_name_record;
+
+/*
+ * Trace collection has an independent bound from match materialization. The
+ * byte cap covers the fixed event records plus copied mark bytes.
+ */
+typedef struct regex_pcre2_trace_limits {
+ uint32_t maximum_events;
+ uint32_t maximum_trace_bytes;
+ uint32_t match_limit;
+ uint32_t depth_limit;
+ uint32_t heap_limit_kib;
+} regex_pcre2_trace_limits;
+
+/*
+ * Every field except the mark slice is copied directly from a PCRE2 callout
+ * block. Positions and lengths are offsets in the 8-bit UTF-8 pattern or
+ * subject supplied to the engine.
+ */
+typedef struct regex_pcre2_trace_event {
+ uint32_t callout_number;
+ uint32_t pattern_position_byte;
+ uint32_t next_item_length_byte;
+ uint32_t subject_position_byte;
+ uint32_t capture_top;
+ uint32_t capture_last;
+ uint32_t mark_offset;
+ uint32_t mark_length;
+} regex_pcre2_trace_event;
+
+/*
+ * A zero status means that a bounded trace was returned. native_match_status
+ * retains pcre2_match()'s exact result, including PCRE2_ERROR_CALLOUT when the
+ * bridge deliberately stopped at a trace cap. last_complete_event is
+ * UINT32_MAX when no complete event was copied.
+ */
+typedef struct regex_pcre2_trace_result {
+ int32_t status;
+ uint32_t error_phase;
+ uint32_t error_offset;
+ int32_t native_match_status;
+ uint32_t effective_options;
+ uint32_t event_count;
+ uint32_t total_event_count;
+ uint32_t mark_bytes_length;
+ uint32_t trace_bytes_length;
+ uint32_t events_truncated;
+ uint32_t marks_truncated;
+ uint32_t last_complete_event;
+ uint32_t matched;
+} regex_pcre2_trace_result;
+
+/*
+ * This fixed 60-byte record contains no native pointers. `status` is zero on
+ * success. Otherwise `error_phase` distinguishes bridge validation, compile,
+ * match and replacement errors; `error_offset` is a UTF-8 byte offset when the
+ * phase provides one.
+ */
+typedef struct regex_pcre2_run_result {
+ int32_t status;
+ uint32_t error_phase;
+ uint32_t error_offset;
+ uint32_t capture_count;
+ uint32_t name_count;
+ uint32_t match_count;
+ uint32_t record_count;
+ uint32_t name_record_count;
+ uint32_t name_bytes_length;
+ uint32_t effective_options;
+ uint32_t results_truncated;
+ uint32_t names_truncated;
+ uint32_t output_length;
+ uint32_t substitution_count;
+ uint32_t output_truncated;
+} regex_pcre2_run_result;
+
+uint32_t regex_pcre2_bridge_abi_version(void);
+uint32_t regex_pcre2_config_flags(void);
+const uint8_t *regex_pcre2_version(void);
+
+int32_t regex_pcre2_error_message(int32_t error_code,
+ uint8_t *output,
+ uint32_t output_capacity);
+
+int32_t regex_pcre2_compile_probe(const uint8_t *pattern,
+ uint32_t pattern_length,
+ uint32_t application_flags,
+ regex_pcre2_compile_result *result);
+
+int32_t regex_pcre2_execute(
+ const uint8_t *pattern,
+ uint32_t pattern_length,
+ const uint8_t *subject,
+ uint32_t subject_length,
+ uint32_t application_flags,
+ const regex_pcre2_limits *limits,
+ regex_pcre2_match_record *records,
+ uint32_t record_capacity,
+ regex_pcre2_name_record *name_records,
+ uint32_t name_record_capacity,
+ uint8_t *name_bytes,
+ uint32_t name_bytes_capacity,
+ regex_pcre2_run_result *result);
+
+int32_t regex_pcre2_substitute(
+ const uint8_t *pattern,
+ uint32_t pattern_length,
+ const uint8_t *subject,
+ uint32_t subject_length,
+ const uint8_t *replacement,
+ uint32_t replacement_length,
+ uint32_t application_flags,
+ const regex_pcre2_limits *limits,
+ regex_pcre2_match_record *records,
+ uint32_t record_capacity,
+ regex_pcre2_name_record *name_records,
+ uint32_t name_record_capacity,
+ uint8_t *name_bytes,
+ uint32_t name_bytes_capacity,
+ /* `output` must have output_capacity + 1 writable bytes for PCRE2's NUL. */
+ uint8_t *output,
+ uint32_t output_capacity,
+ regex_pcre2_run_result *result);
+
+int32_t regex_pcre2_trace(
+ const uint8_t *pattern,
+ uint32_t pattern_length,
+ const uint8_t *subject,
+ uint32_t subject_length,
+ uint32_t application_flags,
+ const regex_pcre2_trace_limits *limits,
+ regex_pcre2_trace_event *events,
+ uint32_t event_capacity,
+ uint8_t *mark_bytes,
+ uint32_t mark_bytes_capacity,
+ regex_pcre2_trace_result *result);
+
+int32_t regex_pcre2_self_test(void);
+
+#endif
diff --git a/eslint.config.mjs b/eslint.config.mjs
index f930094..e08bf99 100644
--- a/eslint.config.mjs
+++ b/eslint.config.mjs
@@ -12,6 +12,7 @@ export default tseslint.config(
"coverage",
"test-results",
"playwright-report",
+ ".engine-build",
],
},
{
diff --git a/package-lock.json b/package-lock.json
index cf94e2b..22e0e3d 100644
--- a/package-lock.json
+++ b/package-lock.json
@@ -1,12 +1,12 @@
{
"name": "regex-tools",
- "version": "0.1.0",
+ "version": "0.2.0",
"lockfileVersion": 3,
"requires": true,
"packages": {
"": {
"name": "regex-tools",
- "version": "0.1.0",
+ "version": "0.2.0",
"license": "GPL-3.0-or-later",
"dependencies": {
"@add-ideas/toolbox-contract": "0.2.2",
@@ -18,6 +18,7 @@
"@codemirror/view": "6.43.6",
"@eslint-community/regexpp": "4.12.2",
"@lezer/highlight": "1.2.3",
+ "fflate": "0.8.3",
"react": "19.2.7",
"react-dom": "19.2.7"
},
@@ -35,7 +36,6 @@
"eslint": "10.7.0",
"eslint-plugin-react-hooks": "7.1.1",
"eslint-plugin-react-refresh": "0.5.2",
- "fflate": "0.8.3",
"globals": "17.7.0",
"jsdom": "29.1.1",
"prettier": "3.9.5",
@@ -2442,7 +2442,6 @@
"version": "0.8.3",
"resolved": "https://registry.npmjs.org/fflate/-/fflate-0.8.3.tgz",
"integrity": "sha512-tbZNuJrLwGUp3zshBtdy4W+ORxZuIh8a5ilyIEQDC5rY1f3U20JMry0Ll3WBzU58EZKsEuJFXhb5gwv8CsPvgA==",
- "dev": true,
"license": "MIT"
},
"node_modules/file-entry-cache": {
diff --git a/package.json b/package.json
index 9105b5e..26da57e 100644
--- a/package.json
+++ b/package.json
@@ -1,6 +1,6 @@
{
"name": "regex-tools",
- "version": "0.1.0",
+ "version": "0.2.0",
"description": "Develop, explain, test and apply regular expressions locally in the browser.",
"license": "GPL-3.0-or-later",
"author": "Albrecht Degering",
@@ -32,6 +32,9 @@
"test:conformance": "vitest run tests/conformance",
"test:browser": "playwright test",
"engines:build": "node scripts/build-engines.mjs",
+ "engines:pcre2:build": "node scripts/build-engines.mjs --pcre2",
+ "engines:pcre2:install": "node scripts/install-pcre2-engine.mjs",
+ "engines:pcre2:verify": "node scripts/verify-engine-assets.mjs --pcre2-pack .engine-build/pcre2",
"engines:verify": "node scripts/verify-engine-assets.mjs",
"manifest:generate": "node scripts/generate-toolbox-manifest.mjs",
"manifest:check": "node scripts/generate-toolbox-manifest.mjs --check",
@@ -51,6 +54,7 @@
"@codemirror/view": "6.43.6",
"@eslint-community/regexpp": "4.12.2",
"@lezer/highlight": "1.2.3",
+ "fflate": "0.8.3",
"react": "19.2.7",
"react-dom": "19.2.7"
},
@@ -68,7 +72,6 @@
"eslint": "10.7.0",
"eslint-plugin-react-hooks": "7.1.1",
"eslint-plugin-react-refresh": "0.5.2",
- "fflate": "0.8.3",
"globals": "17.7.0",
"jsdom": "29.1.1",
"prettier": "3.9.5",
diff --git a/public/engines/README.md b/public/engines/README.md
index a8ecff3..34b814f 100644
--- a/public/engines/README.md
+++ b/public/engines/README.md
@@ -1,7 +1,9 @@
# Engine assets
-Regex Tools 0.1.0 executes ECMAScript through the browser's native `RegExp`
-engine. It therefore ships no external engine binary or WebAssembly pack.
+Regex Tools executes ECMAScript through the browser's native `RegExp` engine.
+It also ships the declared, self-hosted PCRE2 10.47 WebAssembly pack in
+`pcre2/`.
-Future flavour packs will be built from pinned, documented open-source
-revisions and placed in versioned subdirectories here.
+The PCRE2 pack is built from pinned, documented open-source code and
+toolchain revisions. `engine-metadata.json`, `SHA256SUMS` and `LICENSE.txt`
+travel with the runtime files and are verified by the release gate.
diff --git a/public/engines/pcre2/LICENSE.txt b/public/engines/pcre2/LICENSE.txt
new file mode 100644
index 0000000..f6fba35
--- /dev/null
+++ b/public/engines/pcre2/LICENSE.txt
@@ -0,0 +1,104 @@
+PCRE2 Licence
+=============
+
+| SPDX-License-Identifier: | BSD-3-Clause WITH PCRE2-exception |
+|---------|-------|
+
+PCRE2 is a library of functions to support regular expressions whose syntax
+and semantics are as close as possible to those of the Perl 5 language.
+
+Releases 10.00 and above of PCRE2 are distributed under the terms of the "BSD"
+licence, as specified below, with one exemption for certain binary
+redistributions. The documentation for PCRE2, supplied in the "doc" directory,
+is distributed under the same terms as the software itself. The data in the
+testdata directory is not copyrighted and is in the public domain.
+
+The basic library functions are written in C and are freestanding. Also
+included in the distribution is a just-in-time compiler that can be used to
+optimize pattern matching. This is an optional feature that can be omitted when
+the library is built. The just-in-time compiler is separately licensed under the
+"2-clause BSD" licence.
+
+
+COPYRIGHT
+---------
+
+### The basic library functions
+
+ Written by: Philip Hazel
+ Email local part: Philip.Hazel
+ Email domain: gmail.com
+
+ Retired from University of Cambridge Computing Service,
+ Cambridge, England.
+
+ Copyright (c) 1997-2007 University of Cambridge
+ Copyright (c) 2007-2024 Philip Hazel
+ All rights reserved.
+
+### PCRE2 Just-In-Time compilation support
+
+ Written by: Zoltan Herczeg
+ Email local part: hzmester
+ Email domain: freemail.hu
+
+ Copyright (c) 2010-2024 Zoltan Herczeg
+ All rights reserved.
+
+### Stack-less Just-In-Time compiler
+
+ Written by: Zoltan Herczeg
+ Email local part: hzmester
+ Email domain: freemail.hu
+
+ Copyright (c) 2009-2024 Zoltan Herczeg
+ All rights reserved.
+
+The code in the `deps/sljit` directory has its own LICENSE file.
+
+### All other contributions
+
+Many other contributors have participated in the authorship of PCRE2. As PCRE2
+has never required a Contributor Licensing Agreement, or other copyright
+assignment agreement, all contributions have copyright retained by each
+original contributor or their employer.
+
+
+THE "BSD" LICENCE
+-----------------
+
+Redistribution and use in source and binary forms, with or without
+modification, are permitted provided that the following conditions are met:
+
+* Redistributions of source code must retain the above copyright notices,
+ this list of conditions and the following disclaimer.
+
+* Redistributions in binary form must reproduce the above copyright
+ notices, this list of conditions and the following disclaimer in the
+ documentation and/or other materials provided with the distribution.
+
+* Neither the name of the University of Cambridge nor the names of any
+ contributors may be used to endorse or promote products derived from this
+ software without specific prior written permission.
+
+THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
+AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
+IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
+ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE
+LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
+CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
+SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
+INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
+CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
+ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
+POSSIBILITY OF SUCH DAMAGE.
+
+
+EXEMPTION FOR BINARY LIBRARY-LIKE PACKAGES
+------------------------------------------
+
+The second condition in the BSD licence (covering binary redistributions) does
+not apply all the way down a chain of software. If binary package A includes
+PCRE2, it must respect the condition, but if package B is software that
+includes package A, the condition is not imposed on package B unless it uses
+PCRE2 independently.
diff --git a/public/engines/pcre2/SHA256SUMS b/public/engines/pcre2/SHA256SUMS
new file mode 100644
index 0000000..8ed63d6
--- /dev/null
+++ b/public/engines/pcre2/SHA256SUMS
@@ -0,0 +1,4 @@
+197d8a73ffee0d6b09adba2f9c677b5f5aede24edf89258a68e48248d010d811 LICENSE.txt
+8c7f67ad85893fcc545c40d496abcc82d1009b76c9b7bcc7bd9ab92828852d76 engine-metadata.json
+d31a5f1e0839955ae8321430eafe65da8cbdef832a1245bb71ccc5bb0d4b6b9c pcre2.mjs
+cc7c4b2b0038e137c86d2037e73f2302e56432cb09bc082e4f35bba221dd1889 pcre2.wasm
diff --git a/public/engines/pcre2/engine-metadata.json b/public/engines/pcre2/engine-metadata.json
new file mode 100644
index 0000000..8f950fe
--- /dev/null
+++ b/public/engines/pcre2/engine-metadata.json
@@ -0,0 +1,90 @@
+{
+ "schemaVersion": 1,
+ "engine": "pcre2",
+ "engineVersion": "10.47 2025-10-21",
+ "status": "production-runtime",
+ "bridge": {
+ "abiVersion": 3,
+ "maximumPatternBytes": 1048576,
+ "maximumSubjectBytes": 16777216,
+ "maximumReplacementBytes": 262144,
+ "maximumMatches": 10000,
+ "maximumCaptureRows": 100000,
+ "maximumCaptureGroups": 1000,
+ "maximumTraceEvents": 50000,
+ "maximumTraceBytes": 10485760,
+ "maximumTraceMarkBytes": 1024,
+ "sourceFiles": [
+ {
+ "path": "engines/pcre2/CMakeLists.txt",
+ "sha256": "670397b070a908b414d4d8ffa7c0a9ed2549b9ee27815a90de5fc9622c6df8fa",
+ "bytes": 2583
+ },
+ {
+ "path": "engines/pcre2/pcre2_bridge.c",
+ "sha256": "ab5929472d9981eb2d1cc635a4963a7243d5cbb9db6c5ba8460136d4c79b3246",
+ "bytes": 32530
+ },
+ {
+ "path": "engines/pcre2/pcre2_bridge.h",
+ "sha256": "b8767c0c59229a92d092915fc7b862a22d088a7f184485e4d67e14d6071d7d1c",
+ "bytes": 7388
+ }
+ ],
+ "exports": [
+ "_free",
+ "_malloc",
+ "_regex_pcre2_bridge_abi_version",
+ "_regex_pcre2_compile_probe",
+ "_regex_pcre2_config_flags",
+ "_regex_pcre2_error_message",
+ "_regex_pcre2_execute",
+ "_regex_pcre2_self_test",
+ "_regex_pcre2_substitute",
+ "_regex_pcre2_trace",
+ "_regex_pcre2_version"
+ ]
+ },
+ "source": {
+ "repository": "https://github.com/PCRE2Project/pcre2.git",
+ "tag": "pcre2-10.47",
+ "tagObjectSha1": "cd007b4466798f66d479d1442a407099e7c40050",
+ "commitSha1": "f454e231fe5006dd7ff8f4693fd2b8eb94333429",
+ "treeSha1": "81a83a3552bd68d0ea7b7004f8bb6e7892f583ba",
+ "license": "BSD-3-Clause WITH PCRE2-exception"
+ },
+ "toolchain": {
+ "emscriptenVersion": "6.0.4",
+ "emscriptenCompilerRevision": "fe5be6afdff43ad58860d821fcc8572a23f92d19",
+ "emsdkCommitSha1": "224ec5f9f2f72f09f9ce0e26d66bae7dbd8b692f",
+ "cmakeVersion": "4.3.4",
+ "ninjaVersion": "1.13.2"
+ },
+ "configuration": {
+ "codeUnitWidth": 8,
+ "unicode": true,
+ "jit": false,
+ "threads": false,
+ "filesystem": false,
+ "initialMemoryBytes": 16777216,
+ "maximumMemoryBytes": 268435456
+ },
+ "sourceDateEpoch": 1760997684,
+ "files": [
+ {
+ "path": "LICENSE.txt",
+ "sha256": "197d8a73ffee0d6b09adba2f9c677b5f5aede24edf89258a68e48248d010d811",
+ "bytes": 4011
+ },
+ {
+ "path": "pcre2.mjs",
+ "sha256": "d31a5f1e0839955ae8321430eafe65da8cbdef832a1245bb71ccc5bb0d4b6b9c",
+ "bytes": 7513
+ },
+ {
+ "path": "pcre2.wasm",
+ "sha256": "cc7c4b2b0038e137c86d2037e73f2302e56432cb09bc082e4f35bba221dd1889",
+ "bytes": 339494
+ }
+ ]
+}
diff --git a/public/engines/pcre2/pcre2.mjs b/public/engines/pcre2/pcre2.mjs
new file mode 100644
index 0000000..b8f1e04
--- /dev/null
+++ b/public/engines/pcre2/pcre2.mjs
@@ -0,0 +1,2 @@
+async function createRegexPcre2(moduleArg={}){var Module=moduleArg;var ENVIRONMENT_IS_WEB=!!globalThis.window;var ENVIRONMENT_IS_WORKER=!!globalThis.WorkerGlobalScope;var ENVIRONMENT_IS_NODE=globalThis.process?.versions?.node&&globalThis.process?.type!="renderer";if(ENVIRONMENT_IS_NODE){const{createRequire}=await import("node:module");var require=createRequire(import.meta.url)}var programArgs=[];var thisProgram="./this.program";var quit_=(status,toThrow)=>{throw toThrow};var _scriptName=import.meta.url;var scriptDirectory="";function locateFile(path){return scriptDirectory+path}var readAsync,readBinary;if(ENVIRONMENT_IS_NODE){var fs=require("node:fs");if(_scriptName.startsWith("file:")){scriptDirectory=require("node:path").dirname(require("node:url").fileURLToPath(_scriptName))+"/"}readBinary=filename=>{filename=isFileURI(filename)?new URL(filename):filename;var ret=fs.readFileSync(filename);return ret};readAsync=async(filename,binary=true)=>{filename=isFileURI(filename)?new URL(filename):filename;var ret=fs.readFileSync(filename,binary?undefined:"utf8");return ret};if(process.argv.length>1){thisProgram=process.argv[1].replace(/\\/g,"/")}programArgs=process.argv.slice(2);quit_=(status,toThrow)=>{process.exitCode=status;throw toThrow}}else if(ENVIRONMENT_IS_WEB||ENVIRONMENT_IS_WORKER){try{scriptDirectory=new URL(".",_scriptName).href}catch{}{if(ENVIRONMENT_IS_WORKER){readBinary=url=>{var xhr=new XMLHttpRequest;xhr.open("GET",url,false);xhr.responseType="arraybuffer";xhr.send(null);return new Uint8Array(xhr.response)}}readAsync=async url=>{var response=await fetch(url,{credentials:"same-origin"});if(response.ok){return response.arrayBuffer()}throw new Error(response.status+" : "+response.url)}}}else{}var out=console.log.bind(console);var err=console.error.bind(console);var wasmBinary;var ABORT=false;var isFileURI=filename=>filename.startsWith("file://");class EmscriptenEH{}class EmscriptenSjLj extends EmscriptenEH{}var runtimeInitialized=false;function getMemoryBuffer(){return wasmMemory.buffer}function updateMemoryViews(){if(HEAP8?.buffer?.resizable)return;var b=getMemoryBuffer();HEAP8=new Int8Array(b);Module["HEAPU8"]=HEAPU8=new Uint8Array(b)}function preRun(){}function initRuntime(){runtimeInitialized=true;wasmExports["c"]()}function postRun(){}function abort(what){what=`Aborted(${what})`;err(what);ABORT=true;what+=". Build with -sASSERTIONS for more info.";var e=new WebAssembly.RuntimeError(what);throw e}var wasmBinaryFile;function findWasmBinary(){if(Module["locateFile"]){return locateFile("pcre2.wasm")}return new URL("pcre2.wasm",import.meta.url).href}function getBinarySync(file){if(readBinary){return readBinary(file)}throw"both async and sync fetching of the wasm failed"}async function getWasmBinary(binaryFile){if(!wasmBinary){try{var response=await readAsync(binaryFile);return new Uint8Array(response)}catch{}}return getBinarySync(binaryFile)}async function instantiateArrayBuffer(binaryFile,imports){try{var binary=await getWasmBinary(binaryFile);var instance=await WebAssembly.instantiate(binary,imports);return instance}catch(reason){err(`failed to asynchronously prepare wasm: ${reason}`);abort(reason)}}async function instantiateAsync(binary,binaryFile,imports){if(!binary&&!ENVIRONMENT_IS_NODE){try{var response=fetch(binaryFile,{credentials:"same-origin"});var instantiationResult=await WebAssembly.instantiateStreaming(response,imports);return instantiationResult}catch(reason){err(`wasm streaming compile failed: ${reason}`);err("falling back to ArrayBuffer instantiation")}}return instantiateArrayBuffer(binaryFile,imports)}function getWasmImports(){var imports={a:wasmImports};return imports}async function createWasm(){function receiveInstance(instance){wasmExports=instance.exports;assignWasmExports(wasmExports);updateMemoryViews();return wasmExports}function receiveInstantiationResult(result){return receiveInstance(result["instance"])}var info=getWasmImports();wasmBinaryFile??=findWasmBinary();var result=await instantiateAsync(wasmBinary,wasmBinaryFile,info);var exports=receiveInstantiationResult(result);return exports}class ExitStatus{name="ExitStatus";constructor(status){this.message=`Program terminated with exit(${status})`;this.status=status}}var HEAP8;var getHeapMax=()=>268435456;var alignMemory=(size,alignment)=>Math.ceil(size/alignment)*alignment;var growMemory=size=>{var oldHeapSize=wasmMemory.buffer.byteLength;var pages=(size-oldHeapSize+65535)/65536|0;try{wasmMemory.grow(pages);updateMemoryViews();return 1}catch(e){}};var HEAPU8;var _emscripten_resize_heap=requestedSize=>{var oldSize=HEAPU8.length;requestedSize>>>=0;var maxHeapSize=getHeapMax();if(requestedSize>maxHeapSize){return false}for(var cutDown=1;cutDown<=4;cutDown*=2){var overGrownHeapSize=oldSize*(1+.2/cutDown);overGrownHeapSize=Math.min(overGrownHeapSize,requestedSize+100663296);var newSize=Math.min(maxHeapSize,alignMemory(Math.max(requestedSize,overGrownHeapSize),65536));var replacement=growMemory(newSize);if(replacement){return true}}return false};var UTF8Decoder=globalThis.TextDecoder&&new TextDecoder;var findStringEnd=(heapOrArray,idx,maxBytesToRead,ignoreNul)=>{var maxIdx=idx+maxBytesToRead;if(ignoreNul)return maxIdx;while(heapOrArray[idx]&&!(idx>=maxIdx))++idx;return idx};var UTF8ArrayToString=(heapOrArray,idx=0,maxBytesToRead,ignoreNul)=>{var endPtr=findStringEnd(heapOrArray,idx,maxBytesToRead,ignoreNul);if(endPtr-idx>16&&heapOrArray.buffer&&UTF8Decoder){return UTF8Decoder.decode(heapOrArray.subarray(idx,endPtr))}var str="";while(idx>10,56320|ch&1023)}}return str};var UTF8ToString=(ptr,maxBytesToRead,ignoreNul)=>ptr?UTF8ArrayToString(HEAPU8,ptr,maxBytesToRead,ignoreNul):"";{}Module["UTF8ToString"]=UTF8ToString;var _regex_pcre2_bridge_abi_version,_regex_pcre2_config_flags,_regex_pcre2_version,_regex_pcre2_error_message,_regex_pcre2_compile_probe,_regex_pcre2_execute,_regex_pcre2_substitute,_regex_pcre2_trace,_regex_pcre2_self_test,_malloc,_free,memory,__indirect_function_table,wasmMemory;function assignWasmExports(wasmExports){_regex_pcre2_bridge_abi_version=Module["_regex_pcre2_bridge_abi_version"]=wasmExports["d"];_regex_pcre2_config_flags=Module["_regex_pcre2_config_flags"]=wasmExports["e"];_regex_pcre2_version=Module["_regex_pcre2_version"]=wasmExports["f"];_regex_pcre2_error_message=Module["_regex_pcre2_error_message"]=wasmExports["g"];_regex_pcre2_compile_probe=Module["_regex_pcre2_compile_probe"]=wasmExports["h"];_regex_pcre2_execute=Module["_regex_pcre2_execute"]=wasmExports["i"];_regex_pcre2_substitute=Module["_regex_pcre2_substitute"]=wasmExports["j"];_regex_pcre2_trace=Module["_regex_pcre2_trace"]=wasmExports["k"];_regex_pcre2_self_test=Module["_regex_pcre2_self_test"]=wasmExports["l"];_malloc=Module["_malloc"]=wasmExports["m"];_free=Module["_free"]=wasmExports["n"];memory=wasmMemory=wasmExports["b"];__indirect_function_table=wasmExports["__indirect_function_table"]}var wasmImports={a:_emscripten_resize_heap};async function run(){preRun();if(ABORT)return;initRuntime();postRun()}var wasmExports;wasmExports=await createWasm();await run();
+;return Module}export default createRegexPcre2;
diff --git a/public/engines/pcre2/pcre2.wasm b/public/engines/pcre2/pcre2.wasm
new file mode 100644
index 0000000..2dee69e
Binary files /dev/null and b/public/engines/pcre2/pcre2.wasm differ
diff --git a/public/toolbox-app.json b/public/toolbox-app.json
index 2322940..a915b18 100644
--- a/public/toolbox-app.json
+++ b/public/toolbox-app.json
@@ -3,7 +3,7 @@
"schemaVersion": 1,
"id": "de.add-ideas.regex-tools",
"name": "Regex Tools",
- "version": "0.1.0",
+ "version": "0.2.0",
"description": "Develop, explain, test and apply regular expressions locally in the browser.",
"entry": "./",
"icon": "./favicon.svg",
diff --git a/scripts/build-engines.mjs b/scripts/build-engines.mjs
index b254aaf..6409c0e 100644
--- a/scripts/build-engines.mjs
+++ b/scripts/build-engines.mjs
@@ -1,8 +1,37 @@
import { readFile, readdir } from "node:fs/promises";
import path from "node:path";
import { fileURLToPath } from "node:url";
+import {
+ buildPcre2Pack,
+ parsePcre2BuildArguments,
+} from "./pcre2-engine-pack.mjs";
const root = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..");
+const options = parsePcre2BuildArguments(process.argv.slice(2));
+
+if (options.mode === "help") {
+ console.log(`Usage:
+ node scripts/build-engines.mjs
+ node scripts/build-engines.mjs --pcre2 --source-dir PATH --emcc PATH [--force]
+
+The PCRE2 command is an explicit offline build gate. It verifies exact local
+source and compiler identities, writes .engine-build/pcre2, and never
+downloads or publishes an engine pack. Install the verified local pack with
+npm run engines:pcre2:install.`);
+ process.exit(0);
+}
+
+if (options.mode === "pcre2") {
+ const result = await buildPcre2Pack(options, root);
+ console.log(
+ `Built and verified staged PCRE2 ${result.metadata.engineVersion} pack at ${result.output}.`,
+ );
+ console.log(
+ "The verified pack remains in private staging until the explicit local install command is run.",
+ );
+ process.exit(0);
+}
+
const packageJson = JSON.parse(
await readFile(path.join(root, "package.json"), "utf8"),
);
@@ -10,15 +39,16 @@ const engineDirectory = path.join(root, "public", "engines");
const entries = (await readdir(engineDirectory)).sort();
if (
- packageJson.version !== "0.1.0" ||
- entries.length !== 1 ||
- entries[0] !== "README.md"
+ typeof packageJson.version !== "string" ||
+ entries.length !== 2 ||
+ entries[0] !== "README.md" ||
+ entries[1] !== "pcre2"
) {
throw new Error(
- "The ECMAScript release must contain only the documented native-browser engine placeholder.",
+ "The production build requires the documented PCRE2 runtime pack.",
);
}
console.log(
- "Regex Tools 0.1.0 uses the native browser RegExp engine; no external engine pack needs building.",
+ "The checked-in PCRE2 runtime pack is present; no release-time download or engine compilation is needed.",
);
diff --git a/scripts/generate-toolbox-manifest.mjs b/scripts/generate-toolbox-manifest.mjs
index d677ac0..2131d63 100644
--- a/scripts/generate-toolbox-manifest.mjs
+++ b/scripts/generate-toolbox-manifest.mjs
@@ -7,14 +7,23 @@ const root = join(dirname(fileURLToPath(import.meta.url)), "..");
const sourcePath = join(root, "src", "toolbox", "manifest.source.json");
const outputPath = join(root, "public", "toolbox-app.json");
const packagePath = join(root, "package.json");
+const applicationVersionPath = join(root, "src", "version.ts");
const checkOnly = process.argv.includes("--check");
const source = JSON.parse(await readFile(sourcePath, "utf8"));
const packageJson = JSON.parse(await readFile(packagePath, "utf8"));
+const applicationVersionSource = await readFile(applicationVersionPath, "utf8");
+const applicationVersion =
+ /^export const APPLICATION_VERSION = "([^"]+)";$/mu.exec(
+ applicationVersionSource,
+ )?.[1];
-if (source.version !== packageJson.version) {
+if (
+ source.version !== packageJson.version ||
+ applicationVersion !== packageJson.version
+) {
throw new Error(
- `Manifest version ${source.version} differs from package version ${packageJson.version}`,
+ `Version drift: manifest ${source.version}, application ${String(applicationVersion)}, package ${packageJson.version}`,
);
}
diff --git a/scripts/install-pcre2-engine.mjs b/scripts/install-pcre2-engine.mjs
new file mode 100644
index 0000000..3a07874
--- /dev/null
+++ b/scripts/install-pcre2-engine.mjs
@@ -0,0 +1,16 @@
+import path from "node:path";
+import { fileURLToPath } from "node:url";
+import { installPcre2Pack } from "./pcre2-engine-pack.mjs";
+
+const root = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..");
+const arguments_ = process.argv.slice(2);
+if (arguments_.length > 1) {
+ throw new Error(
+ "Usage: node scripts/install-pcre2-engine.mjs [.engine-build/pcre2]",
+ );
+}
+const source = path.resolve(root, arguments_[0] ?? ".engine-build/pcre2");
+const result = await installPcre2Pack(source, root);
+console.log(
+ `Installed verified PCRE2 ${result.metadata.engineVersion} runtime assets at ${result.output}.`,
+);
diff --git a/scripts/package-release.mjs b/scripts/package-release.mjs
index e3d62aa..6fe7b91 100644
--- a/scripts/package-release.mjs
+++ b/scripts/package-release.mjs
@@ -18,6 +18,7 @@ const root = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..");
const zipEpoch = new Date(1980, 0, 1, 0, 0, 0);
const gplVersion3TextSha256 =
"fb981668c18a279e285fc4d83fba1e836cc84dd4daa73c9697d3cfd2d8aca6e0";
+const expectedApplicationVersion = "0.2.0";
const requiredRootFiles = [
"LICENSE",
"README.md",
@@ -332,13 +333,27 @@ async function collectReleaseEntries() {
await readFile(path.join(root, "package.json"), "utf8"),
);
if (
- packageJson.version !== "0.1.0" ||
+ packageJson.version !== expectedApplicationVersion ||
packageJson.license !== "GPL-3.0-or-later" ||
packageJson.repository?.url !==
"git+https://git.add-ideas.de/zemion/regex-tools.git"
) {
throw new Error("Package version, licence or source identity drifted.");
}
+ const sourceIdentity = await readFile(path.join(root, "SOURCE.md"), "utf8");
+ if (
+ !sourceIdentity.includes(
+ `- Release version: \`${expectedApplicationVersion}\``,
+ ) ||
+ !sourceIdentity.includes(
+ `- Release tag: \`v${expectedApplicationVersion}\``,
+ ) ||
+ !sourceIdentity.includes(
+ "- Repository: ",
+ )
+ ) {
+ throw new Error("SOURCE.md release identity drifted.");
+ }
const licence = await readFile(path.join(root, "LICENSE"), "utf8");
if (
!licence.includes("GNU GENERAL PUBLIC LICENSE") ||
@@ -371,6 +386,14 @@ async function collectReleaseEntries() {
"CHANGELOG.md",
"SOURCE.md",
"THIRD_PARTY_NOTICES.md",
+ "engines/README.md",
+ "engines/pcre2/LICENSE.txt",
+ "engines/pcre2/SHA256SUMS",
+ "engines/pcre2/engine-metadata.json",
+ "engines/pcre2/pcre2.mjs",
+ "engines/pcre2/pcre2.wasm",
+ "LICENSES/Emscripten-MIT.txt",
+ "LICENSES/PCRE2.txt",
"LICENSES/build/rolldown-1.1.5/LICENSE",
"LICENSES/build/vite-8.1.5/LICENSE.md",
]) {
diff --git a/scripts/pcre2-codegen-golden.test.ts b/scripts/pcre2-codegen-golden.test.ts
new file mode 100644
index 0000000..7963518
--- /dev/null
+++ b/scripts/pcre2-codegen-golden.test.ts
@@ -0,0 +1,110 @@
+import { createHash } from "node:crypto";
+import { mkdtemp, readFile, rm, writeFile } from "node:fs/promises";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import { spawnSync } from "node:child_process";
+import { describe, expect, it } from "vitest";
+import {
+ generatePcre2C,
+ type Pcre2CGenerationRequest,
+} from "../src/regex/codegen/pcre2-c";
+
+function request(operation: "match" | "replace"): Pcre2CGenerationRequest {
+ return {
+ flavour: "pcre2",
+ flavourVersion: "PCRE2 10.47 8-bit WebAssembly",
+ operation,
+ pattern: String.raw`(?\p{L}+)-(\d+)`,
+ flags: ["g"],
+ options: {
+ matchLimit: 1_000_000,
+ depthLimit: 1_000,
+ heapLimitKib: 32_768,
+ },
+ subject: "x Grüße-42 y alpha-7",
+ scanAll: true,
+ ...(operation === "replace" ? { replacement: "$2:$" } : {}),
+ maximumMatches: 100,
+ maximumCaptureRows: 1_000,
+ maximumOutputBytes: 4_096,
+ };
+}
+
+function sha256(value: string): string {
+ return createHash("sha256").update(value).digest("hex");
+}
+
+const exactToolchainPrefix = process.env.PCRE2_CODEGEN_PREFIX;
+const compileWithExactPcre2 = exactToolchainPrefix ? it : it.skip;
+
+describe("PCRE2 C code-generation golden", () => {
+ it("retains a reviewed deterministic source identity", () => {
+ expect(sha256(generatePcre2C(request("match")).source)).toBe(
+ "22b91d578a449b0ed5c30e6eadad30f19a2bcb26ac9db355c8ec7a69328675da",
+ );
+ expect(sha256(generatePcre2C(request("replace")).source)).toBe(
+ "ab90a0d9831ea7b7145aa0072ac352b42baf9a795319d81f531b0f341b04a636",
+ );
+ });
+
+ compileWithExactPcre2(
+ "compiles and executes against the explicitly supplied PCRE2 10.47 C toolchain",
+ async () => {
+ if (!exactToolchainPrefix) return;
+ const header = path.join(exactToolchainPrefix, "include", "pcre2.h");
+ const library =
+ process.env.PCRE2_CODEGEN_LIBRARY ??
+ path.join(exactToolchainPrefix, "lib64", "libpcre2-8.a");
+ await expect(readFile(header)).resolves.toBeInstanceOf(Buffer);
+ await expect(readFile(library)).resolves.toBeInstanceOf(Buffer);
+ const directory = await mkdtemp(
+ path.join(tmpdir(), "regex-tools-codegen-"),
+ );
+ try {
+ for (const operation of ["match", "replace"] as const) {
+ const generated = generatePcre2C(request(operation));
+ const source = path.join(directory, generated.fileName);
+ const executable = path.join(directory, `pcre2-${operation}`);
+ await writeFile(source, generated.source);
+ const compiled = spawnSync(
+ process.env.CC ?? "cc",
+ [
+ "-std=c17",
+ "-Wall",
+ "-Wextra",
+ "-Werror",
+ `-I${path.join(exactToolchainPrefix, "include")}`,
+ source,
+ library,
+ "-o",
+ executable,
+ ],
+ { encoding: "utf8" },
+ );
+ expect(
+ `${compiled.stdout}${compiled.stderr}`,
+ `${operation} fixture did not compile`,
+ ).toBe("");
+ expect(compiled.status).toBe(0);
+ const executed = spawnSync(executable, [], { encoding: "utf8" });
+ expect(
+ executed.status,
+ `${operation} fixture failed: ${executed.stdout}${executed.stderr}`,
+ ).toBe(0);
+ if (operation === "match") {
+ expect(executed.stdout).toContain("match 1: UTF-8 bytes 2..12");
+ expect(executed.stdout).toContain("group 1 / word");
+ expect(executed.stdout).toContain(
+ "completed: 2 match(es), 4 capture row(s)",
+ );
+ } else {
+ expect(executed.stdout).toBe("x 42:Grüße y 7:alpha");
+ expect(executed.stderr).toContain("completed: 2 substitution(s)");
+ }
+ }
+ } finally {
+ await rm(directory, { recursive: true, force: true });
+ }
+ },
+ );
+});
diff --git a/scripts/pcre2-engine-pack.mjs b/scripts/pcre2-engine-pack.mjs
new file mode 100644
index 0000000..fbad0b6
--- /dev/null
+++ b/scripts/pcre2-engine-pack.mjs
@@ -0,0 +1,1018 @@
+import { spawn } from "node:child_process";
+import { createHash } from "node:crypto";
+import {
+ access,
+ constants as fsConstants,
+ copyFile,
+ lstat,
+ mkdir,
+ mkdtemp,
+ readFile,
+ readdir,
+ realpath,
+ rename,
+ rm,
+ stat,
+ writeFile,
+} from "node:fs/promises";
+import path from "node:path";
+import { pathToFileURL } from "node:url";
+
+export const PCRE2_LOCK = Object.freeze({
+ bridgeAbi: 3,
+ maximumPatternBytes: 1_048_576,
+ maximumSubjectBytes: 16_777_216,
+ maximumReplacementBytes: 262_144,
+ maximumMatches: 10_000,
+ maximumCaptureRows: 100_000,
+ maximumCaptureGroups: 1_000,
+ maximumTraceEvents: 50_000,
+ maximumTraceBytes: 10_485_760,
+ maximumTraceMarkBytes: 1_024,
+ sourceDateEpoch: 1_760_997_684,
+ source: Object.freeze({
+ repository: "https://github.com/PCRE2Project/pcre2.git",
+ tag: "pcre2-10.47",
+ tagObject: "cd007b4466798f66d479d1442a407099e7c40050",
+ commit: "f454e231fe5006dd7ff8f4693fd2b8eb94333429",
+ tree: "81a83a3552bd68d0ea7b7004f8bb6e7892f583ba",
+ version: "10.47 2025-10-21",
+ license: "BSD-3-Clause WITH PCRE2-exception",
+ criticalFiles: Object.freeze({
+ "CMakeLists.txt":
+ "74f65935e2b1120e7d7aeb4be81755e0f7a875732467ae64a9cc6ee7174fa5ba",
+ "LICENCE.md":
+ "197d8a73ffee0d6b09adba2f9c677b5f5aede24edf89258a68e48248d010d811",
+ "src/pcre2.h.generic":
+ "058462dc030e845ee627e19ac36347f561f5cc44d89eaf1c0010f2b363ab68d5",
+ }),
+ }),
+ emscripten: Object.freeze({
+ version: "6.0.4",
+ compilerRevision: "fe5be6afdff43ad58860d821fcc8572a23f92d19",
+ repositoryCommit: "224ec5f9f2f72f09f9ce0e26d66bae7dbd8b692f",
+ emccSha256:
+ "498eba9b6aacffca1c358ba340fc7b37015d7867cb211c7f2d45784ee9310f03",
+ emcmakeSha256:
+ "18c8397cf60835533e0db88d1172e04fa0b753dea68e978149d923a9e6378af3",
+ }),
+ cmakeVersion: "4.3.4",
+ ninjaVersion: "1.13.2",
+});
+
+const PACK_FILES = Object.freeze([
+ "LICENSE.txt",
+ "SHA256SUMS",
+ "engine-metadata.json",
+ "pcre2.mjs",
+ "pcre2.wasm",
+]);
+const CHECKSUM_FILES = Object.freeze([
+ "LICENSE.txt",
+ "engine-metadata.json",
+ "pcre2.mjs",
+ "pcre2.wasm",
+]);
+const BRIDGE_SOURCE_FILES = Object.freeze([
+ "engines/pcre2/CMakeLists.txt",
+ "engines/pcre2/pcre2_bridge.c",
+ "engines/pcre2/pcre2_bridge.h",
+]);
+const EXPORTED_FUNCTIONS = Object.freeze([
+ "_free",
+ "_malloc",
+ "_regex_pcre2_bridge_abi_version",
+ "_regex_pcre2_compile_probe",
+ "_regex_pcre2_config_flags",
+ "_regex_pcre2_error_message",
+ "_regex_pcre2_execute",
+ "_regex_pcre2_self_test",
+ "_regex_pcre2_substitute",
+ "_regex_pcre2_trace",
+ "_regex_pcre2_version",
+]);
+
+function sha256(data) {
+ return createHash("sha256").update(data).digest("hex");
+}
+
+async function sha256File(file) {
+ return sha256(await readFile(file));
+}
+
+function commandLabel(command, arguments_) {
+ return [command, ...arguments_]
+ .map((part) =>
+ /^[A-Za-z0-9_./:=,+-]+$/u.test(part) ? part : JSON.stringify(part),
+ )
+ .join(" ");
+}
+
+async function run(command, arguments_, options = {}) {
+ const child = spawn(command, arguments_, {
+ cwd: options.cwd,
+ env: options.env,
+ stdio: ["ignore", "pipe", "pipe"],
+ });
+ const stdout = [];
+ const stderr = [];
+ child.stdout.on("data", (chunk) => stdout.push(chunk));
+ child.stderr.on("data", (chunk) => stderr.push(chunk));
+
+ const result = await new Promise((resolve, reject) => {
+ child.once("error", reject);
+ child.once("close", (code, signal) => resolve({ code, signal }));
+ });
+ const output = Buffer.concat(stdout).toString("utf8");
+ const errorOutput = Buffer.concat(stderr).toString("utf8");
+ if (result.code !== 0) {
+ const detail = [output, errorOutput].filter(Boolean).join("\n").trim();
+ throw new Error(
+ `${commandLabel(command, arguments_)} failed with ${
+ result.signal ? `signal ${result.signal}` : `exit code ${result.code}`
+ }${detail ? `:\n${detail}` : "."}`,
+ );
+ }
+ return { stdout: output, stderr: errorOutput };
+}
+
+async function assertRealDirectory(candidate, label) {
+ const details = await lstat(candidate).catch(() => null);
+ if (!details?.isDirectory() || details.isSymbolicLink()) {
+ throw new Error(`${label} must be a real directory: ${candidate}`);
+ }
+ return realpath(candidate);
+}
+
+async function assertRegularFile(candidate, label) {
+ const details = await lstat(candidate).catch(() => null);
+ if (!details?.isFile() || details.isSymbolicLink()) {
+ throw new Error(
+ `${label} must be a regular, non-symlink file: ${candidate}`,
+ );
+ }
+ return realpath(candidate);
+}
+
+async function assertHash(file, expected, label) {
+ const actual = await sha256File(file);
+ if (actual !== expected) {
+ throw new Error(
+ `${label} SHA-256 mismatch: expected ${expected}, got ${actual}.`,
+ );
+ }
+}
+
+async function resolveExecutable(name, environment = process.env) {
+ const searchPath = environment.PATH ?? "";
+ for (const directory of searchPath.split(path.delimiter)) {
+ if (!directory) continue;
+ const candidate = path.join(directory, name);
+ try {
+ await access(candidate, fsConstants.X_OK);
+ const details = await stat(candidate);
+ if (details.isFile()) return realpath(candidate);
+ } catch {
+ // Continue through PATH.
+ }
+ }
+ throw new Error(`${name} is required on PATH for the explicit PCRE2 build.`);
+}
+
+function singleVersionLine(output) {
+ return output.trim().split(/\r?\n/u)[0] ?? "";
+}
+
+async function verifyPcre2Source(sourceDirectory) {
+ const source = await assertRealDirectory(sourceDirectory, "PCRE2 source");
+ const git = await resolveExecutable("git");
+ const status = await run(
+ git,
+ ["status", "--porcelain=v1", "--untracked-files=all"],
+ { cwd: source },
+ );
+ if (status.stdout !== "") {
+ throw new Error("The PCRE2 source checkout must be completely clean.");
+ }
+
+ const checks = [
+ [["rev-parse", "HEAD"], PCRE2_LOCK.source.commit, "PCRE2 commit"],
+ [["rev-parse", "HEAD^{tree}"], PCRE2_LOCK.source.tree, "PCRE2 source tree"],
+ [
+ ["rev-parse", `refs/tags/${PCRE2_LOCK.source.tag}^{tag}`],
+ PCRE2_LOCK.source.tagObject,
+ "PCRE2 signed tag object",
+ ],
+ [
+ ["rev-parse", `refs/tags/${PCRE2_LOCK.source.tag}^{commit}`],
+ PCRE2_LOCK.source.commit,
+ "PCRE2 tag target",
+ ],
+ ];
+ for (const [arguments_, expected, label] of checks) {
+ const { stdout } = await run(git, arguments_, { cwd: source });
+ const actual = stdout.trim();
+ if (actual !== expected) {
+ throw new Error(
+ `${label} mismatch: expected ${expected}, got ${actual}.`,
+ );
+ }
+ }
+
+ for (const [relativePath, expected] of Object.entries(
+ PCRE2_LOCK.source.criticalFiles,
+ )) {
+ await assertHash(
+ path.join(source, relativePath),
+ expected,
+ `PCRE2 ${relativePath}`,
+ );
+ }
+ return source;
+}
+
+async function verifyEmscripten(emccFile) {
+ const emcc = await assertRegularFile(emccFile, "emcc");
+ const emcmake = await assertRegularFile(
+ path.join(path.dirname(emcc), "emcmake"),
+ "emcmake",
+ );
+ await assertHash(emcc, PCRE2_LOCK.emscripten.emccSha256, "emcc");
+ await assertHash(emcmake, PCRE2_LOCK.emscripten.emcmakeSha256, "emcmake");
+
+ const version = await run(emcc, ["--version"]);
+ const expectedVersion =
+ `emcc (Emscripten gcc/clang-like replacement + linker emulating GNU ld) ` +
+ `${PCRE2_LOCK.emscripten.version} (${PCRE2_LOCK.emscripten.compilerRevision})`;
+ if (singleVersionLine(version.stdout) !== expectedVersion) {
+ throw new Error(
+ `Emscripten identity mismatch: expected ${expectedVersion}, got ${singleVersionLine(
+ version.stdout,
+ )}.`,
+ );
+ }
+
+ const git = await resolveExecutable("git");
+ const repository = (
+ await run(git, ["rev-parse", "--show-toplevel"], {
+ cwd: path.dirname(emcc),
+ })
+ ).stdout.trim();
+ const commit = (
+ await run(git, ["rev-parse", "HEAD"], { cwd: repository })
+ ).stdout.trim();
+ if (commit !== PCRE2_LOCK.emscripten.repositoryCommit) {
+ throw new Error(
+ `Emscripten repository mismatch: expected ${PCRE2_LOCK.emscripten.repositoryCommit}, got ${commit}.`,
+ );
+ }
+ const tagTarget = (
+ await run(
+ git,
+ ["rev-parse", `refs/tags/${PCRE2_LOCK.emscripten.version}^{commit}`],
+ {
+ cwd: repository,
+ },
+ )
+ ).stdout.trim();
+ if (tagTarget !== PCRE2_LOCK.emscripten.repositoryCommit) {
+ throw new Error(
+ "The Emscripten 6.0.4 tag does not resolve to the pinned commit.",
+ );
+ }
+ const status = await run(
+ git,
+ ["status", "--porcelain=v1", "--untracked-files=all"],
+ { cwd: repository },
+ );
+ if (status.stdout !== "") {
+ throw new Error("The Emscripten checkout must be completely clean.");
+ }
+ return { emcc, emcmake };
+}
+
+async function verifyNativeBuildTools() {
+ const cmake = await resolveExecutable("cmake");
+ const ninja = await resolveExecutable("ninja");
+ const cmakeOutput = await run(cmake, ["--version"]);
+ const ninjaOutput = await run(ninja, ["--version"]);
+ if (
+ singleVersionLine(cmakeOutput.stdout) !==
+ `cmake version ${PCRE2_LOCK.cmakeVersion}`
+ ) {
+ throw new Error(
+ `The PCRE2 pack requires CMake ${PCRE2_LOCK.cmakeVersion}.`,
+ );
+ }
+ if (ninjaOutput.stdout.trim() !== PCRE2_LOCK.ninjaVersion) {
+ throw new Error(
+ `The PCRE2 pack requires Ninja ${PCRE2_LOCK.ninjaVersion}.`,
+ );
+ }
+ return { cmake, ninja };
+}
+
+function parseValue(argv, index, option) {
+ const value = argv[index + 1];
+ if (!value || value.startsWith("--")) {
+ throw new Error(`${option} requires a path.`);
+ }
+ return value;
+}
+
+export function parsePcre2BuildArguments(argv) {
+ if (argv.length === 0) return { mode: "ecmascript" };
+ if (argv.length === 1 && argv[0] === "--help") return { mode: "help" };
+
+ let pcre2 = false;
+ let sourceDirectory;
+ let emccFile;
+ let force = false;
+ for (let index = 0; index < argv.length; index += 1) {
+ const argument = argv[index];
+ if (argument === "--pcre2") {
+ if (pcre2) throw new Error("--pcre2 may be supplied only once.");
+ pcre2 = true;
+ } else if (argument === "--source-dir") {
+ if (sourceDirectory)
+ throw new Error("--source-dir may be supplied only once.");
+ sourceDirectory = parseValue(argv, index, argument);
+ index += 1;
+ } else if (argument === "--emcc") {
+ if (emccFile) throw new Error("--emcc may be supplied only once.");
+ emccFile = parseValue(argv, index, argument);
+ index += 1;
+ } else if (argument === "--force") {
+ if (force) throw new Error("--force may be supplied only once.");
+ force = true;
+ } else {
+ throw new Error(`Unknown engine-build argument: ${argument}`);
+ }
+ }
+ if (!pcre2) {
+ throw new Error(
+ "External engine options require the explicit --pcre2 gate.",
+ );
+ }
+ if (!sourceDirectory || !emccFile) {
+ throw new Error(
+ "The offline PCRE2 build requires both --source-dir and --emcc; it never downloads them.",
+ );
+ }
+ return { mode: "pcre2", sourceDirectory, emccFile, force };
+}
+
+async function bridgeSourceFiles(root) {
+ return Promise.all(
+ BRIDGE_SOURCE_FILES.map(async (relativePath) => {
+ const contents = await readFile(path.join(root, relativePath));
+ return {
+ path: relativePath,
+ sha256: sha256(contents),
+ bytes: contents.byteLength,
+ };
+ }),
+ );
+}
+
+async function describedFile(directory, name) {
+ const contents = await readFile(path.join(directory, name));
+ return { path: name, sha256: sha256(contents), bytes: contents.byteLength };
+}
+
+async function expectedMetadata(root, packDirectory) {
+ const files = await Promise.all(
+ ["LICENSE.txt", "pcre2.mjs", "pcre2.wasm"].map((name) =>
+ describedFile(packDirectory, name),
+ ),
+ );
+ return {
+ schemaVersion: 1,
+ engine: "pcre2",
+ engineVersion: PCRE2_LOCK.source.version,
+ status: "production-runtime",
+ bridge: {
+ abiVersion: PCRE2_LOCK.bridgeAbi,
+ maximumPatternBytes: PCRE2_LOCK.maximumPatternBytes,
+ maximumSubjectBytes: PCRE2_LOCK.maximumSubjectBytes,
+ maximumReplacementBytes: PCRE2_LOCK.maximumReplacementBytes,
+ maximumMatches: PCRE2_LOCK.maximumMatches,
+ maximumCaptureRows: PCRE2_LOCK.maximumCaptureRows,
+ maximumCaptureGroups: PCRE2_LOCK.maximumCaptureGroups,
+ maximumTraceEvents: PCRE2_LOCK.maximumTraceEvents,
+ maximumTraceBytes: PCRE2_LOCK.maximumTraceBytes,
+ maximumTraceMarkBytes: PCRE2_LOCK.maximumTraceMarkBytes,
+ sourceFiles: await bridgeSourceFiles(root),
+ exports: [...EXPORTED_FUNCTIONS],
+ },
+ source: {
+ repository: PCRE2_LOCK.source.repository,
+ tag: PCRE2_LOCK.source.tag,
+ tagObjectSha1: PCRE2_LOCK.source.tagObject,
+ commitSha1: PCRE2_LOCK.source.commit,
+ treeSha1: PCRE2_LOCK.source.tree,
+ license: PCRE2_LOCK.source.license,
+ },
+ toolchain: {
+ emscriptenVersion: PCRE2_LOCK.emscripten.version,
+ emscriptenCompilerRevision: PCRE2_LOCK.emscripten.compilerRevision,
+ emsdkCommitSha1: PCRE2_LOCK.emscripten.repositoryCommit,
+ cmakeVersion: PCRE2_LOCK.cmakeVersion,
+ ninjaVersion: PCRE2_LOCK.ninjaVersion,
+ },
+ configuration: {
+ codeUnitWidth: 8,
+ unicode: true,
+ jit: false,
+ threads: false,
+ filesystem: false,
+ initialMemoryBytes: 16_777_216,
+ maximumMemoryBytes: 268_435_456,
+ },
+ sourceDateEpoch: PCRE2_LOCK.sourceDateEpoch,
+ files,
+ };
+}
+
+function deterministicJson(value) {
+ return `${JSON.stringify(value, null, 2)}\n`;
+}
+
+async function expectedChecksums(packDirectory) {
+ const entries = await Promise.all(
+ CHECKSUM_FILES.map(
+ async (name) =>
+ `${await sha256File(path.join(packDirectory, name))} ${name}`,
+ ),
+ );
+ return `${entries.join("\n")}\n`;
+}
+
+function equalJson(left, right) {
+ return JSON.stringify(left) === JSON.stringify(right);
+}
+
+async function smokePcre2Pack(packDirectory) {
+ const wasm = await readFile(path.join(packDirectory, "pcre2.wasm"));
+ if (
+ wasm.byteLength < 8 ||
+ !wasm.subarray(0, 4).equals(Buffer.from([0x00, 0x61, 0x73, 0x6d])) ||
+ !WebAssembly.validate(wasm)
+ ) {
+ throw new Error("pcre2.wasm is not a valid WebAssembly module.");
+ }
+
+ const moduleUrl = `${pathToFileURL(path.join(packDirectory, "pcre2.mjs")).href}?sha256=${sha256(
+ await readFile(path.join(packDirectory, "pcre2.mjs")),
+ )}`;
+ const imported = await import(moduleUrl);
+ if (typeof imported.default !== "function") {
+ throw new Error("pcre2.mjs does not export its Emscripten factory.");
+ }
+ const diagnostics = [];
+ const module = await imported.default({
+ locateFile(name) {
+ return path.join(packDirectory, name);
+ },
+ print() {},
+ printErr(line) {
+ diagnostics.push(String(line));
+ },
+ });
+
+ if (module._regex_pcre2_bridge_abi_version() !== PCRE2_LOCK.bridgeAbi) {
+ throw new Error("The WebAssembly bridge ABI version is incorrect.");
+ }
+ if (module._regex_pcre2_config_flags() !== 1) {
+ throw new Error(
+ "The WebAssembly pack must have Unicode enabled and JIT disabled.",
+ );
+ }
+ if (
+ module.UTF8ToString(module._regex_pcre2_version()) !==
+ PCRE2_LOCK.source.version
+ ) {
+ throw new Error("The WebAssembly PCRE2 version is incorrect.");
+ }
+ if (module._regex_pcre2_self_test() !== 0) {
+ throw new Error(
+ `The WebAssembly bridge self-test failed: ${diagnostics.join("\n")}`,
+ );
+ }
+
+ const pattern = new TextEncoder().encode("(?\\p{L}+)-(?\\d+)");
+ const patternPointer = module._malloc(pattern.byteLength);
+ const resultPointer = module._malloc(20);
+ if (!patternPointer || !resultPointer) {
+ if (patternPointer) module._free(patternPointer);
+ if (resultPointer) module._free(resultPointer);
+ throw new Error("The WebAssembly smoke test could not allocate memory.");
+ }
+ try {
+ module.HEAPU8.set(pattern, patternPointer);
+ module.HEAPU8.fill(0, resultPointer, resultPointer + 20);
+ const status = module._regex_pcre2_compile_probe(
+ patternPointer,
+ pattern.byteLength,
+ 0x20 | 0x40,
+ resultPointer,
+ );
+ const result = new DataView(module.HEAPU8.buffer, resultPointer, 20);
+ if (
+ status !== 0 ||
+ result.getInt32(0, true) !== 0 ||
+ result.getUint32(8, true) !== 2 ||
+ result.getUint32(12, true) !== 2
+ ) {
+ throw new Error("The WebAssembly compile-probe result is incorrect.");
+ }
+
+ module.HEAPU8.fill(0, resultPointer, resultPointer + 20);
+ const unsupportedStatus = module._regex_pcre2_compile_probe(
+ patternPointer,
+ pattern.byteLength,
+ 0x80000000,
+ resultPointer,
+ );
+ if (unsupportedStatus !== -10002 || result.getInt32(0, true) !== -10002) {
+ throw new Error("The WebAssembly bridge accepted an unknown option bit.");
+ }
+
+ module.HEAPU8[patternPointer] = 0x28;
+ module.HEAPU8.fill(0, resultPointer, resultPointer + 20);
+ const invalidStatus = module._regex_pcre2_compile_probe(
+ patternPointer,
+ 1,
+ 0,
+ resultPointer,
+ );
+ if (
+ invalidStatus <= 0 ||
+ result.getInt32(0, true) !== invalidStatus ||
+ result.getUint32(4, true) > 1
+ ) {
+ throw new Error(
+ "The WebAssembly bridge did not preserve the compile error.",
+ );
+ }
+
+ if (
+ module._regex_pcre2_compile_probe(0, 1, 0, resultPointer) !== -10001 ||
+ module._regex_pcre2_compile_probe(
+ patternPointer,
+ PCRE2_LOCK.maximumPatternBytes + 1,
+ 0,
+ resultPointer,
+ ) !== -10003
+ ) {
+ throw new Error("The WebAssembly bridge input limits are not enforced.");
+ }
+ } finally {
+ module._free(resultPointer);
+ module._free(patternPointer);
+ }
+
+ const runPattern = new TextEncoder().encode("(?😀)|(?\\p{L}+)");
+ const runSubject = new TextEncoder().encode("x😀 Grüße");
+ const runPointers = [];
+ const allocate = (bytes) => {
+ const pointer = module._malloc(Math.max(1, bytes));
+ if (!pointer) throw new Error("The PCRE2 runtime smoke allocation failed.");
+ runPointers.push(pointer);
+ return pointer;
+ };
+ try {
+ const runPatternPointer = allocate(runPattern.byteLength);
+ const runSubjectPointer = allocate(runSubject.byteLength);
+ const limitsPointer = allocate(20);
+ const recordsPointer = allocate(20 * 16);
+ const namesPointer = allocate(10 * 12);
+ const nameBytesPointer = allocate(256);
+ const runResultPointer = allocate(60);
+ module.HEAPU8.set(runPattern, runPatternPointer);
+ module.HEAPU8.set(runSubject, runSubjectPointer);
+ let view = new DataView(module.HEAPU8.buffer);
+ view.setUint32(limitsPointer, 10, true);
+ view.setUint32(limitsPointer + 4, 100, true);
+ view.setUint32(limitsPointer + 8, 1_000_000, true);
+ view.setUint32(limitsPointer + 12, 1_000, true);
+ view.setUint32(limitsPointer + 16, 32_768, true);
+ module.HEAPU8.fill(0, runResultPointer, runResultPointer + 60);
+ const runStatus = module._regex_pcre2_execute(
+ runPatternPointer,
+ runPattern.byteLength,
+ runSubjectPointer,
+ runSubject.byteLength,
+ 0x20 | 0x40 | 0x100,
+ limitsPointer,
+ recordsPointer,
+ 20,
+ namesPointer,
+ 10,
+ nameBytesPointer,
+ 256,
+ runResultPointer,
+ );
+ view = new DataView(module.HEAPU8.buffer);
+ if (
+ runStatus !== 0 ||
+ view.getInt32(runResultPointer, true) !== 0 ||
+ view.getUint32(runResultPointer + 12, true) !== 2 ||
+ view.getUint32(runResultPointer + 20, true) !== 3 ||
+ view.getUint32(runResultPointer + 24, true) !== 9 ||
+ view.getUint32(recordsPointer + 3 * 16 + 8, true) !== 1 ||
+ view.getUint32(recordsPointer + 3 * 16 + 12, true) !== 5
+ ) {
+ throw new Error("The WebAssembly bounded-match smoke result is wrong.");
+ }
+
+ const replacementPattern = new TextEncoder().encode("(?\\p{L}+)");
+ const replacementSubject = new TextEncoder().encode("Grüße 42");
+ const replacement = new TextEncoder().encode("${word}!");
+ const replacementPatternPointer = allocate(replacementPattern.byteLength);
+ const replacementSubjectPointer = allocate(replacementSubject.byteLength);
+ const replacementPointer = allocate(replacement.byteLength);
+ const outputPointer = allocate(65);
+ module.HEAPU8.set(replacementPattern, replacementPatternPointer);
+ module.HEAPU8.set(replacementSubject, replacementSubjectPointer);
+ module.HEAPU8.set(replacement, replacementPointer);
+ module.HEAPU8.fill(0, runResultPointer, runResultPointer + 60);
+ const substituteStatus = module._regex_pcre2_substitute(
+ replacementPatternPointer,
+ replacementPattern.byteLength,
+ replacementSubjectPointer,
+ replacementSubject.byteLength,
+ replacementPointer,
+ replacement.byteLength,
+ 0x20 | 0x40 | 0x100,
+ limitsPointer,
+ recordsPointer,
+ 20,
+ namesPointer,
+ 10,
+ nameBytesPointer,
+ 256,
+ outputPointer,
+ 64,
+ runResultPointer,
+ );
+ view = new DataView(module.HEAPU8.buffer);
+ const outputLength = view.getUint32(runResultPointer + 48, true);
+ const output = new TextDecoder().decode(
+ module.HEAPU8.slice(outputPointer, outputPointer + outputLength),
+ );
+ if (
+ substituteStatus !== 0 ||
+ output !== "Grüße! 42" ||
+ view.getUint32(runResultPointer + 52, true) !== 1 ||
+ view.getUint32(runResultPointer + 56, true) !== 0
+ ) {
+ throw new Error(
+ "The WebAssembly bounded-substitution smoke result is wrong.",
+ );
+ }
+ module.HEAPU8.fill(0, runResultPointer, runResultPointer + 60);
+ const missingOutputStatus = module._regex_pcre2_substitute(
+ replacementPatternPointer,
+ replacementPattern.byteLength,
+ replacementSubjectPointer,
+ replacementSubject.byteLength,
+ replacementPointer,
+ replacement.byteLength,
+ 0x20 | 0x40,
+ limitsPointer,
+ recordsPointer,
+ 20,
+ namesPointer,
+ 10,
+ nameBytesPointer,
+ 256,
+ 0,
+ 0,
+ runResultPointer,
+ );
+ if (
+ missingOutputStatus !== -10001 ||
+ view.getInt32(runResultPointer, true) !== -10001
+ ) {
+ throw new Error(
+ "The WebAssembly substitution bridge accepted a missing NUL byte.",
+ );
+ }
+
+ const traceEventsPointer = allocate(64 * 32);
+ const traceMarksPointer = allocate(256);
+ const traceResultPointer = allocate(52);
+ view.setUint32(limitsPointer, 64, true);
+ view.setUint32(limitsPointer + 4, 4_096, true);
+ module.HEAPU8.fill(0, traceResultPointer, traceResultPointer + 52);
+ const traceStatus = module._regex_pcre2_trace(
+ runPatternPointer,
+ runPattern.byteLength,
+ runSubjectPointer,
+ runSubject.byteLength,
+ 0x20 | 0x40,
+ limitsPointer,
+ traceEventsPointer,
+ 64,
+ traceMarksPointer,
+ 256,
+ traceResultPointer,
+ );
+ view = new DataView(module.HEAPU8.buffer);
+ const traceEventCount = view.getUint32(traceResultPointer + 20, true);
+ if (
+ traceStatus !== 0 ||
+ view.getInt32(traceResultPointer, true) !== 0 ||
+ view.getInt32(traceResultPointer + 12, true) <= 0 ||
+ traceEventCount === 0 ||
+ view.getUint32(traceResultPointer + 24, true) !== traceEventCount ||
+ view.getUint32(traceResultPointer + 36, true) !== 0 ||
+ view.getUint32(traceResultPointer + 44, true) !== traceEventCount - 1 ||
+ view.getUint32(traceResultPointer + 48, true) !== 1
+ ) {
+ throw new Error("The WebAssembly automatic-callout trace is wrong.");
+ }
+
+ view.setUint32(limitsPointer, 1, true);
+ module.HEAPU8.fill(0, traceResultPointer, traceResultPointer + 52);
+ const boundedTraceStatus = module._regex_pcre2_trace(
+ runPatternPointer,
+ runPattern.byteLength,
+ runSubjectPointer,
+ runSubject.byteLength,
+ 0x20 | 0x40,
+ limitsPointer,
+ traceEventsPointer,
+ 64,
+ traceMarksPointer,
+ 256,
+ traceResultPointer,
+ );
+ if (
+ boundedTraceStatus !== 0 ||
+ view.getInt32(traceResultPointer, true) !== 0 ||
+ view.getInt32(traceResultPointer + 12, true) !== -37 ||
+ view.getUint32(traceResultPointer + 20, true) !== 1 ||
+ view.getUint32(traceResultPointer + 24, true) !== 2 ||
+ view.getUint32(traceResultPointer + 36, true) !== 1 ||
+ view.getUint32(traceResultPointer + 44, true) !== 0
+ ) {
+ throw new Error("The WebAssembly trace cap is not authoritative.");
+ }
+ } finally {
+ for (const pointer of runPointers.reverse()) module._free(pointer);
+ }
+}
+
+export async function verifyPcre2Pack(packDirectory, root) {
+ const repositoryRoot = await assertRealDirectory(
+ root,
+ "Regex Tools repository",
+ );
+ const pack = await assertRealDirectory(packDirectory, "PCRE2 pack");
+ const entries = (await readdir(pack, { withFileTypes: true })).sort(
+ (left, right) =>
+ left.name < right.name ? -1 : left.name > right.name ? 1 : 0,
+ );
+ if (
+ entries.length !== PACK_FILES.length ||
+ entries.some(
+ (entry, index) =>
+ entry.name !== PACK_FILES[index] ||
+ !entry.isFile() ||
+ entry.isSymbolicLink(),
+ )
+ ) {
+ throw new Error(
+ `The staged PCRE2 pack must contain only: ${PACK_FILES.join(", ")}.`,
+ );
+ }
+
+ await assertHash(
+ path.join(pack, "LICENSE.txt"),
+ PCRE2_LOCK.source.criticalFiles["LICENCE.md"],
+ "staged PCRE2 licence",
+ );
+ const moduleText = await readFile(path.join(pack, "pcre2.mjs"), "utf8");
+ if (
+ moduleText.includes("sourceMappingURL") ||
+ moduleText.includes("/mnt/") ||
+ moduleText.includes("/home/")
+ ) {
+ throw new Error("pcre2.mjs leaks a build path or source-map reference.");
+ }
+
+ let metadata;
+ try {
+ metadata = JSON.parse(
+ await readFile(path.join(pack, "engine-metadata.json"), "utf8"),
+ );
+ } catch (error) {
+ throw new Error("engine-metadata.json is not valid JSON.", {
+ cause: error,
+ });
+ }
+ const expected = await expectedMetadata(repositoryRoot, pack);
+ if (!equalJson(metadata, expected)) {
+ throw new Error(
+ "The staged PCRE2 metadata does not match the pinned build contract.",
+ );
+ }
+ const checksums = await readFile(path.join(pack, "SHA256SUMS"), "utf8");
+ if (checksums !== (await expectedChecksums(pack))) {
+ throw new Error("The staged PCRE2 SHA256SUMS file is incorrect.");
+ }
+ await smokePcre2Pack(pack);
+ return metadata;
+}
+
+export async function installPcre2Pack(packDirectory, root) {
+ const repositoryRoot = await assertRealDirectory(
+ root,
+ "Regex Tools repository",
+ );
+ const sourcePack = await assertRealDirectory(
+ packDirectory,
+ "verified PCRE2 pack",
+ );
+ const metadata = await verifyPcre2Pack(sourcePack, repositoryRoot);
+ const engineRoot = await assertRealDirectory(
+ path.join(repositoryRoot, "public", "engines"),
+ "public engine directory",
+ );
+ const target = path.join(engineRoot, "pcre2");
+ const stage = await mkdtemp(path.join(engineRoot, ".pcre2-install-"));
+ const targetDetails = await lstat(target).catch(() => null);
+ if (
+ targetDetails &&
+ (!targetDetails.isDirectory() || targetDetails.isSymbolicLink())
+ ) {
+ await rm(stage, { recursive: true, force: true });
+ throw new Error("public/engines/pcre2 must be a real directory.");
+ }
+ try {
+ for (const name of PACK_FILES) {
+ await copyFile(
+ path.join(sourcePack, name),
+ path.join(stage, name),
+ fsConstants.COPYFILE_EXCL,
+ );
+ }
+ await verifyPcre2Pack(stage, repositoryRoot);
+ if (targetDetails) {
+ const backup = path.join(engineRoot, `.pcre2-replaced-${process.pid}`);
+ await rename(target, backup);
+ try {
+ await rename(stage, target);
+ } catch (error) {
+ await rename(backup, target);
+ throw error;
+ }
+ await rm(backup, { recursive: true });
+ } else {
+ await rename(stage, target);
+ }
+ return { output: target, metadata };
+ } finally {
+ await rm(stage, { recursive: true, force: true });
+ }
+}
+
+function cleanBuildEnvironment() {
+ const environment = { ...process.env };
+ for (const variable of [
+ "CFLAGS",
+ "CPPFLAGS",
+ "CXXFLAGS",
+ "DESTDIR",
+ "EMCC_CFLAGS",
+ "EMMAKEN_CFLAGS",
+ "LDFLAGS",
+ ]) {
+ delete environment[variable];
+ }
+ environment.LANG = "C";
+ environment.LC_ALL = "C";
+ environment.SOURCE_DATE_EPOCH = String(PCRE2_LOCK.sourceDateEpoch);
+ environment.TZ = "UTC";
+ environment.ZERO_AR_DATE = "1";
+ return environment;
+}
+
+async function ensurePrivateBuildRoot(root) {
+ const buildRoot = path.join(root, ".engine-build");
+ const existing = await lstat(buildRoot).catch(() => null);
+ if (existing && (!existing.isDirectory() || existing.isSymbolicLink())) {
+ throw new Error(".engine-build must be a real directory.");
+ }
+ await mkdir(buildRoot, { recursive: true });
+ return realpath(buildRoot);
+}
+
+function assertGeneratedChild(buildRoot, candidate) {
+ const relative = path.relative(buildRoot, candidate);
+ if (!relative || relative.startsWith("..") || path.isAbsolute(relative)) {
+ throw new Error(
+ `Refusing to remove a path outside .engine-build: ${candidate}`,
+ );
+ }
+}
+
+export async function buildPcre2Pack(options, root) {
+ const repositoryRoot = await assertRealDirectory(
+ root,
+ "Regex Tools repository",
+ );
+ const source = await verifyPcre2Source(options.sourceDirectory);
+ const { emcmake } = await verifyEmscripten(options.emccFile);
+ const { cmake, ninja } = await verifyNativeBuildTools();
+ const buildRoot = await ensurePrivateBuildRoot(repositoryRoot);
+ const output = path.join(buildRoot, "pcre2");
+ const outputDetails = await lstat(output).catch(() => null);
+ if (outputDetails && !options.force) {
+ throw new Error(
+ `${output} already exists; pass --force to replace that exact staged pack.`,
+ );
+ }
+
+ const work = await mkdtemp(path.join(buildRoot, ".pcre2-work-"));
+ const stage = await mkdtemp(path.join(buildRoot, ".pcre2-stage-"));
+ assertGeneratedChild(buildRoot, work);
+ assertGeneratedChild(buildRoot, stage);
+ const environment = cleanBuildEnvironment();
+ try {
+ await run(
+ emcmake,
+ [
+ cmake,
+ "-S",
+ path.join(repositoryRoot, "engines", "pcre2"),
+ "-B",
+ work,
+ "-G",
+ "Ninja",
+ "-DCMAKE_BUILD_TYPE=Release",
+ `-DCMAKE_MAKE_PROGRAM=${ninja}`,
+ `-DPCRE2_SOURCE_DIR=${source}`,
+ `-DREGEX_PCRE2_OUTPUT_DIR=${stage}`,
+ ],
+ { cwd: repositoryRoot, env: environment },
+ );
+ await run(
+ cmake,
+ ["--build", work, "--target", "regex-pcre2-pack", "--parallel", "1"],
+ {
+ cwd: repositoryRoot,
+ env: environment,
+ },
+ );
+
+ const generatedEntries = (await readdir(stage)).sort();
+ if (
+ generatedEntries.length !== 2 ||
+ generatedEntries[0] !== "pcre2.mjs" ||
+ generatedEntries[1] !== "pcre2.wasm"
+ ) {
+ throw new Error(
+ `Emscripten produced unexpected files: ${generatedEntries.join(", ") || "(none)"}.`,
+ );
+ }
+ await copyFile(
+ path.join(source, "LICENCE.md"),
+ path.join(stage, "LICENSE.txt"),
+ fsConstants.COPYFILE_EXCL,
+ );
+ const metadata = await expectedMetadata(repositoryRoot, stage);
+ await writeFile(
+ path.join(stage, "engine-metadata.json"),
+ deterministicJson(metadata),
+ { encoding: "utf8", flag: "wx" },
+ );
+ await writeFile(
+ path.join(stage, "SHA256SUMS"),
+ await expectedChecksums(stage),
+ { encoding: "utf8", flag: "wx" },
+ );
+ await verifyPcre2Pack(stage, repositoryRoot);
+
+ if (outputDetails) {
+ const backup = path.join(buildRoot, `.pcre2-replaced-${process.pid}`);
+ assertGeneratedChild(buildRoot, backup);
+ await rename(output, backup);
+ try {
+ await rename(stage, output);
+ } catch (error) {
+ await rename(backup, output);
+ throw error;
+ }
+ await rm(backup, { recursive: true });
+ } else {
+ await rename(stage, output);
+ }
+ return { output, metadata };
+ } finally {
+ await rm(work, { recursive: true, force: true });
+ await rm(stage, { recursive: true, force: true });
+ }
+}
diff --git a/scripts/pcre2-engine-pack.test.ts b/scripts/pcre2-engine-pack.test.ts
new file mode 100644
index 0000000..28294fc
--- /dev/null
+++ b/scripts/pcre2-engine-pack.test.ts
@@ -0,0 +1,89 @@
+// @vitest-environment node
+
+import { mkdtemp, rm } from "node:fs/promises";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import { afterEach, describe, expect, it } from "vitest";
+import {
+ PCRE2_LOCK,
+ parsePcre2BuildArguments,
+ verifyPcre2Pack,
+} from "./pcre2-engine-pack.mjs";
+
+const temporaryDirectories: string[] = [];
+
+afterEach(async () => {
+ await Promise.all(
+ temporaryDirectories
+ .splice(0)
+ .map((directory) => rm(directory, { recursive: true, force: true })),
+ );
+});
+
+describe("PCRE2 engine-pack gate", () => {
+ it("keeps the default build on the checked-in production-assets path", () => {
+ expect(parsePcre2BuildArguments([])).toEqual({ mode: "ecmascript" });
+ });
+
+ it("requires explicit offline source and compiler paths", () => {
+ expect(() => parsePcre2BuildArguments(["--pcre2"])).toThrow(
+ "requires both --source-dir and --emcc",
+ );
+ expect(() =>
+ parsePcre2BuildArguments([
+ "--source-dir",
+ "/source",
+ "--emcc",
+ "/compiler",
+ ]),
+ ).toThrow("explicit --pcre2 gate");
+ });
+
+ it("parses the exact staged-build contract", () => {
+ expect(
+ parsePcre2BuildArguments([
+ "--pcre2",
+ "--source-dir",
+ "/source",
+ "--emcc",
+ "/compiler",
+ "--force",
+ ]),
+ ).toEqual({
+ mode: "pcre2",
+ sourceDirectory: "/source",
+ emccFile: "/compiler",
+ force: true,
+ });
+ });
+
+ it.each([
+ ["--unknown"],
+ ["--pcre2", "--pcre2"],
+ ["--pcre2", "--source-dir"],
+ ["--pcre2", "--force", "--force"],
+ ])("rejects ambiguous arguments %j", (...arguments_) => {
+ expect(() => parsePcre2BuildArguments(arguments_)).toThrow();
+ });
+
+ it("pins source, signed-tag object and compiler identities", () => {
+ expect(PCRE2_LOCK.source.tag).toBe("pcre2-10.47");
+ expect(PCRE2_LOCK.source.tagObject).toHaveLength(40);
+ expect(PCRE2_LOCK.source.commit).toHaveLength(40);
+ expect(PCRE2_LOCK.source.tree).toHaveLength(40);
+ expect(PCRE2_LOCK.emscripten.version).toBe("6.0.4");
+ expect(PCRE2_LOCK.emscripten.emccSha256).toHaveLength(64);
+ expect(PCRE2_LOCK.bridgeAbi).toBe(3);
+ expect(PCRE2_LOCK.maximumSubjectBytes).toBe(16 * 1024 * 1024);
+ expect(PCRE2_LOCK.maximumMatches).toBe(10_000);
+ });
+
+ it("rejects an incomplete staged pack before attempting to load it", async () => {
+ const directory = await mkdtemp(path.join(tmpdir(), "regex-pcre2-test-"));
+ temporaryDirectories.push(directory);
+
+ await expect(verifyPcre2Pack(directory, process.cwd())).rejects.toThrow(
+ "must contain only",
+ );
+ });
+});
diff --git a/scripts/serve-test.mjs b/scripts/serve-test.mjs
index f888692..90db867 100644
--- a/scripts/serve-test.mjs
+++ b/scripts/serve-test.mjs
@@ -13,6 +13,7 @@ const mediaTypes = new Map([
[".css", "text/css; charset=utf-8"],
[".html", "text/html; charset=utf-8"],
[".js", "text/javascript; charset=utf-8"],
+ [".mjs", "text/javascript; charset=utf-8"],
[".json", "application/json; charset=utf-8"],
[".svg", "image/svg+xml"],
[".wasm", "application/wasm"],
diff --git a/scripts/verify-engine-assets.mjs b/scripts/verify-engine-assets.mjs
index 363ee04..bacda69 100644
--- a/scripts/verify-engine-assets.mjs
+++ b/scripts/verify-engine-assets.mjs
@@ -1,8 +1,25 @@
import { lstat, readFile, readdir } from "node:fs/promises";
import path from "node:path";
import { fileURLToPath } from "node:url";
+import { verifyPcre2Pack } from "./pcre2-engine-pack.mjs";
const root = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..");
+const arguments_ = process.argv.slice(2);
+
+if (arguments_.length > 0) {
+ if (arguments_.length !== 2 || arguments_[0] !== "--pcre2-pack") {
+ throw new Error(
+ "Usage: node scripts/verify-engine-assets.mjs [--pcre2-pack PATH]",
+ );
+ }
+ const pack = path.resolve(root, arguments_[1]);
+ const metadata = await verifyPcre2Pack(pack, root);
+ console.log(
+ `Verified staged PCRE2 ${metadata.engineVersion} pack and WebAssembly bridge smoke test.`,
+ );
+ process.exit(0);
+}
+
const engineDirectory = path.join(root, "dist", "engines");
const details = await lstat(engineDirectory);
if (details.isSymbolicLink() || !details.isDirectory()) {
@@ -14,21 +31,21 @@ const entries = (await readdir(engineDirectory, { withFileTypes: true })).sort(
left.name < right.name ? -1 : left.name > right.name ? 1 : 0,
);
if (
- entries.length !== 1 ||
+ entries.length !== 2 ||
!entries[0]?.isFile() ||
- entries[0].name !== "README.md"
+ entries[0].name !== "README.md" ||
+ !entries[1]?.isDirectory() ||
+ entries[1].name !== "pcre2"
) {
throw new Error(
- "Regex Tools 0.1.0 must not ship an undeclared external engine pack.",
+ "The production build must contain only README.md and the declared PCRE2 pack.",
);
}
const notice = await readFile(path.join(engineDirectory, "README.md"), "utf8");
-if (
- !notice.includes("native `RegExp`") ||
- !notice.includes("no external engine")
-) {
+if (!notice.includes("native `RegExp`") || !notice.includes("PCRE2 10.47")) {
throw new Error("The engine asset notice does not describe this release.");
}
+await verifyPcre2Pack(path.join(engineDirectory, "pcre2"), root);
-console.log("Verified ECMAScript-only engine assets (no WebAssembly pack).");
+console.log("Verified native ECMAScript and bundled PCRE2 engine assets.");
diff --git a/src/components/AnalysisPanel.css b/src/components/AnalysisPanel.css
new file mode 100644
index 0000000..2408fb1
--- /dev/null
+++ b/src/components/AnalysisPanel.css
@@ -0,0 +1,521 @@
+.analysis-dialog {
+ width: min(90rem, calc(100% - 2rem));
+}
+
+.analysis-panel {
+ --analysis-high: var(--toolbox-danger);
+ --analysis-warning: var(--regex-warning);
+ --analysis-notice: var(--toolbox-accent);
+ min-height: 0;
+}
+
+.analysis-view-tabs {
+ display: flex;
+ gap: 0.25rem;
+ padding: 0.65rem 0.75rem 0;
+ border-bottom: 1px solid var(--toolbox-border);
+ background: var(--toolbox-surface-soft);
+}
+
+.analysis-view-tabs button {
+ border: 1px solid transparent;
+ border-radius: calc(var(--toolbox-radius) * 0.7)
+ calc(var(--toolbox-radius) * 0.7) 0 0;
+ padding: 0.55rem 0.8rem;
+ background: transparent;
+ color: var(--toolbox-muted);
+ font-weight: 750;
+}
+
+.analysis-view-tabs button[aria-selected="true"] {
+ border-color: var(--toolbox-border);
+ border-bottom-color: var(--toolbox-surface);
+ background: var(--toolbox-surface);
+ color: var(--toolbox-accent);
+}
+
+.analysis-content {
+ display: grid;
+ gap: 0.8rem;
+ padding: 0.75rem;
+}
+
+.analysis-section,
+.analysis-result-block {
+ display: grid;
+ gap: 0.75rem;
+ border: 1px solid var(--toolbox-border);
+ border-radius: calc(var(--toolbox-radius) * 0.82);
+ background: var(--toolbox-surface);
+}
+
+.analysis-section > header,
+.analysis-result-block > header {
+ display: flex;
+ align-items: center;
+ justify-content: space-between;
+ gap: 1rem;
+ padding: 0.75rem;
+ border-bottom: 1px solid var(--toolbox-border);
+ background: var(--toolbox-surface-soft);
+}
+
+.analysis-section > header h3,
+.analysis-result-block > header h3 {
+ margin: 0.12rem 0 0;
+ font-size: 0.96rem;
+}
+
+.analysis-section > header > span,
+.analysis-result-block > header > span {
+ color: var(--toolbox-muted);
+ font-size: 0.72rem;
+ font-weight: 700;
+ text-align: right;
+}
+
+.analysis-summary,
+.analysis-advisory,
+.analysis-stop-reason,
+.analysis-error,
+.analysis-unavailable {
+ margin: 0;
+ padding: 0.7rem 0.75rem;
+ color: var(--toolbox-muted);
+ font-size: 0.79rem;
+ line-height: 1.5;
+}
+
+.analysis-advisory,
+.analysis-stop-reason {
+ border-left: 3px solid var(--toolbox-accent);
+ background: var(--toolbox-accent-soft);
+}
+
+.analysis-unavailable,
+.analysis-error {
+ margin: 0.75rem 0.75rem 0;
+ border: 1px solid
+ color-mix(in srgb, var(--toolbox-danger) 45%, var(--toolbox-border));
+ border-radius: calc(var(--toolbox-radius) * 0.72);
+ background: color-mix(
+ in srgb,
+ var(--toolbox-danger) 8%,
+ var(--toolbox-surface)
+ );
+ color: color-mix(in srgb, var(--toolbox-danger) 78%, var(--toolbox-text));
+}
+
+.analysis-status {
+ display: grid;
+ grid-template-columns: auto minmax(0, 1fr);
+ align-items: center;
+ gap: 0.6rem;
+ margin: 0.75rem 0.75rem 0;
+ padding: 0.55rem 0.7rem;
+ border: 1px solid var(--toolbox-border);
+ border-radius: calc(var(--toolbox-radius) * 0.72);
+ background: var(--toolbox-surface-soft);
+ font-size: 0.78rem;
+}
+
+.analysis-status > span {
+ border-radius: 99rem;
+ padding: 0.2rem 0.42rem;
+ background: var(--toolbox-accent-soft);
+ color: var(--toolbox-accent);
+ font-size: 0.65rem;
+ font-weight: 800;
+ text-transform: uppercase;
+}
+
+.analysis-status > strong {
+ overflow: hidden;
+ text-overflow: ellipsis;
+ white-space: nowrap;
+}
+
+.analysis-progress {
+ display: grid;
+ grid-template-columns: 1fr auto;
+ align-items: center;
+ gap: 0.6rem;
+ margin: 0.5rem 0.75rem 0;
+ color: var(--toolbox-muted);
+ font-size: 0.7rem;
+}
+
+.analysis-progress progress {
+ width: 100%;
+ accent-color: var(--toolbox-accent);
+}
+
+.analysis-findings {
+ display: grid;
+ gap: 0.55rem;
+ padding: 0 0.75rem;
+}
+
+.analysis-finding {
+ overflow: hidden;
+ border: 1px solid var(--toolbox-border);
+ border-left-width: 4px;
+ border-radius: calc(var(--toolbox-radius) * 0.72);
+ background: var(--toolbox-surface-soft);
+}
+
+.analysis-finding.is-high {
+ border-left-color: var(--analysis-high);
+}
+
+.analysis-finding.is-warning {
+ border-left-color: var(--analysis-warning);
+}
+
+.analysis-finding.is-notice {
+ border-left-color: var(--analysis-notice);
+}
+
+.analysis-finding-title {
+ display: grid;
+ width: 100%;
+ grid-template-columns: auto minmax(0, 1fr);
+ align-items: start;
+ gap: 0.55rem;
+ border: 0;
+ padding: 0.65rem;
+ background: transparent;
+ color: var(--toolbox-text);
+ text-align: left;
+}
+
+button.analysis-finding-title:hover {
+ background: var(--toolbox-accent-soft);
+}
+
+.analysis-finding-title > span:last-child {
+ display: grid;
+ gap: 0.12rem;
+}
+
+.analysis-finding-title small {
+ color: var(--toolbox-muted);
+ font-size: 0.67rem;
+ font-weight: 500;
+}
+
+.analysis-risk-level,
+.analysis-run-badge {
+ border: 1px solid currentColor;
+ border-radius: 99rem;
+ padding: 0.2rem 0.42rem;
+ font-size: 0.62rem;
+ font-weight: 850;
+ letter-spacing: 0.04em;
+ text-transform: uppercase;
+}
+
+.analysis-risk-level.is-high {
+ color: var(--analysis-high);
+}
+
+.analysis-risk-level.is-warning {
+ color: color-mix(in srgb, var(--analysis-warning) 78%, var(--toolbox-text));
+}
+
+.analysis-risk-level.is-notice {
+ color: var(--analysis-notice);
+}
+
+.analysis-run-badge.is-complete {
+ color: color-mix(in srgb, var(--regex-mint) 72%, var(--toolbox-text));
+}
+
+.analysis-run-badge.is-timeout,
+.analysis-run-badge.is-error {
+ color: var(--toolbox-danger);
+}
+
+.analysis-run-badge.is-cancelled,
+.analysis-run-badge.is-wall-time-limit {
+ color: var(--analysis-warning);
+}
+
+.analysis-finding > p {
+ padding: 0 0.65rem 0.65rem;
+ color: var(--toolbox-text);
+ font-size: 0.78rem;
+ line-height: 1.45;
+}
+
+.analysis-finding dl {
+ display: grid;
+ grid-template-columns: repeat(auto-fit, minmax(13rem, 1fr));
+ margin: 0;
+ border-top: 1px solid var(--toolbox-border);
+}
+
+.analysis-finding dl > div {
+ padding: 0.6rem;
+}
+
+.analysis-finding dt,
+.analysis-result-meta dt,
+.analysis-input-summary dt {
+ color: var(--toolbox-muted);
+ font-size: 0.64rem;
+ font-weight: 800;
+ letter-spacing: 0.04em;
+ text-transform: uppercase;
+}
+
+.analysis-finding dd,
+.analysis-result-meta dd,
+.analysis-input-summary dd {
+ margin: 0.2rem 0 0;
+ overflow-wrap: anywhere;
+ font-size: 0.74rem;
+ line-height: 1.4;
+}
+
+.analysis-details {
+ margin: 0 0.75rem 0.75rem;
+ color: var(--toolbox-muted);
+ font-size: 0.75rem;
+}
+
+.analysis-details summary {
+ cursor: pointer;
+ font-weight: 750;
+}
+
+.analysis-details li,
+.analysis-limitations li {
+ margin-block: 0.35rem;
+ line-height: 1.4;
+}
+
+.analysis-growth-text-controls,
+.analysis-number-controls {
+ display: grid;
+ grid-template-columns: repeat(auto-fit, minmax(10rem, 1fr));
+ gap: 0.55rem;
+ padding: 0 0.75rem;
+}
+
+.analysis-growth-text-controls > label,
+.analysis-number-controls > label {
+ display: grid;
+ align-content: start;
+ gap: 0.25rem;
+}
+
+.analysis-growth-text-controls label > span,
+.analysis-number-controls label > span {
+ color: var(--toolbox-muted);
+ font-size: 0.68rem;
+ font-weight: 750;
+}
+
+.analysis-run-actions {
+ display: flex;
+ flex-wrap: wrap;
+ gap: 0.5rem;
+ padding: 0 0.75rem 0.75rem;
+}
+
+.analysis-result-block {
+ margin: 0 0.75rem 0.75rem;
+}
+
+.analysis-result-meta,
+.analysis-input-summary {
+ display: grid;
+ grid-template-columns: repeat(auto-fit, minmax(12rem, 1fr));
+ margin: 0;
+ padding: 0 0.75rem;
+}
+
+.analysis-result-meta > div,
+.analysis-input-summary > div {
+ padding: 0.55rem;
+ border-bottom: 1px solid var(--toolbox-border);
+}
+
+.analysis-table-scroll {
+ overflow: auto;
+ border-block: 1px solid var(--toolbox-border);
+}
+
+.analysis-table-scroll table {
+ min-width: 48rem;
+}
+
+.analysis-table-scroll th,
+.analysis-table-scroll td {
+ white-space: nowrap;
+}
+
+.analysis-chart-grid {
+ display: grid;
+ grid-template-columns: repeat(auto-fit, minmax(16rem, 1fr));
+ gap: 0.65rem;
+ padding: 0 0.75rem;
+}
+
+.analysis-chart {
+ display: grid;
+ min-width: 0;
+ gap: 0.5rem;
+ margin: 0;
+ padding: 0.65rem;
+ border: 1px solid var(--toolbox-border);
+ border-radius: calc(var(--toolbox-radius) * 0.72);
+ background: var(--toolbox-surface-soft);
+}
+
+.analysis-chart figcaption {
+ display: grid;
+ gap: 0.12rem;
+}
+
+.analysis-chart figcaption strong {
+ font-size: 0.78rem;
+}
+
+.analysis-chart figcaption span {
+ color: var(--toolbox-muted);
+ font-size: 0.65rem;
+}
+
+.analysis-chart svg {
+ width: 100%;
+ height: 10rem;
+ overflow: visible;
+}
+
+.analysis-chart-axis {
+ fill: none;
+ stroke: var(--toolbox-border);
+ stroke-width: 1;
+ vector-effect: non-scaling-stroke;
+}
+
+.analysis-chart-line {
+ fill: none;
+ stroke: var(--toolbox-accent);
+ stroke-width: 2;
+ vector-effect: non-scaling-stroke;
+}
+
+.analysis-chart circle {
+ fill: var(--toolbox-surface);
+ stroke: var(--toolbox-accent);
+ stroke-width: 2;
+}
+
+.analysis-empty-chart {
+ display: grid;
+ min-height: 10rem;
+ place-items: center;
+ color: var(--toolbox-muted);
+ font-size: 0.75rem;
+}
+
+.analysis-outcome-plot {
+ display: flex;
+ min-height: 10rem;
+ align-items: end;
+ gap: 0.22rem;
+}
+
+.analysis-outcome-plot > span {
+ min-width: 0.35rem;
+ max-width: 2rem;
+ height: 45%;
+ flex: 1;
+ border-radius: 0.2rem 0.2rem 0 0;
+ background: var(--toolbox-muted);
+}
+
+.analysis-outcome-plot > span.is-match {
+ height: 88%;
+ background: var(--regex-mint);
+}
+
+.analysis-outcome-plot > span.is-no-match {
+ background: var(--toolbox-accent);
+}
+
+.analysis-outcome-plot > span.is-timeout {
+ height: 100%;
+ background: var(--toolbox-danger);
+}
+
+.analysis-outcome-plot > span.is-crash,
+.analysis-outcome-plot > span.is-compile-error {
+ height: 100%;
+ background: var(--analysis-warning);
+}
+
+.analysis-outcome-key {
+ display: flex;
+ flex-wrap: wrap;
+ gap: 0.4rem 0.7rem;
+ color: var(--toolbox-muted);
+ font-size: 0.64rem;
+}
+
+.analysis-outcome-key span::before {
+ display: inline-block;
+ width: 0.55rem;
+ height: 0.55rem;
+ margin-right: 0.25rem;
+ border-radius: 0.15rem;
+ background: var(--toolbox-muted);
+ content: "";
+}
+
+.analysis-outcome-key .is-match::before {
+ background: var(--regex-mint);
+}
+
+.analysis-outcome-key .is-no-match::before {
+ background: var(--toolbox-accent);
+}
+
+.analysis-outcome-key .is-timeout::before {
+ background: var(--toolbox-danger);
+}
+
+.analysis-outcome-key .is-crash::before {
+ background: var(--analysis-warning);
+}
+
+.analysis-limitations {
+ margin: 0;
+ padding: 0 1.6rem 0.75rem 2rem;
+ color: var(--toolbox-muted);
+ font-size: 0.7rem;
+}
+
+@media (max-width: 40rem) {
+ .analysis-section > header,
+ .analysis-result-block > header {
+ align-items: start;
+ flex-direction: column;
+ }
+
+ .analysis-section > header > span,
+ .analysis-result-block > header > span {
+ text-align: left;
+ }
+
+ .analysis-status {
+ grid-template-columns: 1fr;
+ }
+
+ .analysis-status > strong {
+ white-space: normal;
+ }
+}
diff --git a/src/components/AnalysisPanel.test.tsx b/src/components/AnalysisPanel.test.tsx
new file mode 100644
index 0000000..d89a369
--- /dev/null
+++ b/src/components/AnalysisPanel.test.tsx
@@ -0,0 +1,248 @@
+import { cleanup, render, screen, waitFor } from "@testing-library/react";
+import userEvent from "@testing-library/user-event";
+import { afterEach, describe, expect, it, vi } from "vitest";
+import type { AnalysisWorkerClient } from "../regex/analysis/AnalysisSupervisor";
+import type {
+ BenchmarkWorkerSample,
+ GrowthWorkerSample,
+} from "../regex/analysis/analysis.types";
+import { WorkerRequestError } from "../regex/execution/WorkerSupervisor";
+import { EcmaScriptSyntaxProvider } from "../regex/syntax/providers/ecmascript/EcmaScriptSyntaxProvider";
+import { AnalysisPanel } from "./AnalysisPanel";
+
+const provider = new EcmaScriptSyntaxProvider();
+
+function benchmarkSample(milliseconds = 1): BenchmarkWorkerSample {
+ return {
+ accepted: true,
+ effectiveFlags: "dg",
+ subjectBytes: 5,
+ subjectUtf16: 5,
+ compileMs: milliseconds,
+ firstMatchMs: milliseconds + 1,
+ allMatchesMs: milliseconds + 2,
+ replacementMs: milliseconds + 3,
+ throughputBytesPerSecond: 1024 * 1024,
+ matchCount: 1,
+ matched: true,
+ matchCollectionTruncated: false,
+ replacementOutputUtf16: 1,
+ };
+}
+
+function growthSample(executionMs: number): GrowthWorkerSample {
+ return {
+ accepted: true,
+ effectiveFlags: "d",
+ subjectBytes: 5,
+ subjectUtf16: 5,
+ executionMs,
+ matchCount: 0,
+ matched: false,
+ matchCollectionTruncated: false,
+ };
+}
+
+function worker(
+ overrides: Partial = {},
+): AnalysisWorkerClient {
+ return {
+ identity: vi.fn().mockResolvedValue({
+ flavour: "ecmascript",
+ engineName: "Native ECMAScript RegExp",
+ engineVersion: "Fixture Browser 1",
+ runtimeVersion: "Fixture Browser 1",
+ nativeOffsetUnit: "utf16",
+ }),
+ benchmarkSample: vi.fn().mockResolvedValue(benchmarkSample()),
+ growthProbe: vi.fn().mockResolvedValue(growthSample(1)),
+ cancel: vi.fn(),
+ dispose: vi.fn(),
+ ...overrides,
+ };
+}
+
+async function renderPanel(
+ overrides: Partial[0]> = {},
+) {
+ const pattern = overrides.pattern ?? "(a+)+$";
+ const flavour = overrides.flavour ?? "ecmascript";
+ const syntax =
+ overrides.syntax ??
+ (await provider.parsePattern({
+ flavour: "ecmascript",
+ flavourVersion: "2025",
+ pattern,
+ flags: [],
+ options: {},
+ }));
+ const currentWorker = worker();
+ const rendered = render(
+ currentWorker}
+ {...overrides}
+ />,
+ );
+ return { ...rendered, currentWorker, syntax };
+}
+
+afterEach(() => {
+ cleanup();
+});
+
+describe("AnalysisPanel", () => {
+ it("shows flavour-scoped advisory static findings and selects their source range", async () => {
+ const onSelectPatternRange = vi.fn();
+ await renderPanel({ onSelectPatternRange });
+
+ expect(
+ screen.getByRole("heading", { name: "Potential risk findings" }),
+ ).toBeInTheDocument();
+ expect(screen.getByText("Potential nested-quantifier risk")).toBeVisible();
+ expect(screen.getByText(/potential risks found/u)).toBeVisible();
+ await userEvent.click(
+ screen.getByRole("button", {
+ name: /Potential nested-quantifier risk/u,
+ }),
+ );
+ expect(onSelectPatternRange).toHaveBeenCalledWith({
+ startUtf16: 0,
+ endUtf16: 5,
+ });
+ });
+
+ it("runs bounded cold and warm benchmark samples and renders p95", async () => {
+ const { currentWorker } = await renderPanel();
+ await userEvent.click(screen.getByRole("tab", { name: "Benchmark" }));
+ await userEvent.clear(
+ screen.getByRole("spinbutton", { name: "Measured samples" }),
+ );
+ await userEvent.type(
+ screen.getByRole("spinbutton", { name: "Measured samples" }),
+ "3",
+ );
+ await userEvent.click(
+ screen.getByRole("button", { name: "Run bounded benchmark" }),
+ );
+
+ const table = await screen.findByRole("table", {
+ name: "Cold and warm benchmark metrics",
+ });
+ expect(table).toHaveTextContent("Warm p95");
+ expect(table).toHaveTextContent("Compile");
+ expect(screen.getByText(/Fixture Browser 1/u)).toBeVisible();
+ expect(screen.getAllByText(/3 measured/u)).toHaveLength(2);
+ expect(currentWorker.identity).toHaveBeenCalled();
+ expect(currentWorker.benchmarkSample).toHaveBeenCalledTimes(7);
+ });
+
+ it("renders separate timing and outcome plots for bounded growth", async () => {
+ const samples = [growthSample(0.5), growthSample(10)];
+ const currentWorker = worker({
+ growthProbe: vi
+ .fn()
+ .mockImplementation(() => Promise.resolve(samples.shift()!)),
+ });
+ await renderPanel({ createSupervisor: () => currentWorker });
+ await userEvent.click(
+ screen.getByRole("button", { name: "Run bounded growth probe" }),
+ );
+
+ expect(
+ await screen.findByRole("img", {
+ name: /Execution-time growth plot/u,
+ }),
+ ).toBeVisible();
+ expect(
+ screen.getByRole("img", { name: /Outcome plot with 2 samples/u }),
+ ).toBeVisible();
+ expect(screen.getByText("Observed disproportionate growth")).toBeVisible();
+ expect(
+ screen.getByRole("table", { name: "Dynamic growth samples" }),
+ ).toHaveTextContent("Normalized growth");
+ });
+
+ it("does not apply ECMAScript analysis to PCRE2", async () => {
+ const currentWorker = worker();
+ await renderPanel({
+ flavour: "pcre2",
+ createSupervisor: () => currentWorker,
+ });
+
+ expect(screen.getByRole("alert")).toHaveTextContent(
+ "supports ECMAScript 2025 only",
+ );
+ expect(
+ screen.getByRole("button", { name: "Run bounded growth probe" }),
+ ).toBeDisabled();
+ expect(currentWorker.identity).not.toHaveBeenCalled();
+ });
+
+ it("never renders benchmark results against a changed subject", async () => {
+ const rendered = await renderPanel();
+ await userEvent.click(screen.getByRole("tab", { name: "Benchmark" }));
+ await userEvent.click(
+ screen.getByRole("button", { name: "Run bounded benchmark" }),
+ );
+ expect(
+ await screen.findByRole("table", {
+ name: "Cold and warm benchmark metrics",
+ }),
+ ).toBeVisible();
+
+ rendered.rerender(
+ rendered.currentWorker}
+ />,
+ );
+
+ expect(
+ screen.queryByRole("table", {
+ name: "Cold and warm benchmark metrics",
+ }),
+ ).not.toBeInTheDocument();
+ });
+
+ it("terminates the active worker when the user cancels", async () => {
+ let rejectSample: ((reason: Error) => void) | undefined;
+ const pending = new Promise((_resolve, reject) => {
+ rejectSample = reject;
+ });
+ const currentWorker = worker({
+ benchmarkSample: vi.fn().mockReturnValue(pending),
+ cancel: vi.fn().mockImplementation(() => {
+ rejectSample?.(
+ new WorkerRequestError("cancelled", "fixture cancellation"),
+ );
+ }),
+ });
+ await renderPanel({ createSupervisor: () => currentWorker });
+ await userEvent.click(screen.getByRole("tab", { name: "Benchmark" }));
+ await userEvent.click(
+ screen.getByRole("button", { name: "Run bounded benchmark" }),
+ );
+ await userEvent.click(
+ await screen.findByRole("button", { name: "Cancel benchmark" }),
+ );
+
+ await waitFor(() => expect(currentWorker.cancel).toHaveBeenCalled());
+ expect(await screen.findAllByText(/Benchmark cancelled/u)).toHaveLength(2);
+ expect(currentWorker.dispose).toHaveBeenCalled();
+ });
+});
diff --git a/src/components/AnalysisPanel.tsx b/src/components/AnalysisPanel.tsx
new file mode 100644
index 0000000..6230986
--- /dev/null
+++ b/src/components/AnalysisPanel.tsx
@@ -0,0 +1,1245 @@
+import { useCallback, useEffect, useMemo, useRef, useState } from "react";
+import {
+ DEFAULT_BENCHMARK_SETTINGS,
+ DEFAULT_GROWTH_SETTINGS,
+} from "../regex/analysis/analysis-limits";
+import {
+ runGrowthAnalysis,
+ runRegexBenchmark,
+} from "../regex/analysis/analysis-runner";
+import {
+ AnalysisSupervisor,
+ type AnalysisWorkerClient,
+} from "../regex/analysis/AnalysisSupervisor";
+import type {
+ AnalysisProgress,
+ GrowthAnalysisResult,
+ GrowthSettings,
+ MetricStatistics,
+ RegexBenchmarkResult,
+ RegexRiskFinding,
+} from "../regex/analysis/analysis.types";
+import { analyseStaticRisk } from "../regex/analysis/static-risk";
+import {
+ DEFAULT_REGEX_LIMITS,
+ utf8ByteLength,
+} from "../regex/execution/request-limits";
+import type { RegexFlavourId } from "../regex/model/flavour";
+import type { RegexSyntaxResult, SourceRange } from "../regex/model/syntax";
+import "./AnalysisPanel.css";
+
+type AnalysisView = "risk" | "benchmark";
+type ActiveRun = "benchmark" | "growth";
+
+export interface AnalysisPanelProps {
+ readonly active: boolean;
+ readonly flavour: RegexFlavourId;
+ readonly pattern: string;
+ readonly flags: readonly string[];
+ readonly subject: string;
+ readonly replacement: string;
+ readonly scanAll: boolean;
+ readonly syntax?: RegexSyntaxResult;
+ readonly onClose?: () => void;
+ readonly onSelectPatternRange?: (range: SourceRange) => void;
+ readonly createSupervisor?: () => AnalysisWorkerClient;
+}
+
+function formatMilliseconds(value: number | undefined): string {
+ if (value === undefined) return "—";
+ if (value > 0 && value < 0.01) return "<0.01 ms";
+ return `${value.toFixed(2)} ms`;
+}
+
+function formatThroughput(value: number | undefined): string {
+ if (value === undefined) return "—";
+ return `${(value / (1024 * 1024)).toFixed(2)} MiB/s`;
+}
+
+function metricValue(
+ statistic: MetricStatistics | undefined,
+ key: "minimum" | "median" | "p95" | "maximum",
+ throughput: boolean,
+): string {
+ const value = statistic?.[key];
+ return throughput ? formatThroughput(value) : formatMilliseconds(value);
+}
+
+interface MetricRow {
+ readonly label: string;
+ readonly cold?: number;
+ readonly warm?: MetricStatistics;
+ readonly throughput?: boolean;
+ readonly unavailableReason?: string;
+}
+
+function BenchmarkTable({ result }: { readonly result: RegexBenchmarkResult }) {
+ const cold = result.coldSample;
+ const rows: readonly MetricRow[] = [
+ {
+ label: "Worker cold start",
+ cold: result.coldStartMs,
+ },
+ {
+ label: "Compile",
+ cold: cold?.compileMs,
+ warm: result.warmStatistics.compileMs,
+ },
+ {
+ label: "First match",
+ cold: cold?.firstMatchMs,
+ warm: result.warmStatistics.firstMatchMs,
+ },
+ {
+ label: "All matches",
+ cold: cold?.allMatchesMs,
+ warm: result.warmStatistics.allMatchesMs,
+ },
+ {
+ label: "Replacement",
+ cold: cold?.replacementMs,
+ warm: result.warmStatistics.replacementMs,
+ unavailableReason: cold?.replacementSkippedReason,
+ },
+ {
+ label: "All-match throughput",
+ cold: cold?.throughputBytesPerSecond,
+ warm: result.warmStatistics.throughputBytesPerSecond,
+ throughput: true,
+ },
+ ];
+
+ return (
+
+
+
+
+ Metric
+ Cold / first use
+ Warm minimum
+ Warm median
+ Warm p95
+ Warm maximum
+ Samples
+
+
+
+ {rows.map((row) => (
+
+ {row.label}
+
+ {row.throughput
+ ? formatThroughput(row.cold)
+ : formatMilliseconds(row.cold)}
+
+
+ {metricValue(row.warm, "minimum", row.throughput === true)}
+
+
+ {metricValue(row.warm, "median", row.throughput === true)}
+
+ {metricValue(row.warm, "p95", row.throughput === true)}
+
+ {metricValue(row.warm, "maximum", row.throughput === true)}
+
+ {row.warm?.count.toLocaleString() ?? "—"}
+
+ ))}
+
+
+
+ );
+}
+
+function BenchmarkSummary({
+ result,
+}: {
+ readonly result: RegexBenchmarkResult;
+}) {
+ return (
+
+
+
+
+
Engine
+
+ {result.identity
+ ? `${result.identity.engineName} · ${result.identity.engineVersion}`
+ : "Worker identity unavailable"}
+
+
+
+
Subject
+
+ {result.coldSample
+ ? `${result.coldSample.subjectBytes.toLocaleString()} bytes · ${result.coldSample.subjectUtf16.toLocaleString()} UTF-16 units`
+ : "No completed cold sample"}
+
+
+
+
Output count
+
+ {result.coldSample
+ ? `${result.coldSample.matchCount.toLocaleString()} matches${
+ result.coldSample.matchCollectionTruncated
+ ? " · bounded prefix"
+ : ""
+ }`
+ : "Unavailable"}
+
+
+
+
Sample count
+
+ {result.completedMeasuredIterations.toLocaleString()} measured ·{" "}
+ {result.completedWarmups.toLocaleString()} warm-up
+
+
+
+
Wall time
+ {formatMilliseconds(result.wallTimeMs)}
+
+
+
Effective flags
+
+ {result.coldSample?.effectiveFlags || "(none)"}
+
+
+
+ {result.stoppedReason ? (
+ {result.stoppedReason}
+ ) : null}
+
+
+ {result.warnings.map((warning) => (
+ {warning}
+ ))}
+
+
+ );
+}
+
+function RiskFindingCard({
+ finding,
+ onSelectPatternRange,
+}: {
+ readonly finding: RegexRiskFinding;
+ readonly onSelectPatternRange?: (range: SourceRange) => void;
+}) {
+ const content = (
+ <>
+
+ {finding.severity}
+
+
+ {finding.title}
+
+ {finding.evidence.replaceAll("-", " ")} · {finding.confidence}{" "}
+ confidence · UTF-16 {finding.range.startUtf16}…
+ {finding.range.endUtf16}
+
+
+ >
+ );
+ return (
+
+ {onSelectPatternRange ? (
+ onSelectPatternRange(finding.range)}
+ >
+ {content}
+
+ ) : (
+ {content}
+ )}
+ {finding.explanation}
+
+
+
Example risk
+ {finding.exampleRisk}
+
+
+
Suggested investigation
+ {finding.suggestedInvestigation}
+
+
+
Limitations
+ {finding.limitations}
+
+
+
+ );
+}
+
+function TimeGrowthPlot({ result }: { readonly result: GrowthAnalysisResult }) {
+ const measured = result.samples.filter(
+ (
+ sample,
+ ): sample is (typeof result.samples)[number] & {
+ readonly executionMs: number;
+ } => sample.executionMs !== undefined,
+ );
+ const maximumTime = Math.max(
+ 0.01,
+ ...measured.map((sample) => sample.executionMs),
+ );
+ const points = measured
+ .map((sample, index) => {
+ const x =
+ measured.length === 1 ? 50 : (index / (measured.length - 1)) * 100;
+ const y = 100 - (sample.executionMs / maximumTime) * 92;
+ return `${x.toFixed(2)},${y.toFixed(2)}`;
+ })
+ .join(" ");
+
+ return (
+
+
+ Execution time by generated input
+
+ Maximum observed {formatMilliseconds(maximumTime)} · separate from
+ outcome status
+
+
+ {measured.length > 0 ? (
+
+
+
+ {points.split(" ").map((point) => {
+ const [x, y] = point.split(",");
+ return (
+
+ );
+ })}
+
+ ) : (
+ No completed timing sample.
+ )}
+
+ );
+}
+
+function OutcomeGrowthPlot({
+ result,
+}: {
+ readonly result: GrowthAnalysisResult;
+}) {
+ return (
+
+
+ Match and worker outcome by generated input
+ Timeout is distinct from non-match and crash
+
+
+ {result.samples.map((sample) => (
+
+ ))}
+
+
+ match
+ no match
+ timeout
+ crash
+
+
+ );
+}
+
+function GrowthSummary({
+ result,
+ onSelectPatternRange,
+}: {
+ readonly result: GrowthAnalysisResult;
+ readonly onSelectPatternRange?: (range: SourceRange) => void;
+}) {
+ return (
+
+
+ {result.stoppedReason}
+
+
+
+
+
+
+
+
+ Repetitions
+ Input bytes
+ Time
+ Match result
+ Status
+ Normalized growth
+
+
+
+ {result.samples.map((sample) => (
+
+ {sample.repetitions.toLocaleString()}
+ {sample.inputBytes.toLocaleString()}
+ {formatMilliseconds(sample.executionMs)}
+
+ {sample.matched === undefined
+ ? "—"
+ : sample.matched
+ ? `${sample.matchCount?.toLocaleString() ?? "≥1"} match(es)`
+ : "no match"}
+
+ {sample.status.replaceAll("-", " ")}
+
+ {sample.normalizedGrowth === undefined
+ ? "—"
+ : `${sample.normalizedGrowth.toFixed(2)}×`}
+
+
+ ))}
+
+
+
+ {result.dynamicFindings.map((finding) => (
+
+ ))}
+
+ These observations apply only to this generated input family, browser
+ runtime and selected limits. They are not proof of general complexity or
+ safety.
+
+
+ );
+}
+
+function numericValue(
+ value: string,
+ label: string,
+ minimum: number,
+ maximum: number,
+): number {
+ const parsed = Number(value);
+ if (
+ value.trim() === "" ||
+ !Number.isFinite(parsed) ||
+ parsed < minimum ||
+ parsed > maximum
+ ) {
+ throw new RangeError(`${label} must be from ${minimum} to ${maximum}.`);
+ }
+ return parsed;
+}
+
+export function AnalysisPanel({
+ active,
+ flavour,
+ pattern,
+ flags,
+ subject,
+ replacement,
+ scanAll,
+ syntax,
+ onClose,
+ onSelectPatternRange,
+ createSupervisor = () => new AnalysisSupervisor(),
+}: AnalysisPanelProps) {
+ const [view, setView] = useState("risk");
+ const [activeRun, setActiveRun] = useState();
+ const [progressState, setProgressState] = useState();
+ const [message, setMessage] = useState(
+ "Analysis has not run for this configuration.",
+ );
+ const [error, setError] = useState();
+ const [benchmarkRecord, setBenchmarkRecord] = useState<{
+ readonly configurationKey: symbol;
+ readonly result: RegexBenchmarkResult;
+ }>();
+ const [growthRecord, setGrowthRecord] = useState<{
+ readonly configurationKey: symbol;
+ readonly result: GrowthAnalysisResult;
+ }>();
+ const [warmupIterations, setWarmupIterations] = useState(
+ String(DEFAULT_BENCHMARK_SETTINGS.warmupIterations),
+ );
+ const [measuredIterations, setMeasuredIterations] = useState(
+ String(DEFAULT_BENCHMARK_SETTINGS.measuredIterations),
+ );
+ const [benchmarkTimeoutMs, setBenchmarkTimeoutMs] = useState(
+ String(DEFAULT_BENCHMARK_SETTINGS.sampleTimeoutMs),
+ );
+ const [growthPrefix, setGrowthPrefix] = useState(
+ DEFAULT_GROWTH_SETTINGS.prefix,
+ );
+ const [growthFragment, setGrowthFragment] = useState(
+ DEFAULT_GROWTH_SETTINGS.repeatedFragment,
+ );
+ const [growthSuffix, setGrowthSuffix] = useState(
+ DEFAULT_GROWTH_SETTINGS.suffix,
+ );
+ const [growthStart, setGrowthStart] = useState(
+ String(DEFAULT_GROWTH_SETTINGS.startingRepetitions),
+ );
+ const [growthMaximum, setGrowthMaximum] = useState(
+ String(DEFAULT_GROWTH_SETTINGS.maximumRepetitions),
+ );
+ const [growthMultiplier, setGrowthMultiplier] = useState(
+ String(DEFAULT_GROWTH_SETTINGS.multiplier),
+ );
+ const [growthSteps, setGrowthSteps] = useState(
+ String(DEFAULT_GROWTH_SETTINGS.maximumSteps),
+ );
+ const [growthTimeoutMs, setGrowthTimeoutMs] = useState(
+ String(DEFAULT_GROWTH_SETTINGS.sampleTimeoutMs),
+ );
+ const [growthMaximumBytes, setGrowthMaximumBytes] = useState(
+ String(DEFAULT_GROWTH_SETTINGS.maximumSubjectBytes),
+ );
+ const [growthThreshold, setGrowthThreshold] = useState(
+ String(DEFAULT_GROWTH_SETTINGS.normalizedGrowthThreshold),
+ );
+ const supervisor = useRef(undefined);
+ const abortController = useRef(undefined);
+ const generation = useRef(0);
+
+ const syntaxCurrent =
+ syntax?.accepted === true &&
+ syntax.root.raw === pattern &&
+ syntax.root.support.flavour === flavour;
+ const analysisAvailable = flavour === "ecmascript" && syntaxCurrent;
+ const staticReport = useMemo(
+ () =>
+ syntaxCurrent && syntax
+ ? analyseStaticRisk({
+ flavour,
+ root: syntax.root,
+ flags,
+ scanAll,
+ replacement,
+ })
+ : undefined,
+ [flavour, flags, replacement, scanAll, syntax, syntaxCurrent],
+ );
+ const configurationKey = useMemo(
+ () =>
+ Symbol(
+ `analysis-${flavour}-${flags.join("")}-${pattern.length}-${subject.length}-${replacement.length}-${String(scanAll)}`,
+ ),
+ [flavour, flags, pattern, replacement, scanAll, subject],
+ );
+ const previousConfiguration = useRef(configurationKey);
+ const benchmarkResult =
+ benchmarkRecord?.configurationKey === configurationKey
+ ? benchmarkRecord.result
+ : undefined;
+ const growthResult =
+ growthRecord?.configurationKey === configurationKey
+ ? growthRecord.result
+ : undefined;
+
+ const disposeActive = useCallback((abort = true) => {
+ if (abort) abortController.current?.abort();
+ supervisor.current?.cancel();
+ supervisor.current?.dispose();
+ supervisor.current = undefined;
+ abortController.current = undefined;
+ }, []);
+
+ useEffect(() => {
+ if (active) return;
+ const stoppedGeneration = generation.current + 1;
+ generation.current = stoppedGeneration;
+ disposeActive();
+ globalThis.queueMicrotask(() => {
+ if (generation.current === stoppedGeneration) setActiveRun(undefined);
+ });
+ }, [active, disposeActive]);
+
+ useEffect(() => {
+ if (previousConfiguration.current === configurationKey) return;
+ previousConfiguration.current = configurationKey;
+ const stoppedGeneration = generation.current + 1;
+ generation.current = stoppedGeneration;
+ disposeActive();
+ globalThis.queueMicrotask(() => {
+ if (generation.current !== stoppedGeneration) return;
+ setActiveRun(undefined);
+ setBenchmarkRecord(undefined);
+ setGrowthRecord(undefined);
+ setError(undefined);
+ setMessage(
+ "Configuration changed; previous timing results were cleared.",
+ );
+ });
+ }, [configurationKey, disposeActive]);
+
+ useEffect(
+ () => () => {
+ generation.current += 1;
+ disposeActive();
+ },
+ [disposeActive],
+ );
+
+ const beginRun = useCallback(
+ (kind: ActiveRun) => {
+ disposeActive();
+ const currentSupervisor = createSupervisor();
+ const controller = new AbortController();
+ supervisor.current = currentSupervisor;
+ abortController.current = controller;
+ generation.current += 1;
+ setActiveRun(kind);
+ setError(undefined);
+ setProgressState(undefined);
+ setMessage(
+ kind === "benchmark"
+ ? "Starting bounded benchmark…"
+ : "Starting bounded dynamic growth analysis…",
+ );
+ return {
+ currentSupervisor,
+ controller,
+ runGeneration: generation.current,
+ };
+ },
+ [createSupervisor, disposeActive],
+ );
+
+ const finishRun = useCallback(
+ (runGeneration: number, currentSupervisor: AnalysisWorkerClient) => {
+ currentSupervisor.dispose();
+ if (supervisor.current === currentSupervisor) {
+ supervisor.current = undefined;
+ abortController.current = undefined;
+ }
+ if (generation.current === runGeneration) {
+ setActiveRun(undefined);
+ setProgressState(undefined);
+ }
+ },
+ [],
+ );
+
+ const runBenchmark = async () => {
+ if (!analysisAvailable) return;
+ const run = beginRun("benchmark");
+ try {
+ const result = await runRegexBenchmark(
+ {
+ flavour: "ecmascript",
+ pattern,
+ flags,
+ subject,
+ replacement,
+ scanAll,
+ maximumMatches: DEFAULT_REGEX_LIMITS.maximumMatches,
+ settings: {
+ warmupIterations: numericValue(
+ warmupIterations,
+ "Warm-up iterations",
+ 0,
+ 100,
+ ),
+ measuredIterations: numericValue(
+ measuredIterations,
+ "Measured iterations",
+ 1,
+ 1_000,
+ ),
+ sampleTimeoutMs: numericValue(
+ benchmarkTimeoutMs,
+ "Sample timeout",
+ 25,
+ DEFAULT_REGEX_LIMITS.advancedMaximumTimeoutMs,
+ ),
+ maximumWallTimeMs: DEFAULT_REGEX_LIMITS.maximumBenchmarkWallTimeMs,
+ },
+ },
+ run.currentSupervisor,
+ {
+ signal: run.controller.signal,
+ onProgress: (next) => {
+ if (generation.current !== run.runGeneration) return;
+ setProgressState(next);
+ setMessage(next.message);
+ },
+ },
+ );
+ if (generation.current !== run.runGeneration) return;
+ setBenchmarkRecord({ configurationKey, result });
+ setMessage(
+ result.status === "complete"
+ ? `Benchmark complete with ${result.completedMeasuredIterations.toLocaleString()} measured warm samples.`
+ : (result.stoppedReason ??
+ `Benchmark stopped with status ${result.status}.`),
+ );
+ } catch (cause) {
+ if (generation.current !== run.runGeneration) return;
+ const next =
+ cause instanceof Error ? cause.message : "Benchmark could not start.";
+ setError(next);
+ setMessage("Benchmark settings require review.");
+ } finally {
+ finishRun(run.runGeneration, run.currentSupervisor);
+ }
+ };
+
+ const growthSettings = (): GrowthSettings => ({
+ prefix: growthPrefix,
+ repeatedFragment: growthFragment,
+ suffix: growthSuffix,
+ startingRepetitions: numericValue(
+ growthStart,
+ "Starting repetitions",
+ 1,
+ 10_000_000,
+ ),
+ maximumRepetitions: numericValue(
+ growthMaximum,
+ "Maximum repetitions",
+ 1,
+ 10_000_000,
+ ),
+ multiplier: numericValue(growthMultiplier, "Growth multiplier", 1.1, 10),
+ maximumSteps: numericValue(growthSteps, "Maximum steps", 1, 24),
+ sampleTimeoutMs: numericValue(
+ growthTimeoutMs,
+ "Sample timeout",
+ 25,
+ DEFAULT_REGEX_LIMITS.advancedMaximumTimeoutMs,
+ ),
+ maximumWallTimeMs: DEFAULT_REGEX_LIMITS.maximumBenchmarkWallTimeMs,
+ maximumSubjectBytes: numericValue(
+ growthMaximumBytes,
+ "Maximum generated bytes",
+ 1,
+ DEFAULT_REGEX_LIMITS.interactiveSubjectHardBytes,
+ ),
+ normalizedGrowthThreshold: numericValue(
+ growthThreshold,
+ "Growth threshold",
+ 1.1,
+ 100,
+ ),
+ });
+
+ const runGrowth = async () => {
+ if (!analysisAvailable) return;
+ const run = beginRun("growth");
+ try {
+ const result = await runGrowthAnalysis(
+ {
+ flavour: "ecmascript",
+ pattern,
+ flags,
+ scanAll,
+ maximumMatches: DEFAULT_REGEX_LIMITS.maximumMatches,
+ settings: growthSettings(),
+ },
+ run.currentSupervisor,
+ {
+ signal: run.controller.signal,
+ onProgress: (next) => {
+ if (generation.current !== run.runGeneration) return;
+ setProgressState(next);
+ setMessage(next.message);
+ },
+ },
+ );
+ if (generation.current !== run.runGeneration) return;
+ setGrowthRecord({ configurationKey, result });
+ setMessage(result.stoppedReason);
+ } catch (cause) {
+ if (generation.current !== run.runGeneration) return;
+ const next =
+ cause instanceof Error
+ ? cause.message
+ : "Growth analysis could not start.";
+ setError(next);
+ setMessage("Growth settings require review.");
+ } finally {
+ finishRun(run.runGeneration, run.currentSupervisor);
+ }
+ };
+
+ const cancel = () => {
+ abortController.current?.abort();
+ supervisor.current?.cancel();
+ setMessage(
+ activeRun === "benchmark"
+ ? "Cancelling benchmark and terminating its worker…"
+ : "Cancelling growth analysis and terminating its worker…",
+ );
+ };
+
+ const progressMaximum = Math.max(1, progressState?.total ?? 1);
+ const progressValue = Math.min(
+ progressMaximum,
+ progressState?.completed ?? 0,
+ );
+
+ return (
+
+
+
+
+ setView("risk")}
+ >
+ Risk & growth
+
+ setView("benchmark")}
+ >
+ Benchmark
+
+
+
+ {flavour !== "ecmascript" ? (
+
+ The current analyser supports ECMAScript 2025 only. It will not apply
+ ECMAScript heuristics or native-browser timing claims to {flavour}.
+
+ ) : !syntaxCurrent ? (
+
+ Analysis requires a current pattern accepted by the ECMAScript syntax
+ provider.
+
+ ) : null}
+
+
+ {activeRun ? "running" : "ready"}
+ {message}
+
+ {progressState ? (
+
+
+
+ {progressState.completed.toLocaleString()} /{" "}
+ {progressState.total.toLocaleString()}
+
+
+ ) : null}
+ {error ? (
+
+ {error}
+
+ ) : null}
+
+ {view === "risk" ? (
+
+
+
+ {staticReport ? (
+ <>
+ {staticReport.summary}
+
+ {staticReport.findings.map((finding) => (
+
+ ))}
+
+
+ Static analyser scope and limitations
+
+ {staticReport.limitations.map((limitation) => (
+ {limitation}
+ ))}
+
+
+ >
+ ) : (
+
+ Static findings are unavailable until the active pattern has a
+ current supported syntax tree.
+
+ )}
+
+
+
+
+ ) : (
+
+
+
+
+ Each timing sample measures compile, first match, all matches and
+ replacement separately in the native ECMAScript worker. Worker
+ startup is reported only as cold start. Trace mode is never used.
+
+
+
+ Warm-up samples
+ setWarmupIterations(event.target.value)}
+ disabled={activeRun !== undefined}
+ />
+
+
+ Measured samples
+
+ setMeasuredIterations(event.target.value)
+ }
+ disabled={activeRun !== undefined}
+ />
+
+
+ Per-sample timeout
+
+ setBenchmarkTimeoutMs(event.target.value)
+ }
+ disabled={activeRun !== undefined}
+ >
+ 100 ms
+ 250 ms
+ 1 s
+ 2 s
+ 5 s
+ 10 s
+
+
+
+
+
+
Pattern
+
+ {pattern.length.toLocaleString()} UTF-16 units · flags{" "}
+ {flags.join("") || "(none)"}
+
+
+
+
Subject
+
+ {subject.length.toLocaleString()} UTF-16 units ·{" "}
+ {utf8ByteLength(subject).toLocaleString()} bytes
+
+
+
+
Replacement
+
+ {replacement.length.toLocaleString()} UTF-16 units · output
+ estimate is bounded before timing
+
+
+
+
+ void runBenchmark()}
+ >
+ Run bounded benchmark
+
+ {activeRun === "benchmark" ? (
+
+ Cancel benchmark
+
+ ) : null}
+
+ {benchmarkResult ? (
+
+ ) : null}
+
+
+ )}
+
+ );
+}
diff --git a/src/components/CapabilityPanel.test.tsx b/src/components/CapabilityPanel.test.tsx
index 431bf72..eca41a8 100644
--- a/src/components/CapabilityPanel.test.tsx
+++ b/src/components/CapabilityPanel.test.tsx
@@ -1,5 +1,6 @@
import { render, screen } from "@testing-library/react";
-import { describe, expect, it } from "vitest";
+import userEvent from "@testing-library/user-event";
+import { describe, expect, it, vi } from "vitest";
import type { RegexExecutionResult } from "../regex/model/match";
import type { RegexSyntaxResult } from "../regex/model/syntax";
import { CapabilityPanel } from "./CapabilityPanel";
@@ -78,11 +79,66 @@ describe("CapabilityPanel", () => {
).toBeInTheDocument();
expect(screen.getByText("Fixture engine")).toBeInTheDocument();
expect(screen.getByText("123")).toBeInTheDocument();
+ expect(
+ screen.getByText("Not separately reported by this adapter"),
+ ).toBeInTheDocument();
+ expect(screen.getByText("1", { selector: "dd" })).toBeInTheDocument();
expect(screen.getByText("UTF-8 bytes")).toBeInTheDocument();
+ expect(screen.getByText("fixture")).toBeInTheDocument();
const replacement = screen.getByText("Replacement").closest("div");
expect(replacement).toHaveTextContent("Unavailable");
const compilation = screen.getByText("Compilation").closest("div");
expect(compilation).toHaveTextContent("Available");
+ const generatedCases = screen.getByText("Generated cases").closest("div");
+ expect(generatedCases).toHaveTextContent(
+ "deterministic AST candidates, retained only after actual-engine verification",
+ );
+ const formatting = screen.getByText("Pattern formatting").closest("div");
+ expect(formatting).toHaveTextContent(
+ "grammar-backed literal/control escaping with mandatory exact-snapshot validation",
+ );
+ });
+
+ it("does not advertise ECMAScript generation for the partial PCRE2 syntax provider", () => {
+ render(
+ ,
+ );
+
+ const generatedCases = screen.getByText("Generated cases").closest("div");
+ expect(generatedCases).toHaveTextContent(
+ "Unavailable — the partial PCRE2 provider does not expose a complete generation AST",
+ );
+ const formatting = screen.getByText("Pattern formatting").closest("div");
+ expect(formatting).toHaveTextContent(
+ "Unavailable — no complete PCRE2 grammar-backed formatter is implemented",
+ );
+ });
+
+ it("offers an accessible close control when used as a modal panel", async () => {
+ const onClose = vi.fn();
+ const user = userEvent.setup();
+ render(
+ ,
+ );
+
+ await user.click(
+ screen.getByRole("button", { name: "Close capabilities" }),
+ );
+ expect(onClose).toHaveBeenCalledOnce();
});
});
diff --git a/src/components/CapabilityPanel.tsx b/src/components/CapabilityPanel.tsx
index 2a4e8eb..fe527be 100644
--- a/src/components/CapabilityPanel.tsx
+++ b/src/components/CapabilityPanel.tsx
@@ -1,4 +1,4 @@
-import { REGEXPP_VERSION, SYNTAX_PROFILE } from "../version";
+import { SYNTAX_PROFILE } from "../version";
import type { RegexExecutionResult } from "../regex/model/match";
import type { RegexSyntaxResult } from "../regex/model/syntax";
import type { RegexEngineCapabilities } from "../regex/model/flavour";
@@ -35,23 +35,27 @@ function offsetLabel(
export function CapabilityPanel({
syntax,
execution,
+ onClose,
}: {
readonly syntax?: RegexSyntaxResult;
readonly execution?: RegexExecutionResult;
+ readonly onClose?: () => void;
}) {
const capabilities = execution?.engine.capabilities;
+ const activeFlavour =
+ execution?.engine.flavour ?? syntax?.root.support.flavour;
const rows = [
[
"Flavour",
execution?.engine.flavour ??
syntax?.root.support.flavour ??
- "ECMAScript (awaiting workers)",
+ "Awaiting flavour workers",
],
[
"Syntax provider",
syntax
? `${syntax.provider.id} ${syntax.provider.version}`
- : `regexpp ${REGEXPP_VERSION} (awaiting syntax worker)`,
+ : "Awaiting syntax worker result",
],
["Syntax profile", SYNTAX_PROFILE],
[
@@ -65,9 +69,20 @@ export function CapabilityPanel({
execution?.engine.engineName ?? "Awaiting first engine result",
],
[
- "Engine/runtime version",
+ "Engine version",
execution?.engine.engineVersion ?? "Shown after first execution",
],
+ [
+ "Runtime version",
+ execution?.engine.runtimeVersion ??
+ (execution
+ ? "Not separately reported by this adapter"
+ : "Shown after first execution"),
+ ],
+ [
+ "Adapter version",
+ execution?.engine.adapterVersion ?? "Shown after first execution",
+ ],
["Native offsets", offsetLabel(execution?.engine.offsetUnit)],
["Compilation", capability(capabilities, "compilation")],
["Matching", capability(capabilities, "matching")],
@@ -90,6 +105,38 @@ export function CapabilityPanel({
),
],
["Benchmark", capability(capabilities, "benchmark")],
+ [
+ "Static risk analysis",
+ activeFlavour === "ecmascript"
+ ? "Available — advisory ECMAScript 2025 heuristics"
+ : activeFlavour === "pcre2"
+ ? "Unavailable — ECMAScript heuristics are not applied to PCRE2"
+ : "Awaiting flavour worker result",
+ ],
+ [
+ "Generated cases",
+ activeFlavour === "ecmascript"
+ ? "Available — deterministic AST candidates, retained only after actual-engine verification"
+ : activeFlavour === "pcre2"
+ ? "Unavailable — the partial PCRE2 provider does not expose a complete generation AST"
+ : "Awaiting flavour worker result",
+ ],
+ [
+ "Pattern formatting",
+ activeFlavour === "ecmascript"
+ ? "Available — grammar-backed literal/control escaping with mandatory exact-snapshot validation"
+ : activeFlavour === "pcre2"
+ ? "Unavailable — no complete PCRE2 grammar-backed formatter is implemented"
+ : "Awaiting flavour worker result",
+ ],
+ [
+ "Known syntax gaps",
+ syntax
+ ? syntax.coverage.unsupportedConstructs.length > 0
+ ? syntax.coverage.unsupportedConstructs.join("; ")
+ : "None reported by the active provider"
+ : "Awaiting syntax worker result",
+ ],
] as const;
return (
Current provider and adapter metadata
Capabilities
- Community build
+
+ Community build
+ {onClose ? (
+
+ ×
+
+ ) : null}
+
{rows.map(([term, value]) => (
@@ -112,10 +172,11 @@ export function CapabilityPanel({
))}
- Next flavour: official PCRE2 10.47 WebAssembly with actual callout
- traces. Python, Go, Rust, .NET and Java remain unavailable until their
- named runtimes pass the same worker, offset, conformance and licensing
- gates.
+ PCRE2 10.47 matching and substitution run in the bundled WebAssembly
+ worker. Its separate trace worker exposes bounded reported automatic
+ callouts; movement classifications remain explicitly derived. Python,
+ Go, Rust, .NET and Java remain unavailable until their named runtimes
+ pass the same worker, offset, conformance and licensing gates.
);
diff --git a/src/components/ComparisonCodePanel.css b/src/components/ComparisonCodePanel.css
new file mode 100644
index 0000000..7311987
--- /dev/null
+++ b/src/components/ComparisonCodePanel.css
@@ -0,0 +1,331 @@
+.comparison-code-dialog {
+ width: min(88rem, calc(100% - 2rem));
+}
+
+.comparison-code-panel {
+ max-height: calc(100vh - 2rem);
+ overflow: auto;
+}
+
+.comparison-tabs {
+ display: flex;
+ gap: 0.35rem;
+ padding: 0.7rem 0.8rem 0;
+}
+
+.comparison-tabs button {
+ border: 1px solid var(--toolbox-border);
+ border-radius: calc(var(--toolbox-radius) * 0.72);
+ padding: 0.55rem 0.8rem;
+ background: var(--toolbox-surface);
+ color: var(--toolbox-text);
+ font-weight: 750;
+}
+
+.comparison-tabs button[aria-selected="true"] {
+ border-color: var(--toolbox-accent);
+ background: var(--toolbox-accent-soft);
+}
+
+.comparison-config,
+.comparison-tab-panel {
+ display: grid;
+ gap: 0.8rem;
+ padding: 0.8rem;
+}
+
+.comparison-config {
+ border-bottom: 1px solid var(--toolbox-border);
+ background: var(--toolbox-surface-soft);
+}
+
+.comparison-config-row,
+.comparison-actions {
+ display: flex;
+ flex-wrap: wrap;
+ align-items: end;
+ gap: 0.65rem;
+}
+
+.comparison-config label,
+.comparison-limit-grid label,
+.comparison-text-field {
+ display: grid;
+ gap: 0.3rem;
+ color: var(--toolbox-muted);
+ font-size: 0.75rem;
+ font-weight: 700;
+}
+
+.comparison-config select,
+.comparison-config input,
+.comparison-config textarea {
+ border: 1px solid var(--toolbox-border);
+ border-radius: calc(var(--toolbox-radius) * 0.65);
+ padding: 0.5rem 0.6rem;
+ background: var(--toolbox-surface);
+ color: var(--toolbox-text);
+}
+
+.comparison-config textarea {
+ min-height: 4.8rem;
+ resize: vertical;
+ font-family: ui-monospace, monospace;
+ font-size: 0.78rem;
+ line-height: 1.45;
+}
+
+.comparison-scan-control {
+ display: flex !important;
+ align-items: center;
+ padding-bottom: 0.5rem;
+}
+
+.comparison-variant-grid,
+.comparison-side-grid {
+ display: grid;
+ grid-template-columns: repeat(2, minmax(0, 1fr));
+ gap: 0.8rem;
+}
+
+.comparison-flag-selector,
+.comparison-limit-grid {
+ min-width: 0;
+ margin: 0;
+ border: 1px solid var(--toolbox-border);
+ border-radius: calc(var(--toolbox-radius) * 0.7);
+ padding: 0.55rem;
+}
+
+.comparison-flag-selector legend,
+.comparison-limit-grid legend {
+ padding-inline: 0.3rem;
+ color: var(--toolbox-muted);
+ font-size: 0.72rem;
+ font-weight: 750;
+}
+
+.comparison-flag-selector {
+ display: flex;
+ flex-wrap: wrap;
+ gap: 0.35rem;
+}
+
+.comparison-flag-selector label {
+ display: inline-flex;
+ align-items: center;
+ gap: 0.25rem;
+ border: 1px solid var(--toolbox-border);
+ border-radius: 99rem;
+ padding: 0.25rem 0.45rem;
+ background: var(--toolbox-surface);
+}
+
+.comparison-flag-selector span {
+ font-size: 0.67rem;
+}
+
+.comparison-limit-grid {
+ display: grid;
+ grid-template-columns: repeat(6, minmax(6rem, 1fr));
+ gap: 0.5rem;
+}
+
+.comparison-subject textarea {
+ min-height: 7rem;
+}
+
+.comparison-actions p {
+ margin: 0;
+ color: var(--toolbox-muted);
+ font-size: 0.78rem;
+}
+
+.comparison-panel-status.status-error,
+.comparison-error {
+ color: var(--toolbox-danger);
+}
+
+.comparison-result {
+ display: grid;
+ gap: 0.8rem;
+}
+
+.comparison-verdict,
+.comparison-reasons,
+.comparison-differences,
+.comparison-alignments,
+.generated-code-result {
+ border: 1px solid var(--toolbox-border);
+ border-radius: calc(var(--toolbox-radius) * 0.78);
+ padding: 0.8rem;
+ background: var(--toolbox-surface);
+}
+
+.comparison-verdict h3,
+.comparison-verdict p,
+.comparison-verdict small,
+.comparison-reasons h4,
+.comparison-differences h4,
+.comparison-alignments h4,
+.generated-code-result h3,
+.generated-code-result p {
+ margin: 0;
+}
+
+.comparison-verdict {
+ display: grid;
+ gap: 0.3rem;
+ border-left: 4px solid var(--regex-success);
+}
+
+.comparison-verdict.verdict-not-comparable {
+ border-left-color: var(--regex-warning);
+}
+
+.comparison-verdict.verdict-different-for-current-input {
+ border-left-color: var(--toolbox-danger);
+}
+
+.comparison-verdict p,
+.comparison-verdict small,
+.comparison-reasons,
+.comparison-differences > p,
+.comparison-alignments > p,
+.generated-code-result li {
+ color: var(--toolbox-muted);
+ font-size: 0.77rem;
+ line-height: 1.5;
+}
+
+.comparison-reasons ul,
+.generated-code-result ul {
+ margin-bottom: 0;
+}
+
+.comparison-side-card {
+ min-width: 0;
+ overflow: hidden;
+ border: 1px solid var(--toolbox-border);
+ border-radius: calc(var(--toolbox-radius) * 0.78);
+ background: var(--toolbox-surface);
+}
+
+.comparison-side-card > header,
+.generated-code-result > header {
+ display: flex;
+ align-items: center;
+ justify-content: space-between;
+ gap: 0.7rem;
+ padding: 0.7rem;
+ border-bottom: 1px solid var(--toolbox-border);
+}
+
+.comparison-side-card h4,
+.comparison-side-card p {
+ margin: 0;
+}
+
+.comparison-side-card dl {
+ display: grid;
+ grid-template-columns: repeat(2, minmax(0, 1fr));
+ margin: 0;
+}
+
+.comparison-side-card dl > div {
+ min-width: 0;
+ padding: 0.55rem 0.7rem;
+ border-bottom: 1px solid var(--toolbox-border);
+}
+
+.comparison-side-card dt {
+ color: var(--toolbox-muted);
+ font-size: 0.67rem;
+ font-weight: 750;
+ text-transform: uppercase;
+}
+
+.comparison-side-card dd {
+ margin: 0.15rem 0 0;
+ overflow-wrap: anywhere;
+ font-size: 0.76rem;
+}
+
+.comparison-side-card details,
+.comparison-side-card > .comparison-error {
+ padding: 0.55rem 0.7rem;
+ font-size: 0.74rem;
+}
+
+.comparison-side-card details ul {
+ margin-bottom: 0;
+}
+
+.comparison-differences,
+.comparison-alignments {
+ overflow: auto;
+}
+
+.comparison-differences h4,
+.comparison-alignments h4 {
+ margin-bottom: 0.65rem;
+}
+
+.comparison-differences td code {
+ display: block;
+ max-width: 22rem;
+ overflow-wrap: anywhere;
+ white-space: pre-wrap;
+}
+
+.comparison-limit-note {
+ margin-bottom: 0 !important;
+}
+
+.generated-code-result {
+ min-width: 0;
+ padding: 0;
+ overflow: hidden;
+}
+
+.generated-code-result > p,
+.generated-code-result > ul {
+ margin-inline: 0.8rem;
+}
+
+.generated-code-result pre {
+ max-height: 34rem;
+ overflow: auto;
+ margin: 0;
+ border-top: 1px solid var(--toolbox-border);
+ padding: 0.8rem;
+ background: #111827;
+ color: #e5e7eb;
+ font-size: 0.72rem;
+ line-height: 1.45;
+}
+
+.comparison-empty-state {
+ margin: 0;
+ padding: 1.2rem;
+ color: var(--toolbox-muted);
+ text-align: center;
+}
+
+@media (max-width: 64rem) {
+ .comparison-limit-grid {
+ grid-template-columns: repeat(3, minmax(6rem, 1fr));
+ }
+}
+
+@media (max-width: 48rem) {
+ .comparison-variant-grid,
+ .comparison-side-grid,
+ .comparison-side-card dl {
+ grid-template-columns: 1fr;
+ }
+
+ .comparison-limit-grid {
+ grid-template-columns: repeat(2, minmax(6rem, 1fr));
+ }
+}
diff --git a/src/components/ComparisonCodePanel.test.tsx b/src/components/ComparisonCodePanel.test.tsx
new file mode 100644
index 0000000..4750f33
--- /dev/null
+++ b/src/components/ComparisonCodePanel.test.tsx
@@ -0,0 +1,82 @@
+import { fireEvent, render, screen } from "@testing-library/react";
+import { describe, expect, it, vi } from "vitest";
+import { ComparisonCodePanel } from "./ComparisonCodePanel";
+
+function renderPanel() {
+ const onClose = vi.fn();
+ render(
+ \\p{Letter}+)"}
+ subject="Grüße"
+ replacement="$!"
+ scanAll
+ timeoutMs={2_000}
+ onClose={onClose}
+ />,
+ );
+ return { onClose };
+}
+
+describe("ComparisonCodePanel", () => {
+ it("exposes comparison and reviewed PCRE2 C generation through one usable path", () => {
+ renderPanel();
+
+ expect(
+ screen.getByRole("tab", { name: "Compare engines" }),
+ ).toHaveAttribute("aria-selected", "true");
+ expect(
+ screen.getByRole("textbox", { name: "Shared comparison pattern" }),
+ ).toHaveValue("(?\\p{Letter}+)");
+ expect(
+ screen.getByRole("button", { name: "Run comparison" }),
+ ).toBeEnabled();
+
+ fireEvent.click(screen.getByRole("tab", { name: "PCRE2 C code" }));
+ fireEvent.click(
+ screen.getByRole("button", { name: "Generate reviewed C17" }),
+ );
+
+ expect(screen.getByText("PCRE2 10.47 8-bit · all-matches")).toBeVisible();
+ expect(screen.getByText("Exact UTF-8 byte arrays")).toBeVisible();
+ expect(
+ screen.getByText(/PCRE2_UTF and PCRE2_UCP are mandatory/u),
+ ).toBeVisible();
+ const source = screen.getByText(
+ /This generated program requires PCRE2 10\.47 exactly/u,
+ );
+ expect(source).toHaveTextContent("pcre2_set_match_limit");
+ expect(source).not.toHaveTextContent("(?\\p{Letter}+)");
+ });
+
+ it("uses the explicit PCRE2 variant without rewriting it", () => {
+ renderPanel();
+ fireEvent.change(screen.getByLabelText("Pattern model"), {
+ target: { value: "variants" },
+ });
+ const pcrePattern = screen.getByLabelText("PCRE2 pattern variant");
+ fireEvent.change(pcrePattern, {
+ target: { value: "(?\\w++)" },
+ });
+ fireEvent.click(screen.getByRole("tab", { name: "PCRE2 C code" }));
+ fireEvent.click(
+ screen.getByRole("button", { name: "Generate reviewed C17" }),
+ );
+
+ const source = screen.getByText(
+ /This generated program requires PCRE2 10\.47 exactly/u,
+ );
+ expect(source).toHaveTextContent("0x2b, 0x2b");
+ expect(source).not.toHaveTextContent("(?\\w++)");
+ });
+
+ it("closes through the dialog action", () => {
+ const { onClose } = renderPanel();
+ fireEvent.click(
+ screen.getByRole("button", { name: "Close compare and generate" }),
+ );
+ expect(onClose).toHaveBeenCalledOnce();
+ });
+});
diff --git a/src/components/ComparisonCodePanel.tsx b/src/components/ComparisonCodePanel.tsx
new file mode 100644
index 0000000..c9c7295
--- /dev/null
+++ b/src/components/ComparisonCodePanel.tsx
@@ -0,0 +1,884 @@
+import { useEffect, useMemo, useRef, useState } from "react";
+import { writeClipboardText } from "../browser/clipboard";
+import {
+ AVAILABLE_REGEX_FLAVOURS,
+ defaultRegexOptions,
+} from "../regex/flavours/flavour-registry";
+import { ComparisonOrchestrator } from "../regex/comparison/ComparisonOrchestrator";
+import type {
+ ComparisonFlavour,
+ ComparisonPatternModel,
+ RegexComparisonResult,
+} from "../regex/comparison/comparison.types";
+import {
+ generatePcre2C,
+ type GeneratedPcre2CProgram,
+} from "../regex/codegen/pcre2-c";
+import {
+ DEFAULT_REGEX_LIMITS,
+ utf8ByteLength,
+} from "../regex/execution/request-limits";
+import type {
+ RegexEngineOptions,
+ RegexFlavourId,
+} from "../regex/model/flavour";
+import "./ComparisonCodePanel.css";
+
+const ECMASCRIPT = AVAILABLE_REGEX_FLAVOURS.require("ecmascript");
+const PCRE2 = AVAILABLE_REGEX_FLAVOURS.require("pcre2");
+const MAXIMUM_RENDERED_DIFFERENCES = 250;
+const MAXIMUM_RENDERED_ALIGNMENTS = 100;
+
+type PanelTab = "compare" | "code";
+type PanelStatus =
+ | { readonly kind: "idle"; readonly message: string }
+ | { readonly kind: "running"; readonly message: string }
+ | { readonly kind: "ready"; readonly message: string }
+ | { readonly kind: "error"; readonly message: string };
+
+export interface ComparisonCodePanelProps {
+ readonly activeFlavour: RegexFlavourId;
+ readonly activeFlags: readonly string[];
+ readonly activeOptions: RegexEngineOptions;
+ readonly pattern: string;
+ readonly subject: string;
+ readonly replacement: string;
+ readonly scanAll: boolean;
+ readonly timeoutMs: number;
+ readonly onClose: () => void;
+}
+
+function initialFlags(
+ activeFlavour: RegexFlavourId,
+ activeFlags: readonly string[],
+ flavour: ComparisonFlavour,
+): readonly string[] {
+ const definition = flavour === "ecmascript" ? ECMASCRIPT : PCRE2;
+ return activeFlavour === flavour
+ ? definition.flags
+ .map((flag) => flag.value)
+ .filter((flag) => activeFlags.includes(flag))
+ : definition.defaultFlags;
+}
+
+function initialOptions(
+ activeFlavour: RegexFlavourId,
+ activeOptions: RegexEngineOptions,
+): RegexEngineOptions {
+ return activeFlavour === "pcre2" ? activeOptions : defaultRegexOptions(PCRE2);
+}
+
+function toggleFlag(
+ current: readonly string[],
+ flag: string,
+ enabled: boolean,
+ flavour: ComparisonFlavour,
+): readonly string[] {
+ const definition = flavour === "ecmascript" ? ECMASCRIPT : PCRE2;
+ const selected = new Set(current);
+ if (enabled) {
+ selected.add(flag);
+ for (const group of definition.mutuallyExclusiveFlags ?? []) {
+ if (!group.includes(flag)) continue;
+ for (const incompatible of group) {
+ if (incompatible !== flag) selected.delete(incompatible);
+ }
+ }
+ } else {
+ selected.delete(flag);
+ }
+ return definition.flags
+ .map((candidate) => candidate.value)
+ .filter((candidate) => selected.has(candidate));
+}
+
+function preview(value: string | undefined, maximum = 240): string {
+ if (value === undefined) return "—";
+ return value.length <= maximum
+ ? value
+ : `${value.slice(0, maximum)}… (${value.length.toLocaleString()} UTF-16 units)`;
+}
+
+function downloadSource(program: GeneratedPcre2CProgram): void {
+ const url = URL.createObjectURL(
+ new Blob([program.source], { type: "text/x-c;charset=utf-8" }),
+ );
+ const anchor = document.createElement("a");
+ anchor.href = url;
+ anchor.download = program.fileName;
+ anchor.click();
+ window.setTimeout(() => URL.revokeObjectURL(url), 1_000);
+}
+
+function engineExecution(result: RegexComparisonResult["sides"][number]) {
+ return result.runtime.replacement?.execution ?? result.runtime.execution;
+}
+
+function SideResult({
+ side,
+}: {
+ readonly side: RegexComparisonResult["sides"][number];
+}) {
+ const execution = engineExecution(side);
+ const diagnostics = [
+ ...(side.syntax.pattern?.diagnostics ?? []),
+ ...(side.syntax.replacement?.diagnostics ?? []),
+ ...(execution?.diagnostics ?? []),
+ ];
+ return (
+
+
+
+
+
Pattern syntax
+
+ {side.syntax.pattern
+ ? side.syntax.pattern.accepted
+ ? "Accepted"
+ : "Rejected"
+ : side.syntax.status}
+
+
+
+
Provider
+
+ {side.syntax.pattern
+ ? `${side.syntax.pattern.provider.id} ${side.syntax.pattern.provider.version}`
+ : "Unavailable"}
+
+
+
+
Engine compile
+
+ {execution
+ ? execution.accepted
+ ? "Accepted"
+ : "Rejected"
+ : side.runtime.status}
+
+
+
+
Engine identity
+
+ {execution
+ ? `${execution.engine.engineName} · ${execution.engine.engineVersion}`
+ : "Unavailable"}
+
+
+
+
Exact flags
+
+ user {side.input.flags.join("") || "none"} · effective{" "}
+ {execution?.flags.effectiveFlags || "unavailable"}
+
+
+
+
Options
+
+ {JSON.stringify(side.input.options)}
+
+
+
+
Matches
+ {execution?.matches.length.toLocaleString() ?? "Unavailable"}
+
+
+
Offsets
+
+ native {execution?.engine.offsetUnit ?? "unavailable"} · comparison
+ UTF-16
+
+
+ {side.runtime.replacement ? (
+
+
Replacement
+
+ {side.runtime.replacement.outputBytes.toLocaleString()} bytes
+ {side.runtime.replacement.truncated ? " · incomplete" : ""}
+
+
+ ) : null}
+
+ {side.syntax.pattern?.coverage.unsupportedConstructs.length ? (
+
+ Provider coverage gaps
+
+ {side.syntax.pattern.coverage.unsupportedConstructs.map((gap) => (
+ {gap}
+ ))}
+
+
+ ) : null}
+ {diagnostics.length > 0 ? (
+
+ {diagnostics.length.toLocaleString()} diagnostic(s)
+
+ {diagnostics.slice(0, 100).map((diagnostic) => (
+
+ {diagnostic.severity}: {diagnostic.message}
+
+ ))}
+
+
+ ) : null}
+ {side.syntax.error || side.runtime.error ? (
+
+ {side.syntax.error ?? side.runtime.error}
+
+ ) : null}
+
+ );
+}
+
+function ComparisonResultView({
+ result,
+}: {
+ readonly result: RegexComparisonResult;
+}) {
+ const visibleDifferences = result.differences.slice(
+ 0,
+ MAXIMUM_RENDERED_DIFFERENCES,
+ );
+ const visibleAlignments = result.matchAlignments.slice(
+ 0,
+ MAXIMUM_RENDERED_ALIGNMENTS,
+ );
+ return (
+
+
+ Bounded comparison verdict
+ {result.status.replaceAll("-", " ")}
+ {result.notices[0]?.message}
+
+ {result.elapsedMs.toFixed(1)} ms wall time ·{" "}
+ {result.totalDifferences.toLocaleString()} recorded difference(s)
+
+
+ {result.notComparable.length > 0 ? (
+
+ Why this run is not comparable
+
+ {result.notComparable.map((reason, index) => (
+
+ {reason.code.replaceAll("-", " ")}: {" "}
+ {reason.message}
+
+ ))}
+
+
+ ) : null}
+
+
+
+
+
+ Syntax, execution and replacement differences
+ {visibleDifferences.length === 0 ? (
+
+ No retained semantic differences for this subject. Engine metadata
+ can still differ.
+
+ ) : (
+
+
+
+ Kind
+ Finding
+ ECMAScript
+ PCRE2
+
+
+
+ {visibleDifferences.map((difference, index) => (
+
+ {difference.kind.replaceAll("-", " ")}
+ {difference.summary}
+
+ {preview(difference.left)}
+
+
+ {preview(difference.right)}
+
+
+ ))}
+
+
+ )}
+ {visibleDifferences.length < result.totalDifferences ? (
+
+ Rendering the first {visibleDifferences.length.toLocaleString()} of{" "}
+ {result.totalDifferences.toLocaleString()} differences.
+
+ ) : null}
+
+ {visibleAlignments.length > 0 ? (
+
+ Match alignment by normalized editor range
+
+
+
+ Alignment
+ ECMAScript UTF-16
+ PCRE2 UTF-16
+ Equal
+
+
+
+ {visibleAlignments.map((alignment, index) => (
+
+ {alignment.alignment.replaceAll("-", " ")}
+
+ {alignment.left
+ ? `${alignment.left.range.startUtf16}–${alignment.left.range.endUtf16}`
+ : "—"}
+
+
+ {alignment.right
+ ? `${alignment.right.range.startUtf16}–${alignment.right.range.endUtf16}`
+ : "—"}
+
+ {alignment.equal ? "yes" : "no"}
+
+ ))}
+
+
+ {visibleAlignments.length < result.totalMatchAlignments ? (
+
+ Rendering the first {visibleAlignments.length.toLocaleString()} of{" "}
+ {result.totalMatchAlignments.toLocaleString()} alignments.
+
+ ) : null}
+
+ ) : null}
+
+ );
+}
+
+export function ComparisonCodePanel({
+ activeFlavour,
+ activeFlags,
+ activeOptions,
+ pattern: initialPattern,
+ subject: initialSubject,
+ replacement: initialReplacement,
+ scanAll: initialScanAll,
+ timeoutMs: initialTimeoutMs,
+ onClose,
+}: ComparisonCodePanelProps) {
+ const [tab, setTab] = useState("compare");
+ const [patternModel, setPatternModel] =
+ useState("shared");
+ const [sharedPattern, setSharedPattern] = useState(initialPattern);
+ const [ecmaPattern, setEcmaPattern] = useState(initialPattern);
+ const [pcrePattern, setPcrePattern] = useState(initialPattern);
+ const [subject, setSubject] = useState(initialSubject);
+ const [operation, setOperation] = useState<"match" | "replace">("match");
+ const [ecmaReplacement, setEcmaReplacement] = useState(initialReplacement);
+ const [pcreReplacement, setPcreReplacement] = useState(initialReplacement);
+ const [ecmaFlags, setEcmaFlags] = useState(() =>
+ initialFlags(activeFlavour, activeFlags, "ecmascript"),
+ );
+ const [pcreFlags, setPcreFlags] = useState(() =>
+ initialFlags(activeFlavour, activeFlags, "pcre2"),
+ );
+ const [pcreOptions, setPcreOptions] = useState(() =>
+ initialOptions(activeFlavour, activeOptions),
+ );
+ const [scanAll, setScanAll] = useState(initialScanAll);
+ const [timeoutMs, setTimeoutMs] = useState(initialTimeoutMs);
+ const [maximumMatches, setMaximumMatches] = useState(1_000);
+ const [maximumCaptureRows, setMaximumCaptureRows] = useState(10_000);
+ const [maximumOutputBytes, setMaximumOutputBytes] = useState(1024 * 1024);
+ const [status, setStatus] = useState({
+ kind: "idle",
+ message: "Configure two exact engine requests.",
+ });
+ const [result, setResult] = useState();
+ const [generated, setGenerated] = useState();
+ const [generatedStatus, setGeneratedStatus] = useState("");
+ const orchestrator = useRef(undefined);
+ const comparisonRevision = useRef(0);
+
+ useEffect(
+ () => () => {
+ comparisonRevision.current += 1;
+ orchestrator.current?.dispose();
+ orchestrator.current = undefined;
+ },
+ [],
+ );
+
+ const pcreOptionValues = useMemo(
+ () => ({
+ matchLimit: Number(pcreOptions.matchLimit),
+ depthLimit: Number(pcreOptions.depthLimit),
+ heapLimitKib: Number(pcreOptions.heapLimitKib),
+ }),
+ [pcreOptions],
+ );
+
+ const compare = async () => {
+ const revision = ++comparisonRevision.current;
+ setResult(undefined);
+ setStatus({
+ kind: "running",
+ message: "Running independently killable ECMAScript and PCRE2 workers…",
+ });
+ try {
+ orchestrator.current ??= new ComparisonOrchestrator();
+ const comparison = await orchestrator.current.compare({
+ schemaVersion: 1,
+ patternModel,
+ operation,
+ subject,
+ scanAll,
+ maximumMatches,
+ maximumCaptureRows,
+ maximumOutputBytes,
+ timeoutMs,
+ sides: [
+ {
+ flavour: "ecmascript",
+ flavourVersion: ECMASCRIPT.defaultVersion,
+ pattern: patternModel === "shared" ? sharedPattern : ecmaPattern,
+ flags: ecmaFlags,
+ options: {},
+ ...(operation === "replace"
+ ? { replacement: ecmaReplacement }
+ : {}),
+ },
+ {
+ flavour: "pcre2",
+ flavourVersion: PCRE2.defaultVersion,
+ pattern: patternModel === "shared" ? sharedPattern : pcrePattern,
+ flags: pcreFlags,
+ options: pcreOptionValues,
+ ...(operation === "replace"
+ ? { replacement: pcreReplacement }
+ : {}),
+ },
+ ],
+ });
+ if (revision !== comparisonRevision.current) return;
+ setResult(comparison);
+ setStatus({
+ kind: "ready",
+ message:
+ comparison.status === "not-comparable"
+ ? "Both side outcomes are retained; equivalence is not claimed."
+ : "Both exact requests completed.",
+ });
+ } catch (error) {
+ if (revision !== comparisonRevision.current) return;
+ setStatus({
+ kind: "error",
+ message: error instanceof Error ? error.message : String(error),
+ });
+ }
+ };
+
+ const cancel = () => {
+ comparisonRevision.current += 1;
+ orchestrator.current?.cancel();
+ setStatus({
+ kind: "idle",
+ message: "Comparison cancelled; both active workers were terminated.",
+ });
+ };
+
+ const generate = () => {
+ setGenerated(undefined);
+ setGeneratedStatus("");
+ try {
+ const program = generatePcre2C({
+ flavour: "pcre2",
+ flavourVersion: PCRE2.defaultVersion,
+ operation,
+ pattern: patternModel === "shared" ? sharedPattern : pcrePattern,
+ flags: pcreFlags,
+ options: pcreOptionValues,
+ subject,
+ scanAll,
+ ...(operation === "replace" ? { replacement: pcreReplacement } : {}),
+ maximumMatches,
+ maximumCaptureRows,
+ maximumOutputBytes,
+ });
+ setGenerated(program);
+ setGeneratedStatus(
+ "Generated locally from the exact PCRE2 snapshot shown above.",
+ );
+ } catch (error) {
+ setGeneratedStatus(
+ error instanceof Error ? error.message : String(error),
+ );
+ }
+ };
+
+ const setPcreOption = (name: string, value: number) => {
+ setPcreOptions((current) => ({ ...current, [name]: value }));
+ };
+
+ return (
+
+
+
+ setTab("compare")}
+ >
+ Compare engines
+
+ setTab("code")}
+ >
+ PCRE2 C code
+
+
+
+
+
+ Pattern model
+
+ setPatternModel(event.target.value as ComparisonPatternModel)
+ }
+ >
+ Shared pattern · unchanged
+ Explicit per-flavour variants
+
+
+
+ Operation
+
+ setOperation(event.target.value as "match" | "replace")
+ }
+ >
+ Match
+ Replace
+
+
+
+ setScanAll(event.target.checked)}
+ />
+ Scan all explicitly
+
+
+ {patternModel === "shared" ? (
+
+ Shared pattern sent unchanged to both engines
+
+ ) : (
+
+
+ ECMAScript pattern variant
+
+
+ PCRE2 pattern variant
+
+
+ )}
+
+ {(
+ [
+ ["ecmascript", ECMASCRIPT, ecmaFlags, setEcmaFlags],
+ ["pcre2", PCRE2, pcreFlags, setPcreFlags],
+ ] as const
+ ).map(([flavour, definition, selected, setSelected]) => (
+
+ {definition.label} flags · exact
+ {definition.flags.map((flag) => (
+
+
+ setSelected(
+ toggleFlag(
+ selected,
+ flag.value,
+ event.target.checked,
+ flavour,
+ ),
+ )
+ }
+ />
+ {flag.value}
+ {flag.label}
+
+ ))}
+
+ ))}
+
+
+ PCRE2 native and host bounds
+
+ Match steps
+
+ setPcreOption("matchLimit", Number(event.target.value))
+ }
+ />
+
+
+ Depth
+
+ setPcreOption("depthLimit", Number(event.target.value))
+ }
+ />
+
+
+ Heap KiB
+
+ setPcreOption("heapLimitKib", Number(event.target.value))
+ }
+ />
+
+
+ Matches
+
+ setMaximumMatches(Number(event.target.value))
+ }
+ />
+
+
+ Capture rows
+
+ setMaximumCaptureRows(Number(event.target.value))
+ }
+ />
+
+
+ Worker ms
+ setTimeoutMs(Number(event.target.value))}
+ />
+
+
+ {operation === "replace" ? (
+
+
+ ECMAScript replacement · exact syntax
+
+
+ PCRE2 replacement · exact syntax
+
+
+ Output byte cap
+
+ setMaximumOutputBytes(Number(event.target.value))
+ }
+ />
+
+
+ ) : null}
+
+
+ Shared subject · {utf8ByteLength(subject).toLocaleString()} UTF-8
+ bytes
+
+
+
+ {tab === "compare" ? (
+
+
+
void compare()}
+ >
+ {status.kind === "running" ? "Comparing…" : "Run comparison"}
+
+ {status.kind === "running" ? (
+
+ Cancel both workers
+
+ ) : null}
+
+ {status.message}
+
+
+ {result ?
: null}
+
+ ) : (
+
+
+
+ Generate reviewed C17
+
+ {generated ? (
+ <>
+
{
+ void writeClipboardText(generated.source)
+ .then(() =>
+ setGeneratedStatus("Complete generated source copied."),
+ )
+ .catch((error: unknown) =>
+ setGeneratedStatus(
+ error instanceof Error
+ ? error.message
+ : String(error),
+ ),
+ );
+ }}
+ >
+ Copy complete source
+
+
downloadSource(generated)}
+ >
+ Download .c
+
+ >
+ ) : null}
+
{generatedStatus}
+
+ {generated ? (
+
+
+
+ Compile: {generated.compileCommand}
+
+
+ {generated.caveats.map((caveat) => (
+ {caveat}
+ ))}
+
+
+ {generated.source}
+
+
+ ) : (
+
+ Generation consumes the exact PCRE2 pattern, flags, native limits,
+ subject and optional replacement above. It never powers the
+ browser runtime.
+
+ )}
+
+ )}
+
+ );
+}
diff --git a/src/components/CorpusPanel.test.tsx b/src/components/CorpusPanel.test.tsx
new file mode 100644
index 0000000..1dcde7d
--- /dev/null
+++ b/src/components/CorpusPanel.test.tsx
@@ -0,0 +1,253 @@
+import {
+ act,
+ cleanup,
+ fireEvent,
+ render,
+ screen,
+} from "@testing-library/react";
+import userEvent from "@testing-library/user-event";
+import { afterEach, describe, expect, it, vi } from "vitest";
+import type { RegexExecutionRequest } from "../regex/model/match";
+
+const engineHarness = vi.hoisted(() => ({
+ cancelCalls: 0,
+ disposeCalls: 0,
+ executeRequests: [] as RegexExecutionRequest[],
+}));
+
+vi.mock("../regex/execution/EngineSupervisor", () => ({
+ EngineSupervisor: class {
+ execute(request: RegexExecutionRequest) {
+ engineHarness.executeRequests.push(request);
+ return Promise.resolve({
+ accepted: true,
+ engine: {
+ flavour: "ecmascript",
+ adapterVersion: "test",
+ engineName: "Test engine",
+ engineVersion: "1",
+ runtimeVersion: "1",
+ offsetUnit: "utf16",
+ capabilities: {
+ compilation: true,
+ matching: true,
+ replacement: true,
+ namedCaptures: true,
+ captureHistory: false,
+ actualTrace: false,
+ benchmark: false,
+ },
+ },
+ flags: {
+ userFlags: "",
+ effectiveFlags: "g",
+ internallyAddedIndicesFlag: true,
+ internallyAddedGlobalFlag: true,
+ },
+ matches: [
+ {
+ matchNumber: 1,
+ value: "42",
+ valueStatus: "complete",
+ range: { startUtf16: 0, endUtf16: 2 },
+ nativeRange: { start: 0, end: 2, unit: "utf16" },
+ captures: [],
+ },
+ ],
+ diagnostics: [],
+ elapsedMs: 2,
+ truncated: false,
+ });
+ }
+
+ replace(): Promise {
+ return Promise.reject(new Error("Unexpected replacement call"));
+ }
+
+ cancel(): void {
+ engineHarness.cancelCalls += 1;
+ }
+
+ dispose(): void {
+ engineHarness.disposeCalls += 1;
+ }
+ },
+}));
+
+import { CorpusPanel } from "./CorpusPanel";
+
+afterEach(() => {
+ cleanup();
+ engineHarness.cancelCalls = 0;
+ engineHarness.disposeCalls = 0;
+ engineHarness.executeRequests.length = 0;
+});
+
+function renderPanel(
+ overrides: Partial[0]> = {},
+) {
+ return render(
+ {}}
+ timeoutMs={2_000}
+ {...overrides}
+ />,
+ );
+}
+
+describe("CorpusPanel", () => {
+ it("adds pasted text and produces an exhaustive per-document summary", async () => {
+ renderPanel();
+ await userEvent.type(
+ screen.getByRole("textbox", { name: "Pasted corpus text" }),
+ "42",
+ );
+ await userEvent.click(
+ screen.getByRole("button", { name: "Add pasted text" }),
+ );
+
+ expect(
+ screen.getByRole("list", { name: "Corpus documents" }),
+ ).toHaveTextContent("pasted-text.txt");
+ await userEvent.click(screen.getByRole("button", { name: "Scan corpus" }));
+
+ expect(
+ await screen.findByRole("table", { name: "Corpus document results" }),
+ ).toHaveTextContent("complete");
+ expect(engineHarness.executeRequests).toHaveLength(1);
+ expect(engineHarness.executeRequests[0]).toMatchObject({
+ subject: "42",
+ scanAll: true,
+ });
+ expect(
+ screen.getByRole("button", { name: "Download summary JSON" }),
+ ).toBeEnabled();
+ });
+
+ it("blocks invalid runs and clears results when configuration changes", async () => {
+ const rendered = renderPanel({ patternAccepted: false });
+ fireEvent.click(screen.getByRole("button", { name: "Add pasted text" }));
+ expect(screen.getByRole("button", { name: "Scan corpus" })).toBeDisabled();
+ expect(screen.getByRole("alert")).toHaveTextContent(
+ "syntax provider accepts",
+ );
+
+ rendered.rerender(
+ {}}
+ timeoutMs={2_000}
+ />,
+ );
+ expect(screen.getByRole("button", { name: "Scan corpus" })).toBeEnabled();
+ expect(screen.getByRole("status")).toHaveTextContent(
+ "previous corpus results were cleared",
+ );
+ });
+
+ it("cancels an active batch when the workspace is left", async () => {
+ let resolveExecution: ((value: never) => void) | undefined;
+ const pending = new Promise((resolve) => {
+ resolveExecution = resolve;
+ });
+ const execute = vi
+ .spyOn(
+ (await import("../regex/execution/EngineSupervisor")).EngineSupervisor
+ .prototype,
+ "execute",
+ )
+ .mockReturnValue(pending);
+ const rendered = renderPanel();
+ fireEvent.click(screen.getByRole("button", { name: "Add pasted text" }));
+ fireEvent.click(screen.getByRole("button", { name: "Scan corpus" }));
+ expect(
+ screen.getByRole("button", { name: "Cancel corpus run" }),
+ ).toBeEnabled();
+
+ rendered.rerender(
+ {}}
+ timeoutMs={2_000}
+ />,
+ );
+ expect(engineHarness.cancelCalls).toBeGreaterThan(0);
+ execute.mockRestore();
+ await act(async () => {
+ resolveExecution?.(undefined as never);
+ });
+ });
+
+ it("does not publish stale batch results after configuration changes", async () => {
+ let resolveExecution: ((value: never) => void) | undefined;
+ const pending = new Promise((resolve) => {
+ resolveExecution = resolve;
+ });
+ const execute = vi
+ .spyOn(
+ (await import("../regex/execution/EngineSupervisor")).EngineSupervisor
+ .prototype,
+ "execute",
+ )
+ .mockReturnValue(pending);
+ const rendered = renderPanel();
+ fireEvent.click(screen.getByRole("button", { name: "Add pasted text" }));
+ fireEvent.click(screen.getByRole("button", { name: "Scan corpus" }));
+
+ rendered.rerender(
+ {}}
+ timeoutMs={2_000}
+ />,
+ );
+ await act(async () => {
+ resolveExecution?.(undefined as never);
+ });
+
+ expect(
+ screen.queryByRole("table", { name: "Corpus document results" }),
+ ).not.toBeInTheDocument();
+ expect(screen.getByRole("status")).toHaveTextContent(
+ "previous corpus results were cleared",
+ );
+ execute.mockRestore();
+ });
+});
diff --git a/src/components/CorpusPanel.tsx b/src/components/CorpusPanel.tsx
new file mode 100644
index 0000000..930a1c3
--- /dev/null
+++ b/src/components/CorpusPanel.tsx
@@ -0,0 +1,1024 @@
+import { useEffect, useMemo, useRef, useState } from "react";
+import { EngineSupervisor } from "../regex/execution/EngineSupervisor";
+import {
+ DEFAULT_REGEX_LIMITS,
+ utf8ByteLength,
+} from "../regex/execution/request-limits";
+import type {
+ RegexEngineOptions,
+ RegexFlavourId,
+} from "../regex/model/flavour";
+import type { CaptureDefinition } from "../regex/model/syntax";
+import {
+ canExportAppliedCorpus,
+ corpusAppliedDownloadName,
+ createAppliedCorpusZip,
+ serializeCorpusReport,
+ type CorpusReportFormat,
+} from "../regex/corpus/corpus-export";
+import {
+ corpusTotalBytes,
+ createPastedCorpusDocument,
+ readCorpusFiles,
+} from "../regex/corpus/corpus-input";
+import { runCorpusBatch } from "../regex/corpus/corpus-runner";
+import type {
+ CorpusBatchResult,
+ CorpusDocument,
+ CorpusDocumentResult,
+ CorpusOperation,
+ CorpusProgress,
+ CorpusSubjectMode,
+} from "../regex/corpus/corpus.types";
+
+const CORPUS_FILE_ACCEPT =
+ "text/*,.asc,.cfg,.conf,.csv,.htm,.html,.ini,.json,.jsonl,.log,.md,.ndjson,.properties,.rtf,.sql,.svg,.toml,.tsv,.txt,.xml,.yaml,.yml";
+
+function downloadBlob(blob: Blob, name: string): void {
+ const url = URL.createObjectURL(blob);
+ const anchor = document.createElement("a");
+ anchor.href = url;
+ anchor.download = name;
+ anchor.click();
+ window.setTimeout(() => URL.revokeObjectURL(url), 1_000);
+}
+
+function resultCount(result: CorpusDocumentResult): string {
+ if (result.status === "complete") return result.matchCount.toLocaleString();
+ if (result.status === "partial") {
+ return `≥${result.matchCount.toLocaleString()}`;
+ }
+ return "—";
+}
+
+function corpusMiB(bytes: number): string {
+ return `${(bytes / 1024 / 1024).toFixed(bytes === 0 ? 0 : 2)} MiB`;
+}
+
+function CorpusResultTable({
+ batch,
+ allowContentExport,
+ selectedDocumentId,
+ onInspect,
+}: {
+ readonly batch: CorpusBatchResult;
+ readonly allowContentExport: boolean;
+ readonly selectedDocumentId?: string;
+ readonly onInspect: (result: CorpusDocumentResult) => void;
+}) {
+ return (
+
+
+
+
+ Document
+ Status
+ Matches
+ Captures
+ Lines
+ Input
+ Output
+ Change
+ Engine time
+ Actions
+
+
+
+ {batch.documents.map((result) => {
+ const downloadable =
+ allowContentExport &&
+ result.status === "complete" &&
+ result.output !== undefined;
+ return (
+
+
+ {result.name}
+ {result.message ? {result.message} : null}
+
+
+
+ {result.status.replace("-", " ")}
+
+
+ {resultCount(result)}
+
+ {Math.max(
+ 0,
+ result.captureSummaries.filter(
+ (capture) => capture.groupNumber !== 0,
+ ).length,
+ ).toLocaleString()}
+ {result.captureSummariesTruncated ? "+" : ""}
+
+ {result.lineCount.toLocaleString()}
+ {result.inputBytes.toLocaleString()} B
+
+ {result.outputBytes === undefined
+ ? "—"
+ : `${result.outputBytes.toLocaleString()} B`}
+
+
+ {result.changed === true
+ ? "Changed"
+ : result.changed === false
+ ? "Unchanged"
+ : "—"}
+
+ {result.elapsedMs.toFixed(2)} ms
+
+
+ onInspect(result)}
+ >
+ Inspect
+
+ {batch.operation === "replace" ? (
+ {
+ if (!downloadable || result.output === undefined)
+ return;
+ downloadBlob(
+ new Blob([result.output], {
+ type: "text/plain;charset=utf-8",
+ }),
+ corpusAppliedDownloadName(result),
+ );
+ }}
+ >
+ Download
+
+ ) : null}
+
+
+
+ );
+ })}
+
+
+
+ );
+}
+
+export interface CorpusPanelProps {
+ readonly active: boolean;
+ readonly flavour: RegexFlavourId;
+ readonly flavourVersion?: string;
+ readonly options: RegexEngineOptions;
+ readonly pattern: string;
+ readonly flags: readonly string[];
+ readonly captureMetadata: readonly CaptureDefinition[];
+ readonly patternAccepted: boolean;
+ readonly replacement: string;
+ readonly replacementAccepted: boolean;
+ readonly onReplacementChange: (value: string) => void;
+ readonly timeoutMs: number;
+}
+
+export function CorpusPanel({
+ active,
+ flavour,
+ flavourVersion,
+ options,
+ pattern,
+ flags,
+ captureMetadata,
+ patternAccepted,
+ replacement,
+ replacementAccepted,
+ onReplacementChange,
+ timeoutMs,
+}: CorpusPanelProps) {
+ const [documents, setDocuments] = useState([]);
+ const [operation, setOperation] = useState("match");
+ const [subjectMode, setSubjectMode] = useState("document");
+ const [pastedName, setPastedName] = useState("pasted-text.txt");
+ const [pastedText, setPastedText] = useState("");
+ const [batch, setBatch] = useState();
+ const [progress, setProgress] = useState();
+ const [status, setStatus] = useState(
+ "Corpus documents stay in this browser tab and are never uploaded.",
+ );
+ const [running, setRunning] = useState(false);
+ const [importing, setImporting] = useState(false);
+ const [allowContentExport, setAllowContentExport] = useState(false);
+ const [reportFormat, setReportFormat] = useState("json");
+ const [selectedDocumentId, setSelectedDocumentId] = useState();
+ const [exportingZip, setExportingZip] = useState(false);
+ const fileInput = useRef(null);
+ const engine = useRef(undefined);
+ const runController = useRef(undefined);
+ const runRevision = useRef(0);
+ const importController = useRef(undefined);
+ const zipCancellation = useRef<(() => void) | undefined>(undefined);
+
+ const configurationIdentity = useMemo(
+ () =>
+ JSON.stringify({
+ flavour,
+ flavourVersion,
+ options,
+ pattern,
+ flags,
+ replacement,
+ }),
+ [flags, flavour, flavourVersion, options, pattern, replacement],
+ );
+ const previousConfigurationIdentity = useRef(configurationIdentity);
+
+ const totalBytes = useMemo(() => corpusTotalBytes(documents), [documents]);
+ const canRun =
+ documents.length > 0 &&
+ patternAccepted &&
+ (operation === "match" || replacementAccepted) &&
+ !running &&
+ !importing;
+ const exactAppliedExport = batch ? canExportAppliedCorpus(batch) : false;
+ const selectedResult = batch?.documents.find(
+ (result) => result.documentId === selectedDocumentId,
+ );
+
+ const cancelRun = (message = "Cancelling corpus processing…") => {
+ if (!runController.current) return;
+ runController.current.abort();
+ engine.current?.cancel();
+ setStatus(message);
+ };
+
+ const invalidateResults = (message: string) => {
+ runRevision.current += 1;
+ if (running) cancelRun();
+ setBatch(undefined);
+ setProgress(undefined);
+ setAllowContentExport(false);
+ setSelectedDocumentId(undefined);
+ setStatus(message);
+ };
+
+ useEffect(() => {
+ if (previousConfigurationIdentity.current === configurationIdentity) return;
+ previousConfigurationIdentity.current = configurationIdentity;
+ runRevision.current += 1;
+ if (running) {
+ runController.current?.abort();
+ engine.current?.cancel();
+ }
+ setBatch(undefined);
+ setProgress(undefined);
+ setAllowContentExport(false);
+ setSelectedDocumentId(undefined);
+ setStatus(
+ "Pattern, flavour, flags, options or replacement changed; previous corpus results were cleared.",
+ );
+ }, [configurationIdentity, running]);
+
+ useEffect(() => {
+ if (!active && running) {
+ runController.current?.abort();
+ engine.current?.cancel();
+ }
+ }, [active, running]);
+
+ useEffect(
+ () => () => {
+ runController.current?.abort();
+ importController.current?.abort();
+ zipCancellation.current?.();
+ engine.current?.dispose();
+ engine.current = undefined;
+ },
+ [],
+ );
+
+ const importFiles = async (files: readonly File[]) => {
+ if (files.length === 0) return;
+ const controller = new AbortController();
+ importController.current?.abort();
+ importController.current = controller;
+ setImporting(true);
+ setStatus(`Reading ${files.length.toLocaleString()} local text files…`);
+ try {
+ const imported = await readCorpusFiles(
+ files,
+ documents,
+ controller.signal,
+ (completed, total, name) =>
+ setStatus(`Read ${completed} of ${total}: ${name}`),
+ );
+ if (controller.signal.aborted) return;
+ setDocuments((current) => [...current, ...imported]);
+ setBatch(undefined);
+ setAllowContentExport(false);
+ setStatus(
+ `Added ${imported.length.toLocaleString()} local text document${
+ imported.length === 1 ? "" : "s"
+ }. Nothing was uploaded.`,
+ );
+ } catch (error) {
+ if (controller.signal.aborted) {
+ setStatus("File import cancelled.");
+ } else {
+ setStatus(error instanceof Error ? error.message : String(error));
+ }
+ } finally {
+ if (importController.current === controller) {
+ importController.current = undefined;
+ setImporting(false);
+ }
+ }
+ };
+
+ const run = async () => {
+ if (!canRun) return;
+ const revision = runRevision.current + 1;
+ runRevision.current = revision;
+ const controller = new AbortController();
+ runController.current = controller;
+ engine.current ??= new EngineSupervisor();
+ setRunning(true);
+ setBatch(undefined);
+ setSelectedDocumentId(undefined);
+ setAllowContentExport(false);
+ setProgress({
+ completedDocuments: 0,
+ totalDocuments: documents.length,
+ currentDocumentName: documents[0]?.name,
+ totalMatches: 0,
+ });
+ setStatus(
+ `Processing ${documents.length.toLocaleString()} documents locally…`,
+ );
+ const wallTimer = window.setTimeout(() => {
+ controller.abort("corpus-wall-time");
+ engine.current?.cancel();
+ }, DEFAULT_REGEX_LIMITS.maximumCorpusWallTimeMs);
+ try {
+ const result = await runCorpusBatch(
+ documents,
+ {
+ flavour,
+ ...(flavourVersion === undefined ? {} : { flavourVersion }),
+ options,
+ pattern,
+ flags,
+ captureMetadata,
+ replacement,
+ operation,
+ subjectMode,
+ timeoutMs,
+ },
+ {
+ engine: engine.current,
+ signal: controller.signal,
+ onProgress: setProgress,
+ },
+ );
+ if (runRevision.current !== revision) return;
+ setBatch(result);
+ setSelectedDocumentId(result.documents[0]?.documentId);
+ setStatus(result.message);
+ } catch (error) {
+ if (runRevision.current !== revision) return;
+ setStatus(error instanceof Error ? error.message : String(error));
+ } finally {
+ window.clearTimeout(wallTimer);
+ if (runController.current === controller) {
+ runController.current = undefined;
+ setRunning(false);
+ }
+ }
+ };
+
+ return (
+
+
+
+
+ fileInput.current?.click()}
+ >
+ Add text files
+
+ {
+ const files = Array.from(event.target.files ?? []);
+ event.target.value = "";
+ void importFiles(files);
+ }}
+ />
+ {importing ? (
+ importController.current?.abort()}
+ >
+ Cancel import
+
+ ) : null}
+
+ UTF-8 text only ·{" "}
+ {corpusMiB(DEFAULT_REGEX_LIMITS.maximumCorpusDocumentBytes)} per
+ document · {corpusMiB(DEFAULT_REGEX_LIMITS.corpusHardBytes)} total
+
+
+
+
+ Document name
+ setPastedName(event.target.value)}
+ />
+
+
+ Pasted corpus text
+
+
+
+ {utf8ByteLength(pastedText).toLocaleString()} UTF-8 bytes
+
+ {
+ try {
+ const added = createPastedCorpusDocument(
+ pastedName,
+ pastedText,
+ documents,
+ );
+ setDocuments((current) => [...current, added]);
+ setPastedText("");
+ setBatch(undefined);
+ setAllowContentExport(false);
+ setStatus(
+ `Added ${added.name}. Its content remains only in this browser tab.`,
+ );
+ } catch (error) {
+ setStatus(
+ error instanceof Error ? error.message : String(error),
+ );
+ }
+ }}
+ >
+ Add pasted text
+
+
+
+ {documents.length > 0 ? (
+
+ {documents.map((document) => (
+
+
+ {document.name}
+
+ {document.bytes.toLocaleString()} B ·{" "}
+ {document.source === "file" ? "local file" : "pasted text"}
+
+
+ {
+ setDocuments((current) =>
+ current.filter(
+ (candidate) => candidate.id !== document.id,
+ ),
+ );
+ invalidateResults(
+ `${document.name} was removed; previous corpus results were cleared.`,
+ );
+ }}
+ >
+ ×
+
+
+ ))}
+
+ ) : (
+
+ Add one or more local files, or paste a text document.
+
+ )}
+
+
+
+
+
+ Subject semantics
+
+ {
+ setSubjectMode("document");
+ invalidateResults(
+ "Corpus subject semantics changed; previous results were cleared.",
+ );
+ }}
+ />
+
+ Whole document
+
+ Run once per document. Matches may cross line endings when the
+ pattern and active flags allow it.
+
+
+
+
+ {
+ setSubjectMode("line");
+ invalidateResults(
+ "Corpus subject semantics changed; previous results were cleared.",
+ );
+ }}
+ />
+
+ Independent lines
+
+ Run separately on every logical line. Anchors see one line at a
+ time, matches cannot cross lines, and apply preserves each
+ original line separator.
+
+
+
+
+
+ Corpus operation
+
+ {
+ setOperation("match");
+ invalidateResults(
+ "Corpus operation changed; previous results were cleared.",
+ );
+ }}
+ />
+
+ Find matches
+ Count every returned match per document.
+
+
+
+ {
+ setOperation("replace");
+ invalidateResults(
+ "Corpus operation changed; previous results were cleared.",
+ );
+ }}
+ />
+
+ Apply replacement
+
+ Create bounded local outputs without modifying sources.
+
+
+
+
+ {operation === "replace" ? (
+
+ Replacement template
+
+ ) : null}
+
+
+ Corpus jobs always scan all matches. An internal iteration flag may
+ be added for this run; saved user flags are unchanged.
+
+ {subjectMode === "line" ? (
+
+ Independent-line mode is bounded to{" "}
+ {DEFAULT_REGEX_LIMITS.maximumCorpusLines.toLocaleString()} logical
+ lines across the batch.
+
+ ) : null}
+
+ Documents run one at a time with the selected {timeoutMs} ms
+ per-document limit and a{" "}
+ {DEFAULT_REGEX_LIMITS.maximumCorpusWallTimeMs / 60_000} minute batch
+ wall limit.
+
+
+ {!patternAccepted ? (
+
+ Corpus processing is paused until the syntax provider accepts the
+ current pattern.
+
+ ) : operation === "replace" && !replacementAccepted ? (
+
+ Corpus apply is paused until the replacement template is valid.
+
+ ) : null}
+
+ void run()}
+ >
+ {running
+ ? "Processing…"
+ : operation === "replace"
+ ? "Apply to corpus"
+ : "Scan corpus"}
+
+ {running ? (
+ cancelRun()}
+ >
+ Cancel corpus run
+
+ ) : null}
+ {
+ setDocuments([]);
+ invalidateResults("Corpus documents and results were cleared.");
+ }}
+ >
+ Clear corpus
+
+
+ {progress ? (
+
+
+
+ {progress.completedDocuments.toLocaleString()} of{" "}
+ {progress.totalDocuments.toLocaleString()} documents ·{" "}
+ {progress.totalMatches.toLocaleString()} matches
+ {progress.currentDocumentName
+ ? ` · ${progress.currentDocumentName}`
+ : ""}
+
+
+ ) : null}
+
+ {status}
+
+
+
+
+
+ {batch ? (
+ <>
+ setSelectedDocumentId(result.documentId)}
+ />
+ {selectedResult ? (
+
+
+
+
Bounded local detail
+
{selectedResult.name}
+
+
+ {selectedResult.captureSummaries.length.toLocaleString()}{" "}
+ retained group summaries
+
+
+ {selectedResult.output !== undefined ? (
+
+
Replacement preview
+
{selectedResult.output.slice(0, 2_000) || "∅"}
+
+ Showing{" "}
+ {Math.min(
+ 2_000,
+ selectedResult.output.length,
+ ).toLocaleString()}{" "}
+ of {selectedResult.output.length.toLocaleString()} UTF-16
+ units
+ {selectedResult.status !== "complete"
+ ? " from incomplete output"
+ : ""}
+ .
+
+
+ ) : null}
+ {selectedResult.captureSummaries.length > 0 ? (
+
+
+
+
+ Group
+ Participated
+ Empty
+ Did not participate
+ Unavailable / clipped
+ Bounded samples
+
+
+
+ {selectedResult.captureSummaries.map((capture) => (
+
+
+ {capture.groupNumber === 0
+ ? "Full match"
+ : `${capture.groupNumber}${
+ capture.groupName
+ ? ` · ${capture.groupName}`
+ : ""
+ }`}
+
+ {capture.participated.toLocaleString()}
+ {capture.matchedEmpty.toLocaleString()}
+
+ {capture.didNotParticipate.toLocaleString()}
+
+
+ {(
+ capture.unavailable + capture.truncated
+ ).toLocaleString()}
+
+
+ {capture.samples.length > 0 ? (
+
+ {capture.samples.map((sample, index) => (
+
+ {sample || "∅"}
+
+ ))}
+
+ ) : (
+ "—"
+ )}
+ {capture.samplesClipped ? (
+ One or more samples were clipped.
+ ) : null}
+
+
+ ))}
+
+
+
+ ) : (
+
+ No match or capture samples were retained for this document.
+
+ )}
+
+ ) : null}
+
+
+
+ Summary format
+
+ setReportFormat(event.target.value as CorpusReportFormat)
+ }
+ >
+ JSON
+ CSV
+ NDJSON
+
+
+ {
+ try {
+ const serialized = serializeCorpusReport(
+ batch,
+ {
+ flavour,
+ ...(flavourVersion === undefined
+ ? {}
+ : { flavourVersion }),
+ pattern,
+ flags,
+ },
+ reportFormat,
+ );
+ const mediaType =
+ reportFormat === "csv"
+ ? "text/csv;charset=utf-8"
+ : reportFormat === "ndjson"
+ ? "application/x-ndjson"
+ : "application/json";
+ downloadBlob(
+ new Blob([serialized], { type: mediaType }),
+ `regex-corpus-report.${reportFormat}`,
+ );
+ setStatus(
+ "Downloaded a summary report without source or replacement output content.",
+ );
+ } catch (error) {
+ setStatus(
+ error instanceof Error ? error.message : String(error),
+ );
+ }
+ }}
+ >
+ Download summary {reportFormat.toUpperCase()}
+
+
+ Includes names, pattern and diagnostics, but no document
+ contents or applied output.
+
+
+ {batch.operation === "replace" ? (
+ <>
+
+
+ setAllowContentExport(event.target.checked)
+ }
+ />
+ I understand applied downloads contain corpus content
+
+
{
+ try {
+ const pending = createAppliedCorpusZip(batch);
+ zipCancellation.current = pending.cancel;
+ setExportingZip(true);
+ setStatus("Creating applied-output ZIP locally…");
+ void pending.promise
+ .then((archive) => {
+ const archiveBytes = Uint8Array.from(archive);
+ downloadBlob(
+ new Blob([archiveBytes.buffer], {
+ type: "application/zip",
+ }),
+ "regex-corpus-applied.zip",
+ );
+ setStatus(
+ "Downloaded exact applied outputs. Original files were not modified.",
+ );
+ })
+ .catch((error: unknown) =>
+ setStatus(
+ error instanceof Error
+ ? error.message
+ : String(error),
+ ),
+ )
+ .finally(() => {
+ zipCancellation.current = undefined;
+ setExportingZip(false);
+ });
+ } catch (error) {
+ setStatus(
+ error instanceof Error
+ ? error.message
+ : String(error),
+ );
+ }
+ }}
+ >
+ {exportingZip ? "Creating ZIP…" : "Download applied ZIP"}
+
+ >
+ ) : null}
+
+ {batch.status !== "complete" ? (
+
+ This batch is {batch.status}. Counts marked with ≥ are lower
+ bounds. Applied downloads are disabled for incomplete rows and
+ the all-document ZIP is disabled unless every output is exact.
+
+ ) : null}
+ >
+ ) : (
+
+ Run the corpus to create per-document summaries.
+
+ )}
+
+
+ );
+}
diff --git a/src/components/GenerationPanel.css b/src/components/GenerationPanel.css
new file mode 100644
index 0000000..4849e30
--- /dev/null
+++ b/src/components/GenerationPanel.css
@@ -0,0 +1,314 @@
+.generation-dialog {
+ width: min(90rem, calc(100% - 2rem));
+}
+
+.generation-panel {
+ display: grid;
+ gap: 1rem;
+ min-width: 0;
+ max-height: calc(100vh - 2rem);
+ overflow: auto;
+}
+
+.generation-disclaimer,
+.generation-unavailable,
+.generation-status,
+.generation-summary,
+.generation-coverage,
+.generation-cases,
+.generation-discarded {
+ border: 1px solid var(--toolbox-border);
+ border-radius: calc(var(--toolbox-radius) * 0.82);
+ background: color-mix(
+ in srgb,
+ var(--toolbox-surface) 94%,
+ var(--toolbox-accent) 6%
+ );
+}
+
+.generation-disclaimer {
+ margin: 0;
+ padding: 0.8rem 0.95rem;
+ color: var(--toolbox-muted);
+}
+
+.generation-unavailable {
+ display: grid;
+ gap: 0.35rem;
+ padding: 1rem;
+}
+
+.generation-unavailable span {
+ color: var(--toolbox-muted);
+}
+
+.generation-settings {
+ display: grid;
+ gap: 0.8rem;
+ padding: 1rem;
+ border: 1px solid var(--toolbox-border);
+ border-radius: calc(var(--toolbox-radius) * 0.82);
+ background: var(--toolbox-surface);
+}
+
+.generation-settings-primary,
+.generation-settings-grid {
+ display: grid;
+ grid-template-columns: repeat(3, minmax(0, 1fr));
+ gap: 0.7rem;
+}
+
+.generation-settings label {
+ display: grid;
+ gap: 0.3rem;
+ min-width: 0;
+ color: var(--toolbox-muted);
+ font-size: 0.82rem;
+ font-weight: 700;
+}
+
+.generation-settings input {
+ min-width: 0;
+ width: 100%;
+}
+
+.generation-settings details {
+ border-top: 1px solid var(--toolbox-border);
+ padding-top: 0.65rem;
+}
+
+.generation-settings details summary,
+.generation-discarded summary,
+.generation-coverage details summary {
+ cursor: pointer;
+ font-weight: 750;
+}
+
+.generation-settings details[open] summary {
+ margin-bottom: 0.65rem;
+}
+
+.generation-run-actions,
+.generation-case-actions {
+ display: flex;
+ flex-wrap: wrap;
+ align-items: center;
+ gap: 0.55rem;
+}
+
+.generation-run-actions small {
+ color: var(--toolbox-muted);
+}
+
+.generation-status {
+ display: grid;
+ gap: 0.4rem;
+ padding: 0.75rem 0.9rem;
+}
+
+.generation-status.is-error {
+ border-color: color-mix(
+ in srgb,
+ var(--toolbox-danger) 55%,
+ var(--toolbox-border)
+ );
+}
+
+.generation-status progress {
+ width: 100%;
+}
+
+.generation-status span {
+ color: var(--toolbox-muted);
+ font-size: 0.82rem;
+}
+
+.generation-results {
+ display: grid;
+ gap: 1rem;
+}
+
+.generation-summary,
+.generation-coverage,
+.generation-cases,
+.generation-discarded {
+ padding: 1rem;
+}
+
+.generation-summary dl {
+ display: grid;
+ grid-template-columns: repeat(4, minmax(0, 1fr));
+ gap: 0.55rem;
+ margin: 0;
+}
+
+.generation-summary dl > div {
+ min-width: 0;
+ padding: 0.65rem;
+ border-radius: calc(var(--toolbox-radius) * 0.68);
+ background: var(--toolbox-surface-soft);
+}
+
+.generation-summary dt {
+ color: var(--toolbox-muted);
+ font-size: 0.75rem;
+ font-weight: 750;
+ text-transform: uppercase;
+}
+
+.generation-summary dd {
+ overflow-wrap: anywhere;
+ margin: 0.2rem 0 0;
+}
+
+.generation-warnings {
+ margin: 0.75rem 0 0;
+ color: var(--toolbox-muted);
+}
+
+.generation-coverage > header,
+.generation-cases > header {
+ display: flex;
+ justify-content: space-between;
+ align-items: start;
+ gap: 1rem;
+ margin-bottom: 0.75rem;
+}
+
+.generation-coverage h3,
+.generation-cases h3 {
+ margin: 0;
+}
+
+.generation-coverage > ul,
+.unsupported-generation-list,
+.generated-case-list,
+.generation-discarded ul {
+ display: grid;
+ gap: 0.55rem;
+ margin: 0;
+ padding: 0;
+ list-style: none;
+}
+
+.generation-coverage > ul > li {
+ display: grid;
+ grid-template-columns: 7.5rem minmax(9rem, 0.45fr) minmax(0, 1fr);
+ align-items: start;
+ gap: 0.55rem;
+ padding: 0.6rem;
+ border-radius: calc(var(--toolbox-radius) * 0.68);
+ background: var(--toolbox-surface-soft);
+}
+
+.generation-coverage .status {
+ justify-self: start;
+}
+
+.generation-coverage details {
+ margin-top: 0.8rem;
+}
+
+.unsupported-generation-list {
+ margin-top: 0.65rem;
+}
+
+.unsupported-generation-list li,
+.generation-discarded li {
+ display: grid;
+ grid-template-columns: minmax(10rem, 0.35fr) minmax(8rem, 0.25fr) minmax(
+ 0,
+ 1fr
+ );
+ gap: 0.6rem;
+ align-items: start;
+ padding: 0.55rem;
+ border-radius: calc(var(--toolbox-radius) * 0.68);
+ background: var(--toolbox-surface-soft);
+}
+
+.unsupported-generation-list code,
+.generation-discarded code {
+ overflow-wrap: anywhere;
+ white-space: pre-wrap;
+}
+
+.generated-case-list li {
+ display: grid;
+ gap: 0.5rem;
+ min-width: 0;
+ padding: 0.7rem;
+ border: 1px solid var(--toolbox-border);
+ border-radius: calc(var(--toolbox-radius) * 0.68);
+ background: var(--toolbox-surface-soft);
+}
+
+.generated-case-list label {
+ display: flex;
+ align-items: start;
+ gap: 0.55rem;
+}
+
+.generated-case-list label span {
+ display: grid;
+ gap: 0.15rem;
+}
+
+.generated-case-list small {
+ color: var(--toolbox-muted);
+}
+
+.generated-case-list > li > code {
+ display: block;
+ overflow-wrap: anywhere;
+ white-space: pre-wrap;
+}
+
+.generation-feature-chips {
+ display: flex;
+ flex-wrap: wrap;
+ gap: 0.35rem;
+}
+
+.generation-feature-chips span {
+ padding: 0.2rem 0.45rem;
+ border: 1px solid var(--toolbox-border);
+ border-radius: 999px;
+ color: var(--toolbox-muted);
+ font-size: 0.72rem;
+}
+
+.generation-discarded > p {
+ color: var(--toolbox-muted);
+}
+
+.generation-discarded[open] summary {
+ margin-bottom: 0.5rem;
+}
+
+@media (max-width: 900px) {
+ .generation-settings-primary,
+ .generation-settings-grid,
+ .generation-summary dl {
+ grid-template-columns: repeat(2, minmax(0, 1fr));
+ }
+
+ .generation-coverage > ul > li,
+ .unsupported-generation-list li,
+ .generation-discarded li {
+ grid-template-columns: 1fr;
+ }
+}
+
+@media (max-width: 620px) {
+ .generation-settings-primary,
+ .generation-settings-grid,
+ .generation-summary dl {
+ grid-template-columns: 1fr;
+ }
+
+ .generation-coverage > header,
+ .generation-cases > header {
+ display: grid;
+ }
+}
diff --git a/src/components/GenerationPanel.test.tsx b/src/components/GenerationPanel.test.tsx
new file mode 100644
index 0000000..f53bcda
--- /dev/null
+++ b/src/components/GenerationPanel.test.tsx
@@ -0,0 +1,326 @@
+import { cleanup, render, screen, waitFor } from "@testing-library/react";
+import userEvent from "@testing-library/user-event";
+import { afterEach, describe, expect, it, vi } from "vitest";
+import type { CaseGenerationClient } from "../regex/generation/CaseGenerationOrchestrator";
+import type {
+ CaseGenerationResult,
+ VerifiedGeneratedCase,
+} from "../regex/generation/generation.types";
+import { WorkerRequestError } from "../regex/execution/WorkerSupervisor";
+import { EcmaScriptSyntaxProvider } from "../regex/syntax/providers/ecmascript/EcmaScriptSyntaxProvider";
+import { GenerationPanel } from "./GenerationPanel";
+
+const provider = new EcmaScriptSyntaxProvider();
+
+function generatedCase(
+ id: string,
+ expectation: "should-match" | "should-not-match",
+): VerifiedGeneratedCase {
+ const subject = expectation === "should-match" ? "cat" : "dog";
+ return {
+ id,
+ name: `Generated ${id}`,
+ subject,
+ subjectBytes: subject.length,
+ expectation,
+ features:
+ expectation === "should-match" ? ["shortest"] : ["likely-near-miss"],
+ notes: [],
+ elapsedMs: 0.5,
+ matched: expectation === "should-match",
+ matchCount: expectation === "should-match" ? 1 : 0,
+ provenance: {
+ kind: "generated",
+ generatorId: "regex-tools-ast-cases",
+ generatorVersion: "1",
+ seed: "case-seed-1",
+ candidateId: id,
+ intendedOutcome: expectation === "should-match" ? "match" : "no-match",
+ },
+ };
+}
+
+function fixtureResult(): CaseGenerationResult {
+ return {
+ status: "partial",
+ seed: "case-seed-1",
+ seedHash: "1234abcd",
+ flavour: "ecmascript",
+ cases: [
+ generatedCase("positive", "should-match"),
+ generatedCase("negative", "should-not-match"),
+ ],
+ discarded: [
+ {
+ candidateId: "discarded",
+ intendedOutcome: "no-match",
+ reason: "unexpected-match",
+ detail: "The actual engine matched it.",
+ subjectBytes: 3,
+ subjectPreview: "cat",
+ },
+ ],
+ coverage: [
+ {
+ feature: "shortest",
+ status: "covered",
+ detail: "Shortest path requested.",
+ relevantNodes: 1,
+ },
+ {
+ feature: "alternative-coverage",
+ status: "not-applicable",
+ detail: "No alternatives.",
+ relevantNodes: 0,
+ },
+ ],
+ unsupportedConstructs: [
+ {
+ id: "unsupported-lookahead",
+ kind: "lookahead",
+ range: { startUtf16: 0, endUtf16: 3 },
+ rawPreview: "(?=cat)",
+ reason: "Lookaround is verified but not synthesized.",
+ },
+ ],
+ engine: {
+ flavour: "ecmascript",
+ adapterVersion: "fixture",
+ engineName: "Native ECMAScript RegExp",
+ engineVersion: "Fixture Browser 1",
+ runtimeVersion: "Fixture Browser 1",
+ offsetUnit: "utf16",
+ capabilities: {
+ compilation: true,
+ matching: true,
+ replacement: true,
+ namedCaptures: true,
+ captureHistory: false,
+ actualTrace: false,
+ benchmark: true,
+ },
+ },
+ candidateAttempts: 5,
+ engineExecutions: 3,
+ generatedCandidateBytes: 9,
+ verifiedSubjectBytes: 6,
+ wallTimeMs: 4.5,
+ warnings: [],
+ };
+}
+
+function client(
+ overrides: Partial = {},
+): CaseGenerationClient {
+ return {
+ run: vi.fn().mockImplementation(async (_input, onProgress) => {
+ onProgress?.({
+ phase: "verifying",
+ completed: 1,
+ total: 3,
+ retained: 1,
+ message: "Verifying fixture candidate.",
+ });
+ return fixtureResult();
+ }),
+ cancel: vi.fn(),
+ dispose: vi.fn(),
+ ...overrides,
+ };
+}
+
+async function renderPanel(
+ overrides: Partial[0]> = {},
+) {
+ const pattern = overrides.pattern ?? "cat";
+ const syntax =
+ overrides.syntax ??
+ (await provider.parsePattern({
+ flavour: "ecmascript",
+ flavourVersion: "2025",
+ pattern,
+ flags: [],
+ options: {},
+ }));
+ const currentClient = client();
+ const rendered = render(
+ currentClient}
+ {...overrides}
+ />,
+ );
+ return { ...rendered, currentClient, syntax };
+}
+
+afterEach(() => {
+ cleanup();
+});
+
+describe("GenerationPanel", () => {
+ it("runs deterministic generation, renders honest coverage and adds verified cases to tests", async () => {
+ const onAddToTests = vi.fn();
+ const onSelectPatternRange = vi.fn();
+ const { currentClient } = await renderPanel({
+ onAddToTests,
+ onSelectPatternRange,
+ });
+
+ await userEvent.click(
+ screen.getByRole("button", { name: "Generate & verify" }),
+ );
+
+ expect(
+ await screen.findByText(/Native ECMAScript RegExp · Fixture Browser 1/u),
+ ).toBeVisible();
+ expect(screen.getByText("case-seed-1")).toBeVisible();
+ expect(
+ screen.getByRole("region", { name: "Generation summary" }),
+ ).toHaveTextContent("1234abcd");
+ expect(screen.getByText("Shortest path requested.")).toBeVisible();
+ expect(screen.getAllByText("actual engine verified")).toHaveLength(2);
+ await userEvent.click(
+ screen.getByText(/1 unsupported or partially synthesized construct/u),
+ );
+ expect(currentClient.run).toHaveBeenCalledWith(
+ expect.objectContaining({
+ seed: "case-seed-1",
+ flavour: "ecmascript",
+ pattern: "cat",
+ }),
+ expect.any(Function),
+ );
+
+ await userEvent.click(
+ screen.getByRole("button", {
+ name: /lookahead · 0–3/u,
+ }),
+ );
+ expect(onSelectPatternRange).toHaveBeenCalledWith({
+ startUtf16: 0,
+ endUtf16: 3,
+ });
+
+ await userEvent.click(
+ screen.getByText(/Advanced: 1 rejected candidate diagnostic/u),
+ );
+ expect(screen.getByText(/unexpected match/u)).toBeVisible();
+
+ await userEvent.click(
+ screen.getByRole("button", { name: "Add selected to tests" }),
+ );
+ expect(onAddToTests).toHaveBeenCalledWith(
+ expect.arrayContaining([
+ expect.objectContaining({
+ expectation: "should-match",
+ provenance: expect.objectContaining({ seed: "case-seed-1" }),
+ }),
+ expect.objectContaining({ expectation: "should-not-match" }),
+ ]),
+ );
+ expect(
+ screen.getByText(/Added 2 generated cases to the unit-test suite/u),
+ ).toBeVisible();
+ });
+
+ it("shows strict setting errors without starting worker execution", async () => {
+ const { currentClient } = await renderPanel();
+ const maximumCases = screen.getByRole("spinbutton", {
+ name: "Retain at most",
+ });
+ await userEvent.clear(maximumCases);
+ await userEvent.type(maximumCases, "500");
+ await userEvent.click(screen.getByText("Advanced generation bounds"));
+ const attempts = screen.getByRole("spinbutton", {
+ name: "Candidate attempts",
+ });
+ await userEvent.clear(attempts);
+ await userEvent.type(attempts, "100");
+ await userEvent.click(
+ screen.getByRole("button", { name: "Generate & verify" }),
+ );
+
+ expect(
+ await screen.findByText(/Maximum candidate attempts must be/u),
+ ).toBeVisible();
+ expect(currentClient.run).not.toHaveBeenCalled();
+ });
+
+ it("does not expose the ECMAScript generator for PCRE2", async () => {
+ const { currentClient } = await renderPanel({ flavour: "pcre2" });
+
+ expect(
+ screen.getByText(
+ /current PCRE2 syntax provider is intentionally partial/u,
+ ),
+ ).toBeVisible();
+ expect(
+ screen.queryByRole("button", { name: "Generate & verify" }),
+ ).not.toBeInTheDocument();
+ expect(currentClient.run).not.toHaveBeenCalled();
+ });
+
+ it("clears verified cases immediately when the pattern changes", async () => {
+ const rendered = await renderPanel();
+ await userEvent.click(
+ screen.getByRole("button", { name: "Generate & verify" }),
+ );
+ expect(await screen.findByText("Generated positive")).toBeVisible();
+
+ rendered.rerender(
+ rendered.currentClient}
+ />,
+ );
+
+ expect(screen.queryByText("Generated positive")).not.toBeInTheDocument();
+ expect(
+ await screen.findByText(/previous generated cases were cleared/u),
+ ).toBeVisible();
+ });
+
+ it("terminates synthesis and engine workers on cancellation", async () => {
+ let rejectRun: ((error: Error) => void) | undefined;
+ const pending = new Promise((_resolve, reject) => {
+ rejectRun = reject;
+ });
+ const currentClient = client({
+ run: vi.fn().mockReturnValue(pending),
+ cancel: vi.fn().mockImplementation(() => {
+ rejectRun?.(
+ new WorkerRequestError("cancelled", "fixture cancellation"),
+ );
+ }),
+ });
+ await renderPanel({ createOrchestrator: () => currentClient });
+ await userEvent.click(
+ screen.getByRole("button", { name: "Generate & verify" }),
+ );
+ await userEvent.click(
+ await screen.findByRole("button", { name: "Cancel generation" }),
+ );
+
+ await waitFor(() => expect(currentClient.cancel).toHaveBeenCalled());
+ expect(
+ await screen.findByText(
+ /Generation cancelled; its workers were terminated/u,
+ ),
+ ).toBeVisible();
+ expect(currentClient.dispose).toHaveBeenCalled();
+ });
+});
diff --git a/src/components/GenerationPanel.tsx b/src/components/GenerationPanel.tsx
new file mode 100644
index 0000000..ed0e0e3
--- /dev/null
+++ b/src/components/GenerationPanel.tsx
@@ -0,0 +1,829 @@
+import { useCallback, useEffect, useMemo, useRef, useState } from "react";
+import {
+ CaseGenerationOrchestrator,
+ type CaseGenerationClient,
+} from "../regex/generation/CaseGenerationOrchestrator";
+import {
+ DEFAULT_GENERATION_SETTINGS,
+ GENERATION_LIMITS,
+} from "../regex/generation/generation-limits";
+import type {
+ CaseGenerationProgress,
+ CaseGenerationResult,
+ CandidateGenerationSettings,
+ VerifiedGeneratedCase,
+} from "../regex/generation/generation.types";
+import {
+ GENERATED_CASE_GENERATOR_ID,
+ GENERATED_CASE_GENERATOR_VERSION,
+} from "../regex/generation/generation.types";
+import { DEFAULT_REGEX_LIMITS } from "../regex/execution/request-limits";
+import { WorkerRequestError } from "../regex/execution/WorkerSupervisor";
+import type {
+ RegexEngineOptions,
+ RegexFlavourId,
+} from "../regex/model/flavour";
+import type { RegexSyntaxResult, SourceRange } from "../regex/model/syntax";
+import "./GenerationPanel.css";
+
+const MAXIMUM_SUBJECT_PREVIEW_UTF16 = 240;
+
+function defaultOrchestrator(): CaseGenerationClient {
+ return new CaseGenerationOrchestrator();
+}
+
+function integer(
+ source: string,
+ label: string,
+ minimum: number,
+ maximum: number,
+): number {
+ const value = Number(source);
+ if (
+ source.trim() === "" ||
+ !Number.isSafeInteger(value) ||
+ value < minimum ||
+ value > maximum
+ ) {
+ throw new RangeError(
+ `${label} must be a whole number from ${minimum.toLocaleString()} to ${maximum.toLocaleString()}.`,
+ );
+ }
+ return value;
+}
+
+function subjectPreview(subject: string): string {
+ if (subject.length <= MAXIMUM_SUBJECT_PREVIEW_UTF16) {
+ return JSON.stringify(subject);
+ }
+ return `${JSON.stringify(subject.slice(0, MAXIMUM_SUBJECT_PREVIEW_UTF16))}… (${subject.length.toLocaleString()} UTF-16 units)`;
+}
+
+function downloadCases(
+ result: CaseGenerationResult,
+ configuration: {
+ readonly flavour: RegexFlavourId;
+ readonly flavourVersion?: string;
+ readonly pattern: string;
+ readonly flags: readonly string[];
+ readonly options: RegexEngineOptions;
+ readonly scanAll: boolean;
+ },
+): void {
+ const payload = {
+ schemaVersion: 1,
+ kind: "regex-tools-generated-cases",
+ generator: {
+ id: GENERATED_CASE_GENERATOR_ID,
+ version: GENERATED_CASE_GENERATOR_VERSION,
+ seed: result.seed,
+ seedHash: result.seedHash,
+ },
+ configuration,
+ engine: result.engine,
+ cases: result.cases.map((candidate) => ({
+ id: candidate.id,
+ name: candidate.name,
+ subject: candidate.subject,
+ expectation: { kind: candidate.expectation },
+ features: candidate.features,
+ provenance: candidate.provenance,
+ })),
+ };
+ const url = URL.createObjectURL(
+ new Blob([`${JSON.stringify(payload, null, 2)}\n`], {
+ type: "application/json",
+ }),
+ );
+ const anchor = document.createElement("a");
+ anchor.href = url;
+ anchor.download = `regex-generated-${result.seedHash}.json`;
+ anchor.click();
+ globalThis.setTimeout(() => URL.revokeObjectURL(url), 1_000);
+}
+
+export interface GenerationPanelProps {
+ readonly active: boolean;
+ readonly flavour: RegexFlavourId;
+ readonly flavourVersion?: string;
+ readonly pattern: string;
+ readonly flags: readonly string[];
+ readonly options: RegexEngineOptions;
+ readonly scanAll: boolean;
+ readonly syntax?: RegexSyntaxResult;
+ readonly onClose?: () => void;
+ readonly onAddToTests?: (cases: readonly VerifiedGeneratedCase[]) => void;
+ readonly onSelectPatternRange?: (range: SourceRange) => void;
+ readonly createOrchestrator?: () => CaseGenerationClient;
+}
+
+export function GenerationPanel({
+ active,
+ flavour,
+ flavourVersion,
+ pattern,
+ flags,
+ options,
+ scanAll,
+ syntax,
+ onClose,
+ onAddToTests,
+ onSelectPatternRange,
+ createOrchestrator = defaultOrchestrator,
+}: GenerationPanelProps) {
+ const [seed, setSeed] = useState("case-seed-1");
+ const [maximumCases, setMaximumCases] = useState(
+ String(DEFAULT_GENERATION_SETTINGS.maximumCases),
+ );
+ const [maximumCandidateAttempts, setMaximumCandidateAttempts] = useState(
+ String(DEFAULT_GENERATION_SETTINGS.maximumCandidateAttempts),
+ );
+ const [randomVariantCount, setRandomVariantCount] = useState(
+ String(DEFAULT_GENERATION_SETTINGS.randomVariantCount),
+ );
+ const [maximumQuantifierRepetitions, setMaximumQuantifierRepetitions] =
+ useState(String(DEFAULT_GENERATION_SETTINGS.maximumQuantifierRepetitions));
+ const [maximumSubjectBytes, setMaximumSubjectBytes] = useState(
+ String(DEFAULT_GENERATION_SETTINGS.maximumSubjectBytes),
+ );
+ const [maximumTotalSubjectBytes, setMaximumTotalSubjectBytes] = useState(
+ String(DEFAULT_GENERATION_SETTINGS.maximumTotalSubjectBytes),
+ );
+ const [perCaseTimeoutMs, setPerCaseTimeoutMs] = useState(
+ String(DEFAULT_GENERATION_SETTINGS.perCaseTimeoutMs),
+ );
+ const [maximumWallTimeMs, setMaximumWallTimeMs] = useState(
+ String(DEFAULT_GENERATION_SETTINGS.maximumWallTimeMs),
+ );
+ const [running, setRunning] = useState(false);
+ const [progress, setProgress] = useState();
+ const [message, setMessage] = useState(
+ "No cases have been generated for this configuration.",
+ );
+ const [error, setError] = useState();
+ const [resultRecord, setResultRecord] = useState<{
+ readonly configurationKey: symbol;
+ readonly result: CaseGenerationResult;
+ }>();
+ const [selectedIds, setSelectedIds] = useState>(
+ new Set(),
+ );
+ const orchestrator = useRef(undefined);
+ const runRevision = useRef(0);
+
+ const syntaxCurrent =
+ syntax?.accepted === true &&
+ syntax.root.raw === pattern &&
+ syntax.root.support.flavour === flavour;
+ const generationAvailable = flavour === "ecmascript" && syntaxCurrent;
+ const configurationKey = useMemo(
+ () =>
+ Symbol(
+ `generation-${flavour}-${flavourVersion ?? ""}-${flags.join("")}-${JSON.stringify(options)}-${pattern.length}-${String(scanAll)}`,
+ ),
+ [flags, flavour, flavourVersion, options, pattern, scanAll],
+ );
+ const previousConfiguration = useRef(configurationKey);
+ const result =
+ resultRecord?.configurationKey === configurationKey
+ ? resultRecord.result
+ : undefined;
+
+ const disposeActive = useCallback(() => {
+ orchestrator.current?.cancel();
+ orchestrator.current?.dispose();
+ orchestrator.current = undefined;
+ }, []);
+
+ useEffect(() => {
+ if (active) return;
+ runRevision.current += 1;
+ disposeActive();
+ queueMicrotask(() => setRunning(false));
+ }, [active, disposeActive]);
+
+ useEffect(() => {
+ if (previousConfiguration.current === configurationKey) return;
+ previousConfiguration.current = configurationKey;
+ const revision = ++runRevision.current;
+ disposeActive();
+ queueMicrotask(() => {
+ if (runRevision.current !== revision) return;
+ setRunning(false);
+ setProgress(undefined);
+ setResultRecord(undefined);
+ setSelectedIds(new Set());
+ setError(undefined);
+ setMessage(
+ "Pattern or engine configuration changed; previous generated cases were cleared.",
+ );
+ });
+ }, [configurationKey, disposeActive]);
+
+ useEffect(
+ () => () => {
+ runRevision.current += 1;
+ disposeActive();
+ },
+ [disposeActive],
+ );
+
+ const settings = (): CandidateGenerationSettings => {
+ const retained = integer(
+ maximumCases,
+ "Maximum retained cases",
+ 1,
+ DEFAULT_REGEX_LIMITS.maximumGeneratedCases,
+ );
+ return {
+ maximumCases: retained,
+ maximumCandidateAttempts: integer(
+ maximumCandidateAttempts,
+ "Maximum candidate attempts",
+ retained,
+ GENERATION_LIMITS.maximumCandidateAttempts,
+ ),
+ maximumTotalSubjectBytes: integer(
+ maximumTotalSubjectBytes,
+ "Maximum aggregate subject bytes",
+ 1,
+ GENERATION_LIMITS.maximumTotalSubjectBytes,
+ ),
+ maximumSubjectBytes: integer(
+ maximumSubjectBytes,
+ "Maximum bytes per subject",
+ 1,
+ GENERATION_LIMITS.maximumSubjectBytes,
+ ),
+ maximumAstNodes: DEFAULT_GENERATION_SETTINGS.maximumAstNodes,
+ maximumAstDepth: DEFAULT_GENERATION_SETTINGS.maximumAstDepth,
+ maximumQuantifierRepetitions: integer(
+ maximumQuantifierRepetitions,
+ "Maximum synthesized repetitions",
+ 1,
+ GENERATION_LIMITS.maximumQuantifierRepetitions,
+ ),
+ randomVariantCount: integer(
+ randomVariantCount,
+ "Random variants",
+ 0,
+ DEFAULT_REGEX_LIMITS.maximumGeneratedCases,
+ ),
+ perCaseTimeoutMs: integer(
+ perCaseTimeoutMs,
+ "Per-case timeout",
+ GENERATION_LIMITS.minimumPerCaseTimeoutMs,
+ GENERATION_LIMITS.maximumPerCaseTimeoutMs,
+ ),
+ maximumWallTimeMs: integer(
+ maximumWallTimeMs,
+ "Aggregate wall time",
+ 1,
+ GENERATION_LIMITS.maximumWallTimeMs,
+ ),
+ };
+ };
+
+ const generate = async () => {
+ if (!generationAvailable || !syntax) return;
+ const revision = ++runRevision.current;
+ disposeActive();
+ const client = createOrchestrator();
+ orchestrator.current = client;
+ setRunning(true);
+ setProgress(undefined);
+ setError(undefined);
+ setResultRecord(undefined);
+ setSelectedIds(new Set());
+ setMessage(
+ "Starting deterministic synthesis and actual-engine verification…",
+ );
+ try {
+ const next = await client.run(
+ {
+ flavour,
+ ...(flavourVersion === undefined ? {} : { flavourVersion }),
+ pattern,
+ flags,
+ options,
+ scanAll,
+ root: syntax.root,
+ captureMetadata: syntax.captures,
+ syntaxAccepted: syntax.accepted,
+ seed,
+ settings: settings(),
+ },
+ (value) => {
+ if (runRevision.current !== revision) return;
+ setProgress(value);
+ setMessage(value.message);
+ },
+ );
+ if (runRevision.current !== revision) return;
+ setResultRecord({ configurationKey, result: next });
+ setSelectedIds(new Set(next.cases.map((candidate) => candidate.id)));
+ setMessage(
+ next.stoppedReason ??
+ `Generation ${next.status}: ${next.cases.length.toLocaleString()} actual-engine-verified cases retained.`,
+ );
+ } catch (cause) {
+ if (runRevision.current !== revision) return;
+ if (cause instanceof WorkerRequestError && cause.kind === "cancelled") {
+ setMessage("Generation cancelled; its workers were terminated.");
+ } else {
+ const detail =
+ cause instanceof Error
+ ? cause.message
+ : "Generated cases could not be produced.";
+ setError(detail);
+ setMessage("Generation settings or worker execution require review.");
+ }
+ } finally {
+ client.dispose();
+ if (orchestrator.current === client) orchestrator.current = undefined;
+ if (runRevision.current === revision) {
+ setRunning(false);
+ setProgress(undefined);
+ }
+ }
+ };
+
+ const cancel = () => {
+ orchestrator.current?.cancel();
+ setMessage(
+ "Cancelling synthesis and verification; active workers are being terminated…",
+ );
+ };
+
+ const selectedCases =
+ result?.cases.filter((candidate) => selectedIds.has(candidate.id)) ?? [];
+ const progressMaximum = Math.max(1, progress?.total ?? 1);
+ const progressValue = Math.min(progressMaximum, progress?.completed ?? 0);
+
+ return (
+
+
+
+
+ Generation samples supported structure; it is not proof of complete
+ coverage. A positive or negative label appears only after the selected
+ actual engine confirms that exact subject. Rejected candidates remain
+ diagnostics, never tests.
+
+
+ {!generationAvailable ? (
+
+
+ Generated cases are unavailable for this configuration.
+
+
+ {flavour === "pcre2"
+ ? "The current PCRE2 syntax provider is intentionally partial and does not expose a complete normalized AST. It is not reinterpreted as ECMAScript."
+ : syntaxCurrent
+ ? "This flavour does not have an implemented generator."
+ : "Wait for an accepted, current syntax result before generating."}
+
+
+ ) : (
+
+ )}
+
+
+
{message}
+ {progress ? (
+ <>
+
+
+ {progress.phase} · {progress.completed.toLocaleString()} /{" "}
+ {progress.total.toLocaleString()} ·{" "}
+ {progress.retained.toLocaleString()} retained
+
+ >
+ ) : null}
+ {error ?
{error} : null}
+
+
+ {result ? (
+
+
+
+
+
Status
+ {result.status.replaceAll("-", " ")}
+
+
+
Seed
+
+ {result.seed} · {result.seedHash}
+
+
+
+
Actual engine
+
+ {result.engine
+ ? `${result.engine.engineName} · ${result.engine.engineVersion}`
+ : "No candidate reached engine verification"}
+
+
+
+
Verified cases
+ {result.cases.length.toLocaleString()}
+
+
+
Engine executions
+ {result.engineExecutions.toLocaleString()}
+
+
+
Candidate bytes
+ {result.generatedCandidateBytes.toLocaleString()}
+
+
+
Wall time
+ {result.wallTimeMs.toFixed(1)} ms
+
+
+ {result.warnings.length > 0 ? (
+
+ {result.warnings.map((warning) => (
+ {warning}
+ ))}
+
+ ) : null}
+
+
+
+
+
+ {result.coverage.map((entry) => (
+
+
+ {entry.status}
+
+ {entry.feature.replaceAll("-", " ")}
+ {entry.detail}
+
+ ))}
+
+ {result.unsupportedConstructs.length > 0 ? (
+
+
+ {result.unsupportedConstructs.length.toLocaleString()}{" "}
+ unsupported or partially synthesized construct(s)
+
+
+ {result.unsupportedConstructs.map((construct) => (
+
+ onSelectPatternRange?.(construct.range)}
+ >
+ {construct.kind} · {construct.range.startUtf16}–
+ {construct.range.endUtf16}
+
+ {construct.rawPreview || "empty"}
+ {construct.reason}
+
+ ))}
+
+
+ ) : null}
+
+
+
+
+
+
Exact verified subjects
+
+ Retained cases{" "}
+ ({result.cases.length.toLocaleString()})
+
+
+
+
+ setSelectedIds(
+ selectedIds.size === result.cases.length
+ ? new Set()
+ : new Set(
+ result.cases.map((candidate) => candidate.id),
+ ),
+ )
+ }
+ >
+ {selectedIds.size === result.cases.length
+ ? "Select none"
+ : "Select all"}
+
+
+ downloadCases(result, {
+ flavour,
+ ...(flavourVersion === undefined
+ ? {}
+ : { flavourVersion }),
+ pattern,
+ flags,
+ options,
+ scanAll,
+ })
+ }
+ >
+ Export verified JSON
+
+ {onAddToTests ? (
+ {
+ try {
+ onAddToTests(selectedCases);
+ setError(undefined);
+ setMessage(
+ `Added ${selectedCases.length.toLocaleString()} generated cases to the unit-test suite with seed provenance.`,
+ );
+ } catch (cause) {
+ setError(
+ cause instanceof Error
+ ? cause.message
+ : "Generated cases could not be added.",
+ );
+ }
+ }}
+ >
+ Add selected to tests
+
+ ) : null}
+
+
+ {result.cases.length === 0 ? (
+
+ No candidate satisfied its intended property within the
+ configured bounds.
+
+ ) : (
+
+ )}
+
+
+ {result.discarded.length > 0 ? (
+
+
+ Advanced: {result.discarded.length.toLocaleString()} rejected
+ candidate diagnostic(s)
+
+
+ These candidates are not tests. Subjects are clipped here; the
+ diagnostic records only why each intended property failed.
+
+
+ {result.discarded.map((candidate) => (
+
+ {candidate.reason.replaceAll("-", " ")}
+ {subjectPreview(candidate.subjectPreview)}
+ {candidate.detail}
+
+ ))}
+
+
+ ) : null}
+
+ ) : null}
+
+ );
+}
diff --git a/src/components/HelpDialog.tsx b/src/components/HelpDialog.tsx
index 0ad9934..57affca 100644
--- a/src/components/HelpDialog.tsx
+++ b/src/components/HelpDialog.tsx
@@ -1,5 +1,9 @@
import { useEffect, useRef } from "react";
import { DEFAULT_REGEX_LIMITS } from "../regex/execution/request-limits";
+import {
+ DEFAULT_GENERATION_SETTINGS,
+ GENERATION_LIMITS,
+} from "../regex/generation/generation-limits";
import { manifest } from "../toolbox/manifest";
const releaseSourceUrl = `https://git.add-ideas.de/zemion/regex-tools/src/tag/v${manifest.version}`;
@@ -43,10 +47,11 @@ export function HelpDialog({
What runs where
- Syntax parsing and actual ECMAScript execution run in separate,
- killable browser workers. Patterns, subjects and projects are never
- uploaded. There is no backend, account, telemetry, CDN dependency,
- or executable replacement function.
+ Syntax parsing and actual ECMAScript or PCRE2 execution run in
+ separate, killable browser workers. PCRE2 is the bundled,
+ self-hosted 10.47 WebAssembly engine. Patterns, subjects and
+ projects are never uploaded. There is no backend, account,
+ telemetry, CDN dependency, or executable replacement function.
@@ -56,8 +61,47 @@ export function HelpDialog({
syntax provider. The extraction tree contains actual browser-engine
matches and captures. Replacement mode adds a typed token tree and
bounded per-match contribution/range mapping over those actual
- matches. None is an internal V8, SpiderMonkey or JavaScriptCore
- execution trace.
+ matches. These views are not internal engine traces. The separate
+ PCRE trace viewer shows actual bounded automatic callouts reported
+ by PCRE2; its adjacent-position movement labels are derived and it
+ is not a complete record of every internal action. Browser
+ ECMAScript exposes no corresponding V8, SpiderMonkey or
+ JavaScriptCore trace API.
+
+
+
+ Advisory analysis
+
+ Static risk findings are ECMAScript-2025-specific heuristics, and
+ generated-input growth observations describe only the selected
+ bounded samples. Neither proves safety or general complexity. Growth
+ and cold/warm benchmark timing run in a separate killable worker and
+ never include the PCRE2 trace path.
+
+
+
+ Generated cases
+
+ ECMAScript generator v1 deterministically samples supported
+ normalized-AST paths from an explicit seed. Every intended positive
+ and negative subject runs through the selected actual engine before
+ it can become a test; candidates with the wrong outcome are
+ discarded. The coverage report names unsupported constructs.
+ Sampling is not proof of complete language coverage. PCRE2 remains
+ unavailable until its structural provider exposes the required AST;
+ its syntax is never reinterpreted as ECMAScript.
+
+
+
+ Pattern formatting
+
+ The ECMAScript formatter makes only grammar-backed literal/control
+ escape changes; it does not insert layout whitespace or reinterpret
+ PCRE2 syntax. Applying a preview requires independent source and
+ candidate reparsing, actual-engine match/replacement comparison,
+ every applicable exact unit test, and an explicit confirmation.
+ Passing bounded checks is evidence for those snapshots, not proof
+ for every possible subject.
@@ -105,6 +149,78 @@ export function HelpDialog({
Unit-test suite wall time
60 seconds
+
+
Corpus input
+
+ {DEFAULT_REGEX_LIMITS.maximumCorpusDocuments.toLocaleString()}{" "}
+ documents · {DEFAULT_REGEX_LIMITS.corpusHardBytes / 1024 / 1024}{" "}
+ MiB
+
+
+
+
Corpus wall time
+
+ {DEFAULT_REGEX_LIMITS.maximumCorpusWallTimeMs / 60_000} minutes
+
+
+
+
Analysis wall time
+
+ {DEFAULT_REGEX_LIMITS.maximumBenchmarkWallTimeMs / 1_000}{" "}
+ seconds
+
+
+
+
Benchmark iteration cap
+
+ {DEFAULT_REGEX_LIMITS.maximumBenchmarkIterations.toLocaleString()}
+
+
+
+
Growth probe steps
+ 24
+
+
+
Generated cases
+
+ {DEFAULT_REGEX_LIMITS.maximumGeneratedCases.toLocaleString()}{" "}
+ retained ·{" "}
+ {GENERATION_LIMITS.maximumCandidateAttempts.toLocaleString()}{" "}
+ attempts
+
+
+
+
Generation defaults
+
+ {DEFAULT_GENERATION_SETTINGS.perCaseTimeoutMs} ms per case ·{" "}
+ {DEFAULT_GENERATION_SETTINGS.maximumWallTimeMs / 1_000} seconds
+ aggregate
+
+
+
+
Generated subject text
+
+ {GENERATION_LIMITS.maximumTotalSubjectBytes / 1024 / 1024} MiB
+ aggregate hard cap
+
+
+
+
Formatter validation
+ 1,000 tests · 60 seconds aggregate
+
+
+
Independent corpus lines
+
+ {DEFAULT_REGEX_LIMITS.maximumCorpusLines.toLocaleString()}
+
+
+
+
Corpus applied output
+
+ {DEFAULT_REGEX_LIMITS.maximumCorpusOutputBytes / 1024 / 1024}{" "}
+ MiB
+
+
A timeout terminates the worker. It means “timed out”, never “no
@@ -116,7 +232,15 @@ export function HelpDialog({
Ad-hoc test text is excluded from local saves and exports by
default. Test definitions are deliberate project data. Import is
- validated and paused until you review and run it.
+ validated and paused until you review and run it. Corpus documents,
+ contents, outputs and results are always ephemeral and excluded from
+ projects and IndexedDB. Analysis settings and results are likewise
+ ephemeral. Generated-case settings and review results are ephemeral;
+ only cases explicitly added to the test suite persist, together with
+ their bounded seed and generator provenance. Test subjects remain
+ excluded unless separately opted in. Formatter previews and
+ validation observations are ephemeral; only an explicitly applied
+ pattern enters workbench/project state.
diff --git a/src/components/MinimizePanel.css b/src/components/MinimizePanel.css
new file mode 100644
index 0000000..96bf68a
--- /dev/null
+++ b/src/components/MinimizePanel.css
@@ -0,0 +1,241 @@
+.minimize-dialog {
+ width: min(90rem, calc(100% - 2rem));
+}
+
+.minimize-panel {
+ max-height: calc(100vh - 2rem);
+ overflow: auto;
+}
+
+.minimizer-body {
+ display: grid;
+ gap: 0.8rem;
+ padding: 0.8rem;
+}
+
+.minimizer-advisory {
+ margin: 0;
+ border-left: 3px solid var(--toolbox-accent);
+ padding: 0.7rem 0.8rem;
+ background: var(--toolbox-accent-soft);
+ color: var(--toolbox-muted);
+ font-size: 0.78rem;
+ line-height: 1.5;
+}
+
+.minimizer-config,
+.minimizer-result {
+ display: grid;
+ gap: 0.75rem;
+ overflow: hidden;
+ border: 1px solid var(--toolbox-border);
+ border-radius: calc(var(--toolbox-radius) * 0.8);
+ background: var(--toolbox-surface);
+}
+
+.minimizer-config > header,
+.minimizer-result > header {
+ display: flex;
+ align-items: center;
+ justify-content: space-between;
+ gap: 0.8rem;
+ border-bottom: 1px solid var(--toolbox-border);
+ padding: 0.7rem 0.8rem;
+ background: var(--toolbox-surface-soft);
+}
+
+.minimizer-config h3,
+.minimizer-result h3,
+.minimizer-result header p {
+ margin: 0;
+}
+
+.minimizer-config h3,
+.minimizer-result h3 {
+ font-size: 0.95rem;
+}
+
+.minimizer-config > header span {
+ color: var(--toolbox-muted);
+ font-size: 0.72rem;
+}
+
+.minimizer-control-grid,
+.minimizer-variant-grid {
+ display: grid;
+ grid-template-columns: repeat(auto-fit, minmax(11rem, 1fr));
+ align-items: end;
+ gap: 0.65rem;
+ padding: 0 0.8rem;
+}
+
+.minimizer-variant-grid {
+ grid-template-columns: repeat(2, minmax(0, 1fr));
+}
+
+.minimizer-control-grid label:not(.check-control),
+.minimizer-variant-grid label,
+.minimizer-subject,
+.minimizer-output {
+ display: grid;
+ gap: 0.3rem;
+ color: var(--toolbox-muted);
+ font-size: 0.72rem;
+ font-weight: 700;
+}
+
+.minimizer-wide-control {
+ grid-column: 1 / -1;
+}
+
+.minimizer-check {
+ min-height: 2.25rem;
+ align-items: center;
+}
+
+.minimizer-subject {
+ padding: 0 0.8rem 0.8rem;
+}
+
+.minimizer-subject textarea {
+ min-height: 8rem;
+}
+
+.minimizer-variant-grid textarea {
+ min-height: 4.5rem;
+}
+
+.minimizer-actions {
+ display: flex;
+ flex-wrap: wrap;
+ align-items: center;
+ gap: 0.55rem;
+}
+
+.minimizer-actions p,
+.minimizer-status {
+ margin: 0;
+ color: var(--toolbox-muted);
+ font-size: 0.76rem;
+}
+
+.minimizer-progress {
+ display: grid;
+ grid-template-columns: minmax(8rem, 1fr) auto;
+ align-items: center;
+ gap: 0.65rem;
+ color: var(--toolbox-muted);
+ font-size: 0.7rem;
+}
+
+.minimizer-progress progress {
+ width: 100%;
+ accent-color: var(--toolbox-accent);
+}
+
+.minimizer-result > dl {
+ display: grid;
+ grid-template-columns: repeat(auto-fit, minmax(9rem, 1fr));
+ margin: 0;
+ padding: 0 0.8rem;
+}
+
+.minimizer-result > dl > div {
+ padding: 0.55rem;
+ border-bottom: 1px solid var(--toolbox-border);
+}
+
+.minimizer-result dt {
+ color: var(--toolbox-muted);
+ font-size: 0.66rem;
+ font-weight: 750;
+ text-transform: uppercase;
+}
+
+.minimizer-result dd {
+ margin: 0.15rem 0 0;
+ font-size: 0.78rem;
+}
+
+.minimizer-result-badge {
+ border: 1px solid currentColor;
+ border-radius: 99rem;
+ padding: 0.2rem 0.45rem;
+ color: var(--toolbox-accent);
+ font-size: 0.65rem;
+ font-weight: 800;
+ text-transform: uppercase;
+}
+
+.minimizer-result-badge.is-partial,
+.minimizer-result-badge.is-cancelled {
+ color: var(--regex-warning);
+}
+
+.minimizer-result-badge.is-unsupported {
+ color: var(--toolbox-danger);
+}
+
+.minimizer-observation-grid {
+ display: grid;
+ grid-template-columns: repeat(2, minmax(0, 1fr));
+ gap: 0.65rem;
+ padding: 0 0.8rem;
+}
+
+.minimizer-observation {
+ display: grid;
+ gap: 0.3rem;
+ min-width: 0;
+ border: 1px solid var(--toolbox-border);
+ border-left: 4px solid var(--toolbox-accent);
+ border-radius: calc(var(--toolbox-radius) * 0.7);
+ padding: 0.65rem;
+ background: var(--toolbox-surface-soft);
+}
+
+.minimizer-observation.is-timeout {
+ border-left-color: var(--regex-warning);
+}
+
+.minimizer-observation.is-crash,
+.minimizer-observation.is-worker-error {
+ border-left-color: var(--toolbox-danger);
+}
+
+.minimizer-observation span,
+.minimizer-observation p,
+.minimizer-observation small {
+ margin: 0;
+ color: var(--toolbox-muted);
+ font-size: 0.72rem;
+ line-height: 1.45;
+}
+
+.minimizer-observation code {
+ overflow-wrap: anywhere;
+ font-size: 0.68rem;
+}
+
+.minimizer-output {
+ padding: 0 0.8rem;
+}
+
+.minimizer-output textarea {
+ min-height: 6rem;
+}
+
+.minimizer-result > .minimizer-actions {
+ padding: 0 0.8rem 0.8rem;
+}
+
+@media (max-width: 720px) {
+ .minimizer-variant-grid,
+ .minimizer-observation-grid {
+ grid-template-columns: 1fr;
+ }
+
+ .minimizer-progress {
+ grid-template-columns: 1fr;
+ }
+}
diff --git a/src/components/MinimizePanel.test.tsx b/src/components/MinimizePanel.test.tsx
new file mode 100644
index 0000000..4a55a92
--- /dev/null
+++ b/src/components/MinimizePanel.test.tsx
@@ -0,0 +1,184 @@
+import { fireEvent, render, screen, waitFor } from "@testing-library/react";
+import { describe, expect, it, vi } from "vitest";
+import type {
+ SubjectMinimizationProgress,
+ SubjectMinimizationRequest,
+ SubjectMinimizationResult,
+} from "../regex/minimization/minimization.types";
+import { MinimizePanel, type SubjectMinimizerClient } from "./MinimizePanel";
+
+function result(): SubjectMinimizationResult {
+ const observation = {
+ status: "complete" as const,
+ reproduced: true,
+ fingerprint: ["expected-no-match:present"],
+ summary: "The forbidden match is still present.",
+ sides: [
+ {
+ flavour: "ecmascript" as const,
+ status: "complete" as const,
+ engineIdentity: "ecmascript · Fixture · 1 · utf16",
+ },
+ ],
+ };
+ return {
+ schemaVersion: 1,
+ targetIdentity: "minimization-fixture",
+ status: "complete",
+ stopReason: "locally-minimal",
+ locallyMinimal: true,
+ originalSubject: "abXcd",
+ minimizedSubject: "X",
+ originalScalars: 5,
+ minimizedScalars: 1,
+ originalUtf8Bytes: 5,
+ minimizedUtf8Bytes: 1,
+ evaluations: 12,
+ acceptedReductions: 3,
+ elapsedMs: 4,
+ budgets: {
+ maximumEvaluations: 500,
+ maximumWallTimeMs: 15_000,
+ candidateTimeoutMs: 2_000,
+ },
+ baseline: observation,
+ final: observation,
+ retainedHistory: [],
+ historyTruncated: false,
+ };
+}
+
+function renderPanel(client: SubjectMinimizerClient) {
+ const onUseSubject = vi.fn();
+ const onClose = vi.fn();
+ render(
+ client}
+ />,
+ );
+ return { onUseSubject, onClose };
+}
+
+function completingClient(
+ requests: SubjectMinimizationRequest[],
+): SubjectMinimizerClient {
+ return {
+ minimize: vi.fn(
+ async (
+ request: SubjectMinimizationRequest,
+ onProgress?: (progress: SubjectMinimizationProgress) => void,
+ ) => {
+ requests.push(request);
+ onProgress?.({
+ phase: "local-sweep",
+ evaluations: 12,
+ maximumEvaluations: 500,
+ acceptedReductions: 3,
+ currentScalars: 1,
+ currentUtf8Bytes: 1,
+ elapsedMs: 4,
+ maximumWallTimeMs: 15_000,
+ });
+ return result();
+ },
+ ),
+ cancel: vi.fn(),
+ dispose: vi.fn(),
+ };
+}
+
+describe("MinimizePanel", () => {
+ it("runs a failing unit-test predicate and can apply the verified result", async () => {
+ const requests: SubjectMinimizationRequest[] = [];
+ const { onUseSubject } = renderPanel(completingClient(requests));
+
+ expect(
+ screen.getByText(/does not claim a globally or semantically minimal/u),
+ ).toBeVisible();
+ fireEvent.click(screen.getByRole("button", { name: "Verify & minimize" }));
+
+ expect(
+ await screen.findByRole("heading", {
+ name: "Locally minimal under the documented transforms",
+ }),
+ ).toBeVisible();
+ expect(screen.getByLabelText("Minimized subject")).toHaveValue("X");
+ expect(requests[0]?.target).toMatchObject({
+ kind: "engine",
+ oracle: {
+ kind: "unit-test-failure",
+ expectation: { kind: "should-not-match" },
+ },
+ });
+ fireEvent.click(
+ screen.getByRole("button", { name: "Use minimized subject" }),
+ );
+ expect(onUseSubject).toHaveBeenCalledWith("X");
+ });
+
+ it("exposes capture-range, timeout and two-engine mismatch targets", async () => {
+ const requests: SubjectMinimizationRequest[] = [];
+ renderPanel(completingClient(requests));
+ const kind = screen.getByLabelText("Minimization failure kind");
+
+ expect(kind).toContainHTML("capture-range-mismatch");
+ expect(kind).toContainHTML("engine-timeout");
+ expect(kind).toContainHTML("comparison-mismatch");
+
+ fireEvent.change(kind, { target: { value: "engine-timeout" } });
+ fireEvent.click(screen.getByRole("button", { name: "Verify & minimize" }));
+ await waitFor(() => expect(requests).toHaveLength(1));
+ expect(requests[0]?.target).toMatchObject({
+ kind: "engine",
+ oracle: { kind: "engine-timeout" },
+ });
+
+ fireEvent.change(kind, {
+ target: { value: "comparison-mismatch" },
+ });
+ fireEvent.click(screen.getByRole("button", { name: "Verify & minimize" }));
+ await waitFor(() => expect(requests).toHaveLength(2));
+ expect(requests[1]?.target).toMatchObject({
+ kind: "comparison-mismatch",
+ operation: "match",
+ sides: [
+ { flavour: "ecmascript", pattern: "X" },
+ { flavour: "pcre2", pattern: "X" },
+ ],
+ });
+ });
+
+ it("cancels the active worker set and closes independently", async () => {
+ const client: SubjectMinimizerClient = {
+ minimize: vi.fn(
+ () => new Promise(() => undefined),
+ ),
+ cancel: vi.fn(),
+ dispose: vi.fn(),
+ };
+ const { onClose } = renderPanel(client);
+
+ fireEvent.click(screen.getByRole("button", { name: "Verify & minimize" }));
+ fireEvent.click(await screen.findByRole("button", { name: "Cancel" }));
+ expect(client.cancel).toHaveBeenCalledOnce();
+ expect(
+ screen.getByText(/active syntax and engine workers were terminated/u),
+ ).toBeVisible();
+
+ fireEvent.click(
+ screen.getByRole("button", { name: "Close subject minimizer" }),
+ );
+ expect(onClose).toHaveBeenCalledOnce();
+ });
+});
diff --git a/src/components/MinimizePanel.tsx b/src/components/MinimizePanel.tsx
new file mode 100644
index 0000000..214555c
--- /dev/null
+++ b/src/components/MinimizePanel.tsx
@@ -0,0 +1,848 @@
+import { useEffect, useMemo, useRef, useState } from "react";
+import { writeClipboardText } from "../browser/clipboard";
+import {
+ AVAILABLE_REGEX_FLAVOURS,
+ defaultRegexOptions,
+} from "../regex/flavours/flavour-registry";
+import {
+ DEFAULT_REGEX_LIMITS,
+ utf8ByteLength,
+} from "../regex/execution/request-limits";
+import type {
+ RegexEngineOptions,
+ RegexFlavourId,
+} from "../regex/model/flavour";
+import type { RegexTestExpectation } from "../regex/tests/test-case.types";
+import { SubjectMinimizer } from "../regex/minimization/SubjectMinimizer";
+import type {
+ EngineFailureOracle,
+ SubjectMinimizationProgress,
+ SubjectMinimizationRequest,
+ SubjectMinimizationResult,
+} from "../regex/minimization/minimization.types";
+import "./MinimizePanel.css";
+
+type MinimizationMode =
+ | "unit-test-failure"
+ | "capture-range-mismatch"
+ | "engine-timeout"
+ | "comparison-mismatch";
+
+type UnitExpectationKind = RegexTestExpectation["kind"];
+
+export interface SubjectMinimizerClient {
+ minimize(
+ request: SubjectMinimizationRequest,
+ onProgress?: (progress: SubjectMinimizationProgress) => void,
+ ): Promise;
+ cancel(): void;
+ dispose(): void;
+}
+
+export interface MinimizePanelProps {
+ readonly activeFlavour: RegexFlavourId;
+ readonly activeFlavourVersion: string;
+ readonly activeFlags: readonly string[];
+ readonly activeOptions: RegexEngineOptions;
+ readonly pattern: string;
+ readonly subject: string;
+ readonly replacement: string;
+ readonly scanAll: boolean;
+ readonly timeoutMs: number;
+ readonly onUseSubject: (subject: string) => void;
+ readonly onClose: () => void;
+ readonly createMinimizer?: () => SubjectMinimizerClient;
+}
+
+const ECMASCRIPT = AVAILABLE_REGEX_FLAVOURS.require("ecmascript");
+const PCRE2 = AVAILABLE_REGEX_FLAVOURS.require("pcre2");
+
+function flagsFor(
+ activeFlavour: RegexFlavourId,
+ activeFlags: readonly string[],
+ flavour: "ecmascript" | "pcre2",
+): readonly string[] {
+ const definition = flavour === "ecmascript" ? ECMASCRIPT : PCRE2;
+ return activeFlavour === flavour
+ ? definition.flags
+ .map((flag) => flag.value)
+ .filter((flag) => activeFlags.includes(flag))
+ : definition.defaultFlags;
+}
+
+function optionsFor(
+ activeFlavour: RegexFlavourId,
+ activeOptions: RegexEngineOptions,
+ flavour: "ecmascript" | "pcre2",
+): RegexEngineOptions {
+ if (activeFlavour === flavour) return activeOptions;
+ return defaultRegexOptions(flavour === "ecmascript" ? ECMASCRIPT : PCRE2);
+}
+
+function parseCaptureSelector(value: string): {
+ readonly groupNumber?: number;
+ readonly groupName?: string;
+} {
+ const trimmed = value.trim();
+ const number = Number(trimmed);
+ return trimmed !== "" && Number.isSafeInteger(number) && number > 0
+ ? { groupNumber: number }
+ : { groupName: trimmed };
+}
+
+function resultLabel(result: SubjectMinimizationResult): string {
+ if (result.locallyMinimal) {
+ return "Locally minimal under the documented transforms";
+ }
+ return result.stopReason.replaceAll("-", " ");
+}
+
+function Observation({
+ label,
+ observation,
+}: {
+ readonly label: string;
+ readonly observation: SubjectMinimizationResult["baseline"];
+}) {
+ return (
+
+ {label}
+
+ {observation.status} ·{" "}
+ {observation.reproduced ? "predicate reproduced" : "not reproduced"}
+
+ {observation.summary}
+ {observation.fingerprint.length > 0 ? (
+ {observation.fingerprint.join(" · ")}
+ ) : null}
+ {observation.sides?.length ? (
+
+ {observation.sides
+ .map(
+ (side) =>
+ `${side.flavour}: ${side.status}${
+ side.engineIdentity ? ` · ${side.engineIdentity}` : ""
+ }`,
+ )
+ .join(" | ")}
+
+ ) : null}
+
+ );
+}
+
+export function MinimizePanel({
+ activeFlavour,
+ activeFlavourVersion,
+ activeFlags,
+ activeOptions,
+ pattern: initialPattern,
+ subject: initialSubject,
+ replacement: initialReplacement,
+ scanAll: initialScanAll,
+ timeoutMs: initialTimeoutMs,
+ onUseSubject,
+ onClose,
+ createMinimizer = () => new SubjectMinimizer(),
+}: MinimizePanelProps) {
+ const [mode, setMode] = useState("unit-test-failure");
+ const [subject, setSubject] = useState(initialSubject);
+ const [scanAll, setScanAll] = useState(initialScanAll);
+ const [unitKind, setUnitKind] =
+ useState("should-not-match");
+ const [expectedCount, setExpectedCount] = useState(1);
+ const [matchIndex, setMatchIndex] = useState(0);
+ const [expectedValue, setExpectedValue] = useState("");
+ const [captureSelector, setCaptureSelector] = useState("1");
+ const [captureStatus, setCaptureStatus] = useState<
+ "participated" | "did-not-participate" | "matched-empty"
+ >("participated");
+ const [checkCaptureValue, setCheckCaptureValue] = useState(false);
+ const [expectedStart, setExpectedStart] = useState(0);
+ const [expectedEnd, setExpectedEnd] = useState(0);
+ const [durationLimitMs, setDurationLimitMs] = useState(10);
+ const [comparisonOperation, setComparisonOperation] = useState<
+ "match" | "replace"
+ >("match");
+ const [ecmaPattern, setEcmaPattern] = useState(initialPattern);
+ const [pcrePattern, setPcrePattern] = useState(initialPattern);
+ const [ecmaReplacement, setEcmaReplacement] = useState(initialReplacement);
+ const [pcreReplacement, setPcreReplacement] = useState(initialReplacement);
+ const [maximumEvaluations, setMaximumEvaluations] = useState(500);
+ const [maximumWallTimeMs, setMaximumWallTimeMs] = useState(15_000);
+ const [candidateTimeoutMs, setCandidateTimeoutMs] = useState(
+ Math.min(
+ DEFAULT_REGEX_LIMITS.advancedMaximumTimeoutMs,
+ Math.max(1, initialTimeoutMs),
+ ),
+ );
+ const [running, setRunning] = useState(false);
+ const [status, setStatus] = useState(
+ "Select an exact failure predicate, then verify the baseline.",
+ );
+ const [progress, setProgress] = useState();
+ const [result, setResult] = useState();
+ const [copyStatus, setCopyStatus] = useState("");
+ const minimizer = useRef(null);
+ const revision = useRef(0);
+
+ useEffect(
+ () => () => {
+ revision.current += 1;
+ minimizer.current?.dispose();
+ minimizer.current = null;
+ },
+ [],
+ );
+
+ const ecmaFlags = useMemo(
+ () => flagsFor(activeFlavour, activeFlags, "ecmascript"),
+ [activeFlags, activeFlavour],
+ );
+ const pcreFlags = useMemo(
+ () => flagsFor(activeFlavour, activeFlags, "pcre2"),
+ [activeFlags, activeFlavour],
+ );
+ const ecmaOptions = useMemo(
+ () => optionsFor(activeFlavour, activeOptions, "ecmascript"),
+ [activeFlavour, activeOptions],
+ );
+ const pcreOptions = useMemo(
+ () => optionsFor(activeFlavour, activeOptions, "pcre2"),
+ [activeFlavour, activeOptions],
+ );
+
+ const unitExpectation = (): RegexTestExpectation => {
+ switch (unitKind) {
+ case "should-match":
+ case "should-not-match":
+ case "must-time-out":
+ return { kind: unitKind };
+ case "match-count":
+ return { kind: unitKind, count: expectedCount };
+ case "full-match":
+ return {
+ kind: unitKind,
+ matchIndex,
+ value: expectedValue,
+ };
+ case "capture":
+ return {
+ kind: unitKind,
+ matchIndex,
+ ...parseCaptureSelector(captureSelector),
+ status: captureStatus,
+ ...(checkCaptureValue ? { value: expectedValue } : {}),
+ };
+ case "replacement":
+ return { kind: unitKind, expected: expectedValue };
+ case "must-complete-within":
+ return { kind: unitKind, milliseconds: durationLimitMs };
+ }
+ };
+
+ const activeSide = {
+ flavour:
+ activeFlavour === "pcre2" ? ("pcre2" as const) : ("ecmascript" as const),
+ flavourVersion: activeFlavourVersion,
+ pattern: initialPattern,
+ flags: activeFlags,
+ options: activeOptions,
+ ...(unitKind === "replacement" ? { replacement: initialReplacement } : {}),
+ };
+
+ const request = (): SubjectMinimizationRequest => {
+ let target: SubjectMinimizationRequest["target"];
+ if (mode === "comparison-mismatch") {
+ target = {
+ kind: "comparison-mismatch",
+ operation: comparisonOperation,
+ scanAll,
+ sides: [
+ {
+ flavour: "ecmascript",
+ flavourVersion: ECMASCRIPT.defaultVersion,
+ pattern: ecmaPattern,
+ flags: ecmaFlags,
+ options: ecmaOptions,
+ ...(comparisonOperation === "replace"
+ ? { replacement: ecmaReplacement }
+ : {}),
+ },
+ {
+ flavour: "pcre2",
+ flavourVersion: PCRE2.defaultVersion,
+ pattern: pcrePattern,
+ flags: pcreFlags,
+ options: pcreOptions,
+ ...(comparisonOperation === "replace"
+ ? { replacement: pcreReplacement }
+ : {}),
+ },
+ ],
+ };
+ } else {
+ let oracle: EngineFailureOracle;
+ if (mode === "engine-timeout") {
+ oracle = { kind: "engine-timeout" };
+ } else if (mode === "capture-range-mismatch") {
+ oracle = {
+ kind: "capture-range-mismatch",
+ matchIndex,
+ ...parseCaptureSelector(captureSelector),
+ expectedRange: {
+ startUtf16: expectedStart,
+ endUtf16: expectedEnd,
+ },
+ };
+ } else {
+ oracle = {
+ kind: "unit-test-failure",
+ expectation: unitExpectation(),
+ };
+ }
+ target = {
+ kind: "engine",
+ side: activeSide,
+ scanAll,
+ oracle,
+ };
+ }
+ return {
+ schemaVersion: 1,
+ subject,
+ target,
+ budgets: {
+ maximumEvaluations,
+ maximumWallTimeMs,
+ candidateTimeoutMs,
+ },
+ maximumMatches: 1_000,
+ maximumCaptureRows: 10_000,
+ maximumOutputBytes: 1024 * 1024,
+ };
+ };
+
+ const run = async () => {
+ const currentRevision = ++revision.current;
+ setRunning(true);
+ setResult(undefined);
+ setCopyStatus("");
+ setStatus("Verifying the baseline with the selected real engine worker…");
+ try {
+ minimizer.current ??= createMinimizer();
+ const minimized = await minimizer.current.minimize(request(), (value) => {
+ if (currentRevision !== revision.current) return;
+ setProgress(value);
+ setStatus(
+ `${value.phase.replaceAll("-", " ")} · ${value.evaluations}/${value.maximumEvaluations} evaluations · ${value.acceptedReductions} accepted`,
+ );
+ });
+ if (currentRevision !== revision.current) return;
+ setResult(minimized);
+ setStatus(resultLabel(minimized));
+ } catch (error) {
+ if (currentRevision !== revision.current) return;
+ setStatus(error instanceof Error ? error.message : String(error));
+ } finally {
+ if (currentRevision === revision.current) setRunning(false);
+ }
+ };
+
+ const cancel = () => {
+ revision.current += 1;
+ minimizer.current?.cancel();
+ setRunning(false);
+ setStatus("Cancelled; active syntax and engine workers were terminated.");
+ };
+
+ return (
+
+
+
+
+
+ This reducer proves only local minimality under deterministic
+ Unicode-scalar chunk deletion, single-scalar deletion and canonical
+ replacement. It does not claim a globally or semantically minimal
+ example.
+
+
+
+
+
+
+
+
void run()}
+ >
+ {running ? "Minimizing…" : "Verify & minimize"}
+
+ {running ? (
+
+ Cancel
+
+ ) : null}
+
+ {status}
+
+
+
+ {progress ? (
+
+
+
+ {progress.currentScalars.toLocaleString()} scalars ·{" "}
+ {progress.currentUtf8Bytes.toLocaleString()} bytes ·{" "}
+ {progress.elapsedMs.toFixed(0)}/
+ {progress.maximumWallTimeMs.toLocaleString()} ms
+
+
+ ) : null}
+
+ {result ? (
+
+
+
+
+
Scalars
+
+ {result.originalScalars.toLocaleString()} →{" "}
+ {result.minimizedScalars.toLocaleString()}
+
+
+
+
UTF-8 bytes
+
+ {result.originalUtf8Bytes.toLocaleString()} →{" "}
+ {result.minimizedUtf8Bytes.toLocaleString()}
+
+
+
+
Evaluations
+
+ {result.evaluations.toLocaleString()} /{" "}
+ {result.budgets.maximumEvaluations.toLocaleString()}
+
+
+
+
Accepted reductions
+ {result.acceptedReductions.toLocaleString()}
+
+
+
Wall time
+ {result.elapsedMs.toFixed(1)} ms
+
+
+
Stop reason
+ {result.stopReason.replaceAll("-", " ")}
+
+
+
+
+
+
+
+ Minimized subject
+
+
+
+
onUseSubject(result.minimizedSubject)}
+ >
+ Use minimized subject
+
+
{
+ void writeClipboardText(result.minimizedSubject)
+ .then(() => setCopyStatus("Copied locally."))
+ .catch((error: unknown) =>
+ setCopyStatus(
+ error instanceof Error ? error.message : String(error),
+ ),
+ );
+ }}
+ >
+ Copy
+
+
{copyStatus}
+
+
+ ) : null}
+
+
+ );
+}
diff --git a/src/components/ModalDialog.tsx b/src/components/ModalDialog.tsx
new file mode 100644
index 0000000..76e077b
--- /dev/null
+++ b/src/components/ModalDialog.tsx
@@ -0,0 +1,71 @@
+import { useEffect, useRef, type ReactNode } from "react";
+
+export function ModalDialog({
+ id,
+ open,
+ labelledBy,
+ onClose,
+ children,
+ className = "",
+}: {
+ readonly id: string;
+ readonly open: boolean;
+ readonly labelledBy: string;
+ readonly onClose: () => void;
+ readonly children: ReactNode;
+ readonly className?: string;
+}) {
+ const dialog = useRef(null);
+
+ useEffect(() => {
+ const element = dialog.current;
+ if (!element) return;
+
+ if (open && !element.open) {
+ if (typeof element.showModal === "function") {
+ element.showModal();
+ } else {
+ // jsdom and older browsers can still expose the content for a usable
+ // non-modal fallback.
+ element.setAttribute("open", "");
+ }
+ (
+ element.querySelector("[data-dialog-initial-focus]") ??
+ element
+ ).focus();
+ return;
+ }
+
+ if (!open && element.open) {
+ if (typeof element.close === "function") {
+ element.close();
+ } else {
+ element.removeAttribute("open");
+ }
+ }
+ }, [open]);
+
+ return (
+ {
+ event.preventDefault();
+ onClose();
+ }}
+ onKeyDownCapture={(event) => {
+ if (event.key !== "Escape") return;
+ event.preventDefault();
+ event.stopPropagation();
+ onClose();
+ }}
+ >
+ {children}
+
+ );
+}
diff --git a/src/components/PatternFormatterPanel.css b/src/components/PatternFormatterPanel.css
new file mode 100644
index 0000000..f242cef
--- /dev/null
+++ b/src/components/PatternFormatterPanel.css
@@ -0,0 +1,342 @@
+.formatter-dialog {
+ width: min(90rem, calc(100% - 2rem));
+}
+
+.formatter-panel {
+ max-height: calc(100vh - 2rem);
+ overflow: auto;
+}
+
+.formatter-body {
+ display: grid;
+ gap: 0.8rem;
+ padding: 0.8rem;
+}
+
+.formatter-advisory,
+.formatter-unavailable,
+.formatter-empty,
+.formatter-result-note {
+ margin: 0;
+ color: var(--toolbox-muted);
+ font-size: 0.78rem;
+ line-height: 1.5;
+}
+
+.formatter-advisory {
+ border-left: 3px solid var(--toolbox-accent);
+ padding: 0.7rem 0.8rem;
+ background: var(--toolbox-accent-soft);
+}
+
+.formatter-unavailable {
+ margin: 0.8rem;
+ border: 1px solid
+ color-mix(in srgb, var(--toolbox-danger) 45%, var(--toolbox-border));
+ border-radius: calc(var(--toolbox-radius) * 0.72);
+ padding: 0.8rem;
+ background: color-mix(
+ in srgb,
+ var(--toolbox-danger) 8%,
+ var(--toolbox-surface)
+ );
+ color: color-mix(in srgb, var(--toolbox-danger) 78%, var(--toolbox-text));
+}
+
+.formatter-preview,
+.formatter-validation {
+ display: grid;
+ gap: 0.75rem;
+ overflow: hidden;
+ border: 1px solid var(--toolbox-border);
+ border-radius: calc(var(--toolbox-radius) * 0.82);
+ background: var(--toolbox-surface);
+}
+
+.formatter-preview > header,
+.formatter-validation > header {
+ display: flex;
+ align-items: center;
+ justify-content: space-between;
+ gap: 1rem;
+ padding: 0.7rem 0.8rem;
+ border-bottom: 1px solid var(--toolbox-border);
+ background: var(--toolbox-surface-soft);
+}
+
+.formatter-preview > header p,
+.formatter-preview > header h3,
+.formatter-validation > header p,
+.formatter-validation > header h3 {
+ margin: 0;
+}
+
+.formatter-preview > header h3,
+.formatter-validation > header h3 {
+ font-size: 0.95rem;
+}
+
+.formatter-preview > header > span,
+.formatter-validation > header > span {
+ color: var(--toolbox-muted);
+ font-size: 0.7rem;
+ font-weight: 750;
+ text-transform: uppercase;
+}
+
+.formatter-pattern-grid {
+ display: grid;
+ grid-template-columns: repeat(2, minmax(0, 1fr));
+ gap: 0.7rem;
+ padding: 0 0.8rem;
+}
+
+.formatter-pattern-grid article {
+ display: grid;
+ min-width: 0;
+ gap: 0.4rem;
+ color: var(--toolbox-muted);
+ font-size: 0.72rem;
+}
+
+.formatter-pattern-grid pre {
+ min-height: 5rem;
+ max-height: 16rem;
+ overflow: auto;
+ margin: 0;
+ border: 1px solid var(--toolbox-border);
+ border-radius: calc(var(--toolbox-radius) * 0.66);
+ padding: 0.65rem;
+ background: var(--toolbox-surface-soft);
+ color: var(--toolbox-text);
+ font-size: 0.75rem;
+ line-height: 1.5;
+ white-space: pre-wrap;
+ overflow-wrap: anywhere;
+}
+
+.formatter-table-scroll {
+ overflow: auto;
+ margin: 0 0.8rem;
+ border: 1px solid var(--toolbox-border);
+ border-radius: calc(var(--toolbox-radius) * 0.66);
+}
+
+.formatter-table-scroll table {
+ width: 100%;
+ border-collapse: collapse;
+ font-size: 0.72rem;
+}
+
+.formatter-table-scroll th,
+.formatter-table-scroll td {
+ padding: 0.5rem 0.6rem;
+ border-bottom: 1px solid var(--toolbox-border);
+ text-align: left;
+ vertical-align: top;
+}
+
+.formatter-table-scroll th {
+ position: sticky;
+ top: 0;
+ background: var(--toolbox-surface-soft);
+ color: var(--toolbox-muted);
+}
+
+.formatter-table-scroll code {
+ overflow-wrap: anywhere;
+}
+
+.formatter-table-scroll > p,
+.formatter-empty {
+ padding: 0.65rem;
+}
+
+.formatter-notices {
+ display: grid;
+ gap: 0.35rem;
+ margin: 0;
+ padding: 0 1.8rem 0.8rem 2.2rem;
+ color: var(--toolbox-muted);
+ font-size: 0.72rem;
+ line-height: 1.45;
+}
+
+.formatter-status {
+ margin: 0 0.8rem;
+ border-left: 3px solid var(--toolbox-accent);
+ padding: 0.65rem 0.75rem;
+ background: var(--toolbox-surface-soft);
+ font-size: 0.76rem;
+}
+
+.formatter-status.is-different {
+ border-left-color: var(--toolbox-danger);
+}
+
+.formatter-status.is-inconclusive {
+ border-left-color: var(--regex-warning);
+}
+
+.formatter-status p {
+ margin: 0.25rem 0 0;
+ color: var(--toolbox-danger);
+}
+
+.formatter-progress {
+ display: grid;
+ grid-template-columns: minmax(8rem, 1fr) auto;
+ align-items: center;
+ gap: 0.65rem;
+ padding: 0 0.8rem;
+ color: var(--toolbox-muted);
+ font-size: 0.7rem;
+}
+
+.formatter-progress progress {
+ width: 100%;
+ accent-color: var(--toolbox-accent);
+}
+
+.formatter-actions {
+ display: flex;
+ flex-wrap: wrap;
+ gap: 0.55rem;
+ padding: 0 0.8rem 0.8rem;
+}
+
+.formatter-result {
+ display: grid;
+ gap: 0.75rem;
+ border-top: 1px solid var(--toolbox-border);
+ padding-top: 0.75rem;
+}
+
+.formatter-result > dl {
+ display: grid;
+ grid-template-columns: repeat(auto-fit, minmax(9rem, 1fr));
+ margin: 0;
+ padding: 0 0.8rem;
+}
+
+.formatter-result > dl > div {
+ padding: 0.5rem;
+ border-bottom: 1px solid var(--toolbox-border);
+}
+
+.formatter-result dt {
+ color: var(--toolbox-muted);
+ font-size: 0.65rem;
+ font-weight: 750;
+ text-transform: uppercase;
+}
+
+.formatter-result dd {
+ margin: 0.15rem 0 0;
+ font-size: 0.77rem;
+}
+
+.formatter-checks {
+ display: grid;
+ grid-template-columns: repeat(auto-fit, minmax(16rem, 1fr));
+ gap: 0.6rem;
+ padding: 0 0.8rem;
+}
+
+.formatter-check {
+ display: grid;
+ gap: 0.35rem;
+ min-width: 0;
+ border: 1px solid var(--toolbox-border);
+ border-left: 4px solid var(--toolbox-accent);
+ border-radius: calc(var(--toolbox-radius) * 0.7);
+ padding: 0.65rem;
+ background: var(--toolbox-surface-soft);
+}
+
+.formatter-check.is-different {
+ border-left-color: var(--toolbox-danger);
+}
+
+.formatter-check.is-inconclusive {
+ border-left-color: var(--regex-warning);
+}
+
+.formatter-check header {
+ display: flex;
+ justify-content: space-between;
+ gap: 0.6rem;
+}
+
+.formatter-check header span {
+ color: var(--toolbox-muted);
+ font-size: 0.65rem;
+ font-weight: 800;
+ text-transform: uppercase;
+}
+
+.formatter-check p,
+.formatter-check small {
+ margin: 0;
+ color: var(--toolbox-muted);
+ font-size: 0.71rem;
+ line-height: 1.45;
+}
+
+.formatter-check code {
+ overflow-wrap: anywhere;
+ font-size: 0.67rem;
+}
+
+.formatter-result-note {
+ padding: 0 0.8rem;
+}
+
+.formatter-differences {
+ margin: 0 0.8rem;
+ border: 1px solid
+ color-mix(in srgb, var(--toolbox-danger) 35%, var(--toolbox-border));
+ border-radius: calc(var(--toolbox-radius) * 0.7);
+ padding: 0.7rem;
+ background: color-mix(
+ in srgb,
+ var(--toolbox-danger) 6%,
+ var(--toolbox-surface)
+ );
+}
+
+.formatter-differences h4 {
+ margin: 0 0 0.45rem;
+ font-size: 0.82rem;
+}
+
+.formatter-differences ul {
+ display: grid;
+ gap: 0.35rem;
+ margin: 0;
+ padding-left: 1.25rem;
+ color: var(--toolbox-muted);
+ font-size: 0.71rem;
+ line-height: 1.45;
+}
+
+.formatter-confirmation {
+ margin: 0 0.8rem;
+ border: 1px solid var(--toolbox-border);
+ border-radius: calc(var(--toolbox-radius) * 0.7);
+ padding: 0.65rem;
+ background: var(--toolbox-surface-soft);
+ color: var(--toolbox-muted);
+ font-size: 0.74rem;
+ line-height: 1.4;
+}
+
+@media (max-width: 720px) {
+ .formatter-pattern-grid {
+ grid-template-columns: 1fr;
+ }
+
+ .formatter-progress {
+ grid-template-columns: 1fr;
+ }
+}
diff --git a/src/components/PatternFormatterPanel.test.tsx b/src/components/PatternFormatterPanel.test.tsx
new file mode 100644
index 0000000..75ef03f
--- /dev/null
+++ b/src/components/PatternFormatterPanel.test.tsx
@@ -0,0 +1,245 @@
+import { cleanup, render, screen, waitFor } from "@testing-library/react";
+import userEvent from "@testing-library/user-event";
+import { afterEach, describe, expect, it, vi } from "vitest";
+import { EcmaScriptSyntaxProvider } from "../regex/syntax/providers/ecmascript/EcmaScriptSyntaxProvider";
+import type {
+ PatternFormatSnapshotCheck,
+ PatternFormatValidationResult,
+} from "../regex/formatting/formatting.types";
+import {
+ PatternFormatterPanel,
+ type PatternFormatValidatorClient,
+ type PatternFormatterPanelProps,
+} from "./PatternFormatterPanel";
+
+const provider = new EcmaScriptSyntaxProvider();
+
+function sameCheck(): PatternFormatSnapshotCheck {
+ return {
+ comparison: "same",
+ before: {
+ status: "complete",
+ engineIdentity: "Native ECMAScript RegExp · fixture",
+ message: "Source completed.",
+ },
+ after: {
+ status: "complete",
+ engineIdentity: "Native ECMAScript RegExp · fixture",
+ message: "Candidate completed.",
+ },
+ summary: "Exact normalized match and replacement output is unchanged.",
+ mismatchKinds: [],
+ };
+}
+
+function validationResult(
+ overrides: Partial = {},
+): PatternFormatValidationResult {
+ return {
+ status: "equivalent",
+ canApply: true,
+ sourceSyntaxAccepted: true,
+ formattedSyntaxAccepted: true,
+ captureShapePreserved: true,
+ current: sameCheck(),
+ tests: [],
+ applicableTestCount: 0,
+ skippedTestCount: 0,
+ completedTestCount: 0,
+ differences: [],
+ elapsedMs: 3,
+ notices: [
+ "No enabled unit test has the exact active pattern; the current snapshot was compared.",
+ ],
+ ...overrides,
+ };
+}
+
+function validator(
+ result: PatternFormatValidationResult = validationResult(),
+): PatternFormatValidatorClient {
+ return {
+ validate: vi.fn().mockImplementation(async (_input, onProgress) => {
+ onProgress?.(0, 0);
+ return result;
+ }),
+ cancel: vi.fn(),
+ dispose: vi.fn(),
+ };
+}
+
+async function renderFormatter(
+ overrides: Partial = {},
+ result: PatternFormatValidationResult = validationResult(),
+) {
+ const pattern = overrides.pattern ?? "a/b";
+ const syntax =
+ overrides.syntax ??
+ (await provider.parsePattern({
+ flavour: "ecmascript",
+ flavourVersion: "2025",
+ pattern,
+ flags: [],
+ options: {},
+ }));
+ const currentValidator = validator(result);
+ const onApply = vi.fn();
+ const rendered = render(
+ currentValidator}
+ {...overrides}
+ />,
+ );
+ return { ...rendered, currentValidator, onApply, syntax };
+}
+
+afterEach(cleanup);
+
+describe("PatternFormatterPanel", () => {
+ it("shows an exact preview and requires validation plus confirmation before apply", async () => {
+ const user = userEvent.setup();
+ const { currentValidator, onApply } = await renderFormatter();
+
+ expect(screen.getByTestId("formatted-pattern-preview").textContent).toBe(
+ "a\\/b",
+ );
+ expect(
+ screen.queryByRole("button", { name: "Apply formatted pattern" }),
+ ).not.toBeInTheDocument();
+
+ await user.click(
+ screen.getByRole("button", { name: "Validate exact snapshots" }),
+ );
+ await waitFor(() => {
+ expect(currentValidator.validate).toHaveBeenCalledTimes(1);
+ });
+ expect(
+ await screen.findByText(
+ "All completed safety gates retained the same observable behavior.",
+ ),
+ ).toBeVisible();
+ expect(screen.getByRole("checkbox")).toBeEnabled();
+ expect(
+ screen.getByRole("button", { name: "Apply formatted pattern" }),
+ ).toBeDisabled();
+
+ await user.click(screen.getByRole("checkbox"));
+ await user.click(
+ screen.getByRole("button", { name: "Apply formatted pattern" }),
+ );
+
+ expect(onApply).toHaveBeenCalledWith("a\\/b");
+ });
+
+ it("keeps a behavior-changing candidate blocked", async () => {
+ const user = userEvent.setup();
+ await renderFormatter(
+ {},
+ validationResult({
+ status: "different",
+ canApply: false,
+ current: {
+ ...sameCheck(),
+ comparison: "different",
+ summary: "Match count changed.",
+ mismatchKinds: ["match-count"],
+ },
+ differences: [
+ {
+ scope: "current-subject",
+ code: "current-behavior-changed",
+ summary: "Match count changed.",
+ },
+ ],
+ }),
+ );
+
+ await user.click(
+ screen.getByRole("button", { name: "Validate exact snapshots" }),
+ );
+
+ expect(await screen.findByText("current-behavior-changed")).toBeVisible();
+ expect(screen.getByRole("checkbox")).toBeDisabled();
+ expect(
+ screen.getByRole("button", { name: "Apply formatted pattern" }),
+ ).toBeDisabled();
+ });
+
+ it("reports PCRE2 and stale syntax snapshots as unavailable", async () => {
+ const { rerender, syntax } = await renderFormatter({
+ flavour: "pcre2",
+ syntax: undefined,
+ });
+
+ expect(
+ screen.getByText(/currently available only for ECMAScript/),
+ ).toBeVisible();
+ expect(
+ screen.queryByRole("button", { name: "Validate exact snapshots" }),
+ ).not.toBeInTheDocument();
+
+ rerender(
+ ,
+ );
+ expect(
+ screen.getByText(/accepted, non-recovered syntax-provider snapshot/),
+ ).toBeVisible();
+ });
+
+ it("clears a validation result when an exact input snapshot changes", async () => {
+ const user = userEvent.setup();
+ const rendered = await renderFormatter();
+ await user.click(
+ screen.getByRole("button", { name: "Validate exact snapshots" }),
+ );
+ expect(await screen.findByRole("checkbox")).toBeVisible();
+
+ rendered.rerender(
+ rendered.currentValidator}
+ />,
+ );
+
+ expect(screen.queryByRole("checkbox")).not.toBeInTheDocument();
+ expect(
+ screen.getByRole("button", { name: "Validate exact snapshots" }),
+ ).toBeEnabled();
+ });
+});
diff --git a/src/components/PatternFormatterPanel.tsx b/src/components/PatternFormatterPanel.tsx
new file mode 100644
index 0000000..f8e9db4
--- /dev/null
+++ b/src/components/PatternFormatterPanel.tsx
@@ -0,0 +1,617 @@
+import { useEffect, useMemo, useRef, useState } from "react";
+import type {
+ RegexEngineOptions,
+ RegexFlavourId,
+} from "../regex/model/flavour";
+import type { RegexSyntaxResult } from "../regex/model/syntax";
+import type { RegexTestCase } from "../regex/tests/test-case.types";
+import { WorkerRequestError } from "../regex/execution/WorkerSupervisor";
+import { formatEcmaScriptPattern } from "../regex/formatting/ecmascript-format";
+import { PatternFormatValidator } from "../regex/formatting/PatternFormatValidator";
+import type {
+ PatternFormatPreview,
+ PatternFormatSnapshotCheck,
+ PatternFormatValidationInput,
+ PatternFormatValidationResult,
+} from "../regex/formatting/formatting.types";
+import "./PatternFormatterPanel.css";
+
+const MAXIMUM_VISIBLE_CHANGES = 200;
+const MAXIMUM_VISIBLE_CHECKS = 200;
+
+export interface PatternFormatValidatorClient {
+ validate(
+ input: PatternFormatValidationInput,
+ onProgress?: (completed: number, total: number) => void,
+ ): Promise;
+ cancel(): void;
+ dispose(): void;
+}
+
+export interface PatternFormatterPanelProps {
+ readonly flavour: RegexFlavourId;
+ readonly flavourVersion?: string;
+ readonly pattern: string;
+ readonly flags: readonly string[];
+ readonly options: RegexEngineOptions;
+ readonly syntax?: RegexSyntaxResult;
+ readonly subject: string;
+ readonly replacement: string;
+ readonly scanAll: boolean;
+ readonly timeoutMs: number;
+ readonly tests: readonly RegexTestCase[];
+ readonly onApply: (formattedPattern: string) => void;
+ readonly onClose?: () => void;
+ readonly createValidator?: () => PatternFormatValidatorClient;
+}
+
+type PreviewState =
+ | { readonly status: "available"; readonly preview: PatternFormatPreview }
+ | { readonly status: "unavailable"; readonly reason: string };
+
+interface ValidationConfiguration {
+ readonly flavourVersion?: string;
+ readonly preview: PatternFormatPreview;
+ readonly flags: readonly string[];
+ readonly options: RegexEngineOptions;
+ readonly subject: string;
+ readonly replacement: string;
+ readonly scanAll: boolean;
+ readonly timeoutMs: number;
+ readonly tests: readonly RegexTestCase[];
+}
+
+interface ValidationRecord {
+ readonly configuration: ValidationConfiguration;
+ readonly result: PatternFormatValidationResult;
+}
+
+interface ConfirmationRecord {
+ readonly configuration: ValidationConfiguration;
+ readonly checked: boolean;
+}
+
+function previewState(
+ flavour: RegexFlavourId,
+ pattern: string,
+ syntax: RegexSyntaxResult | undefined,
+): PreviewState {
+ if (flavour !== "ecmascript") {
+ return {
+ status: "unavailable",
+ reason:
+ "Canonical formatting is currently available only for ECMAScript. PCRE2 and other flavours remain unchanged because no complete grammar-backed formatter is implemented for them.",
+ };
+ }
+ if (!syntax) {
+ return {
+ status: "unavailable",
+ reason:
+ "Wait for the ECMAScript syntax provider to finish parsing the current pattern.",
+ };
+ }
+ try {
+ return {
+ status: "available",
+ preview: formatEcmaScriptPattern(pattern, syntax),
+ };
+ } catch (cause) {
+ return {
+ status: "unavailable",
+ reason:
+ cause instanceof Error
+ ? cause.message
+ : "The current syntax snapshot cannot be formatted.",
+ };
+ }
+}
+
+function formatDuration(milliseconds: number): string {
+ if (milliseconds < 1) return "<1 ms";
+ if (milliseconds < 1_000) return `${milliseconds.toFixed(0)} ms`;
+ return `${(milliseconds / 1_000).toFixed(2)} s`;
+}
+
+function observationIdentity(check: PatternFormatSnapshotCheck): string {
+ const before = check.before.engineIdentity ?? check.before.status;
+ const after = check.after.engineIdentity ?? check.after.status;
+ return before === after ? before : `${before} → ${after}`;
+}
+
+function ValidationCheck({
+ title,
+ check,
+}: {
+ readonly title: string;
+ readonly check: PatternFormatSnapshotCheck;
+}) {
+ return (
+
+
+ {title}
+ {check.comparison}
+
+ {check.summary}
+ {observationIdentity(check)}
+ {check.mismatchKinds.length > 0 ? (
+ {check.mismatchKinds.join(" · ")}
+ ) : null}
+
+ );
+}
+
+export function PatternFormatterPanel({
+ flavour,
+ flavourVersion,
+ pattern,
+ flags,
+ options,
+ syntax,
+ subject,
+ replacement,
+ scanAll,
+ timeoutMs,
+ tests,
+ onApply,
+ onClose,
+ createValidator = () => new PatternFormatValidator(),
+}: PatternFormatterPanelProps) {
+ const preview = useMemo(
+ () => previewState(flavour, pattern, syntax),
+ [flavour, pattern, syntax],
+ );
+ const configuration = useMemo(
+ () =>
+ preview.status === "available"
+ ? {
+ flavourVersion,
+ preview: preview.preview,
+ flags,
+ options,
+ subject,
+ replacement,
+ scanAll,
+ timeoutMs,
+ tests,
+ }
+ : undefined,
+ [
+ flags,
+ flavourVersion,
+ options,
+ preview,
+ replacement,
+ scanAll,
+ subject,
+ tests,
+ timeoutMs,
+ ],
+ );
+ const validator = useRef(undefined);
+ const revision = useRef(0);
+ const [running, setRunning] = useState(false);
+ const [progress, setProgress] = useState({ completed: 0, total: 0 });
+ const [message, setMessage] = useState(
+ "Review the exact preview before running bounded validation.",
+ );
+ const [error, setError] = useState();
+ const [record, setRecord] = useState();
+ const [confirmation, setConfirmation] = useState();
+
+ const currentRecord =
+ record?.configuration === configuration ? record : undefined;
+ const confirmed =
+ confirmation !== undefined &&
+ confirmation.configuration === configuration &&
+ confirmation.checked;
+
+ useEffect(
+ () => () => {
+ revision.current += 1;
+ validator.current?.dispose();
+ validator.current = undefined;
+ },
+ [],
+ );
+
+ const cancel = () => {
+ revision.current += 1;
+ validator.current?.cancel();
+ setRunning(false);
+ setMessage("Validation cancelled; formatter workers were terminated.");
+ };
+
+ const runValidation = async () => {
+ if (
+ !configuration ||
+ !configuration.preview.changed ||
+ !configuration.preview.withinPatternLimit
+ ) {
+ return;
+ }
+ const currentValidator =
+ validator.current ?? (validator.current = createValidator());
+ currentValidator.cancel();
+ revision.current += 1;
+ const runRevision = revision.current;
+ setRunning(true);
+ setConfirmation(undefined);
+ setError(undefined);
+ setRecord(undefined);
+ setProgress({ completed: 0, total: 0 });
+ setMessage(
+ "Reparsing, compiling and comparing exact source/candidate snapshots…",
+ );
+ try {
+ const result = await currentValidator.validate(
+ {
+ flavour: "ecmascript",
+ ...(configuration.flavourVersion === undefined
+ ? {}
+ : { flavourVersion: configuration.flavourVersion }),
+ sourcePattern: configuration.preview.sourcePattern,
+ formattedPattern: configuration.preview.formattedPattern,
+ flags: configuration.flags,
+ options: configuration.options,
+ subject: configuration.subject,
+ replacement: configuration.replacement,
+ scanAll: configuration.scanAll,
+ timeoutMs: configuration.timeoutMs,
+ tests: configuration.tests,
+ },
+ (completed, total) => {
+ if (revision.current !== runRevision) return;
+ setProgress({ completed, total });
+ setMessage(
+ total === 0
+ ? "The current subject/replacement snapshot is being compared."
+ : `Validated ${completed.toLocaleString()} of ${total.toLocaleString()} applicable unit tests.`,
+ );
+ },
+ );
+ if (revision.current !== runRevision) return;
+ setRecord({ configuration, result });
+ setMessage(
+ result.status === "equivalent"
+ ? "All completed safety gates retained the same observable behavior."
+ : result.status === "different"
+ ? "Formatting changed at least one checked observation."
+ : "Validation was inconclusive; the candidate cannot be applied.",
+ );
+ } catch (cause) {
+ if (revision.current !== runRevision) return;
+ const cancelled =
+ cause instanceof WorkerRequestError && cause.kind === "cancelled";
+ setError(
+ cancelled
+ ? "Validation was cancelled."
+ : cause instanceof Error
+ ? cause.message
+ : "Formatter validation failed.",
+ );
+ setMessage(
+ cancelled
+ ? "Validation cancelled; no candidate can be applied."
+ : "Validation failed; no candidate can be applied.",
+ );
+ } finally {
+ if (revision.current === runRevision) setRunning(false);
+ }
+ };
+
+ const close = () => {
+ if (running) cancel();
+ onClose?.();
+ };
+
+ const result = currentRecord?.result;
+ const canApply =
+ result?.canApply === true &&
+ confirmed &&
+ configuration?.preview.changed === true;
+
+ return (
+
+
+
+ {preview.status === "unavailable" ? (
+
+ {preview.reason}
+
+ ) : (
+
+
+ This deliberately narrow formatter escapes literal slashes and raw
+ control/line-separator characters. It does not insert insignificant
+ whitespace—ECMAScript patterns generally have no such whitespace.
+
+
+
+
+
+
+ Source
+
+ {preview.preview.sourcePattern}
+
+
+
+ Formatted candidate
+
+ {preview.preview.formattedPattern}
+
+
+
+
+ {preview.preview.changed ? (
+
+
+
+
+ UTF-16 range
+ Kind
+ Before
+ After
+
+
+
+ {preview.preview.changes
+ .slice(0, MAXIMUM_VISIBLE_CHANGES)
+ .map((change) => (
+
+
+ {change.startUtf16}–{change.endUtf16}
+
+ {change.kind.replaceAll("-", " ")}
+
+ {JSON.stringify(change.before)}
+
+
+ {JSON.stringify(change.after)}
+
+
+ ))}
+
+
+ {preview.preview.totalChanges > MAXIMUM_VISIBLE_CHANGES ? (
+
+ The table shows the first{" "}
+ {MAXIMUM_VISIBLE_CHANGES.toLocaleString()} transformations.
+
+ ) : null}
+
+ ) : (
+
+ This pattern already uses the formatter’s canonical literal and
+ control escapes.
+
+ )}
+
+
+ {preview.preview.warnings.map((warning) => (
+ {warning}
+ ))}
+
+
+
+ {preview.preview.changed ? (
+
+
+
+
Required before apply
+
Safety validation
+
+
+ {result?.status ?? (running ? "running" : "not run")}
+
+
+
+
{message}
+ {error ?
{error}
: null}
+
+ {running ? (
+
+
+
+ {progress.completed.toLocaleString()} /{" "}
+ {progress.total.toLocaleString()} tests
+
+
+ ) : null}
+
+ void runValidation()}
+ >
+ Validate exact snapshots
+
+ {running ? (
+
+ Cancel
+
+ ) : null}
+
+
+ {result ? (
+
+
+
+
Syntax
+
+ {result.sourceSyntaxAccepted &&
+ result.formattedSyntaxAccepted
+ ? "both accepted"
+ : "rejected"}
+
+
+
+
Capture shape
+
+ {result.captureShapePreserved ? "preserved" : "changed"}
+
+
+
+
Unit tests
+
+ {result.completedTestCount.toLocaleString()} /{" "}
+ {result.applicableTestCount.toLocaleString()} applicable
+
+
+
+
Unrelated / disabled
+ {result.skippedTestCount.toLocaleString()}
+
+
+
Elapsed
+ {formatDuration(result.elapsedMs)}
+
+
+
+
+
+ {result.tests
+ .slice(0, MAXIMUM_VISIBLE_CHECKS)
+ .map((test) => (
+
+ ))}
+
+ {result.tests.length > MAXIMUM_VISIBLE_CHECKS ? (
+
+ The panel shows the first{" "}
+ {MAXIMUM_VISIBLE_CHECKS.toLocaleString()} completed test
+ comparisons.
+
+ ) : null}
+
+ {result.differences.length > 0 ? (
+
+
+
+ {result.differences
+ .slice(0, MAXIMUM_VISIBLE_CHECKS)
+ .map((difference, index) => (
+
+ {difference.code} {" "}
+ {difference.summary}
+
+ ))}
+
+
+ ) : null}
+
+ {result.notices.map((notice) => (
+ {notice}
+ ))}
+
+
+
+ {
+ if (!configuration) return;
+ setConfirmation({
+ configuration,
+ checked: event.target.checked,
+ });
+ }}
+ />
+
+ I understand that this bounded validation is not a proof
+ for every possible subject, and I want to apply the exact
+ candidate shown above.
+
+
+
+ {
+ if (!canApply || !configuration) return;
+ onApply(configuration.preview.formattedPattern);
+ }}
+ >
+ Apply formatted pattern
+
+
+
+ ) : null}
+
+ ) : null}
+
+ )}
+
+ );
+}
diff --git a/src/components/Pcre2TracePanel.test.tsx b/src/components/Pcre2TracePanel.test.tsx
new file mode 100644
index 0000000..2b2fba0
--- /dev/null
+++ b/src/components/Pcre2TracePanel.test.tsx
@@ -0,0 +1,59 @@
+import { render, screen } from "@testing-library/react";
+import { describe, expect, it, vi } from "vitest";
+import { Pcre2TracePanel } from "./Pcre2TracePanel";
+
+function renderPanel(flavour: "ecmascript" | "pcre2") {
+ return render(
+ ,
+ );
+}
+
+describe("Pcre2TracePanel", () => {
+ it("discloses reported versus derived trace semantics and hard caps", () => {
+ renderPanel("pcre2");
+
+ expect(
+ screen.getByRole("heading", { name: "Execution trace" }),
+ ).toBeInTheDocument();
+ expect(
+ screen.getByText(/actual stream reported by PCRE2_AUTO_CALLOUT/u),
+ ).toBeInTheDocument();
+ expect(
+ screen.getByText(/Movement labels are derived only/u),
+ ).toBeInTheDocument();
+ expect(
+ screen.getByRole("spinbutton", { name: "Maximum events" }),
+ ).toHaveAttribute("max", "50000");
+ expect(
+ screen.getByRole("spinbutton", {
+ name: "Maximum serialized bytes",
+ }),
+ ).toHaveAttribute("max", String(10 * 1024 * 1024));
+ expect(screen.getByRole("button", { name: "Run trace" })).toBeEnabled();
+ });
+
+ it("does not offer an invented ECMAScript trace", () => {
+ renderPanel("ecmascript");
+
+ expect(screen.getByRole("button", { name: "Run trace" })).toBeDisabled();
+ expect(screen.getByText(/Select PCRE2 to enable/u)).toBeInTheDocument();
+ });
+});
diff --git a/src/components/Pcre2TracePanel.tsx b/src/components/Pcre2TracePanel.tsx
new file mode 100644
index 0000000..641e943
--- /dev/null
+++ b/src/components/Pcre2TracePanel.tsx
@@ -0,0 +1,407 @@
+import { useEffect, useRef, useState } from "react";
+import { Pcre2TraceSupervisor } from "../regex/execution/Pcre2TraceSupervisor";
+import { WorkerRequestError } from "../regex/execution/WorkerSupervisor";
+import { DEFAULT_REGEX_LIMITS } from "../regex/execution/request-limits";
+import type {
+ RegexEngineOptions,
+ RegexFlavourId,
+} from "../regex/model/flavour";
+import type { Pcre2TraceEvent, Pcre2TraceResult } from "../regex/model/trace";
+import type { SourceRange } from "../regex/model/syntax";
+
+const DEFAULT_TRACE_EVENTS = 5_000;
+const DEFAULT_TRACE_BYTES = 2 * 1024 * 1024;
+const MAXIMUM_RENDERED_TRACE_EVENTS = 2_000;
+
+type TraceState =
+ | { readonly status: "idle"; readonly message: string }
+ | { readonly status: "running"; readonly message: string }
+ | { readonly status: "ready"; readonly message: string }
+ | { readonly status: "timeout"; readonly message: string }
+ | { readonly status: "error"; readonly message: string };
+
+function movementLabel(event: Pcre2TraceEvent): string {
+ switch (event.movement.classification) {
+ case "first-event":
+ return "first";
+ case "same-position":
+ return "same position";
+ case "apparent-backtrack":
+ return "apparent backtrack";
+ case "forward":
+ return "forward";
+ }
+}
+
+export function Pcre2TracePanel({
+ active,
+ flavour,
+ pattern,
+ flags,
+ options,
+ subject,
+ timeoutMs,
+ onClose,
+ onSelectRanges,
+}: {
+ readonly active: boolean;
+ readonly flavour: RegexFlavourId;
+ readonly pattern: string;
+ readonly flags: readonly string[];
+ readonly options: RegexEngineOptions;
+ readonly subject: string;
+ readonly timeoutMs: number;
+ readonly onClose: () => void;
+ readonly onSelectRanges: (
+ patternRange: SourceRange,
+ subjectRange: SourceRange,
+ ) => void;
+}) {
+ const supervisor = useRef(null);
+ const requestRevision = useRef(0);
+ const [maximumEvents, setMaximumEvents] = useState(DEFAULT_TRACE_EVENTS);
+ const [maximumBytes, setMaximumBytes] = useState(DEFAULT_TRACE_BYTES);
+ const [completed, setCompleted] = useState<{
+ readonly result: Pcre2TraceResult;
+ readonly input: {
+ readonly pattern: string;
+ readonly subject: string;
+ readonly flags: string;
+ readonly options: string;
+ };
+ }>();
+ const [state, setState] = useState({
+ status: "idle",
+ message: "Run an isolated PCRE2 automatic-callout trace.",
+ });
+
+ const traceEngine = () => {
+ supervisor.current ??= new Pcre2TraceSupervisor();
+ return supervisor.current;
+ };
+
+ useEffect(() => {
+ return () => supervisor.current?.dispose();
+ }, []);
+
+ useEffect(() => {
+ if (active) return;
+ supervisor.current?.cancel();
+ }, [active]);
+
+ const limitsValid =
+ Number.isSafeInteger(maximumEvents) &&
+ maximumEvents >= 1 &&
+ maximumEvents <= DEFAULT_REGEX_LIMITS.maximumTraceEvents &&
+ Number.isSafeInteger(maximumBytes) &&
+ maximumBytes >= 32 &&
+ maximumBytes <= DEFAULT_REGEX_LIMITS.maximumTraceBytes;
+
+ const runTrace = async () => {
+ if (flavour !== "pcre2" || !limitsValid) return;
+ const revision = (requestRevision.current += 1);
+ setCompleted(undefined);
+ setState({
+ status: "running",
+ message: "Collecting reported callouts in a dedicated worker…",
+ });
+ try {
+ const next = await traceEngine().trace(
+ {
+ flavour,
+ pattern,
+ flags,
+ options,
+ subject,
+ maximumTraceEvents: maximumEvents,
+ maximumTraceBytes: maximumBytes,
+ },
+ timeoutMs,
+ );
+ if (revision !== requestRevision.current) return;
+ setCompleted({
+ result: next,
+ input: {
+ pattern,
+ subject,
+ flags: flags.join(""),
+ options: JSON.stringify(options),
+ },
+ });
+ const error = next.diagnostics.find(
+ (diagnostic) => diagnostic.severity === "error",
+ );
+ setState(
+ next.accepted
+ ? {
+ status: "ready",
+ message: next.truncated
+ ? "Bounded trace complete; collection stopped at a configured cap."
+ : "Bounded trace complete.",
+ }
+ : {
+ status: "error",
+ message: error?.message ?? "PCRE2 rejected the trace request.",
+ },
+ );
+ } catch (error) {
+ if (revision !== requestRevision.current) return;
+ if (error instanceof WorkerRequestError && error.kind === "timeout") {
+ setState({
+ status: "timeout",
+ message: `${error.message}; the trace worker was terminated.`,
+ });
+ } else if (
+ error instanceof WorkerRequestError &&
+ error.kind === "cancelled"
+ ) {
+ setState({
+ status: "idle",
+ message: "Trace cancelled; its worker was terminated.",
+ });
+ } else {
+ setState({
+ status: "error",
+ message:
+ error instanceof Error
+ ? error.message
+ : "The PCRE2 trace worker failed.",
+ });
+ }
+ }
+ };
+
+ const resultIsCurrent =
+ completed?.input.pattern === pattern &&
+ completed.input.subject === subject &&
+ completed.input.flags === flags.join("") &&
+ completed.input.options === JSON.stringify(options);
+ const visibleResult = resultIsCurrent ? completed.result : undefined;
+ const visibleState =
+ resultIsCurrent || state.status === "running"
+ ? state
+ : {
+ status: "idle" as const,
+ message: "Inputs changed; run a new isolated trace.",
+ };
+ const renderedEvents =
+ visibleResult?.events.slice(0, MAXIMUM_RENDERED_TRACE_EVENTS) ?? [];
+ return (
+
+
+
+ This is an actual stream reported by PCRE2_AUTO_CALLOUT for one{" "}
+ pcre2_match() invocation. Movement labels are derived only
+ from adjacent reported subject positions; the stream is not a complete
+ record of every internal engine action.
+
+
+
+ Maximum events
+ setMaximumEvents(Number(event.target.value))}
+ />
+
+
+ Maximum serialized bytes
+ setMaximumBytes(Number(event.target.value))}
+ />
+
+ void runTrace()}
+ >
+ {visibleState.status === "running" ? "Tracing…" : "Run trace"}
+
+ {visibleState.status === "running" ? (
+ {
+ requestRevision.current += 1;
+ supervisor.current?.cancel();
+ setState({
+ status: "idle",
+ message: "Trace cancelled; its worker was terminated.",
+ });
+ }}
+ >
+ Cancel trace
+
+ ) : null}
+
+ {flavour !== "pcre2" ? (
+
+ Select PCRE2 to enable its automatic-callout trace.
+
+ ) : null}
+
+ {visibleState.message}
+ {visibleResult?.accepted ? (
+
+ {visibleResult.events.length.toLocaleString()} complete retained
+ events of {visibleResult.totalEventCount.toLocaleString()} callback
+ event(s) observed before return ·{" "}
+ {visibleResult.traceBytes.toLocaleString()} serialized bytes · last
+ complete event{" "}
+ {visibleResult.lastCompleteEvent?.toLocaleString() ?? "none"} ·{" "}
+ native status {visibleResult.nativeMatchStatus} ·{" "}
+ {visibleResult.matched ? "matched" : "not matched or bounded stop"}{" "}
+ · {visibleResult.elapsedMs.toFixed(2)} ms
+
+ ) : null}
+
+ {visibleResult?.diagnostics.length ? (
+
+ {visibleResult.diagnostics.map((diagnostic) => (
+
+ {diagnostic.severity} {diagnostic.message}
+
+ ))}
+
+ ) : null}
+ {visibleResult && visibleResult.events.length > renderedEvents.length ? (
+
+ Trace viewer limited. Showing the first{" "}
+ {renderedEvents.length.toLocaleString()} of{" "}
+ {visibleResult.events.length.toLocaleString()} complete retained
+ events. The worker result remains bounded by both configured caps.
+
+ ) : null}
+ {renderedEvents.length > 0 ? (
+
+
+
+
+ Event
+ Callout
+ Pattern
+ Next item
+ Subject
+ Captures
+ Mark
+ Movement
+
+
+
+ {renderedEvents.map((event) => (
+
+
+
+ onSelectRanges(
+ {
+ startUtf16:
+ event.reported.nextPatternItem.startUtf16,
+ endUtf16: event.reported.nextPatternItem.endUtf16,
+ },
+ {
+ startUtf16: event.reported.subjectPosition.utf16,
+ endUtf16: event.reported.subjectPosition.utf16,
+ },
+ )
+ }
+ >
+ {event.eventNumber}
+
+
+ {event.reported.calloutNumber}
+
+ {event.reported.patternPosition.utf16} UTF-16
+
+ {event.reported.patternPosition.nativeByte} byte
+
+
+
+
+ {pattern.slice(
+ event.reported.nextPatternItem.startUtf16,
+ event.reported.nextPatternItem.endUtf16,
+ ) || "end"}
+
+
+ {event.reported.nextPatternItem.startNativeByte}…
+ {event.reported.nextPatternItem.endNativeByte} bytes
+
+
+
+ {event.reported.subjectPosition.utf16} UTF-16
+
+ {event.reported.subjectPosition.nativeByte} byte
+
+
+
+ top {event.reported.captureTop} · last{" "}
+ {event.reported.captureLast}
+
+
+ {event.reported.mark ? (
+ {event.reported.mark}
+ ) : (
+ "—"
+ )}
+
+
+
+ {movementLabel(event)}
+
+
+ {event.movement.subjectDeltaBytes > 0 ? "+" : ""}
+ {event.movement.subjectDeltaBytes} bytes · derived
+
+
+
+ ))}
+
+
+
+ ) : visibleResult?.accepted ? (
+ No callout event was retained.
+ ) : null}
+
+ );
+}
diff --git a/src/components/ProjectPanel.tsx b/src/components/ProjectPanel.tsx
index ac4a4d9..8ec5a0b 100644
--- a/src/components/ProjectPanel.tsx
+++ b/src/components/ProjectPanel.tsx
@@ -65,6 +65,10 @@ export function ProjectPanel({
{project.name} · schema {project.schemaVersion}
+
+ Corpus documents, applied outputs and corpus results are always excluded
+ from project JSON and local saves.
+
{
it("shows flavour, flags, example and authoritative source before insertion", async () => {
- const onInsert = vi.fn();
+ const onUsePattern = vi.fn();
const user = userEvent.setup();
- render( );
+ render( );
await user.type(
screen.getByLabelText("Filter constructs"),
@@ -20,7 +20,37 @@ describe("QuickReference", () => {
screen.getByRole("link", { name: "ECMAScript Atom grammar" }),
).toHaveAttribute("href", expect.stringContaining("tc39.es"));
- await user.click(screen.getByRole("button", { name: "Insert example" }));
- expect(onInsert).toHaveBeenCalledWith("a.b");
+ await user.click(
+ screen.getByRole("button", { name: "Use example as pattern" }),
+ );
+ expect(onUsePattern).toHaveBeenCalledWith("a.b");
+ });
+
+ it("shows an explicit empty state for a filter without matches", async () => {
+ const user = userEvent.setup();
+ render( );
+
+ await user.type(
+ screen.getByLabelText("Filter constructs"),
+ "no such construct",
+ );
+
+ expect(screen.getByRole("status")).toHaveTextContent(
+ "No reference constructs match",
+ );
+ expect(screen.getByText("0 of 12 constructs")).toBeInTheDocument();
+ });
+
+ it("switches to the authoritative PCRE2 construct set", async () => {
+ const user = userEvent.setup();
+ render( );
+
+ expect(screen.getByText("Original PCRE2 reference")).toBeInTheDocument();
+ expect(screen.getByText("8 of 8 constructs")).toBeInTheDocument();
+ await user.type(screen.getByLabelText("Filter constructs"), "branch reset");
+ expect(screen.getByText("Branch-reset group")).toBeInTheDocument();
+ expect(
+ screen.getByRole("link", { name: "PCRE2 pattern documentation" }),
+ ).toHaveAttribute("href", expect.stringContaining("pcre.org"));
});
});
diff --git a/src/components/QuickReference.tsx b/src/components/QuickReference.tsx
index e28fdac..fd664eb 100644
--- a/src/components/QuickReference.tsx
+++ b/src/components/QuickReference.tsx
@@ -1,30 +1,44 @@
import { useMemo, useState } from "react";
-import { ECMASCRIPT_REFERENCE } from "../regex/reference/token-reference";
+import type { RegexFlavourId } from "../regex/model/flavour";
+import {
+ ECMASCRIPT_REFERENCE,
+ PCRE2_REFERENCE,
+} from "../regex/reference/token-reference";
export function QuickReference({
- onInsert,
+ onUsePattern,
+ onClose,
+ flavour = "ecmascript",
}: {
- readonly onInsert: (syntax: string) => void;
+ readonly onUsePattern: (syntax: string) => void;
+ readonly onClose?: () => void;
+ readonly flavour?: RegexFlavourId;
}) {
const [query, setQuery] = useState("");
+ const reference =
+ flavour === "pcre2" ? PCRE2_REFERENCE : ECMASCRIPT_REFERENCE;
const entries = useMemo(() => {
- const needle = query.toLocaleLowerCase();
+ const normalizeSearchText = (value: string) =>
+ value
+ .toLocaleLowerCase()
+ .replaceAll(/[^\p{L}\p{N}]+/gu, " ")
+ .trim();
+ const needle = normalizeSearchText(query);
return needle
- ? ECMASCRIPT_REFERENCE.filter((entry) =>
- [
- entry.construct,
- entry.syntax,
- entry.explanation,
- entry.example,
- entry.flags.join(" "),
- entry.source,
- ]
- .join(" ")
- .toLocaleLowerCase()
- .includes(needle),
+ ? reference.filter((entry) =>
+ normalizeSearchText(
+ [
+ entry.construct,
+ entry.syntax,
+ entry.explanation,
+ entry.example,
+ entry.flags.join(" "),
+ entry.source,
+ ].join(" "),
+ ).includes(needle),
)
- : ECMASCRIPT_REFERENCE;
- }, [query]);
+ : reference;
+ }, [query, reference]);
return (
-
Original ECMAScript reference
+
+ Original {flavour === "pcre2" ? "PCRE2" : "ECMAScript"} reference
+
Quick reference
+
+
+ {entries.length} of {reference.length} constructs
+
+ {onClose ? (
+
+ ×
+
+ ) : null}
+
Filter constructs
@@ -44,51 +76,57 @@ export function QuickReference({
onChange={(event) => setQuery(event.target.value)}
/>
-
+ {entry.explanation}
+
+
+
Flavour
+ {entry.flavours.join(", ")}
+
+
+
Relevant flags
+
+ {entry.flags.length > 0 ? entry.flags.join(", ") : "None"}
+
+
+
+
Example
+
+ {entry.example}
+
+
+
+
+ {entry.caveat ? {entry.caveat} : null}
+ onUsePattern(entry.example)}
+ >
+ Use example as pattern
+
+
+ ))}
+
+ ) : (
+
+ No reference constructs match “{query.trim()}”.
+
+ )}
);
}
diff --git a/src/components/ReplacementPanel.tsx b/src/components/ReplacementPanel.tsx
index 5ccc7f2..90845ce 100644
--- a/src/components/ReplacementPanel.tsx
+++ b/src/components/ReplacementPanel.tsx
@@ -9,6 +9,7 @@ import type {
ReplacementContributionPreview,
ReplacementPreview,
} from "../regex/replacement/replacement-preview";
+import type { RegexFlavourId } from "../regex/model/flavour";
import {
DEFAULT_REGEX_LIMITS,
TEMPLATE_PRESENTATION_LIMITS,
@@ -50,6 +51,7 @@ function ReplacementTokenRow({
}
export function ReplacementPanel({
+ flavour,
replacement,
onReplacementChange,
syntax,
@@ -61,6 +63,7 @@ export function ReplacementPanel({
onSelectToken,
onSelectContribution,
}: {
+ readonly flavour?: RegexFlavourId;
readonly replacement: string;
readonly onReplacementChange: (value: string) => void;
readonly syntax?: ReplacementSyntaxResult;
@@ -74,6 +77,9 @@ export function ReplacementPanel({
contribution: ReplacementContributionPreview,
) => void;
}) {
+ const activeFlavour =
+ flavour ?? result?.execution.engine.flavour ?? "ecmascript";
+ const engineLabel = activeFlavour === "pcre2" ? "PCRE2" : "ECMAScript";
const retainedTokens = syntax?.tokens ?? [];
const editorTokens = retainedTokens.slice(
0,
@@ -94,7 +100,7 @@ export function ReplacementPanel({
Flags
- {FLAG_OPTIONS.map(([flag, label]) => (
-
+ {flavourDefinition.flags.map((flag) => (
+
{
setFlags((current) => {
const next = new Set(current);
if (event.target.checked) {
- next.add(flag);
- if (flag === "u") next.delete("v");
- if (flag === "v") next.delete("u");
+ next.add(flag.value);
+ for (const group of flavourDefinition.mutuallyExclusiveFlags ??
+ []) {
+ if (!group.includes(flag.value)) continue;
+ for (const incompatible of group) {
+ if (incompatible !== flag.value) {
+ next.delete(incompatible);
+ }
+ }
+ }
} else {
- next.delete(flag);
+ next.delete(flag.value);
}
- return [...next];
+ return flavourDefinition.flags
+ .map((definition) => definition.value)
+ .filter((value) => next.has(value));
});
}}
/>
- {flag}
+ {flag.value}
))}
@@ -1041,6 +1044,9 @@ export function UnitTestPanel({
/{test.pattern}/{test.flags.join("")} ·{" "}
{expectationSummary(test.expectation)} ·{" "}
{test.subject.length.toLocaleString()} UTF-16 units
+ {test.generation
+ ? ` · generated · seed ${test.generation.seed}`
+ : ""}
diff --git a/src/components/Workbench.test.tsx b/src/components/Workbench.test.tsx
index 70bb77e..3ba48d5 100644
--- a/src/components/Workbench.test.tsx
+++ b/src/components/Workbench.test.tsx
@@ -240,6 +240,16 @@ describe("Workbench syntax coordination", () => {
fireEvent.click(screen.getByRole("checkbox", { name: "Live" }));
await advanceParseDebounce();
+ expect(syntaxHarness.patterns[0]?.request).toMatchObject({
+ flavour: "ecmascript",
+ flavourVersion: "2025",
+ flags: ["g", "u"],
+ options: {},
+ });
+ expect(
+ screen.getByRole("combobox", { name: "Flavour" }),
+ ).not.toBeDisabled();
+ expect(screen.getByRole("combobox", { name: "Version" })).toBeDisabled();
await resolvePattern(0, [
{
number: 1,
@@ -308,6 +318,9 @@ describe("Workbench syntax coordination", () => {
});
expect(engineHarness.requests).toHaveLength(1);
expect(engineHarness.requests[0]).toMatchObject({
+ flavour: "ecmascript",
+ flavourVersion: "ECMAScript 2025 syntax / current browser runtime",
+ options: {},
pattern: "(?b)",
captureMetadata: [{ number: 1, name: "new" }],
});
@@ -380,4 +393,47 @@ describe("Workbench syntax coordination", () => {
)?.cancelCalls,
).toBeGreaterThan(0);
});
+
+ it("opens the bounded subject minimizer with the current editor snapshot", () => {
+ vi.useFakeTimers();
+ render( );
+
+ fireEvent.click(screen.getByRole("button", { name: "Minimize" }));
+
+ expect(
+ screen.getByRole("heading", { name: "Minimize a reproducer" }),
+ ).toBeVisible();
+ expect(screen.getByLabelText("Subject to minimize")).toHaveValue(
+ "2026-07-24 alice\nInvalid row\n2025-12-31 Bérénice",
+ );
+ expect(screen.getByLabelText("Minimization failure kind")).toHaveValue(
+ "unit-test-failure",
+ );
+ fireEvent.click(
+ screen.getByRole("button", { name: "Close subject minimizer" }),
+ );
+ expect(
+ screen.queryByRole("heading", { name: "Minimize a reproducer" }),
+ ).not.toBeInTheDocument();
+ });
+
+ it("opens and closes the grammar-backed formatter workspace", () => {
+ vi.useFakeTimers();
+ render( );
+
+ fireEvent.click(screen.getByRole("button", { name: "Format" }));
+
+ expect(
+ screen.getByRole("heading", { name: "Format pattern" }),
+ ).toBeVisible();
+ expect(
+ screen.getByText(/Wait for the ECMAScript syntax provider/),
+ ).toBeVisible();
+ fireEvent.click(
+ screen.getByRole("button", { name: "Close pattern formatter" }),
+ );
+ expect(
+ screen.queryByRole("heading", { name: "Format pattern" }),
+ ).not.toBeInTheDocument();
+ });
});
diff --git a/src/components/Workbench.tsx b/src/components/Workbench.tsx
index 4658623..312ef10 100644
--- a/src/components/Workbench.tsx
+++ b/src/components/Workbench.tsx
@@ -9,6 +9,10 @@ import { EngineSupervisor } from "../regex/execution/EngineSupervisor";
import { SyntaxSupervisor } from "../regex/execution/SyntaxSupervisor";
import { WorkerRequestError } from "../regex/execution/WorkerSupervisor";
import type { RegexDiagnostic } from "../regex/model/diagnostics";
+import type {
+ RegexEngineOptions,
+ RegexFlavourId,
+} from "../regex/model/flavour";
import type {
ExtractionNode,
RegexExecutionRequest,
@@ -67,6 +71,20 @@ import { UnitTestPanel, type RegexTestDraft } from "./UnitTestPanel";
import { QuickReference } from "./QuickReference";
import { CapabilityPanel } from "./CapabilityPanel";
import { ProjectPanel } from "./ProjectPanel";
+import { ModalDialog } from "./ModalDialog";
+import { CorpusPanel } from "./CorpusPanel";
+import { Pcre2TracePanel } from "./Pcre2TracePanel";
+import { AnalysisPanel } from "./AnalysisPanel";
+import { ComparisonCodePanel } from "./ComparisonCodePanel";
+import { MinimizePanel } from "./MinimizePanel";
+import { GenerationPanel } from "./GenerationPanel";
+import { PatternFormatterPanel } from "./PatternFormatterPanel";
+import type { VerifiedGeneratedCase } from "../regex/generation/generation.types";
+import {
+ AVAILABLE_REGEX_FLAVOURS,
+ defaultRegexOptions,
+ resolveRegexFlavourVersion,
+} from "../regex/flavours/flavour-registry";
const INITIAL_PATTERN =
"(?\\d{4}-\\d{2}-\\d{2})\\s+(?[\\p{Letter}._-]+)";
@@ -77,16 +95,7 @@ const INITIAL_REPLACEMENT = "$ — $";
const MAXIMUM_PATTERN_EDITOR_MARKS = 2_000;
const MAXIMUM_SUBJECT_EDITOR_MARKS = 2_000;
const MAXIMUM_TEST_SUITE_WALL_TIME_MS = 60_000;
-const FLAG_OPTIONS = [
- ["g", "Global"],
- ["i", "Ignore case"],
- ["m", "Multiline"],
- ["s", "Dot all"],
- ["u", "Unicode"],
- ["v", "Unicode sets"],
- ["y", "Sticky"],
- ["d", "Expose indices"],
-] as const;
+const INITIAL_FLAVOUR = AVAILABLE_REGEX_FLAVOURS.require("ecmascript");
type RunState =
| { readonly status: "idle"; readonly message: string }
@@ -314,16 +323,23 @@ function testCaseFromDraft(
id: string,
enabled: boolean,
draft: RegexTestDraft,
+ configuration: {
+ readonly flavour: RegexFlavourId;
+ readonly flavourVersion?: string;
+ readonly options: RegexEngineOptions;
+ },
): RegexTestCase {
return {
id,
name: draft.name,
enabled,
- flavour: "ecmascript",
- flavourVersion: "ECMAScript 2025 syntax / current browser runtime",
+ flavour: configuration.flavour,
+ ...(configuration.flavourVersion === undefined
+ ? {}
+ : { flavourVersion: configuration.flavourVersion }),
pattern: draft.pattern,
flags: draft.flags,
- options: {},
+ options: configuration.options,
scanAll: draft.scanAll,
subject: draft.subject,
...(draft.replacement === undefined
@@ -333,6 +349,7 @@ function testCaseFromDraft(
...(draft.resourceLimits === undefined
? {}
: { resourceLimits: draft.resourceLimits }),
+ ...(draft.generation === undefined ? {} : { generation: draft.generation }),
};
}
@@ -361,8 +378,17 @@ function failedTestResult(testId: string, message: string): RegexTestResult {
}
export function Workbench() {
+ const [flavour, setFlavour] = useState(INITIAL_FLAVOUR.id);
+ const [flavourVersion, setFlavourVersion] = useState(
+ INITIAL_FLAVOUR.defaultVersion,
+ );
+ const [options, setOptions] = useState(() =>
+ defaultRegexOptions(INITIAL_FLAVOUR),
+ );
const [pattern, setPattern] = useState(INITIAL_PATTERN);
- const [flags, setFlags] = useState(["g", "u"]);
+ const [flags, setFlags] = useState(
+ INITIAL_FLAVOUR.defaultFlags,
+ );
const [subject, setSubject] = useState(INITIAL_SUBJECT);
const [replacement, setReplacement] = useState(INITIAL_REPLACEMENT);
const [listTemplate, setListTemplate] = useState("${user} — ${date}");
@@ -406,8 +432,15 @@ export function Workbench() {
const [testsRunning, setTestsRunning] = useState(false);
const [showReference, setShowReference] = useState(false);
const [showCapabilities, setShowCapabilities] = useState(false);
+ const [showTrace, setShowTrace] = useState(false);
+ const [showAnalysis, setShowAnalysis] = useState(false);
+ const [showComparison, setShowComparison] = useState(false);
+ const [showMinimizer, setShowMinimizer] = useState(false);
+ const [showGeneration, setShowGeneration] = useState(false);
+ const [showFormatter, setShowFormatter] = useState(false);
const [showProject, setShowProject] = useState(false);
const [showSubjectWhitespace, setShowSubjectWhitespace] = useState(false);
+ const [corpusSession, setCorpusSession] = useState(0);
const [subjectCursor, setSubjectCursor] = useState({ line: 1, column: 1 });
const [projectIdentity, setProjectIdentity] = useState(() => ({
id: projectId(),
@@ -423,6 +456,11 @@ export function Workbench() {
const executionRevision = useRef(0);
const testRunRevision = useRef(0);
const liveExecutionTimer = useRef(undefined);
+ const flavourDefinition = AVAILABLE_REGEX_FLAVOURS.require(flavour);
+ const activeTestConfiguration = useMemo(
+ () => ({ flavour, flavourVersion, options }),
+ [flavour, flavourVersion, options],
+ );
const patternSyntaxEngine = () => {
patternSyntaxSupervisor.current ??= new SyntaxSupervisor(
@@ -528,7 +566,22 @@ export function Workbench() {
[],
);
- const syntax = isPatternSyntaxForInput(syntaxSnapshot, pattern, flags)
+ const syntaxRequest = useMemo(() => {
+ const definition = AVAILABLE_REGEX_FLAVOURS.require(flavour);
+ const version = resolveRegexFlavourVersion(
+ flavourVersion,
+ definition,
+ "Selected flavour version",
+ ).definition;
+ return {
+ flavour,
+ flavourVersion: version.syntaxVersion,
+ pattern,
+ flags,
+ options,
+ };
+ }, [flags, flavour, flavourVersion, options, pattern]);
+ const syntax = isPatternSyntaxForInput(syntaxSnapshot, syntaxRequest)
? syntaxSnapshot.result
: undefined;
@@ -545,19 +598,12 @@ export function Workbench() {
return;
}
void patternSyntaxEngine()
- .parsePattern({
- flavour: "ecmascript",
- flavourVersion: "2025",
- pattern,
- flags,
- options: {},
- })
+ .parsePattern(syntaxRequest)
.then((result) => {
if (syntaxRevision.current !== currentRevision) return;
setSyntaxSnapshot({
revision: currentRevision,
- pattern,
- flags,
+ request: syntaxRequest,
result,
});
setSelectedSyntax((selected) =>
@@ -598,14 +644,14 @@ export function Workbench() {
});
}, DEFAULT_REGEX_LIMITS.liveParseDebounceMs);
return () => window.clearTimeout(timer);
- }, [clearExecutionResults, flags, pattern]);
+ }, [clearExecutionResults, pattern, syntaxRequest]);
useEffect(() => {
if (!syntax) return;
const currentRevision = ++replacementSyntaxRevision.current;
void replacementSyntaxEngine()
.parseReplacement({
- flavour: "ecmascript",
+ flavour,
replacement,
captureMetadata: syntax.captures,
})
@@ -622,22 +668,33 @@ export function Workbench() {
console.warn("Replacement parsing failed", error);
}
});
- }, [replacement, syntax]);
+ }, [flavour, replacement, syntax]);
const subjectBytes = useMemo(() => utf8ByteLength(subject), [subject]);
const makeRequest = useCallback(
(): RegexExecutionRequest => ({
- flavour: "ecmascript",
+ flavour,
+ flavourVersion,
pattern,
flags,
+ options,
subject,
captureMetadata: syntax?.captures ?? [],
scanAll,
maximumMatches: DEFAULT_REGEX_LIMITS.maximumMatches,
maximumCaptureRows: DEFAULT_REGEX_LIMITS.maximumCaptureRows,
}),
- [flags, pattern, scanAll, subject, syntax?.captures],
+ [
+ flags,
+ flavour,
+ flavourVersion,
+ options,
+ pattern,
+ scanAll,
+ subject,
+ syntax?.captures,
+ ],
);
const run = useCallback(
@@ -653,8 +710,7 @@ export function Workbench() {
!isPatternSyntaxCurrent(
syntaxSnapshot,
syntaxRevision.current,
- pattern,
- flags,
+ syntaxRequest,
)
) {
setRunState({
@@ -744,6 +800,7 @@ export function Workbench() {
"current-result-size-check",
true,
candidate,
+ activeTestConfiguration,
),
]);
return true;
@@ -827,6 +884,7 @@ export function Workbench() {
}
},
[
+ activeTestConfiguration,
clearExecutionResults,
makeRequest,
mode,
@@ -838,6 +896,7 @@ export function Workbench() {
syntax?.accepted,
syntax?.captures.length,
syntaxSnapshot,
+ syntaxRequest,
flags,
pattern,
tests,
@@ -855,6 +914,7 @@ export function Workbench() {
!liveWithinLimits ||
!syntax?.accepted ||
mode === "tests" ||
+ mode === "corpus" ||
(mode === "replace" && replacementSyntax?.accepted !== true)
) {
return;
@@ -935,11 +995,11 @@ export function Workbench() {
name: "Regex workbench",
createdAt: projectIdentity.createdAt,
updatedAt: new Date().toISOString(),
- flavour: "ecmascript",
- flavourVersion: "ECMAScript 2025 syntax / current browser runtime",
+ flavour,
+ flavourVersion,
pattern,
flags,
- options: {},
+ options,
replacement,
mode,
testText: { included: true, value: subject },
@@ -1171,6 +1231,9 @@ export function Workbench() {
}
const parsed = await testSyntaxEngine().parsePattern({
flavour: test.flavour,
+ ...(test.flavourVersion === undefined
+ ? {}
+ : { flavourVersion: test.flavourVersion }),
pattern: test.pattern,
flags: test.flags,
options: test.options,
@@ -1250,14 +1313,55 @@ export function Workbench() {
`A suite is limited to ${MAXIMUM_REGEX_TESTS.toLocaleString()} tests.`,
);
}
- const next = [...tests, testCaseFromDraft(projectId(), true, draft)];
+ const next = [
+ ...tests,
+ testCaseFromDraft(projectId(), true, draft, activeTestConfiguration),
+ ];
+ assertRegexTestSuiteWithinLimit(next);
+ setTests(next);
+ };
+
+ const addGeneratedTests = (
+ generatedCases: readonly VerifiedGeneratedCase[],
+ ) => {
+ if (generatedCases.length === 0) return;
+ if (tests.length + generatedCases.length > MAXIMUM_REGEX_TESTS) {
+ throw new Error(
+ `Adding ${generatedCases.length.toLocaleString()} generated cases would exceed the ${MAXIMUM_REGEX_TESTS.toLocaleString()} test limit.`,
+ );
+ }
+ const appended = generatedCases.map((candidate) =>
+ testCaseFromDraft(
+ projectId(),
+ true,
+ {
+ name: candidate.name.slice(0, MAXIMUM_REGEX_TEST_NAME_UTF16),
+ pattern,
+ flags,
+ scanAll,
+ subject: candidate.subject,
+ expectation: { kind: candidate.expectation },
+ generation: candidate.provenance,
+ },
+ activeTestConfiguration,
+ ),
+ );
+ const next = [...tests, ...appended];
assertRegexTestSuiteWithinLimit(next);
setTests(next);
};
const updateTest = (id: string, draft: RegexTestDraft) => {
const next = tests.map((test) =>
- test.id === id ? testCaseFromDraft(test.id, test.enabled, draft) : test,
+ test.id === id
+ ? testCaseFromDraft(test.id, test.enabled, draft, {
+ flavour: test.flavour,
+ ...(test.flavourVersion === undefined
+ ? {}
+ : { flavourVersion: test.flavourVersion }),
+ options: test.options,
+ })
+ : test,
);
assertRegexTestSuiteWithinLimit(next);
setTests(next);
@@ -1308,10 +1412,20 @@ export function Workbench() {
};
const importProject = (project: RegexProjectV1) => {
+ const importedFlavour = AVAILABLE_REGEX_FLAVOURS.require(project.flavour);
invalidatePatternSyntax();
executionRevision.current += 1;
engineSupervisor.current?.cancel();
cancelTestRun();
+ setFlavour(importedFlavour.id);
+ setFlavourVersion(
+ resolveRegexFlavourVersion(
+ project.flavourVersion,
+ importedFlavour,
+ "Project flavour version",
+ ).definition.value,
+ );
+ setOptions(project.options);
setPattern(project.pattern);
setProjectIdentity({ id: project.id, createdAt: project.createdAt });
setFlags(project.flags);
@@ -1322,6 +1436,7 @@ export function Workbench() {
setListTemplate(project.listTemplate?.source ?? "$0");
setMode(project.mode);
setTests(project.tests);
+ setCorpusSession((current) => current + 1);
setCurrentResultDraft(undefined);
setTestResults(new Map());
setLive(false);
@@ -1336,51 +1451,142 @@ export function Workbench() {
});
};
+ const changeFlavour = (value: string) => {
+ const next = AVAILABLE_REGEX_FLAVOURS.parse(value, "Selected flavour");
+ if (next.id === flavour) return;
+ invalidateExecution();
+ invalidatePatternSyntax();
+ cancelTestRun();
+ setFlavour(next.id);
+ setFlavourVersion(next.defaultVersion);
+ setOptions(defaultRegexOptions(next));
+ setFlags(next.defaultFlags);
+ };
+
+ const changeFlavourVersion = (value: string) => {
+ if (
+ value === flavourVersion ||
+ !flavourDefinition.versions.some((version) => version.value === value)
+ ) {
+ return;
+ }
+ invalidateExecution();
+ invalidatePatternSyntax();
+ setFlavourVersion(value);
+ };
+
+ const changeEngineOption = (name: string, value: number) => {
+ const definition = flavourDefinition.options.find(
+ (candidate) => candidate.name === name,
+ );
+ if (
+ !definition ||
+ definition.kind !== "integer" ||
+ !Number.isSafeInteger(value) ||
+ value < definition.minimum ||
+ value > definition.maximum ||
+ options[name] === value
+ ) {
+ return;
+ }
+ invalidateExecution();
+ invalidatePatternSyntax();
+ setOptions((current) => ({ ...current, [name]: value }));
+ };
+
return (
Flavour
-
- ECMAScript
+ changeFlavour(event.target.value)}
+ >
+ {AVAILABLE_REGEX_FLAVOURS.definitions.map((definition) => (
+
+ {definition.label}
+
+ ))}
Version
-
- 2025 syntax · current browser
+ changeFlavourVersion(event.target.value)}
+ >
+ {flavourDefinition.versions.map((version) => (
+
+ {version.label}
+
+ ))}
Flags
- {FLAG_OPTIONS.map(([flag, label]) => (
-
+ {flavourDefinition.flags.map((flag) => (
+
{
invalidateExecution();
invalidatePatternSyntax();
setFlags((current) => {
const next = new Set(current);
if (event.target.checked) {
- next.add(flag);
- if (flag === "u") next.delete("v");
- if (flag === "v") next.delete("u");
+ next.add(flag.value);
+ for (const group of flavourDefinition.mutuallyExclusiveFlags ??
+ []) {
+ if (!group.includes(flag.value)) continue;
+ for (const incompatible of group) {
+ if (incompatible !== flag.value) {
+ next.delete(incompatible);
+ }
+ }
+ }
} else {
- next.delete(flag);
+ next.delete(flag.value);
}
- return FLAG_OPTIONS.map(([value]) => value).filter(
- (value) => next.has(value),
- );
+ return flavourDefinition.flags
+ .map((definition) => definition.value)
+ .filter((value) => next.has(value));
});
}}
/>
- {flag}
+ {flag.value}
))}
+ {flavourDefinition.options.length > 0 ? (
+
+ Engine limits
+ {flavourDefinition.options.map((option) => (
+
+ {option.label}
+ {option.kind === "integer" ? (
+
+ changeEngineOption(
+ option.name,
+ Number(event.target.value),
+ )
+ }
+ />
+ ) : null}
+
+ ))}
+
+ ) : null}
flag === "g" || flag === "y") ? (
- Scan all is explicit: the worker adds an internal g only
- for iteration. The pattern’s saved user flags remain unchanged.
+ Scan all is explicit: the worker adds an internal global-iteration
+ flag only for this request. The pattern’s saved user flags remain
+ unchanged.
) : null}
@@ -1462,6 +1670,7 @@ export function Workbench() {
["match", "Match"],
["replace", "Replace"],
["list", "List"],
+ ["corpus", "Corpus / Apply"],
["tests", "Unit tests"],
] as const
).map(([value, label]) => (
@@ -1481,11 +1690,77 @@ export function Workbench() {
))}
+ setShowGeneration(true)}
+ aria-expanded={showGeneration}
+ aria-controls="generation-dialog"
+ >
+ Generate cases
+
+ setShowFormatter(true)}
+ aria-expanded={showFormatter}
+ aria-controls="formatter-dialog"
+ title={
+ flavour === "ecmascript"
+ ? "Preview canonical literal/control escaping and validate it against exact snapshots"
+ : "Open the formatter to see current flavour availability"
+ }
+ >
+ Format
+
+ setShowMinimizer(true)}
+ aria-expanded={showMinimizer}
+ aria-controls="minimize-dialog"
+ >
+ Minimize
+
+ setShowComparison(true)}
+ aria-expanded={showComparison}
+ aria-controls="comparison-code-dialog"
+ >
+ Compare & code
+
+ setShowAnalysis(true)}
+ aria-expanded={showAnalysis}
+ aria-controls="analysis-dialog"
+ >
+ Analysis
+
+ setShowTrace(true)}
+ aria-expanded={showTrace}
+ aria-controls="trace-dialog"
+ >
+ PCRE trace
+
setShowReference((value) => !value)}
aria-expanded={showReference}
+ aria-controls="quick-reference-dialog"
>
Quick reference
@@ -1494,6 +1769,7 @@ export function Workbench() {
className="secondary-button"
onClick={() => setShowCapabilities((value) => !value)}
aria-expanded={showCapabilities}
+ aria-controls="capabilities-dialog"
>
Capabilities
@@ -1526,6 +1802,199 @@ export function Workbench() {
}}
/>
+
setShowGeneration(false)}
+ className="generation-dialog"
+ >
+ {showGeneration ? (
+ {
+ const node = syntax
+ ? findSmallestNode(syntax.root, range)
+ : undefined;
+ if (node) {
+ selectSyntaxNode(node);
+ } else {
+ setRequestedPatternRange(range);
+ }
+ setShowGeneration(false);
+ }}
+ onClose={() => setShowGeneration(false)}
+ />
+ ) : null}
+
+
setShowFormatter(false)}
+ className="formatter-dialog"
+ >
+ {showFormatter ? (
+ {
+ changePattern(value);
+ setShowFormatter(false);
+ }}
+ onClose={() => setShowFormatter(false)}
+ />
+ ) : null}
+
+
setShowMinimizer(false)}
+ className="minimize-dialog"
+ >
+ {showMinimizer ? (
+ {
+ changeSubject(value);
+ setShowMinimizer(false);
+ }}
+ onClose={() => setShowMinimizer(false)}
+ />
+ ) : null}
+
+
setShowComparison(false)}
+ className="comparison-code-dialog"
+ >
+ {showComparison ? (
+ setShowComparison(false)}
+ />
+ ) : null}
+
+
setShowReference(false)}
+ className="reference-dialog"
+ >
+ {
+ changePattern(example);
+ setShowReference(false);
+ }}
+ onClose={() => setShowReference(false)}
+ />
+
+
setShowCapabilities(false)}
+ className="capabilities-dialog"
+ >
+ setShowCapabilities(false)}
+ />
+
+
setShowTrace(false)}
+ className="trace-dialog"
+ >
+ setShowTrace(false)}
+ onSelectRanges={(patternRange, subjectRange) => {
+ setSelectedSyntax(undefined);
+ setSelectedExtraction(undefined);
+ setSelectedCaptureRow(undefined);
+ setRequestedPatternRange(patternRange);
+ setRequestedSubjectRange(subjectRange);
+ setShowTrace(false);
+ }}
+ />
+
+
setShowAnalysis(false)}
+ className="analysis-dialog"
+ >
+ setShowAnalysis(false)}
+ onSelectPatternRange={(range) => {
+ const node = syntax
+ ? findSmallestNode(syntax.root, range)
+ : undefined;
+ if (node) {
+ selectSyntaxNode(node);
+ } else {
+ setRequestedPatternRange(range);
+ }
+ setShowAnalysis(false);
+ }}
+ />
+
selectSyntaxNode(node)}
/>
- {mode !== "tests" ? (
+ {mode !== "tests" && mode !== "corpus" ? (
@@ -1715,6 +2184,7 @@ export function Workbench() {
) : null}
{mode === "replace" ? (
) : null}
+
+
+
{mode === "tests" ? (
- {showReference ? : null}
- {showCapabilities ? (
-
- ) : null}
);
}
diff --git a/src/project/project.test.ts b/src/project/project.test.ts
index 59c068c..4685365 100644
--- a/src/project/project.test.ts
+++ b/src/project/project.test.ts
@@ -5,6 +5,11 @@ import {
} from "./project.serialization";
import type { RegexProjectV1 } from "./project.types";
import { DEFAULT_REGEX_LIMITS } from "../regex/execution/request-limits";
+import { parseRegexProject, parseRegexTests } from "./project.validation";
+import {
+ RegexFlavourRegistry,
+ type RegexFlavourDefinition,
+} from "../regex/flavours/flavour-registry";
const project: RegexProjectV1 = {
schemaVersion: 1,
@@ -58,6 +63,55 @@ describe("project import and export", () => {
expect(serialized).toContain('"subject": "42"');
});
+ it("retains validated generation provenance when private subjects are omitted", () => {
+ const generatedTest = {
+ ...project.tests[0]!,
+ generation: {
+ kind: "generated" as const,
+ generatorId: "regex-tools-ast-cases" as const,
+ generatorVersion: "1" as const,
+ seed: "project-seed-17",
+ candidateId: "alternative:root:1",
+ intendedOutcome: "match" as const,
+ },
+ };
+ const imported = importProjectDocument(
+ serializeProject(
+ { ...project, tests: [generatedTest] },
+ {
+ includeTestText: false,
+ includeUnitTestSubjects: false,
+ },
+ ),
+ );
+
+ expect(imported.tests[0]?.subject).toBe("");
+ expect(imported.tests[0]?.generation).toEqual(generatedTest.generation);
+ });
+
+ it("persists corpus mode without accepting ephemeral corpus content", () => {
+ const imported = importProjectDocument(
+ serializeProject(
+ { ...project, mode: "corpus" },
+ {
+ includeTestText: false,
+ includeUnitTestSubjects: false,
+ },
+ ),
+ );
+ expect(imported.mode).toBe("corpus");
+
+ expect(() =>
+ importProjectDocument(
+ JSON.stringify({
+ ...project,
+ mode: "corpus",
+ corpus: [{ name: "secret.txt", text: "secret" }],
+ }),
+ ),
+ ).toThrow(/ephemeral/iu);
+ });
+
it("preflights aggregate export size after applying privacy options", () => {
const escapedSubject = "\0".repeat(2 * 1024 * 1024);
const aggregateProject: RegexProjectV1 = {
@@ -171,10 +225,39 @@ describe("project import and export", () => {
).toThrow(/Test 1 subject exceeds.*UTF-8 byte limit/u);
});
- it("rejects unsupported flavour and option semantics instead of rewriting them", () => {
+ it("accepts PCRE2 and rejects unsupported flavour or option semantics without rewriting", () => {
+ expect(
+ importProjectDocument(
+ JSON.stringify({
+ ...project,
+ flavour: "pcre2",
+ flavourVersion: "PCRE2 10.47",
+ flags: ["g"],
+ options: {
+ matchLimit: 50_000,
+ depthLimit: 500,
+ heapLimitKib: 4_096,
+ },
+ tests: [],
+ }),
+ ),
+ ).toEqual(
+ expect.objectContaining({
+ flavour: "pcre2",
+ flavourVersion: "PCRE2 10.47",
+ flags: ["g"],
+ options: {
+ matchLimit: 50_000,
+ depthLimit: 500,
+ heapLimitKib: 4_096,
+ },
+ }),
+ );
expect(() =>
- importProjectDocument(JSON.stringify({ ...project, flavour: "pcre2" })),
- ).toThrow(/only supports ECMAScript/u);
+ importProjectDocument(
+ JSON.stringify({ ...project, flavourVersion: "ECMAScript 2026" }),
+ ),
+ ).toThrow(/flavour version.*unsupported/iu);
expect(() =>
importProjectDocument(
JSON.stringify({
@@ -182,7 +265,20 @@ describe("project import and export", () => {
tests: [{ ...project.tests[0], flavour: "python" }],
}),
),
- ).toThrow(/only supports ECMAScript/u);
+ ).toThrow(/only supports ECMAScript and PCRE2/u);
+ expect(() =>
+ importProjectDocument(
+ JSON.stringify({
+ ...project,
+ tests: [
+ {
+ ...project.tests[0],
+ flavourVersion: "ECMAScript 2026",
+ },
+ ],
+ }),
+ ),
+ ).toThrow(/flavour version.*unsupported/iu);
expect(() =>
importProjectDocument(
JSON.stringify({ ...project, options: { ignoreCase: true } }),
@@ -214,6 +310,84 @@ describe("project import and export", () => {
).toThrow(/test id.*duplicated/iu);
});
+ it("keeps schema v1 while delegating future flavour fields to a contract", () => {
+ const fixtureFlavour: RegexFlavourDefinition = {
+ id: "pcre2",
+ label: "Fixture flavour",
+ versions: [
+ {
+ value: "fixture-version",
+ label: "Fixture version",
+ syntaxVersion: "fixture-version",
+ },
+ ],
+ defaultVersion: "fixture-version",
+ flags: [{ value: "i", label: "Caseless" }],
+ defaultFlags: [],
+ options: [
+ {
+ name: "ungreedy",
+ label: "Ungreedy",
+ kind: "boolean",
+ defaultValue: false,
+ },
+ ],
+ };
+ const registry = new RegexFlavourRegistry([fixtureFlavour]);
+ const imported = parseRegexProject(
+ {
+ ...project,
+ flavour: "pcre2",
+ flavourVersion: "fixture-version",
+ flags: ["i"],
+ options: { ungreedy: true },
+ tests: [],
+ },
+ registry,
+ );
+ const importedTests = parseRegexTests(
+ [
+ {
+ ...project.tests[0],
+ flavour: "pcre2",
+ flavourVersion: "fixture-version",
+ flags: ["i"],
+ options: { ungreedy: true },
+ },
+ ],
+ registry,
+ );
+
+ expect(imported).toEqual(
+ expect.objectContaining({
+ schemaVersion: 1,
+ flavour: "pcre2",
+ flavourVersion: "fixture-version",
+ flags: ["i"],
+ options: { ungreedy: true },
+ }),
+ );
+ expect(importedTests[0]).toEqual(
+ expect.objectContaining({
+ flavour: "pcre2",
+ flavourVersion: "fixture-version",
+ flags: ["i"],
+ options: { ungreedy: true },
+ }),
+ );
+ });
+
+ it("imports legacy schema-v1 projects without a flavour-version field", () => {
+ const imported = parseRegexProject(project);
+
+ expect(imported.schemaVersion).toBe(1);
+ expect(imported.flavour).toBe("ecmascript");
+ expect(imported.flavourVersion).toBe(
+ "ECMAScript 2025 syntax / current browser runtime",
+ );
+ expect(imported.tests[0]?.flavourVersion).toBeUndefined();
+ });
+
it("validates and preserves the supported per-test timeout limit", () => {
const imported = importProjectDocument(
JSON.stringify({
diff --git a/src/project/project.types.ts b/src/project/project.types.ts
index ecfc74d..55523c1 100644
--- a/src/project/project.types.ts
+++ b/src/project/project.types.ts
@@ -1,9 +1,12 @@
-import type { RegexFlavourId } from "../regex/model/flavour";
+import type {
+ RegexEngineOptions,
+ RegexFlavourId,
+} from "../regex/model/flavour";
import type { RegexResourceLimits } from "../regex/execution/request-limits";
import type { ListTemplate } from "../regex/list/list-template";
import type { RegexTestCase } from "../regex/tests/test-case.types";
-export type WorkspaceMode = "match" | "replace" | "list" | "tests";
+export type WorkspaceMode = "match" | "replace" | "list" | "corpus" | "tests";
export interface RegexProjectV1 {
readonly schemaVersion: 1;
@@ -15,7 +18,7 @@ export interface RegexProjectV1 {
readonly flavourVersion?: string;
readonly pattern: string;
readonly flags: readonly string[];
- readonly options: Readonly>;
+ readonly options: RegexEngineOptions;
readonly replacement?: string;
readonly mode: WorkspaceMode;
readonly testText?: {
diff --git a/src/project/project.validation.ts b/src/project/project.validation.ts
index b193656..3b4b67a 100644
--- a/src/project/project.validation.ts
+++ b/src/project/project.validation.ts
@@ -17,14 +17,22 @@ import {
utf8ByteLength,
} from "../regex/execution/request-limits";
import { assertProjectDocumentWithinLimit } from "./project-limits";
+import {
+ AVAILABLE_REGEX_FLAVOURS,
+ parseRegexFlags,
+ parseRegexOptions,
+ resolveRegexFlavourVersion,
+ type RegexFlavourRegistry,
+} from "../regex/flavours/flavour-registry";
+import { parseGeneratedCaseProvenance } from "../regex/generation/generation-provenance";
const IMPLEMENTED_MODES = new Set([
"match",
"replace",
"list",
+ "corpus",
"tests",
]);
-const VALID_FLAGS = new Set("dgimsuvy".split(""));
const MAXIMUM_PATTERN_UTF16 = MAXIMUM_REGEX_TEST_PATTERN_UTF16;
const MAXIMUM_SUBJECT_UTF16 = MAXIMUM_REGEX_TEST_SUBJECT_UTF16;
@@ -58,44 +66,6 @@ function subjectValue(value: unknown, label: string): string {
return subject;
}
-function parseFlags(value: unknown): readonly string[] {
- if (!Array.isArray(value)) throw new Error("Project flags must be an array.");
- const flags = value.map((flag) => stringValue(flag, "Flag", 1));
- if (
- new Set(flags).size !== flags.length ||
- flags.some((flag) => !VALID_FLAGS.has(flag))
- ) {
- throw new Error("Project flags contain duplicates or unsupported values.");
- }
- if (flags.includes("u") && flags.includes("v")) {
- throw new Error("ECMAScript flags u and v cannot be combined.");
- }
- return flags;
-}
-
-function parseFlavour(value: unknown, label: string): "ecmascript" {
- const flavour = stringValue(value, label, 32);
- if (flavour !== "ecmascript") {
- throw new Error(
- `${label} ${JSON.stringify(flavour)} is unsupported; version 0.1.0 only supports ECMAScript.`,
- );
- }
- return flavour;
-}
-
-function parseOptions(
- value: unknown,
- label: string,
-): Readonly> {
- const options = value === undefined ? {} : objectValue(value, label);
- if (Object.keys(options).length > 0) {
- throw new Error(
- `${label} contains unsupported entries; version 0.1.0 does not implement ECMAScript engine options.`,
- );
- }
- return {};
-}
-
function parseTestResourceLimits(
value: unknown,
label: string,
@@ -125,7 +95,10 @@ function parseTestResourceLimits(
return { manualExecutionTimeoutMs: timeout };
}
-export function parseRegexTests(value: unknown): readonly RegexTestCase[] {
+export function parseRegexTests(
+ value: unknown,
+ registry: RegexFlavourRegistry = AVAILABLE_REGEX_FLAVOURS,
+): readonly RegexTestCase[] {
if (!Array.isArray(value)) throw new Error("Project tests must be an array.");
if (value.length > MAXIMUM_REGEX_TESTS) {
throw new Error(`Project contains more than ${MAXIMUM_REGEX_TESTS} tests.`);
@@ -133,8 +106,22 @@ export function parseRegexTests(value: unknown): readonly RegexTestCase[] {
const testIds = new Set();
return value.map((candidate, index) => {
const test = objectValue(candidate, `Test ${index + 1}`);
- const flavour = parseFlavour(test.flavour, `Test ${index + 1} flavour`);
- const options = parseOptions(test.options, `Test ${index + 1} options`);
+ const flavour = registry.parse(test.flavour, `Test ${index + 1} flavour`);
+ const flags = parseRegexFlags(
+ test.flags,
+ flavour,
+ `Test ${index + 1} flags`,
+ );
+ const options = parseRegexOptions(
+ test.options,
+ flavour,
+ `Test ${index + 1} options`,
+ );
+ const flavourVersion = resolveRegexFlavourVersion(
+ test.flavourVersion,
+ flavour,
+ `Test ${index + 1} flavour version`,
+ ).value;
const resourceLimits = parseTestResourceLimits(
test.resourceLimits,
`Test ${index + 1} resource limits`,
@@ -261,6 +248,29 @@ export function parseRegexTests(value: unknown): readonly RegexTestCase[] {
`Imported test ${index + 1} uses unsupported expectation ${kind}.`,
);
}
+ const generation =
+ test.generation === undefined
+ ? undefined
+ : parseGeneratedCaseProvenance(
+ test.generation,
+ `Test ${index + 1} generation`,
+ );
+ if (generation) {
+ if (flavour.id !== "ecmascript") {
+ throw new Error(
+ `Test ${index + 1} generation provenance is supported only for ECMAScript generator v1.`,
+ );
+ }
+ const expectedKind =
+ generation.intendedOutcome === "match"
+ ? "should-match"
+ : "should-not-match";
+ if (parsedExpectation.kind !== expectedKind) {
+ throw new Error(
+ `Test ${index + 1} generation intendedOutcome does not match its ${parsedExpectation.kind} expectation.`,
+ );
+ }
+ }
const id = stringValue(test.id, `Test ${index + 1} id`, 128);
if (id.length === 0) {
throw new Error(`Test ${index + 1} id must not be empty.`);
@@ -277,22 +287,14 @@ export function parseRegexTests(value: unknown): readonly RegexTestCase[] {
MAXIMUM_REGEX_TEST_NAME_UTF16,
),
enabled: test.enabled !== false,
- flavour,
- ...(test.flavourVersion === undefined
- ? {}
- : {
- flavourVersion: stringValue(
- test.flavourVersion,
- `Test ${index + 1} flavour version`,
- 128,
- ),
- }),
+ flavour: flavour.id,
+ ...(test.flavourVersion === undefined ? {} : { flavourVersion }),
pattern: stringValue(
test.pattern,
`Test ${index + 1} pattern`,
MAXIMUM_REGEX_TEST_PATTERN_UTF16,
),
- flags: parseFlags(test.flags),
+ flags,
options,
...(test.scanAll === undefined
? {}
@@ -313,28 +315,44 @@ export function parseRegexTests(value: unknown): readonly RegexTestCase[] {
}),
...(resourceLimits === undefined ? {} : { resourceLimits }),
expectation: parsedExpectation,
+ ...(generation === undefined ? {} : { generation }),
};
});
}
-export function parseRegexProject(value: unknown): RegexProjectV1 {
+export function parseRegexProject(
+ value: unknown,
+ registry: RegexFlavourRegistry = AVAILABLE_REGEX_FLAVOURS,
+): RegexProjectV1 {
const input = objectValue(value);
if (input.schemaVersion !== 1) {
throw new Error("Only regex-tools project schema version 1 is supported.");
}
- const flavour = parseFlavour(input.flavour, "Project flavour");
- const options = parseOptions(input.options, "Project options");
+ const flavour = registry.parse(input.flavour, "Project flavour");
+ const flavourVersion = resolveRegexFlavourVersion(
+ input.flavourVersion,
+ flavour,
+ "Project flavour version",
+ ).value;
+ const flags = parseRegexFlags(input.flags, flavour, "Project flags");
+ const options = parseRegexOptions(input.options, flavour, "Project options");
if (input.resourceOverrides !== undefined) {
const overrides = objectValue(
input.resourceOverrides,
"Project resource overrides",
);
if (Object.keys(overrides).length > 0) {
- throw new Error(
- "Project resource overrides are not implemented in version 0.1.0.",
- );
+ throw new Error("Project resource overrides are not implemented.");
}
}
+ const corpusFields = Object.keys(input).filter((key) =>
+ key.toLocaleLowerCase("en-US").startsWith("corpus"),
+ );
+ if (corpusFields.length > 0) {
+ throw new Error(
+ `Corpus documents and results are ephemeral and cannot be imported from a schema-v1 project (${corpusFields.join(", ")}).`,
+ );
+ }
const mode = stringValue(input.mode, "Project mode", 16) as WorkspaceMode;
if (!IMPLEMENTED_MODES.has(mode)) {
throw new Error(`Project mode ${mode} is not implemented in this release.`);
@@ -358,10 +376,10 @@ export function parseRegexProject(value: unknown): RegexProjectV1 {
typeof input.createdAt === "string" ? input.createdAt.slice(0, 64) : now,
updatedAt:
typeof input.updatedAt === "string" ? input.updatedAt.slice(0, 64) : now,
- flavour,
- flavourVersion: "ECMAScript 2025 syntax / current browser runtime",
+ flavour: flavour.id,
+ flavourVersion,
pattern: stringValue(input.pattern, "Pattern", MAXIMUM_PATTERN_UTF16),
- flags: parseFlags(input.flags),
+ flags,
options,
replacement:
input.replacement === undefined
@@ -381,7 +399,7 @@ export function parseRegexProject(value: unknown): RegexProjectV1 {
: {}),
},
listTemplate: parseListTemplate(listSource),
- tests: parseRegexTests(input.tests ?? []),
+ tests: parseRegexTests(input.tests ?? [], registry),
ui: {
live:
typeof input.ui === "object" &&
diff --git a/src/regex/analysis/AnalysisSupervisor.test.ts b/src/regex/analysis/AnalysisSupervisor.test.ts
new file mode 100644
index 0000000..f242cf9
--- /dev/null
+++ b/src/regex/analysis/AnalysisSupervisor.test.ts
@@ -0,0 +1,161 @@
+import { describe, expect, it } from "vitest";
+import type {
+ WorkerLike,
+ WorkerRequestError,
+} from "../execution/WorkerSupervisor";
+import {
+ WORKER_PROTOCOL_VERSION,
+ type WorkerRequest,
+ type WorkerResponse,
+} from "../execution/worker-protocol";
+import { AnalysisSupervisor } from "./AnalysisSupervisor";
+import type {
+ AnalysisWorkerOperation,
+ AnalysisWorkerResult,
+} from "./analysis.types";
+
+class ResponsiveWorker implements WorkerLike {
+ onmessage: ((event: MessageEvent) => void) | null = null;
+ onerror: ((event: ErrorEvent) => void) | null = null;
+ onmessageerror: ((event: MessageEvent) => void) | null = null;
+ terminated = false;
+
+ postMessage(value: unknown): void {
+ const request = value as WorkerRequest;
+ let payload: AnalysisWorkerResult;
+ if (request.payload.kind === "identity") {
+ payload = {
+ kind: "identity",
+ result: {
+ flavour: "ecmascript",
+ engineName: "Native ECMAScript RegExp",
+ engineVersion: "Fixture runtime",
+ runtimeVersion: "Fixture runtime",
+ nativeOffsetUnit: "utf16",
+ },
+ };
+ } else if (request.payload.kind === "benchmark-sample") {
+ payload = {
+ kind: "benchmark-sample",
+ result: {
+ accepted: true,
+ effectiveFlags: "dg",
+ subjectBytes: 1,
+ subjectUtf16: 1,
+ compileMs: 1,
+ firstMatchMs: 1,
+ allMatchesMs: 1,
+ replacementMs: 1,
+ throughputBytesPerSecond: 1_000,
+ matchCount: 1,
+ matched: true,
+ matchCollectionTruncated: false,
+ replacementOutputUtf16: 1,
+ },
+ };
+ } else {
+ payload = {
+ kind: "growth-probe",
+ result: {
+ accepted: true,
+ effectiveFlags: "d",
+ subjectBytes: 1,
+ subjectUtf16: 1,
+ executionMs: 1,
+ matchCount: 0,
+ matched: false,
+ matchCollectionTruncated: false,
+ },
+ };
+ }
+ const response: WorkerResponse = {
+ protocolVersion: WORKER_PROTOCOL_VERSION,
+ requestId: request.requestId,
+ generation: request.generation,
+ ok: true,
+ payload,
+ };
+ globalThis.queueMicrotask(() => {
+ if (!this.terminated) {
+ this.onmessage?.(new MessageEvent("message", { data: response }));
+ }
+ });
+ }
+
+ terminate(): void {
+ this.terminated = true;
+ }
+}
+
+class SilentWorker implements WorkerLike {
+ onmessage: ((event: MessageEvent) => void) | null = null;
+ onerror: ((event: ErrorEvent) => void) | null = null;
+ onmessageerror: ((event: MessageEvent) => void) | null = null;
+ terminated = false;
+
+ postMessage(): void {}
+
+ terminate(): void {
+ this.terminated = true;
+ }
+}
+
+describe("AnalysisSupervisor", () => {
+ it("routes typed identity, benchmark and growth responses", async () => {
+ const created: ResponsiveWorker[] = [];
+ const supervisor = new AnalysisSupervisor(() => {
+ const next = new ResponsiveWorker();
+ created.push(next);
+ return next;
+ });
+
+ await expect(supervisor.identity(100)).resolves.toEqual(
+ expect.objectContaining({ engineVersion: "Fixture runtime" }),
+ );
+ await expect(
+ supervisor.benchmarkSample(
+ {
+ flavour: "ecmascript",
+ pattern: "a",
+ flags: [],
+ subject: "a",
+ replacement: "x",
+ scanAll: false,
+ maximumMatches: 10,
+ },
+ 100,
+ ),
+ ).resolves.toEqual(expect.objectContaining({ compileMs: 1 }));
+ await expect(
+ supervisor.growthProbe(
+ {
+ flavour: "ecmascript",
+ pattern: "a",
+ flags: [],
+ subject: "b",
+ scanAll: false,
+ maximumMatches: 10,
+ },
+ 100,
+ ),
+ ).resolves.toEqual(expect.objectContaining({ matched: false }));
+
+ expect(created).toHaveLength(1);
+ supervisor.dispose();
+ expect(created[0]?.terminated).toBe(true);
+ });
+
+ it("terminates and rejects an active request on cancellation", async () => {
+ const worker = new SilentWorker();
+ const supervisor = new AnalysisSupervisor(() => worker);
+ const pending = supervisor.identity(1_000);
+
+ supervisor.cancel();
+
+ await expect(pending).rejects.toMatchObject({
+ kind: "cancelled",
+ } satisfies Partial);
+ expect(worker.terminated).toBe(true);
+ supervisor.dispose();
+ });
+});
diff --git a/src/regex/analysis/AnalysisSupervisor.ts b/src/regex/analysis/AnalysisSupervisor.ts
new file mode 100644
index 0000000..376d969
--- /dev/null
+++ b/src/regex/analysis/AnalysisSupervisor.ts
@@ -0,0 +1,94 @@
+import {
+ WorkerSupervisor,
+ type WorkerFactory,
+} from "../execution/WorkerSupervisor";
+import type {
+ AnalysisEngineIdentity,
+ AnalysisWorkerOperation,
+ AnalysisWorkerResult,
+ BenchmarkSubject,
+ BenchmarkWorkerSample,
+ GrowthProbeSubject,
+ GrowthWorkerSample,
+} from "./analysis.types";
+
+export interface AnalysisWorkerClient {
+ identity(timeoutMs: number): Promise;
+ benchmarkSample(
+ request: BenchmarkSubject,
+ timeoutMs: number,
+ ): Promise;
+ growthProbe(
+ request: GrowthProbeSubject,
+ timeoutMs: number,
+ ): Promise;
+ cancel(): void;
+ dispose(): void;
+}
+
+const createAnalysisWorker: WorkerFactory = () =>
+ new Worker(new URL("../../workers/analysis.worker.ts", import.meta.url), {
+ type: "module",
+ name: "regex-tools-ecmascript-analysis",
+ });
+
+export class AnalysisSupervisor implements AnalysisWorkerClient {
+ readonly #supervisor: WorkerSupervisor<
+ AnalysisWorkerOperation,
+ AnalysisWorkerResult
+ >;
+
+ constructor(workerFactory: WorkerFactory = createAnalysisWorker) {
+ this.#supervisor = new WorkerSupervisor(
+ "ECMAScript analysis worker",
+ workerFactory,
+ );
+ }
+
+ async identity(timeoutMs: number): Promise {
+ const response = await this.#supervisor.run(
+ { kind: "identity" },
+ timeoutMs,
+ );
+ if (response.kind !== "identity") {
+ throw new Error("Analysis worker returned the wrong identity response.");
+ }
+ return response.result;
+ }
+
+ async benchmarkSample(
+ request: BenchmarkSubject,
+ timeoutMs: number,
+ ): Promise {
+ const response = await this.#supervisor.run(
+ { kind: "benchmark-sample", request },
+ timeoutMs,
+ );
+ if (response.kind !== "benchmark-sample") {
+ throw new Error("Analysis worker returned the wrong benchmark response.");
+ }
+ return response.result;
+ }
+
+ async growthProbe(
+ request: GrowthProbeSubject,
+ timeoutMs: number,
+ ): Promise {
+ const response = await this.#supervisor.run(
+ { kind: "growth-probe", request },
+ timeoutMs,
+ );
+ if (response.kind !== "growth-probe") {
+ throw new Error("Analysis worker returned the wrong growth response.");
+ }
+ return response.result;
+ }
+
+ cancel(): void {
+ this.#supervisor.cancel();
+ }
+
+ dispose(): void {
+ this.#supervisor.dispose();
+ }
+}
diff --git a/src/regex/analysis/analysis-limits.ts b/src/regex/analysis/analysis-limits.ts
new file mode 100644
index 0000000..fb1e16b
--- /dev/null
+++ b/src/regex/analysis/analysis-limits.ts
@@ -0,0 +1,36 @@
+import { DEFAULT_REGEX_LIMITS } from "../execution/request-limits";
+import type { BenchmarkSettings, GrowthSettings } from "./analysis.types";
+
+export const ANALYSIS_LIMITS = {
+ maximumWarmupIterations: 100,
+ maximumMeasuredIterations: 1_000,
+ maximumGrowthSteps: 24,
+ maximumGrowthFragmentUtf16: 4_096,
+ maximumGrowthAffixUtf16: 16_384,
+ maximumGrowthRepetitions: 10_000_000,
+ maximumReplacementBenchmarkOutputUtf16: 8 * 1024 * 1024,
+ minimumSampleTimeoutMs: 25,
+ maximumSampleTimeoutMs: DEFAULT_REGEX_LIMITS.advancedMaximumTimeoutMs,
+ maximumAnalysisWallTimeMs: DEFAULT_REGEX_LIMITS.maximumBenchmarkWallTimeMs,
+} as const;
+
+export const DEFAULT_BENCHMARK_SETTINGS: BenchmarkSettings = {
+ warmupIterations: 3,
+ measuredIterations: 15,
+ sampleTimeoutMs: 2_000,
+ maximumWallTimeMs: DEFAULT_REGEX_LIMITS.maximumBenchmarkWallTimeMs,
+};
+
+export const DEFAULT_GROWTH_SETTINGS: GrowthSettings = {
+ prefix: "",
+ repeatedFragment: "a",
+ suffix: "!",
+ startingRepetitions: 16,
+ maximumRepetitions: 16_384,
+ multiplier: 2,
+ maximumSteps: 11,
+ sampleTimeoutMs: 1_000,
+ maximumWallTimeMs: DEFAULT_REGEX_LIMITS.maximumBenchmarkWallTimeMs,
+ maximumSubjectBytes: 1024 * 1024,
+ normalizedGrowthThreshold: 4,
+};
diff --git a/src/regex/analysis/analysis-runner.test.ts b/src/regex/analysis/analysis-runner.test.ts
new file mode 100644
index 0000000..06ce2fa
--- /dev/null
+++ b/src/regex/analysis/analysis-runner.test.ts
@@ -0,0 +1,336 @@
+import { describe, expect, it, vi } from "vitest";
+import { WorkerRequestError } from "../execution/WorkerSupervisor";
+import {
+ runGrowthAnalysis,
+ runRegexBenchmark,
+ validateBenchmarkRequest,
+ validateGrowthRequest,
+} from "./analysis-runner";
+import type { AnalysisWorkerClient } from "./AnalysisSupervisor";
+import type {
+ BenchmarkWorkerSample,
+ GrowthWorkerSample,
+ RegexBenchmarkRequest,
+ GrowthAnalysisRequest,
+} from "./analysis.types";
+
+function benchmarkSample(
+ milliseconds: number,
+ overrides: Partial = {},
+): BenchmarkWorkerSample {
+ return {
+ accepted: true,
+ effectiveFlags: "dg",
+ subjectBytes: 3,
+ subjectUtf16: 3,
+ compileMs: milliseconds,
+ firstMatchMs: milliseconds + 1,
+ allMatchesMs: milliseconds + 2,
+ replacementMs: milliseconds + 3,
+ throughputBytesPerSecond: milliseconds * 100,
+ matchCount: 1,
+ matched: true,
+ matchCollectionTruncated: false,
+ replacementOutputUtf16: 3,
+ ...overrides,
+ };
+}
+
+function growthSample(
+ executionMs: number,
+ overrides: Partial = {},
+): GrowthWorkerSample {
+ return {
+ accepted: true,
+ effectiveFlags: "d",
+ subjectBytes: 1,
+ subjectUtf16: 1,
+ executionMs,
+ matchCount: 0,
+ matched: false,
+ matchCollectionTruncated: false,
+ ...overrides,
+ };
+}
+
+function client(
+ overrides: Partial = {},
+): AnalysisWorkerClient {
+ return {
+ identity: vi.fn().mockResolvedValue({
+ flavour: "ecmascript",
+ engineName: "Native ECMAScript RegExp",
+ engineVersion: "Test Browser 1",
+ runtimeVersion: "Test Browser 1",
+ nativeOffsetUnit: "utf16",
+ }),
+ benchmarkSample: vi.fn().mockResolvedValue(benchmarkSample(1)),
+ growthProbe: vi.fn().mockResolvedValue(growthSample(1)),
+ cancel: vi.fn(),
+ dispose: vi.fn(),
+ ...overrides,
+ };
+}
+
+function benchmarkRequest(
+ overrides: Partial = {},
+): RegexBenchmarkRequest {
+ return {
+ flavour: "ecmascript",
+ pattern: "a+",
+ flags: ["g"],
+ subject: "aaa",
+ replacement: "x",
+ scanAll: true,
+ maximumMatches: 10_000,
+ settings: {
+ warmupIterations: 1,
+ measuredIterations: 3,
+ sampleTimeoutMs: 100,
+ maximumWallTimeMs: 1_000,
+ },
+ ...overrides,
+ };
+}
+
+function growthRequest(
+ overrides: Partial = {},
+): GrowthAnalysisRequest {
+ return {
+ flavour: "ecmascript",
+ pattern: "(a+)+$",
+ flags: [],
+ scanAll: false,
+ maximumMatches: 10_000,
+ settings: {
+ prefix: "",
+ repeatedFragment: "a",
+ suffix: "!",
+ startingRepetitions: 16,
+ maximumRepetitions: 64,
+ multiplier: 2,
+ maximumSteps: 3,
+ sampleTimeoutMs: 100,
+ maximumWallTimeMs: 1_000,
+ maximumSubjectBytes: 1_024,
+ normalizedGrowthThreshold: 4,
+ },
+ ...overrides,
+ };
+}
+
+describe("bounded analysis runner", () => {
+ it("keeps cold use, warm-up and measured statistics separate", async () => {
+ const samples = [
+ benchmarkSample(9),
+ benchmarkSample(99),
+ benchmarkSample(1),
+ benchmarkSample(2),
+ benchmarkSample(10),
+ ];
+ const worker = client({
+ benchmarkSample: vi
+ .fn()
+ .mockImplementation(() => Promise.resolve(samples.shift()!)),
+ });
+ const phases: string[] = [];
+
+ const result = await runRegexBenchmark(benchmarkRequest(), worker, {
+ onProgress: (value) => phases.push(value.phase),
+ });
+
+ expect(result.status).toBe("complete");
+ expect(result.coldSample?.compileMs).toBe(9);
+ expect(result.completedWarmups).toBe(1);
+ expect(result.completedMeasuredIterations).toBe(3);
+ expect(result.warmStatistics.compileMs).toEqual({
+ count: 3,
+ minimum: 1,
+ median: 2,
+ p95: 10,
+ maximum: 10,
+ });
+ expect(result.identity?.engineVersion).toBe("Test Browser 1");
+ expect(phases).toEqual([
+ "starting-worker",
+ "cold-sample",
+ "warming-up",
+ "measuring",
+ "measuring",
+ "measuring",
+ ]);
+ });
+
+ it("returns partial measured samples and a distinct timeout state", async () => {
+ let calls = 0;
+ const worker = client({
+ benchmarkSample: vi.fn().mockImplementation(() => {
+ calls += 1;
+ if (calls === 3) {
+ return Promise.reject(
+ new WorkerRequestError("timeout", "sample timed out"),
+ );
+ }
+ return Promise.resolve(benchmarkSample(calls));
+ }),
+ });
+
+ const result = await runRegexBenchmark(
+ benchmarkRequest({
+ settings: {
+ warmupIterations: 0,
+ measuredIterations: 3,
+ sampleTimeoutMs: 100,
+ maximumWallTimeMs: 1_000,
+ },
+ }),
+ worker,
+ );
+
+ expect(result.status).toBe("timeout");
+ expect(result.completedMeasuredIterations).toBe(1);
+ expect(result.stoppedReason).toContain("worker was terminated");
+ });
+
+ it("does not mislabel an aggregate wall limit as a regex timeout", async () => {
+ let time = 0;
+ const clock = { now: () => time };
+ const worker = client({
+ identity: vi.fn().mockImplementation(async () => {
+ time = 950;
+ return {
+ flavour: "ecmascript",
+ engineName: "Native ECMAScript RegExp",
+ engineVersion: "Test Browser 1",
+ runtimeVersion: "Test Browser 1",
+ nativeOffsetUnit: "utf16",
+ } as const;
+ }),
+ benchmarkSample: vi
+ .fn()
+ .mockRejectedValue(new WorkerRequestError("timeout", "wall stop")),
+ });
+
+ const result = await runRegexBenchmark(
+ benchmarkRequest({
+ settings: {
+ warmupIterations: 0,
+ measuredIterations: 1,
+ sampleTimeoutMs: 100,
+ maximumWallTimeMs: 1_000,
+ },
+ }),
+ worker,
+ {},
+ clock,
+ );
+
+ expect(result.status).toBe("wall-time-limit");
+ expect(result.stoppedReason).toContain("aggregate wall-time");
+ });
+
+ it("rejects out-of-bound benchmark and growth settings before worker use", () => {
+ expect(() =>
+ validateBenchmarkRequest(
+ benchmarkRequest({
+ settings: {
+ warmupIterations: 101,
+ measuredIterations: 1,
+ sampleTimeoutMs: 100,
+ maximumWallTimeMs: 1_000,
+ },
+ }),
+ ),
+ ).toThrow(/Warm-up iterations/u);
+ expect(() =>
+ validateGrowthRequest(
+ growthRequest({
+ settings: {
+ ...growthRequest().settings,
+ repeatedFragment: "",
+ },
+ }),
+ ),
+ ).toThrow(/must not be empty/u);
+ });
+
+ it("stops on observed normalized growth and records dynamic evidence", async () => {
+ const probes = [growthSample(0.5), growthSample(10)];
+ const worker = client({
+ growthProbe: vi
+ .fn()
+ .mockImplementation(() => Promise.resolve(probes.shift()!)),
+ });
+
+ const result = await runGrowthAnalysis(growthRequest(), worker);
+
+ expect(result.status).toBe("complete");
+ expect(result.stopReason).toBe("growth-threshold");
+ expect(result.samples).toHaveLength(2);
+ expect(result.samples[1]?.normalizedGrowth).toBeGreaterThan(4);
+ expect(result.dynamicFindings[0]).toEqual(
+ expect.objectContaining({
+ rule: "observed-growth",
+ evidence: "dynamically-observed",
+ confidence: "medium",
+ }),
+ );
+ });
+
+ it("distinguishes an observed timeout from a crash", async () => {
+ const worker = client({
+ growthProbe: vi
+ .fn()
+ .mockRejectedValue(
+ new WorkerRequestError("timeout", "probe timed out"),
+ ),
+ });
+
+ const result = await runGrowthAnalysis(growthRequest(), worker);
+
+ expect(result.status).toBe("timeout");
+ expect(result.stopReason).toBe("timeout");
+ expect(result.samples[0]?.status).toBe("timeout");
+ expect(result.dynamicFindings[0]?.rule).toBe("observed-timeout");
+ });
+
+ it("reports a worker crash without relabelling it as a timeout", async () => {
+ const worker = client({
+ growthProbe: vi
+ .fn()
+ .mockRejectedValue(new WorkerRequestError("crash", "fixture crash")),
+ });
+
+ const result = await runGrowthAnalysis(growthRequest(), worker);
+
+ expect(result.status).toBe("error");
+ expect(result.stopReason).toBe("crash");
+ expect(result.samples[0]?.status).toBe("crash");
+ expect(result.dynamicFindings).toEqual([]);
+ expect(result.stoppedReason).toContain("fixture crash");
+ });
+
+ it("preflights generated bytes and honours cancellation without allocating a subject", async () => {
+ const worker = client();
+ const tooLarge = await runGrowthAnalysis(
+ growthRequest({
+ settings: {
+ ...growthRequest().settings,
+ prefix: "prefix",
+ maximumSubjectBytes: 8,
+ },
+ }),
+ worker,
+ );
+ expect(tooLarge.stopReason).toBe("maximum-subject-size");
+ expect(worker.growthProbe).not.toHaveBeenCalled();
+
+ const controller = new AbortController();
+ controller.abort();
+ const cancelled = await runGrowthAnalysis(growthRequest(), client(), {
+ signal: controller.signal,
+ });
+ expect(cancelled.status).toBe("cancelled");
+ expect(cancelled.stopReason).toBe("cancelled");
+ });
+});
diff --git a/src/regex/analysis/analysis-runner.ts b/src/regex/analysis/analysis-runner.ts
new file mode 100644
index 0000000..02bbc1f
--- /dev/null
+++ b/src/regex/analysis/analysis-runner.ts
@@ -0,0 +1,819 @@
+import { WorkerRequestError } from "../execution/WorkerSupervisor";
+import {
+ DEFAULT_REGEX_LIMITS,
+ utf8ByteLength,
+} from "../execution/request-limits";
+import { ANALYSIS_LIMITS } from "./analysis-limits";
+import type { AnalysisWorkerClient } from "./AnalysisSupervisor";
+import type {
+ AnalysisProgress,
+ AnalysisRunOptions,
+ AnalysisRunStatus,
+ BenchmarkWorkerSample,
+ GrowthAnalysisRequest,
+ GrowthAnalysisResult,
+ GrowthSample,
+ GrowthStopReason,
+ RegexBenchmarkRequest,
+ RegexBenchmarkResult,
+ RegexRiskFinding,
+} from "./analysis.types";
+import { summarizeBenchmarkSamples } from "./statistics";
+
+interface Clock {
+ now(): number;
+}
+
+const SYSTEM_CLOCK: Clock = {
+ now: () => performance.now(),
+};
+
+function assertInteger(
+ value: number,
+ label: string,
+ minimum: number,
+ maximum: number,
+): void {
+ if (!Number.isSafeInteger(value) || value < minimum || value > maximum) {
+ throw new RangeError(
+ `${label} must be an integer from ${minimum.toLocaleString()} to ${maximum.toLocaleString()}.`,
+ );
+ }
+}
+
+function assertFiniteRange(
+ value: number,
+ label: string,
+ minimum: number,
+ maximum: number,
+): void {
+ if (!Number.isFinite(value) || value < minimum || value > maximum) {
+ throw new RangeError(
+ `${label} must be from ${minimum.toLocaleString()} to ${maximum.toLocaleString()}.`,
+ );
+ }
+}
+
+export function validateBenchmarkRequest(request: RegexBenchmarkRequest): void {
+ if (request.flavour !== "ecmascript") {
+ throw new RangeError("Benchmarking currently supports ECMAScript only.");
+ }
+ if (request.pattern.length > DEFAULT_REGEX_LIMITS.patternHardLengthUtf16) {
+ throw new RangeError("Pattern exceeds the configured hard limit.");
+ }
+ if (
+ utf8ByteLength(request.subject) >
+ DEFAULT_REGEX_LIMITS.interactiveSubjectHardBytes
+ ) {
+ throw new RangeError(
+ "Benchmark subject exceeds the configured hard limit.",
+ );
+ }
+ if (
+ request.replacement.length >
+ DEFAULT_REGEX_LIMITS.maximumReplacementTemplateUtf16
+ ) {
+ throw new RangeError(
+ "Benchmark replacement exceeds the configured hard limit.",
+ );
+ }
+ assertInteger(
+ request.maximumMatches,
+ "Maximum matches",
+ 1,
+ DEFAULT_REGEX_LIMITS.maximumMatches,
+ );
+ assertInteger(
+ request.settings.warmupIterations,
+ "Warm-up iterations",
+ 0,
+ ANALYSIS_LIMITS.maximumWarmupIterations,
+ );
+ assertInteger(
+ request.settings.measuredIterations,
+ "Measured iterations",
+ 1,
+ ANALYSIS_LIMITS.maximumMeasuredIterations,
+ );
+ assertInteger(
+ request.settings.sampleTimeoutMs,
+ "Sample timeout",
+ ANALYSIS_LIMITS.minimumSampleTimeoutMs,
+ ANALYSIS_LIMITS.maximumSampleTimeoutMs,
+ );
+ assertInteger(
+ request.settings.maximumWallTimeMs,
+ "Benchmark wall time",
+ request.settings.sampleTimeoutMs,
+ ANALYSIS_LIMITS.maximumAnalysisWallTimeMs,
+ );
+ if (
+ request.settings.warmupIterations +
+ request.settings.measuredIterations +
+ 1 >
+ DEFAULT_REGEX_LIMITS.maximumBenchmarkIterations
+ ) {
+ throw new RangeError(
+ `Benchmark exceeds the ${DEFAULT_REGEX_LIMITS.maximumBenchmarkIterations.toLocaleString()}-iteration limit.`,
+ );
+ }
+}
+
+export function validateGrowthRequest(request: GrowthAnalysisRequest): void {
+ if (request.flavour !== "ecmascript") {
+ throw new RangeError(
+ "Dynamic growth analysis currently supports ECMAScript only.",
+ );
+ }
+ if (request.pattern.length > DEFAULT_REGEX_LIMITS.patternHardLengthUtf16) {
+ throw new RangeError("Pattern exceeds the configured hard limit.");
+ }
+ assertInteger(
+ request.maximumMatches,
+ "Maximum matches",
+ 1,
+ DEFAULT_REGEX_LIMITS.maximumMatches,
+ );
+ if (request.settings.repeatedFragment.length === 0) {
+ throw new RangeError("Repeated fragment must not be empty.");
+ }
+ if (
+ request.settings.repeatedFragment.length >
+ ANALYSIS_LIMITS.maximumGrowthFragmentUtf16
+ ) {
+ throw new RangeError(
+ `Repeated fragment exceeds ${ANALYSIS_LIMITS.maximumGrowthFragmentUtf16.toLocaleString()} UTF-16 units.`,
+ );
+ }
+ for (const [label, value] of [
+ ["Growth prefix", request.settings.prefix],
+ ["Growth suffix", request.settings.suffix],
+ ] as const) {
+ if (value.length > ANALYSIS_LIMITS.maximumGrowthAffixUtf16) {
+ throw new RangeError(
+ `${label} exceeds ${ANALYSIS_LIMITS.maximumGrowthAffixUtf16.toLocaleString()} UTF-16 units.`,
+ );
+ }
+ }
+ assertInteger(
+ request.settings.startingRepetitions,
+ "Starting repetitions",
+ 1,
+ ANALYSIS_LIMITS.maximumGrowthRepetitions,
+ );
+ assertInteger(
+ request.settings.maximumRepetitions,
+ "Maximum repetitions",
+ request.settings.startingRepetitions,
+ ANALYSIS_LIMITS.maximumGrowthRepetitions,
+ );
+ assertFiniteRange(request.settings.multiplier, "Growth multiplier", 1.1, 10);
+ assertInteger(
+ request.settings.maximumSteps,
+ "Maximum growth steps",
+ 1,
+ ANALYSIS_LIMITS.maximumGrowthSteps,
+ );
+ assertInteger(
+ request.settings.sampleTimeoutMs,
+ "Sample timeout",
+ ANALYSIS_LIMITS.minimumSampleTimeoutMs,
+ ANALYSIS_LIMITS.maximumSampleTimeoutMs,
+ );
+ assertInteger(
+ request.settings.maximumWallTimeMs,
+ "Growth wall time",
+ request.settings.sampleTimeoutMs,
+ ANALYSIS_LIMITS.maximumAnalysisWallTimeMs,
+ );
+ assertInteger(
+ request.settings.maximumSubjectBytes,
+ "Maximum generated subject bytes",
+ 1,
+ DEFAULT_REGEX_LIMITS.interactiveSubjectHardBytes,
+ );
+ assertFiniteRange(
+ request.settings.normalizedGrowthThreshold,
+ "Normalized growth threshold",
+ 1.1,
+ 100,
+ );
+}
+
+function errorKind(
+ error: unknown,
+): "timeout" | "cancelled" | "crash" | "error" {
+ if (error instanceof WorkerRequestError) {
+ if (error.kind === "timeout") return "timeout";
+ if (error.kind === "cancelled") return "cancelled";
+ if (error.kind === "crash") return "crash";
+ return "error";
+ }
+ if (
+ error &&
+ typeof error === "object" &&
+ "kind" in error &&
+ typeof error.kind === "string"
+ ) {
+ if (
+ error.kind === "timeout" ||
+ error.kind === "cancelled" ||
+ error.kind === "crash"
+ ) {
+ return error.kind;
+ }
+ }
+ return "error";
+}
+
+function errorMessage(error: unknown): string {
+ return error instanceof Error ? error.message : String(error);
+}
+
+function aborted(signal: AbortSignal | undefined): boolean {
+ return signal?.aborted === true;
+}
+
+function timeoutWithinDeadline(
+ configuredTimeoutMs: number,
+ deadline: number,
+ clock: Clock,
+): number {
+ return Math.max(
+ 1,
+ Math.min(configuredTimeoutMs, Math.floor(deadline - clock.now())),
+ );
+}
+
+function benchmarkResult(
+ request: RegexBenchmarkRequest,
+ started: number,
+ clock: Clock,
+ values: {
+ readonly status: AnalysisRunStatus;
+ readonly identity?: RegexBenchmarkResult["identity"];
+ readonly coldStartMs?: number;
+ readonly coldSample?: BenchmarkWorkerSample;
+ readonly warmSamples: readonly BenchmarkWorkerSample[];
+ readonly completedWarmups: number;
+ readonly stoppedReason?: string;
+ readonly warnings: readonly string[];
+ },
+): RegexBenchmarkResult {
+ return {
+ status: values.status,
+ ...(values.identity ? { identity: values.identity } : {}),
+ ...(values.coldStartMs === undefined
+ ? {}
+ : { coldStartMs: values.coldStartMs }),
+ ...(values.coldSample ? { coldSample: values.coldSample } : {}),
+ warmSamples: values.warmSamples,
+ warmStatistics: summarizeBenchmarkSamples(values.warmSamples),
+ completedWarmups: values.completedWarmups,
+ completedMeasuredIterations: values.warmSamples.length,
+ requestedSettings: request.settings,
+ wallTimeMs: Math.max(0, clock.now() - started),
+ ...(values.stoppedReason ? { stoppedReason: values.stoppedReason } : {}),
+ warnings: values.warnings,
+ };
+}
+
+function progress(options: AnalysisRunOptions, value: AnalysisProgress): void {
+ options.onProgress?.(value);
+}
+
+export async function runRegexBenchmark(
+ request: RegexBenchmarkRequest,
+ client: AnalysisWorkerClient,
+ options: AnalysisRunOptions = {},
+ clock: Clock = SYSTEM_CLOCK,
+): Promise {
+ validateBenchmarkRequest(request);
+ const started = clock.now();
+ const deadline = started + request.settings.maximumWallTimeMs;
+ let identity: RegexBenchmarkResult["identity"];
+ let coldStartMs: number | undefined;
+ let coldSample: BenchmarkWorkerSample | undefined;
+ let completedWarmups = 0;
+ const warmSamples: BenchmarkWorkerSample[] = [];
+ let requestWasWallLimited = false;
+ const nextRequestTimeout = () => {
+ requestWasWallLimited =
+ deadline - clock.now() < request.settings.sampleTimeoutMs;
+ return timeoutWithinDeadline(
+ request.settings.sampleTimeoutMs,
+ deadline,
+ clock,
+ );
+ };
+ const warnings = [
+ "One browser, pattern and subject are not a complete performance characterization.",
+ "Native browser and WebAssembly engine timings are not directly comparable.",
+ "Displayed precision does not imply nanosecond measurement accuracy.",
+ ];
+
+ const partial = (
+ status: AnalysisRunStatus,
+ stoppedReason: string,
+ ): RegexBenchmarkResult =>
+ benchmarkResult(request, started, clock, {
+ status,
+ identity,
+ coldStartMs,
+ coldSample,
+ warmSamples,
+ completedWarmups,
+ stoppedReason,
+ warnings,
+ });
+
+ try {
+ if (aborted(options.signal)) {
+ return partial("cancelled", "Benchmark cancelled before it started.");
+ }
+ progress(options, {
+ phase: "starting-worker",
+ completed: 0,
+ total: 1,
+ message: "Starting a fresh ECMAScript analysis worker…",
+ });
+ const coldStart = clock.now();
+ identity = await client.identity(nextRequestTimeout());
+ coldStartMs = Math.max(0, clock.now() - coldStart);
+
+ if (clock.now() >= deadline) {
+ return partial(
+ "wall-time-limit",
+ "Benchmark stopped at its aggregate wall-time limit.",
+ );
+ }
+ progress(options, {
+ phase: "cold-sample",
+ completed: 0,
+ total: 1,
+ message: "Measuring the first engine use separately…",
+ });
+ coldSample = await client.benchmarkSample(request, nextRequestTimeout());
+ if (!coldSample.accepted) {
+ return partial(
+ "error",
+ coldSample.compileError ??
+ "The native ECMAScript engine rejected the benchmark pattern.",
+ );
+ }
+
+ for (let index = 0; index < request.settings.warmupIterations; index += 1) {
+ if (aborted(options.signal)) {
+ return partial("cancelled", "Benchmark cancelled during warm-up.");
+ }
+ if (clock.now() >= deadline) {
+ return partial(
+ "wall-time-limit",
+ "Benchmark stopped at its aggregate wall-time limit.",
+ );
+ }
+ progress(options, {
+ phase: "warming-up",
+ completed: index,
+ total: request.settings.warmupIterations,
+ message: `Warm-up ${index + 1} of ${request.settings.warmupIterations}…`,
+ });
+ const sample = await client.benchmarkSample(
+ request,
+ nextRequestTimeout(),
+ );
+ if (!sample.accepted) {
+ return partial(
+ "error",
+ sample.compileError ??
+ "The native ECMAScript engine rejected a warm-up sample.",
+ );
+ }
+ completedWarmups += 1;
+ }
+
+ for (
+ let index = 0;
+ index < request.settings.measuredIterations;
+ index += 1
+ ) {
+ if (aborted(options.signal)) {
+ return partial("cancelled", "Benchmark cancelled while measuring.");
+ }
+ if (clock.now() >= deadline) {
+ return partial(
+ "wall-time-limit",
+ "Benchmark stopped at its aggregate wall-time limit.",
+ );
+ }
+ progress(options, {
+ phase: "measuring",
+ completed: index,
+ total: request.settings.measuredIterations,
+ message: `Measured sample ${index + 1} of ${request.settings.measuredIterations}…`,
+ });
+ const sample = await client.benchmarkSample(
+ request,
+ nextRequestTimeout(),
+ );
+ if (!sample.accepted) {
+ return partial(
+ "error",
+ sample.compileError ??
+ "The native ECMAScript engine rejected a measured sample.",
+ );
+ }
+ warmSamples.push(sample);
+ }
+ } catch (error) {
+ const kind = errorKind(error);
+ if (kind === "timeout") {
+ if (requestWasWallLimited || clock.now() >= deadline) {
+ return partial(
+ "wall-time-limit",
+ "Benchmark stopped at its aggregate wall-time limit.",
+ );
+ }
+ return partial(
+ "timeout",
+ "A benchmark sample timed out; its worker was terminated.",
+ );
+ }
+ if (kind === "cancelled" || aborted(options.signal)) {
+ return partial(
+ "cancelled",
+ "Benchmark cancelled; its worker was terminated.",
+ );
+ }
+ return partial("error", `Benchmark worker failed: ${errorMessage(error)}`);
+ }
+
+ for (const sample of [coldSample, ...warmSamples]) {
+ if (sample?.matchCollectionTruncated) {
+ warnings.push(
+ `Match collection reached the ${request.maximumMatches.toLocaleString()}-match benchmark limit.`,
+ );
+ break;
+ }
+ }
+ const skippedReplacement = [coldSample, ...warmSamples].find(
+ (sample) => sample?.replacementSkippedReason,
+ )?.replacementSkippedReason;
+ if (skippedReplacement) {
+ warnings.push(`Replacement timing skipped: ${skippedReplacement}`);
+ }
+ return benchmarkResult(request, started, clock, {
+ status: "complete",
+ identity,
+ coldStartMs,
+ coldSample,
+ warmSamples,
+ completedWarmups,
+ warnings,
+ });
+}
+
+function generatedSubjectSize(
+ request: GrowthAnalysisRequest,
+ repetitions: number,
+): { readonly utf16: number; readonly bytes: number } {
+ return {
+ utf16:
+ request.settings.prefix.length +
+ request.settings.repeatedFragment.length * repetitions +
+ request.settings.suffix.length,
+ bytes:
+ utf8ByteLength(request.settings.prefix) +
+ utf8ByteLength(request.settings.repeatedFragment) * repetitions +
+ utf8ByteLength(request.settings.suffix),
+ };
+}
+
+function dynamicFinding(
+ request: GrowthAnalysisRequest,
+ rule: "observed-growth" | "observed-timeout",
+ details: {
+ readonly confidence: "medium" | "high";
+ readonly severity: "warning" | "high";
+ readonly explanation: string;
+ readonly exampleRisk: string;
+ },
+): RegexRiskFinding {
+ return {
+ id: `risk-${rule}-0-${request.pattern.length}`,
+ flavour: "ecmascript",
+ range: { startUtf16: 0, endUtf16: request.pattern.length },
+ rule,
+ title:
+ rule === "observed-timeout"
+ ? "Observed timeout under selected limits"
+ : "Observed disproportionate growth",
+ explanation: details.explanation,
+ exampleRisk: details.exampleRisk,
+ confidence: details.confidence,
+ severity: details.severity,
+ limitations:
+ "This observation applies only to the generated subjects, browser runtime and limits shown in this result.",
+ suggestedInvestigation:
+ "Repeat with representative production boundaries, retain strict worker limits, and simplify ambiguous repetition where possible.",
+ evidence: "dynamically-observed",
+ };
+}
+
+function growthResult(
+ request: GrowthAnalysisRequest,
+ started: number,
+ clock: Clock,
+ values: {
+ readonly status: AnalysisRunStatus;
+ readonly identity?: GrowthAnalysisResult["identity"];
+ readonly samples: readonly GrowthSample[];
+ readonly stopReason: GrowthStopReason;
+ readonly stoppedReason: string;
+ readonly dynamicFindings: readonly RegexRiskFinding[];
+ },
+): GrowthAnalysisResult {
+ return {
+ status: values.status,
+ ...(values.identity ? { identity: values.identity } : {}),
+ samples: values.samples,
+ stopReason: values.stopReason,
+ stoppedReason: values.stoppedReason,
+ wallTimeMs: Math.max(0, clock.now() - started),
+ dynamicFindings: values.dynamicFindings,
+ settings: request.settings,
+ };
+}
+
+export async function runGrowthAnalysis(
+ request: GrowthAnalysisRequest,
+ client: AnalysisWorkerClient,
+ options: AnalysisRunOptions = {},
+ clock: Clock = SYSTEM_CLOCK,
+): Promise {
+ validateGrowthRequest(request);
+ const started = clock.now();
+ const deadline = started + request.settings.maximumWallTimeMs;
+ const samples: GrowthSample[] = [];
+ const dynamicFindings: RegexRiskFinding[] = [];
+ let identity: GrowthAnalysisResult["identity"];
+ let currentRepetitions = request.settings.startingRepetitions;
+ let requestWasWallLimited = false;
+ const nextRequestTimeout = () => {
+ requestWasWallLimited =
+ deadline - clock.now() < request.settings.sampleTimeoutMs;
+ return timeoutWithinDeadline(
+ request.settings.sampleTimeoutMs,
+ deadline,
+ clock,
+ );
+ };
+
+ const finish = (
+ status: AnalysisRunStatus,
+ stopReason: GrowthStopReason,
+ stoppedReason: string,
+ ) =>
+ growthResult(request, started, clock, {
+ status,
+ identity,
+ samples,
+ stopReason,
+ stoppedReason,
+ dynamicFindings,
+ });
+
+ try {
+ if (aborted(options.signal)) {
+ return finish(
+ "cancelled",
+ "cancelled",
+ "Growth analysis cancelled before it started.",
+ );
+ }
+ identity = await client.identity(nextRequestTimeout());
+
+ while (samples.length < request.settings.maximumSteps) {
+ if (aborted(options.signal)) {
+ return finish(
+ "cancelled",
+ "cancelled",
+ "Growth analysis cancelled; its worker was terminated.",
+ );
+ }
+ if (clock.now() >= deadline) {
+ return finish(
+ "wall-time-limit",
+ "wall-time-limit",
+ "Growth analysis stopped at its aggregate wall-time limit.",
+ );
+ }
+ if (currentRepetitions > request.settings.maximumRepetitions) {
+ return finish(
+ "complete",
+ "maximum-repetitions",
+ "Growth analysis reached the configured repetition bound.",
+ );
+ }
+ const size = generatedSubjectSize(request, currentRepetitions);
+ if (
+ !Number.isSafeInteger(size.bytes) ||
+ !Number.isSafeInteger(size.utf16) ||
+ size.bytes > request.settings.maximumSubjectBytes
+ ) {
+ return finish(
+ "complete",
+ "maximum-subject-size",
+ "Growth analysis stopped before exceeding the generated-subject byte limit.",
+ );
+ }
+
+ progress(options, {
+ phase: "growth-probe",
+ completed: samples.length,
+ total: request.settings.maximumSteps,
+ message: `Probing ${currentRepetitions.toLocaleString()} fragment repetitions…`,
+ });
+ const subject =
+ request.settings.prefix +
+ request.settings.repeatedFragment.repeat(currentRepetitions) +
+ request.settings.suffix;
+ let workerSample;
+ try {
+ workerSample = await client.growthProbe(
+ {
+ flavour: "ecmascript",
+ pattern: request.pattern,
+ flags: request.flags,
+ subject,
+ scanAll: request.scanAll,
+ maximumMatches: request.maximumMatches,
+ },
+ nextRequestTimeout(),
+ );
+ } catch (error) {
+ const kind = errorKind(error);
+ if (kind === "timeout") {
+ if (requestWasWallLimited || clock.now() >= deadline) {
+ return finish(
+ "wall-time-limit",
+ "wall-time-limit",
+ "Growth analysis stopped at its aggregate wall-time limit.",
+ );
+ }
+ samples.push({
+ repetitions: currentRepetitions,
+ inputUtf16: size.utf16,
+ inputBytes: size.bytes,
+ status: "timeout",
+ message: "Worker timed out and was terminated.",
+ });
+ dynamicFindings.push(
+ dynamicFinding(request, "observed-timeout", {
+ confidence: "high",
+ severity: "high",
+ explanation: `Execution did not finish within ${request.settings.sampleTimeoutMs.toLocaleString()} ms at ${currentRepetitions.toLocaleString()} generated repetitions.`,
+ exampleRisk:
+ "The selected generated near-miss exhausted the configured per-sample wall time.",
+ }),
+ );
+ return finish(
+ "timeout",
+ "timeout",
+ "Observed timeout under selected limits; the worker was terminated.",
+ );
+ }
+ if (kind === "cancelled" || aborted(options.signal)) {
+ return finish(
+ "cancelled",
+ "cancelled",
+ "Growth analysis cancelled; its worker was terminated.",
+ );
+ }
+ samples.push({
+ repetitions: currentRepetitions,
+ inputUtf16: size.utf16,
+ inputBytes: size.bytes,
+ status: "crash",
+ message: errorMessage(error),
+ });
+ return finish(
+ "error",
+ "crash",
+ `Growth worker failed: ${errorMessage(error)}`,
+ );
+ }
+
+ if (!workerSample.accepted) {
+ samples.push({
+ repetitions: currentRepetitions,
+ inputUtf16: size.utf16,
+ inputBytes: size.bytes,
+ status: "compile-error",
+ message: workerSample.compileError,
+ });
+ return finish(
+ "error",
+ "compile-error",
+ workerSample.compileError ??
+ "The native ECMAScript engine rejected the pattern.",
+ );
+ }
+
+ const previous = samples.at(-1);
+ const executionMs = workerSample.executionMs ?? 0;
+ let normalizedGrowth: number | undefined;
+ if (
+ previous?.status === "complete" &&
+ previous.executionMs !== undefined &&
+ previous.executionMs >= 0.25 &&
+ executionMs >= 1 &&
+ previous.inputBytes > 0
+ ) {
+ const inputRatio = size.bytes / previous.inputBytes;
+ const timeRatio = executionMs / previous.executionMs;
+ if (inputRatio > 0) normalizedGrowth = timeRatio / inputRatio;
+ }
+ samples.push({
+ repetitions: currentRepetitions,
+ inputUtf16: size.utf16,
+ inputBytes: size.bytes,
+ status: "complete",
+ executionMs,
+ matched: workerSample.matched,
+ matchCount: workerSample.matchCount,
+ matchCollectionTruncated: workerSample.matchCollectionTruncated,
+ ...(normalizedGrowth === undefined ? {} : { normalizedGrowth }),
+ });
+
+ if (
+ normalizedGrowth !== undefined &&
+ normalizedGrowth >= request.settings.normalizedGrowthThreshold
+ ) {
+ dynamicFindings.push(
+ dynamicFinding(request, "observed-growth", {
+ confidence: "medium",
+ severity: "warning",
+ explanation: `Observed time growth was ${normalizedGrowth.toFixed(2)}× the input growth between the last two generated samples.`,
+ exampleRisk:
+ "Execution time increased disproportionately for the selected generated subject family.",
+ }),
+ );
+ return finish(
+ "complete",
+ "growth-threshold",
+ "Growth analysis stopped at the configured normalized-growth threshold.",
+ );
+ }
+ if (currentRepetitions >= request.settings.maximumRepetitions) {
+ return finish(
+ "complete",
+ "maximum-repetitions",
+ "Growth analysis reached the configured repetition bound.",
+ );
+ }
+ const next = Math.min(
+ request.settings.maximumRepetitions,
+ Math.max(
+ currentRepetitions + 1,
+ Math.ceil(currentRepetitions * request.settings.multiplier),
+ ),
+ );
+ currentRepetitions = next;
+ }
+ } catch (error) {
+ const kind = errorKind(error);
+ if (kind === "timeout") {
+ if (requestWasWallLimited || clock.now() >= deadline) {
+ return finish(
+ "wall-time-limit",
+ "wall-time-limit",
+ "Growth analysis stopped at its aggregate wall-time limit.",
+ );
+ }
+ return finish(
+ "timeout",
+ "timeout",
+ "The analysis worker timed out while starting and was terminated.",
+ );
+ }
+ if (kind === "cancelled" || aborted(options.signal)) {
+ return finish(
+ "cancelled",
+ "cancelled",
+ "Growth analysis cancelled; its worker was terminated.",
+ );
+ }
+ return finish(
+ "error",
+ "crash",
+ `Growth worker failed: ${errorMessage(error)}`,
+ );
+ }
+
+ return finish(
+ "complete",
+ "maximum-steps",
+ "Growth analysis reached the configured step limit.",
+ );
+}
diff --git a/src/regex/analysis/analysis.types.ts b/src/regex/analysis/analysis.types.ts
new file mode 100644
index 0000000..0bb4211
--- /dev/null
+++ b/src/regex/analysis/analysis.types.ts
@@ -0,0 +1,266 @@
+import type { RegexFlavourId } from "../model/flavour";
+import type { NormalizedRegexNode, SourceRange } from "../model/syntax";
+
+export type RegexRiskConfidence = "low" | "medium" | "high";
+
+export type RegexRiskSeverity = "notice" | "warning" | "high";
+
+export type RegexRiskEvidence = "static" | "dynamically-observed";
+
+export type RegexRiskRule =
+ | "nested-quantifier"
+ | "ambiguous-repeated-alternatives"
+ | "overlapping-alternatives"
+ | "repeated-wildcard"
+ | "backreference-in-repetition"
+ | "nullable-repeated-expression"
+ | "repeated-lookaround-body"
+ | "unanchored-expensive-prefix"
+ | "excessive-captures"
+ | "excessive-nesting"
+ | "potential-replacement-expansion"
+ | "observed-growth"
+ | "observed-timeout";
+
+export interface RegexRiskFinding {
+ readonly id: string;
+ readonly flavour: RegexFlavourId;
+ readonly range: SourceRange;
+ readonly rule: RegexRiskRule;
+ readonly title: string;
+ readonly explanation: string;
+ readonly exampleRisk: string;
+ readonly confidence: RegexRiskConfidence;
+ readonly severity: RegexRiskSeverity;
+ readonly limitations: string;
+ readonly suggestedInvestigation: string;
+ readonly evidence: RegexRiskEvidence;
+}
+
+export interface StaticRiskRequest {
+ readonly flavour: RegexFlavourId;
+ readonly root: NormalizedRegexNode;
+ readonly flags: readonly string[];
+ readonly scanAll: boolean;
+ readonly replacement?: string;
+}
+
+export interface StaticRiskReport {
+ readonly flavour: RegexFlavourId;
+ readonly analyser: {
+ readonly id: "regex-tools-ecmascript-heuristics";
+ readonly version: "1";
+ readonly support: "experimental-ecmascript-2025" | "unsupported-flavour";
+ };
+ readonly findings: readonly RegexRiskFinding[];
+ readonly analysedNodes: number;
+ readonly findingsTruncated: boolean;
+ readonly traversalTruncated: boolean;
+ readonly summary: string;
+ readonly limitations: readonly string[];
+}
+
+export interface AnalysisEngineIdentity {
+ readonly flavour: "ecmascript";
+ readonly engineName: "Native ECMAScript RegExp";
+ readonly engineVersion: string;
+ readonly runtimeVersion: string;
+ readonly nativeOffsetUnit: "utf16";
+}
+
+export interface BenchmarkSubject {
+ readonly flavour: "ecmascript";
+ readonly pattern: string;
+ readonly flags: readonly string[];
+ readonly subject: string;
+ readonly replacement: string;
+ readonly scanAll: boolean;
+ readonly maximumMatches: number;
+}
+
+export interface BenchmarkWorkerSample {
+ readonly accepted: boolean;
+ readonly compileError?: string;
+ readonly effectiveFlags: string;
+ readonly subjectBytes: number;
+ readonly subjectUtf16: number;
+ readonly compileMs: number;
+ readonly firstMatchMs?: number;
+ readonly allMatchesMs?: number;
+ readonly replacementMs?: number;
+ readonly replacementSkippedReason?: string;
+ readonly throughputBytesPerSecond?: number;
+ readonly matchCount: number;
+ readonly matched: boolean;
+ readonly matchCollectionTruncated: boolean;
+ readonly replacementOutputUtf16?: number;
+}
+
+export interface BenchmarkSettings {
+ readonly warmupIterations: number;
+ readonly measuredIterations: number;
+ readonly sampleTimeoutMs: number;
+ readonly maximumWallTimeMs: number;
+}
+
+export interface RegexBenchmarkRequest extends BenchmarkSubject {
+ readonly settings: BenchmarkSettings;
+}
+
+export interface MetricStatistics {
+ readonly count: number;
+ readonly minimum: number;
+ readonly median: number;
+ readonly p95: number;
+ readonly maximum: number;
+}
+
+export interface BenchmarkMetricSummary {
+ readonly compileMs?: MetricStatistics;
+ readonly firstMatchMs?: MetricStatistics;
+ readonly allMatchesMs?: MetricStatistics;
+ readonly replacementMs?: MetricStatistics;
+ readonly throughputBytesPerSecond?: MetricStatistics;
+}
+
+export type AnalysisRunStatus =
+ "complete" | "timeout" | "cancelled" | "wall-time-limit" | "error";
+
+export interface RegexBenchmarkResult {
+ readonly status: AnalysisRunStatus;
+ readonly identity?: AnalysisEngineIdentity;
+ readonly coldStartMs?: number;
+ readonly coldSample?: BenchmarkWorkerSample;
+ readonly warmSamples: readonly BenchmarkWorkerSample[];
+ readonly warmStatistics: BenchmarkMetricSummary;
+ readonly completedWarmups: number;
+ readonly completedMeasuredIterations: number;
+ readonly requestedSettings: BenchmarkSettings;
+ readonly wallTimeMs: number;
+ readonly stoppedReason?: string;
+ readonly warnings: readonly string[];
+}
+
+export interface GrowthProbeSubject {
+ readonly flavour: "ecmascript";
+ readonly pattern: string;
+ readonly flags: readonly string[];
+ readonly subject: string;
+ readonly scanAll: boolean;
+ readonly maximumMatches: number;
+}
+
+export interface GrowthWorkerSample {
+ readonly accepted: boolean;
+ readonly compileError?: string;
+ readonly effectiveFlags: string;
+ readonly subjectBytes: number;
+ readonly subjectUtf16: number;
+ readonly executionMs?: number;
+ readonly matchCount: number;
+ readonly matched: boolean;
+ readonly matchCollectionTruncated: boolean;
+}
+
+export interface GrowthSettings {
+ readonly prefix: string;
+ readonly repeatedFragment: string;
+ readonly suffix: string;
+ readonly startingRepetitions: number;
+ readonly maximumRepetitions: number;
+ readonly multiplier: number;
+ readonly maximumSteps: number;
+ readonly sampleTimeoutMs: number;
+ readonly maximumWallTimeMs: number;
+ readonly maximumSubjectBytes: number;
+ readonly normalizedGrowthThreshold: number;
+}
+
+export type GrowthSampleStatus =
+ "complete" | "timeout" | "crash" | "compile-error";
+
+export interface GrowthSample {
+ readonly repetitions: number;
+ readonly inputUtf16: number;
+ readonly inputBytes: number;
+ readonly status: GrowthSampleStatus;
+ readonly executionMs?: number;
+ readonly matched?: boolean;
+ readonly matchCount?: number;
+ readonly matchCollectionTruncated?: boolean;
+ readonly normalizedGrowth?: number;
+ readonly message?: string;
+}
+
+export type GrowthStopReason =
+ | "maximum-repetitions"
+ | "maximum-steps"
+ | "maximum-subject-size"
+ | "growth-threshold"
+ | "timeout"
+ | "compile-error"
+ | "crash"
+ | "cancelled"
+ | "wall-time-limit";
+
+export interface GrowthAnalysisRequest {
+ readonly flavour: "ecmascript";
+ readonly pattern: string;
+ readonly flags: readonly string[];
+ readonly scanAll: boolean;
+ readonly maximumMatches: number;
+ readonly settings: GrowthSettings;
+}
+
+export interface GrowthAnalysisResult {
+ readonly status: AnalysisRunStatus;
+ readonly identity?: AnalysisEngineIdentity;
+ readonly samples: readonly GrowthSample[];
+ readonly stopReason: GrowthStopReason;
+ readonly stoppedReason: string;
+ readonly wallTimeMs: number;
+ readonly dynamicFindings: readonly RegexRiskFinding[];
+ readonly settings: GrowthSettings;
+}
+
+export type AnalysisWorkerOperation =
+ | { readonly kind: "identity" }
+ | {
+ readonly kind: "benchmark-sample";
+ readonly request: BenchmarkSubject;
+ }
+ | {
+ readonly kind: "growth-probe";
+ readonly request: GrowthProbeSubject;
+ };
+
+export type AnalysisWorkerResult =
+ | {
+ readonly kind: "identity";
+ readonly result: AnalysisEngineIdentity;
+ }
+ | {
+ readonly kind: "benchmark-sample";
+ readonly result: BenchmarkWorkerSample;
+ }
+ | {
+ readonly kind: "growth-probe";
+ readonly result: GrowthWorkerSample;
+ };
+
+export interface AnalysisProgress {
+ readonly phase:
+ | "starting-worker"
+ | "cold-sample"
+ | "warming-up"
+ | "measuring"
+ | "growth-probe";
+ readonly completed: number;
+ readonly total: number;
+ readonly message: string;
+}
+
+export interface AnalysisRunOptions {
+ readonly signal?: AbortSignal;
+ readonly onProgress?: (progress: AnalysisProgress) => void;
+}
diff --git a/src/regex/analysis/ecmascript-analysis.test.ts b/src/regex/analysis/ecmascript-analysis.test.ts
new file mode 100644
index 0000000..6e0ece4
--- /dev/null
+++ b/src/regex/analysis/ecmascript-analysis.test.ts
@@ -0,0 +1,180 @@
+import { describe, expect, it } from "vitest";
+import {
+ benchmarkEcmaScriptSample,
+ ecmaScriptAnalysisIdentity,
+ probeEcmaScriptGrowth,
+} from "./ecmascript-analysis";
+
+function benchmarkRequest(
+ overrides: Partial[0]> = {},
+) {
+ return {
+ flavour: "ecmascript" as const,
+ pattern: "\\d+",
+ flags: ["g"],
+ subject: "12 345",
+ replacement: "[$&]",
+ scanAll: true,
+ maximumMatches: 10_000,
+ ...overrides,
+ };
+}
+
+describe("native ECMAScript analysis worker primitives", () => {
+ it("retains exact runtime identity and separates benchmark phases", () => {
+ const identity = ecmaScriptAnalysisIdentity();
+ const sample = benchmarkEcmaScriptSample(benchmarkRequest());
+
+ expect(identity).toEqual(
+ expect.objectContaining({
+ flavour: "ecmascript",
+ engineName: "Native ECMAScript RegExp",
+ nativeOffsetUnit: "utf16",
+ }),
+ );
+ expect(sample).toEqual(
+ expect.objectContaining({
+ accepted: true,
+ effectiveFlags: expect.stringContaining("g"),
+ matchCount: 2,
+ matched: true,
+ matchCollectionTruncated: false,
+ replacementOutputUtf16: 10,
+ }),
+ );
+ expect(sample.compileMs).toBeGreaterThanOrEqual(0);
+ expect(sample.firstMatchMs).toBeGreaterThanOrEqual(0);
+ expect(sample.allMatchesMs).toBeGreaterThanOrEqual(0);
+ expect(sample.replacementMs).toBeGreaterThanOrEqual(0);
+ expect(sample.throughputBytesPerSecond).toBeGreaterThanOrEqual(0);
+ });
+
+ it("bounds zero-length global iteration", () => {
+ const sample = benchmarkEcmaScriptSample(
+ benchmarkRequest({
+ pattern: "(?=.)",
+ flags: ["g", "u"],
+ subject: "😀x",
+ replacement: "",
+ }),
+ );
+
+ expect(sample.matchCount).toBe(2);
+ expect(sample.matchCollectionTruncated).toBe(false);
+ });
+
+ it("preserves single sticky execution unless scan-all is explicit", () => {
+ const single = benchmarkEcmaScriptSample(
+ benchmarkRequest({
+ pattern: "a",
+ flags: ["y"],
+ subject: "aa",
+ replacement: "x",
+ scanAll: false,
+ }),
+ );
+ const scanAll = benchmarkEcmaScriptSample(
+ benchmarkRequest({
+ pattern: "a",
+ flags: ["y"],
+ subject: "aa",
+ replacement: "x",
+ scanAll: true,
+ }),
+ );
+
+ expect(single.matchCount).toBe(1);
+ expect(scanAll.matchCount).toBe(2);
+ expect(scanAll.replacementMs).toBeUndefined();
+ expect(scanAll.replacementSkippedReason).toContain(
+ "multi-match sticky replacement",
+ );
+ });
+
+ it("returns compile rejection as data", () => {
+ const sample = benchmarkEcmaScriptSample(
+ benchmarkRequest({ pattern: "(", subject: "" }),
+ );
+ const growth = probeEcmaScriptGrowth({
+ flavour: "ecmascript",
+ pattern: "(",
+ flags: [],
+ subject: "",
+ scanAll: false,
+ maximumMatches: 10,
+ });
+
+ expect(sample.accepted).toBe(false);
+ expect(sample.compileError).toBeTruthy();
+ expect(growth).toEqual(
+ expect.objectContaining({
+ accepted: false,
+ compileError: expect.any(String),
+ }),
+ );
+ });
+
+ it("skips replacement timing before a conservative output can grow beyond its bound", () => {
+ const sample = benchmarkEcmaScriptSample(
+ benchmarkRequest({
+ pattern: "a",
+ flags: ["g"],
+ subject: "a".repeat(4_096),
+ replacement: "$`$'",
+ maximumMatches: 10_000,
+ }),
+ );
+
+ expect(sample.replacementMs).toBeUndefined();
+ expect(sample.replacementSkippedReason).toContain(
+ "Conservative output estimate",
+ );
+ });
+
+ it("does not run unbounded replacement work after match collection truncates", () => {
+ const sample = benchmarkEcmaScriptSample(
+ benchmarkRequest({
+ pattern: "a",
+ flags: ["g"],
+ subject: "aaaa",
+ maximumMatches: 1,
+ }),
+ );
+
+ expect(sample.matchCollectionTruncated).toBe(true);
+ expect(sample.replacementMs).toBeUndefined();
+ expect(sample.replacementSkippedReason).toContain(
+ "Match collection reached its bound",
+ );
+ });
+
+ it("rejects flags outside the supported ECMAScript set", () => {
+ expect(() =>
+ benchmarkEcmaScriptSample(
+ benchmarkRequest({ flags: ["g", "unsupported"] }),
+ ),
+ ).toThrow(/Unsupported ECMAScript analysis flag/u);
+ });
+
+ it("reports bounded growth probes without materializing match values", () => {
+ const sample = probeEcmaScriptGrowth({
+ flavour: "ecmascript",
+ pattern: "a+$",
+ flags: [],
+ subject: "aaaa!",
+ scanAll: false,
+ maximumMatches: 10,
+ });
+
+ expect(sample).toEqual(
+ expect.objectContaining({
+ accepted: true,
+ subjectBytes: 5,
+ subjectUtf16: 5,
+ matched: false,
+ matchCount: 0,
+ }),
+ );
+ expect(sample.executionMs).toBeGreaterThanOrEqual(0);
+ });
+});
diff --git a/src/regex/analysis/ecmascript-analysis.ts b/src/regex/analysis/ecmascript-analysis.ts
new file mode 100644
index 0000000..1bb1d17
--- /dev/null
+++ b/src/regex/analysis/ecmascript-analysis.ts
@@ -0,0 +1,299 @@
+import { utf8ByteLength } from "../execution/request-limits";
+import { ANALYSIS_LIMITS } from "./analysis-limits";
+import type {
+ AnalysisEngineIdentity,
+ BenchmarkSubject,
+ BenchmarkWorkerSample,
+ GrowthProbeSubject,
+ GrowthWorkerSample,
+} from "./analysis.types";
+
+const FLAG_ORDER = "dgimsuvy";
+
+function runtimeIdentity(): string {
+ return typeof navigator === "undefined"
+ ? "Unknown ECMAScript runtime"
+ : navigator.userAgent;
+}
+
+export function ecmaScriptAnalysisIdentity(): AnalysisEngineIdentity {
+ const identity = runtimeIdentity();
+ return {
+ flavour: "ecmascript",
+ engineName: "Native ECMAScript RegExp",
+ engineVersion: identity,
+ runtimeVersion: identity,
+ nativeOffsetUnit: "utf16",
+ };
+}
+
+function hasIndicesSupport(): boolean {
+ try {
+ return new RegExp("", "d").hasIndices;
+ } catch {
+ return false;
+ }
+}
+
+function effectiveFlags(flags: readonly string[], scanAll: boolean): string {
+ for (const flag of flags) {
+ if (!FLAG_ORDER.includes(flag)) {
+ throw new RangeError(
+ `Unsupported ECMAScript analysis flag ${JSON.stringify(flag)}.`,
+ );
+ }
+ }
+ const unique = new Set(flags);
+ if (hasIndicesSupport()) unique.add("d");
+ if (scanAll && !unique.has("g") && !unique.has("y")) unique.add("g");
+ return FLAG_ORDER.split("")
+ .filter((flag) => unique.has(flag))
+ .join("");
+}
+
+function advanceStringIndex(
+ subject: string,
+ index: number,
+ unicode: boolean,
+): number {
+ if (!unicode) return index + 1;
+ const first = subject.charCodeAt(index);
+ if (first < 0xd800 || first > 0xdbff || index + 1 >= subject.length) {
+ return index + 1;
+ }
+ const second = subject.charCodeAt(index + 1);
+ return second >= 0xdc00 && second <= 0xdfff ? index + 2 : index + 1;
+}
+
+function collectMatches(
+ expression: RegExp,
+ subject: string,
+ iterate: boolean,
+ maximumMatches: number,
+): {
+ readonly count: number;
+ readonly matched: boolean;
+ readonly truncated: boolean;
+} {
+ const unicode = expression.unicode || expression.flags.includes("v");
+ expression.lastIndex = 0;
+ let count = 0;
+ while (true) {
+ const match = expression.exec(subject);
+ if (!match) break;
+ count += 1;
+ if (count >= maximumMatches) {
+ return { count, matched: true, truncated: iterate };
+ }
+ if (!iterate) break;
+ if (match[0].length === 0) {
+ expression.lastIndex = advanceStringIndex(
+ subject,
+ expression.lastIndex,
+ unicode,
+ );
+ if (expression.lastIndex > subject.length) break;
+ }
+ }
+ return { count, matched: count > 0, truncated: false };
+}
+
+function assertWorkerInput(
+ request: BenchmarkSubject | GrowthProbeSubject,
+): void {
+ if (request.flavour !== "ecmascript") {
+ throw new RangeError("The analysis worker supports ECMAScript only.");
+ }
+ if (request.pattern.length > 64 * 1024) {
+ throw new RangeError("Pattern exceeds the 64 Ki UTF-16 analysis limit.");
+ }
+ if (utf8ByteLength(request.subject) > 16 * 1024 * 1024) {
+ throw new RangeError("Subject exceeds the 16 MiB analysis limit.");
+ }
+ if (
+ !Number.isSafeInteger(request.maximumMatches) ||
+ request.maximumMatches < 1 ||
+ request.maximumMatches > 10_000
+ ) {
+ throw new RangeError("Analysis match limit must be between 1 and 10,000.");
+ }
+}
+
+function replacementWorstCaseUtf16(
+ replacement: string,
+ subjectUtf16: number,
+ matchCount: number,
+): number {
+ let references = 0;
+ for (let index = 0; index < replacement.length - 1; index += 1) {
+ if (
+ replacement[index] === "$" &&
+ /[`'&0-9<]/u.test(replacement[index + 1] ?? "")
+ ) {
+ references += 1;
+ index += 1;
+ }
+ }
+ const literal = replacement.length;
+ return (
+ subjectUtf16 +
+ matchCount * (literal + references * Math.max(1, subjectUtf16))
+ );
+}
+
+function compileFailure(
+ effective: string,
+ subject: string,
+ compileMs: number,
+ error: unknown,
+): BenchmarkWorkerSample {
+ return {
+ accepted: false,
+ compileError:
+ error instanceof Error
+ ? error.message
+ : "The native ECMAScript engine rejected the pattern.",
+ effectiveFlags: effective,
+ subjectBytes: utf8ByteLength(subject),
+ subjectUtf16: subject.length,
+ compileMs,
+ matchCount: 0,
+ matched: false,
+ matchCollectionTruncated: false,
+ };
+}
+
+export function benchmarkEcmaScriptSample(
+ request: BenchmarkSubject,
+): BenchmarkWorkerSample {
+ assertWorkerInput(request);
+ if (request.replacement.length > 64 * 1024) {
+ throw new RangeError(
+ "Replacement exceeds the 64 Ki UTF-16 analysis limit.",
+ );
+ }
+ const flags = effectiveFlags(request.flags, request.scanAll);
+ const compileStart = performance.now();
+ let compiled: RegExp;
+ try {
+ compiled = new RegExp(request.pattern, flags);
+ } catch (error) {
+ return compileFailure(
+ flags,
+ request.subject,
+ performance.now() - compileStart,
+ error,
+ );
+ }
+ const compileMs = performance.now() - compileStart;
+ const iterate = flags.includes("g") || request.scanAll;
+
+ const firstStart = performance.now();
+ compiled.exec(request.subject);
+ const firstMatchMs = performance.now() - firstStart;
+
+ const allExpression = new RegExp(request.pattern, flags);
+ const allStart = performance.now();
+ const collection = collectMatches(
+ allExpression,
+ request.subject,
+ iterate,
+ request.maximumMatches,
+ );
+ const allMatchesMs = performance.now() - allStart;
+ const subjectBytes = utf8ByteLength(request.subject);
+ const throughputBytesPerSecond =
+ allMatchesMs > 0 ? (subjectBytes / allMatchesMs) * 1_000 : undefined;
+
+ const worstCaseOutput = replacementWorstCaseUtf16(
+ request.replacement,
+ request.subject.length,
+ collection.count,
+ );
+ let replacementMs: number | undefined;
+ let replacementOutputUtf16: number | undefined;
+ let replacementSkippedReason: string | undefined;
+ if (flags.includes("y") && request.scanAll) {
+ replacementSkippedReason =
+ "Explicit multi-match sticky replacement is not timed with native String.replace because that API replaces only one sticky match.";
+ } else if (collection.truncated) {
+ replacementSkippedReason =
+ "Match collection reached its bound, so total replacement work and output are not known.";
+ } else if (
+ worstCaseOutput > ANALYSIS_LIMITS.maximumReplacementBenchmarkOutputUtf16
+ ) {
+ replacementSkippedReason = `Conservative output estimate exceeds ${ANALYSIS_LIMITS.maximumReplacementBenchmarkOutputUtf16.toLocaleString()} UTF-16 units.`;
+ } else {
+ const replacementExpression = new RegExp(request.pattern, flags);
+ const replacementStart = performance.now();
+ const output = request.subject.replace(
+ replacementExpression,
+ request.replacement,
+ );
+ replacementMs = performance.now() - replacementStart;
+ replacementOutputUtf16 = output.length;
+ }
+
+ return {
+ accepted: true,
+ effectiveFlags: flags,
+ subjectBytes,
+ subjectUtf16: request.subject.length,
+ compileMs,
+ firstMatchMs,
+ allMatchesMs,
+ ...(replacementMs === undefined ? {} : { replacementMs }),
+ ...(replacementSkippedReason ? { replacementSkippedReason } : {}),
+ ...(throughputBytesPerSecond === undefined
+ ? {}
+ : { throughputBytesPerSecond }),
+ matchCount: collection.count,
+ matched: collection.matched,
+ matchCollectionTruncated: collection.truncated,
+ ...(replacementOutputUtf16 === undefined ? {} : { replacementOutputUtf16 }),
+ };
+}
+
+export function probeEcmaScriptGrowth(
+ request: GrowthProbeSubject,
+): GrowthWorkerSample {
+ assertWorkerInput(request);
+ const flags = effectiveFlags(request.flags, request.scanAll);
+ let expression: RegExp;
+ try {
+ expression = new RegExp(request.pattern, flags);
+ } catch (error) {
+ return {
+ accepted: false,
+ compileError:
+ error instanceof Error
+ ? error.message
+ : "The native ECMAScript engine rejected the pattern.",
+ effectiveFlags: flags,
+ subjectBytes: utf8ByteLength(request.subject),
+ subjectUtf16: request.subject.length,
+ matchCount: 0,
+ matched: false,
+ matchCollectionTruncated: false,
+ };
+ }
+ const iterate = flags.includes("g") || request.scanAll;
+ const started = performance.now();
+ const collection = collectMatches(
+ expression,
+ request.subject,
+ iterate,
+ request.maximumMatches,
+ );
+ const executionMs = performance.now() - started;
+ return {
+ accepted: true,
+ effectiveFlags: flags,
+ subjectBytes: utf8ByteLength(request.subject),
+ subjectUtf16: request.subject.length,
+ executionMs,
+ matchCount: collection.count,
+ matched: collection.matched,
+ matchCollectionTruncated: collection.truncated,
+ };
+}
diff --git a/src/regex/analysis/static-risk.test.ts b/src/regex/analysis/static-risk.test.ts
new file mode 100644
index 0000000..fcccde2
--- /dev/null
+++ b/src/regex/analysis/static-risk.test.ts
@@ -0,0 +1,178 @@
+import { describe, expect, it } from "vitest";
+import { EcmaScriptSyntaxProvider } from "../syntax/providers/ecmascript/EcmaScriptSyntaxProvider";
+import type { NormalizedRegexNode } from "../model/syntax";
+import {
+ analyseStaticRisk,
+ MAXIMUM_STATIC_ANALYSIS_NODES,
+} from "./static-risk";
+
+const provider = new EcmaScriptSyntaxProvider();
+
+async function analyse(
+ pattern: string,
+ overrides: {
+ readonly flags?: readonly string[];
+ readonly scanAll?: boolean;
+ readonly replacement?: string;
+ } = {},
+) {
+ const flags = overrides.flags ?? [];
+ const syntax = await provider.parsePattern({
+ flavour: "ecmascript",
+ flavourVersion: "2025",
+ pattern,
+ flags,
+ options: {},
+ });
+ expect(syntax.accepted).toBe(true);
+ return analyseStaticRisk({
+ flavour: "ecmascript",
+ root: syntax.root,
+ flags,
+ scanAll: overrides.scanAll ?? false,
+ replacement: overrides.replacement,
+ });
+}
+
+describe("ECMAScript static risk analysis", () => {
+ it("identifies nested, nullable and unanchored repetition without claiming vulnerability", async () => {
+ const report = await analyse("(a*)+$");
+ const rules = report.findings.map((finding) => finding.rule);
+
+ expect(rules).toEqual(
+ expect.arrayContaining([
+ "nested-quantifier",
+ "nullable-repeated-expression",
+ "unanchored-expensive-prefix",
+ ]),
+ );
+ expect(report.findings[0]).toEqual(
+ expect.objectContaining({
+ flavour: "ecmascript",
+ evidence: "static",
+ confidence: expect.stringMatching(/low|medium|high/u),
+ limitations: expect.any(String),
+ suggestedInvestigation: expect.any(String),
+ }),
+ );
+ expect(report.summary).toMatch(/potential/u);
+ expect(JSON.stringify(report)).not.toMatch(
+ /\bdefinitely vulnerable\b|\bguaranteed linear\b/iu,
+ );
+ });
+
+ it("detects ambiguous alternatives, wildcards and repeated backreferences", async () => {
+ const alternatives = await analyse("(?:a|aa)+$");
+ expect(alternatives.findings.map((finding) => finding.rule)).toEqual(
+ expect.arrayContaining([
+ "ambiguous-repeated-alternatives",
+ "overlapping-alternatives",
+ ]),
+ );
+
+ const wildcard = await analyse("(?:.*)+x");
+ expect(wildcard.findings.map((finding) => finding.rule)).toContain(
+ "repeated-wildcard",
+ );
+
+ const backreference = await analyse("(a)(?:\\1+)+$");
+ expect(backreference.findings.map((finding) => finding.rule)).toContain(
+ "backreference-in-repetition",
+ );
+ });
+
+ it("reports repeated lookaround bodies and replacement expansion", async () => {
+ const report = await analyse("(?=(a+)+)a", {
+ flags: ["g"],
+ replacement: "$`$'",
+ });
+
+ expect(report.findings.map((finding) => finding.rule)).toEqual(
+ expect.arrayContaining([
+ "repeated-lookaround-body",
+ "potential-replacement-expansion",
+ ]),
+ );
+ });
+
+ it("uses the required cautious empty result wording", async () => {
+ const report = await analyse("^\\d{4}-\\d{2}-\\d{2}$");
+
+ expect(report.findings).toEqual([]);
+ expect(report.summary).toBe(
+ "No issue found by this analyser. This is not a safety guarantee.",
+ );
+ });
+
+ it("does not apply ECMAScript heuristics to another flavour", async () => {
+ const syntax = await provider.parsePattern({
+ flavour: "ecmascript",
+ pattern: "(a+)+",
+ flags: [],
+ options: {},
+ });
+ const report = analyseStaticRisk({
+ flavour: "pcre2",
+ root: syntax.root,
+ flags: [],
+ scanAll: false,
+ });
+
+ expect(report.analyser.support).toBe("unsupported-flavour");
+ expect(report.findings).toEqual([]);
+ expect(report.summary).toContain("unavailable for pcre2");
+ });
+
+ it("bounds normalized-tree traversal independently of render limits", () => {
+ const literal = (index: number): NormalizedRegexNode => ({
+ id: `literal-${index}`,
+ kind: "literal",
+ range: { startUtf16: index, endUtf16: index + 1 },
+ raw: "a",
+ explanation: "literal",
+ children: [],
+ properties: {
+ zeroWidth: false,
+ nullable: false,
+ minimumLength: 1,
+ maximumLength: 1,
+ consumesInput: true,
+ },
+ support: {
+ flavour: "ecmascript",
+ status: "supported",
+ notes: [],
+ },
+ provenance: {
+ provider: "fixture",
+ providerVersion: "1",
+ source: "parsed",
+ },
+ });
+ const root: NormalizedRegexNode = {
+ ...literal(0),
+ id: "root",
+ kind: "pattern",
+ range: {
+ startUtf16: 0,
+ endUtf16: MAXIMUM_STATIC_ANALYSIS_NODES + 10,
+ },
+ raw: "a".repeat(MAXIMUM_STATIC_ANALYSIS_NODES + 10),
+ children: Array.from(
+ { length: MAXIMUM_STATIC_ANALYSIS_NODES + 10 },
+ (_, index) => literal(index),
+ ),
+ };
+
+ const report = analyseStaticRisk({
+ flavour: "ecmascript",
+ root,
+ flags: [],
+ scanAll: false,
+ });
+
+ expect(report.analysedNodes).toBe(MAXIMUM_STATIC_ANALYSIS_NODES);
+ expect(report.traversalTruncated).toBe(true);
+ expect(report.limitations.at(-1)).toContain("Traversal stopped");
+ });
+});
diff --git a/src/regex/analysis/static-risk.ts b/src/regex/analysis/static-risk.ts
new file mode 100644
index 0000000..bd767f9
--- /dev/null
+++ b/src/regex/analysis/static-risk.ts
@@ -0,0 +1,573 @@
+import type { RegexFlavourId } from "../model/flavour";
+import type { NormalizedRegexNode } from "../model/syntax";
+import type {
+ RegexRiskConfidence,
+ RegexRiskFinding,
+ RegexRiskRule,
+ RegexRiskSeverity,
+ StaticRiskReport,
+ StaticRiskRequest,
+} from "./analysis.types";
+
+export const MAXIMUM_STATIC_ANALYSIS_NODES = 20_000;
+export const MAXIMUM_STATIC_RISK_FINDINGS = 100;
+const EXCESSIVE_CAPTURE_THRESHOLD = 100;
+const EXCESSIVE_NESTING_THRESHOLD = 32;
+const MAXIMUM_FIRST_SIGNATURES = 8;
+
+type FirstSignature =
+ `literal:${string}` | `class:${string}` | "any" | "unknown";
+
+interface NodeRecord {
+ readonly node: NormalizedRegexNode;
+ readonly depth: number;
+ readonly parent?: NormalizedRegexNode;
+}
+
+interface NodeSummary {
+ readonly containsRepeatableQuantifier: boolean;
+ readonly firstRepeatableQuantifier?: NormalizedRegexNode;
+ readonly firstBackreference?: NormalizedRegexNode;
+ readonly firstWildcard?: NormalizedRegexNode;
+ readonly firstAmbiguousAlternation?: NormalizedRegexNode;
+ readonly firstSignatures: readonly FirstSignature[];
+}
+
+interface FindingDetails {
+ readonly node: NormalizedRegexNode;
+ readonly rule: RegexRiskRule;
+ readonly title: string;
+ readonly explanation: string;
+ readonly exampleRisk: string;
+ readonly confidence: RegexRiskConfidence;
+ readonly severity: RegexRiskSeverity;
+ readonly limitations: string;
+ readonly suggestedInvestigation: string;
+}
+
+function isRepeatableQuantifier(node: NormalizedRegexNode): boolean {
+ return (
+ node.kind === "quantifier" &&
+ (node.quantifier?.maximum === null || (node.quantifier?.maximum ?? 0) > 1)
+ );
+}
+
+function isLookaround(node: NormalizedRegexNode): boolean {
+ return (
+ node.kind === "lookahead" ||
+ node.kind === "negative-lookahead" ||
+ node.kind === "lookbehind" ||
+ node.kind === "negative-lookbehind"
+ );
+}
+
+function limitedUnique(values: readonly T[], maximum: number): readonly T[] {
+ return [...new Set(values)].slice(0, maximum);
+}
+
+function ownFirstSignature(node: NormalizedRegexNode): FirstSignature[] {
+ if (node.kind === "dot") return ["any"];
+ if (node.kind === "literal") {
+ return [`literal:${node.raw.slice(0, 2)}`];
+ }
+ if (node.kind === "escaped-literal") {
+ if (/^\\[dDsSwWpP]/u.test(node.raw)) {
+ return [`class:${node.raw.slice(0, 2)}`];
+ }
+ return [`literal:${node.raw}`];
+ }
+ if (
+ node.kind === "character-class" ||
+ node.kind === "character-class-range" ||
+ node.kind === "unicode-property"
+ ) {
+ return [`class:${node.raw}`];
+ }
+ if (node.kind === "backreference") return ["unknown"];
+ return [];
+}
+
+function signaturesOverlap(
+ left: readonly FirstSignature[],
+ right: readonly FirstSignature[],
+): boolean {
+ for (const leftSignature of left) {
+ for (const rightSignature of right) {
+ if (leftSignature === "unknown" || rightSignature === "unknown") {
+ continue;
+ }
+ if (leftSignature === "any" || rightSignature === "any") return true;
+ if (leftSignature === rightSignature) return true;
+ }
+ }
+ return false;
+}
+
+function hasOverlappingChildren(
+ node: NormalizedRegexNode,
+ summaries: ReadonlyMap,
+): boolean {
+ const alternativesContainer =
+ node.kind === "disjunction" ||
+ node.kind === "capture-group" ||
+ node.kind === "named-capture-group" ||
+ node.kind === "noncapture-group" ||
+ node.kind === "inline-flags" ||
+ isLookaround(node);
+ if (!alternativesContainer || node.children.length < 2) return false;
+ for (let leftIndex = 0; leftIndex < node.children.length; leftIndex += 1) {
+ const left =
+ summaries.get(node.children[leftIndex]!)?.firstSignatures ?? [];
+ for (
+ let rightIndex = leftIndex + 1;
+ rightIndex < node.children.length;
+ rightIndex += 1
+ ) {
+ const right =
+ summaries.get(node.children[rightIndex]!)?.firstSignatures ?? [];
+ if (signaturesOverlap(left, right)) return true;
+ }
+ }
+ return false;
+}
+
+function firstSignaturesFor(
+ node: NormalizedRegexNode,
+ summaries: ReadonlyMap,
+): readonly FirstSignature[] {
+ const own = ownFirstSignature(node);
+ if (own.length > 0) return own;
+ if (node.children.length === 0) return [];
+
+ const isSequence = node.kind === "sequence" || node.kind === "alternative";
+ if (isSequence) {
+ const signatures: FirstSignature[] = [];
+ for (const child of node.children) {
+ signatures.push(...(summaries.get(child)?.firstSignatures ?? []));
+ if (child.properties.nullable !== true) break;
+ }
+ return limitedUnique(signatures, MAXIMUM_FIRST_SIGNATURES);
+ }
+
+ return limitedUnique(
+ node.children.flatMap(
+ (child) => summaries.get(child)?.firstSignatures ?? [],
+ ),
+ MAXIMUM_FIRST_SIGNATURES,
+ );
+}
+
+function collectRecords(root: NormalizedRegexNode): {
+ readonly records: readonly NodeRecord[];
+ readonly truncated: boolean;
+} {
+ const records: NodeRecord[] = [];
+ const stack: NodeRecord[] = [{ node: root, depth: 0 }];
+ while (stack.length > 0 && records.length < MAXIMUM_STATIC_ANALYSIS_NODES) {
+ const record = stack.pop();
+ if (!record) break;
+ records.push(record);
+ for (let index = record.node.children.length - 1; index >= 0; index -= 1) {
+ const child = record.node.children[index];
+ if (!child) continue;
+ stack.push({
+ node: child,
+ depth: record.depth + 1,
+ parent: record.node,
+ });
+ }
+ }
+ return { records, truncated: stack.length > 0 };
+}
+
+function buildSummaries(
+ records: readonly NodeRecord[],
+): ReadonlyMap {
+ const summaries = new Map();
+ for (let index = records.length - 1; index >= 0; index -= 1) {
+ const node = records[index]?.node;
+ if (!node) continue;
+ const childSummaries = node.children
+ .map((child) => summaries.get(child))
+ .filter((summary): summary is NodeSummary => summary !== undefined);
+ const overlapping = hasOverlappingChildren(node, summaries);
+ summaries.set(node, {
+ containsRepeatableQuantifier:
+ isRepeatableQuantifier(node) ||
+ childSummaries.some((summary) => summary.containsRepeatableQuantifier),
+ firstRepeatableQuantifier:
+ (isRepeatableQuantifier(node) ? node : undefined) ??
+ childSummaries.find((summary) => summary.firstRepeatableQuantifier)
+ ?.firstRepeatableQuantifier,
+ firstBackreference:
+ (node.kind === "backreference" ? node : undefined) ??
+ childSummaries.find((summary) => summary.firstBackreference)
+ ?.firstBackreference,
+ firstWildcard:
+ (node.kind === "dot" ? node : undefined) ??
+ childSummaries.find((summary) => summary.firstWildcard)?.firstWildcard,
+ firstAmbiguousAlternation:
+ (overlapping ? node : undefined) ??
+ childSummaries.find((summary) => summary.firstAmbiguousAlternation)
+ ?.firstAmbiguousAlternation,
+ firstSignatures: firstSignaturesFor(node, summaries),
+ });
+ }
+ return summaries;
+}
+
+function finding(
+ flavour: RegexFlavourId,
+ sequence: number,
+ details: FindingDetails,
+): RegexRiskFinding {
+ return {
+ id: `risk-${details.rule}-${details.node.range.startUtf16}-${details.node.range.endUtf16}-${sequence}`,
+ flavour,
+ range: details.node.range,
+ rule: details.rule,
+ title: details.title,
+ explanation: details.explanation,
+ exampleRisk: details.exampleRisk,
+ confidence: details.confidence,
+ severity: details.severity,
+ limitations: details.limitations,
+ suggestedInvestigation: details.suggestedInvestigation,
+ evidence: "static",
+ };
+}
+
+function startsAtInput(root: NormalizedRegexNode): boolean {
+ return root.raw.startsWith("^");
+}
+
+function severityOrder(severity: RegexRiskSeverity): number {
+ if (severity === "high") return 0;
+ if (severity === "warning") return 1;
+ return 2;
+}
+
+function unsupportedReport(request: StaticRiskRequest): StaticRiskReport {
+ return {
+ flavour: request.flavour,
+ analyser: {
+ id: "regex-tools-ecmascript-heuristics",
+ version: "1",
+ support: "unsupported-flavour",
+ },
+ findings: [],
+ analysedNodes: 0,
+ findingsTruncated: false,
+ traversalTruncated: false,
+ summary: `Static risk analysis is unavailable for ${request.flavour}.`,
+ limitations: [
+ "The current analyser is scoped only to the normalized ECMAScript 2025 grammar.",
+ "No ECMAScript heuristic is applied to another flavour.",
+ ],
+ };
+}
+
+export function analyseStaticRisk(
+ request: StaticRiskRequest,
+): StaticRiskReport {
+ if (
+ request.flavour !== "ecmascript" ||
+ request.root.support.flavour !== "ecmascript"
+ ) {
+ return unsupportedReport(request);
+ }
+
+ const { records, truncated: traversalTruncated } = collectRecords(
+ request.root,
+ );
+ const summaries = buildSummaries(records);
+ const details: FindingDetails[] = [];
+ let captureCount = 0;
+ let deepest = records[0];
+ const expensivePrefixCandidates: NormalizedRegexNode[] = [];
+
+ for (const record of records) {
+ const { node, depth } = record;
+ if (node.capture) captureCount += 1;
+ if (!deepest || depth > deepest.depth) deepest = record;
+ const summary = summaries.get(node);
+ if (!summary) continue;
+
+ if (isRepeatableQuantifier(node)) {
+ const body = node.children[0];
+ const bodySummary = body ? summaries.get(body) : undefined;
+ const innerQuantifier =
+ bodySummary?.firstRepeatableQuantifier === node
+ ? undefined
+ : bodySummary?.firstRepeatableQuantifier;
+ if (innerQuantifier) {
+ details.push({
+ node,
+ rule: "nested-quantifier",
+ title: "Potential nested-quantifier risk",
+ explanation:
+ "A repeated expression contains another repeatable expression. Some failing inputs can make a backtracking engine revisit the same partitions.",
+ exampleRisk:
+ "Long near-misses may take disproportionately longer than matching inputs.",
+ confidence: "high",
+ severity: "high",
+ limitations:
+ "This structural heuristic does not model engine optimizations, atomicity or every surrounding constraint.",
+ suggestedInvestigation:
+ "Run bounded growth analysis with a representative near-miss and consider removing one repetition layer.",
+ });
+ expensivePrefixCandidates.push(node);
+ }
+ if (body?.properties.nullable === true) {
+ details.push({
+ node,
+ rule: "nullable-repeated-expression",
+ title: "Potential nullable-repetition risk",
+ explanation:
+ "The repeated body can match an empty string, so many repetition paths can describe the same input position.",
+ exampleRisk:
+ "Backtracking work or zero-length iteration can grow without consuming input.",
+ confidence: "high",
+ severity: "high",
+ limitations:
+ "Native engines can simplify some nullable repetitions; this analyser does not inspect the compiled program.",
+ suggestedInvestigation:
+ "Require the body to consume input, or test the exact pattern under strict timeout limits.",
+ });
+ expensivePrefixCandidates.push(node);
+ }
+ if (bodySummary?.firstAmbiguousAlternation) {
+ details.push({
+ node,
+ rule: "ambiguous-repeated-alternatives",
+ title: "Potential ambiguous repeated alternatives",
+ explanation:
+ "Alternatives under this repetition can begin with overlapping input, leaving multiple possible partitions.",
+ exampleRisk:
+ "A long suffix failure can force the engine to reconsider many alternative boundaries.",
+ confidence: "medium",
+ severity: "high",
+ limitations:
+ "Overlap is approximated from first consuming constructs; later tokens and engine optimizations are not proven.",
+ suggestedInvestigation:
+ "Make alternatives prefix-distinct, factor shared prefixes, and probe representative near-misses.",
+ });
+ expensivePrefixCandidates.push(node);
+ }
+ if (bodySummary?.firstWildcard) {
+ details.push({
+ node: bodySummary.firstWildcard,
+ rule: "repeated-wildcard",
+ title: "Potential repeated-wildcard risk",
+ explanation:
+ "A wildcard occurs inside a repeated region and may consume the same text using multiple boundaries.",
+ exampleRisk:
+ "Later required text can trigger broad backtracking across the wildcard.",
+ confidence: "medium",
+ severity: "warning",
+ limitations:
+ "Anchors, following literals and engine search optimizations can substantially change the observed cost.",
+ suggestedInvestigation:
+ "Use a more specific character class or a bounded repetition, then compare bounded growth results.",
+ });
+ expensivePrefixCandidates.push(node);
+ }
+ if (bodySummary?.firstBackreference) {
+ details.push({
+ node: bodySummary.firstBackreference,
+ rule: "backreference-in-repetition",
+ title: "Potential repeated-backreference risk",
+ explanation:
+ "A backreference is evaluated inside repetition, coupling later work to previously captured text.",
+ exampleRisk:
+ "Multiple capture and repetition choices can amplify failing-input work.",
+ confidence: "medium",
+ severity: "warning",
+ limitations:
+ "The analyser does not resolve every capture path or prove that competing paths are reachable.",
+ suggestedInvestigation:
+ "Test realistic capture lengths and near-misses with the bounded growth runner.",
+ });
+ expensivePrefixCandidates.push(node);
+ }
+ }
+
+ if (
+ node.children.length > 1 &&
+ summary.firstAmbiguousAlternation === node
+ ) {
+ details.push({
+ node,
+ rule: "overlapping-alternatives",
+ title: "Possibly overlapping alternatives",
+ explanation:
+ "At least two alternatives can begin with the same approximated input class.",
+ exampleRisk:
+ "The engine may need to try more than one branch when later tokens fail.",
+ confidence: "low",
+ severity: "notice",
+ limitations:
+ "Only first-token overlap is checked. This is neither an equivalence test nor proof that costly backtracking occurs.",
+ suggestedInvestigation:
+ "Review whether the alternatives can be made prefix-distinct or whether their shared prefix can be factored out.",
+ });
+ }
+
+ if (isLookaround(node) && summary.containsRepeatableQuantifier) {
+ details.push({
+ node,
+ rule: "repeated-lookaround-body",
+ title: "Potential expensive lookaround",
+ explanation:
+ "This lookaround evaluates a body containing repetition without consuming input at the assertion site.",
+ exampleRisk:
+ "An unanchored search may re-evaluate the repeated assertion at many candidate positions.",
+ confidence: "medium",
+ severity: "warning",
+ limitations:
+ "The analyser does not know which assertion results the runtime caches or optimizes.",
+ suggestedInvestigation:
+ "Anchor or narrow the assertion where possible and test a representative non-match.",
+ });
+ expensivePrefixCandidates.push(node);
+ }
+ }
+
+ if (captureCount > EXCESSIVE_CAPTURE_THRESHOLD) {
+ details.push({
+ node: request.root,
+ rule: "excessive-captures",
+ title: "Large capture set",
+ explanation: `The pattern contains ${captureCount.toLocaleString()} capturing groups, above this analyser’s ${EXCESSIVE_CAPTURE_THRESHOLD.toLocaleString()}-group review threshold.`,
+ exampleRisk:
+ "Capturing can increase result materialization, offset conversion and replacement work.",
+ confidence: "high",
+ severity: "notice",
+ limitations:
+ "The threshold is an application review limit, not an engine vulnerability boundary.",
+ suggestedInvestigation:
+ "Convert groups that are not consumed by results or backreferences to non-capturing groups.",
+ });
+ }
+
+ if (deepest && deepest.depth > EXCESSIVE_NESTING_THRESHOLD) {
+ details.push({
+ node: deepest.node,
+ rule: "excessive-nesting",
+ title: "Deeply nested pattern",
+ explanation: `The normalized tree reaches depth ${deepest.depth.toLocaleString()}, above this analyser’s ${EXCESSIVE_NESTING_THRESHOLD.toLocaleString()}-level review threshold.`,
+ exampleRisk:
+ "Deep nesting can increase compiler, stack or explanation work even when matching remains quick.",
+ confidence: "high",
+ severity: "notice",
+ limitations:
+ "Normalized AST depth is not identical to the native engine’s internal program depth.",
+ suggestedInvestigation:
+ "Simplify unnecessary grouping and retain strict compile and execution limits.",
+ });
+ }
+
+ if (!startsAtInput(request.root) && expensivePrefixCandidates[0]) {
+ details.push({
+ node: expensivePrefixCandidates[0],
+ rule: "unanchored-expensive-prefix",
+ title: "Potential unanchored search amplification",
+ explanation:
+ "The pattern is not start-anchored and contains a structurally expensive region, so the engine may retry it at many subject positions.",
+ exampleRisk:
+ "A long non-match can multiply per-position backtracking by the subject length.",
+ confidence: request.flags.includes("y") ? "low" : "medium",
+ severity: "warning",
+ limitations:
+ "Sticky execution, runtime prefix scans and calling-code start offsets can avoid retries that are possible from the pattern alone.",
+ suggestedInvestigation:
+ "Use an appropriate start anchor or sticky execution when the intended match must begin at a known position.",
+ });
+ }
+
+ const replacement = request.replacement ?? "";
+ const repeatedExecution = request.scanAll || request.flags.includes("g");
+ if (
+ replacement.length > 0 &&
+ repeatedExecution &&
+ (replacement.includes("$`") ||
+ replacement.includes("$'") ||
+ replacement.length > 4_096)
+ ) {
+ details.push({
+ node: request.root,
+ rule: "potential-replacement-expansion",
+ title: "Potentially large replacement output",
+ explanation:
+ replacement.includes("$`") || replacement.includes("$'")
+ ? "The replacement refers to an entire subject prefix or suffix and can do so for every selected match."
+ : "A long replacement template can be emitted once for every selected match.",
+ exampleRisk:
+ "Output size and replacement time can grow much faster than the input preview suggests.",
+ confidence: "medium",
+ severity: "warning",
+ limitations:
+ "The finding uses pattern and template structure; it does not predict the actual match count for every subject.",
+ suggestedInvestigation:
+ "Use bounded replacement previews and test output size on representative maximum inputs.",
+ });
+ }
+
+ const seen = new Set();
+ const unique = details.filter((item) => {
+ const key = `${item.rule}:${item.node.range.startUtf16}:${item.node.range.endUtf16}`;
+ if (seen.has(key)) return false;
+ seen.add(key);
+ return true;
+ });
+ unique.sort(
+ (left, right) =>
+ severityOrder(left.severity) - severityOrder(right.severity) ||
+ left.node.range.startUtf16 - right.node.range.startUtf16 ||
+ left.rule.localeCompare(right.rule),
+ );
+ const findingsTruncated = unique.length > MAXIMUM_STATIC_RISK_FINDINGS;
+ const findings = unique
+ .slice(0, MAXIMUM_STATIC_RISK_FINDINGS)
+ .map((item, index) => finding(request.flavour, index + 1, item))
+ .sort(
+ (left, right) =>
+ severityOrder(left.severity) - severityOrder(right.severity) ||
+ left.range.startUtf16 - right.range.startUtf16 ||
+ left.rule.localeCompare(right.rule),
+ );
+
+ return {
+ flavour: request.flavour,
+ analyser: {
+ id: "regex-tools-ecmascript-heuristics",
+ version: "1",
+ support: "experimental-ecmascript-2025",
+ },
+ findings,
+ analysedNodes: records.length,
+ findingsTruncated,
+ traversalTruncated,
+ summary:
+ findings.length === 0
+ ? "No issue found by this analyser. This is not a safety guarantee."
+ : `${findings.length.toLocaleString()} potential ${
+ findings.length === 1 ? "risk" : "risks"
+ } found by ECMAScript-specific structural heuristics.`,
+ limitations: [
+ "Findings are advisory and do not model the native engine’s complete compiled program or optimizations.",
+ "Absence of a finding does not mean that a pattern is safe or linear.",
+ "Dynamic probes characterize only the generated inputs and limits selected by the user.",
+ ...(traversalTruncated
+ ? [
+ `Traversal stopped at ${MAXIMUM_STATIC_ANALYSIS_NODES.toLocaleString()} normalized nodes.`,
+ ]
+ : []),
+ ...(findingsTruncated
+ ? [
+ `Findings stopped at ${MAXIMUM_STATIC_RISK_FINDINGS.toLocaleString()} entries.`,
+ ]
+ : []),
+ ],
+ };
+}
diff --git a/src/regex/analysis/statistics.test.ts b/src/regex/analysis/statistics.test.ts
new file mode 100644
index 0000000..44a6fef
--- /dev/null
+++ b/src/regex/analysis/statistics.test.ts
@@ -0,0 +1,75 @@
+import { describe, expect, it } from "vitest";
+import { metricStatistics, summarizeBenchmarkSamples } from "./statistics";
+import type { BenchmarkWorkerSample } from "./analysis.types";
+
+function sample(
+ compileMs: number,
+ overrides: Partial = {},
+): BenchmarkWorkerSample {
+ return {
+ accepted: true,
+ effectiveFlags: "g",
+ subjectBytes: 10,
+ subjectUtf16: 10,
+ compileMs,
+ firstMatchMs: compileMs + 1,
+ allMatchesMs: compileMs + 2,
+ replacementMs: compileMs + 3,
+ throughputBytesPerSecond: 100 / (compileMs + 1),
+ matchCount: 1,
+ matched: true,
+ matchCollectionTruncated: false,
+ replacementOutputUtf16: 10,
+ ...overrides,
+ };
+}
+
+describe("analysis statistics", () => {
+ it("computes a conventional median and nearest-rank p95", () => {
+ expect(metricStatistics([4, 1, 3, 2])).toEqual({
+ count: 4,
+ minimum: 1,
+ median: 2.5,
+ p95: 4,
+ maximum: 4,
+ });
+ expect(
+ metricStatistics(Array.from({ length: 100 }, (_, index) => index + 1))
+ ?.p95,
+ ).toBe(95);
+ });
+
+ it("ignores absent, negative and non-finite measurements", () => {
+ expect(
+ metricStatistics([
+ undefined,
+ Number.NaN,
+ -1,
+ 0,
+ Number.POSITIVE_INFINITY,
+ ]),
+ ).toEqual({
+ count: 1,
+ minimum: 0,
+ median: 0,
+ p95: 0,
+ maximum: 0,
+ });
+ expect(metricStatistics([])).toBeUndefined();
+ });
+
+ it("summarizes only measurements actually produced by the worker", () => {
+ const summary = summarizeBenchmarkSamples([
+ sample(1),
+ sample(3, {
+ replacementMs: undefined,
+ replacementSkippedReason: "bounded output estimate",
+ }),
+ ]);
+
+ expect(summary.compileMs?.median).toBe(2);
+ expect(summary.replacementMs).toEqual(
+ expect.objectContaining({ count: 1, median: 4 }),
+ );
+ });
+});
diff --git a/src/regex/analysis/statistics.ts b/src/regex/analysis/statistics.ts
new file mode 100644
index 0000000..7d41064
--- /dev/null
+++ b/src/regex/analysis/statistics.ts
@@ -0,0 +1,61 @@
+import type {
+ BenchmarkMetricSummary,
+ BenchmarkWorkerSample,
+ MetricStatistics,
+} from "./analysis.types";
+
+function finite(values: readonly (number | undefined)[]): number[] {
+ return values.filter(
+ (value): value is number =>
+ value !== undefined && Number.isFinite(value) && value >= 0,
+ );
+}
+
+export function metricStatistics(
+ values: readonly (number | undefined)[],
+): MetricStatistics | undefined {
+ const ordered = finite(values).sort((left, right) => left - right);
+ if (ordered.length === 0) return undefined;
+ const middle = Math.floor(ordered.length / 2);
+ const median =
+ ordered.length % 2 === 0
+ ? ((ordered[middle - 1] ?? 0) + (ordered[middle] ?? 0)) / 2
+ : (ordered[middle] ?? 0);
+ const p95Index = Math.max(0, Math.ceil(ordered.length * 0.95) - 1);
+ return {
+ count: ordered.length,
+ minimum: ordered[0] ?? 0,
+ median,
+ p95: ordered[p95Index] ?? ordered.at(-1) ?? 0,
+ maximum: ordered.at(-1) ?? 0,
+ };
+}
+
+export function summarizeBenchmarkSamples(
+ samples: readonly BenchmarkWorkerSample[],
+): BenchmarkMetricSummary {
+ const metric = (
+ select: (sample: BenchmarkWorkerSample) => number | undefined,
+ ) => metricStatistics(samples.map(select));
+ return {
+ ...(metric((sample) => sample.compileMs)
+ ? { compileMs: metric((sample) => sample.compileMs) }
+ : {}),
+ ...(metric((sample) => sample.firstMatchMs)
+ ? { firstMatchMs: metric((sample) => sample.firstMatchMs) }
+ : {}),
+ ...(metric((sample) => sample.allMatchesMs)
+ ? { allMatchesMs: metric((sample) => sample.allMatchesMs) }
+ : {}),
+ ...(metric((sample) => sample.replacementMs)
+ ? { replacementMs: metric((sample) => sample.replacementMs) }
+ : {}),
+ ...(metric((sample) => sample.throughputBytesPerSecond)
+ ? {
+ throughputBytesPerSecond: metric(
+ (sample) => sample.throughputBytesPerSecond,
+ ),
+ }
+ : {}),
+ };
+}
diff --git a/src/regex/codegen/pcre2-c.test.ts b/src/regex/codegen/pcre2-c.test.ts
new file mode 100644
index 0000000..6c268a5
--- /dev/null
+++ b/src/regex/codegen/pcre2-c.test.ts
@@ -0,0 +1,116 @@
+import { describe, expect, it } from "vitest";
+import { generatePcre2C, type Pcre2CGenerationRequest } from "./pcre2-c";
+
+function request(
+ values: Partial = {},
+): Pcre2CGenerationRequest {
+ return {
+ flavour: "pcre2",
+ flavourVersion: "PCRE2 10.47 8-bit WebAssembly",
+ operation: "match",
+ pattern: "(?.+)",
+ flags: ["g", "i", "m", "s", "x", "U", "J"],
+ options: {
+ matchLimit: 1_000_000,
+ depthLimit: 1_000,
+ heapLimitKib: 32_768,
+ },
+ subject: "value",
+ scanAll: true,
+ maximumMatches: 1_000,
+ maximumCaptureRows: 10_000,
+ maximumOutputBytes: 1024 * 1024,
+ ...values,
+ };
+}
+
+describe("PCRE2 10.47 C generator", () => {
+ it("emits exact UTF-8 byte arrays instead of interpolating host literals", () => {
+ const pattern = '(?["\\\\${}]+\\R)';
+ const subject = "“quote” \\\nGrüße ${value}\0";
+ const replacement = '${x} "$" \\\\';
+ const generated = generatePcre2C(
+ request({
+ operation: "replace",
+ pattern,
+ subject,
+ replacement,
+ }),
+ );
+
+ expect(generated).toMatchObject({
+ target: "c17-pcre2-8",
+ fileName: "regex-tools-pcre2-replace.c",
+ engineIdentity: "PCRE2 10.47 8-bit",
+ represents: "replacement",
+ });
+ expect(generated.source).toContain(
+ '#error "This generated program requires PCRE2 10.47 exactly."',
+ );
+ expect(generated.source).toContain(
+ "PCRE2_UTF | PCRE2_UCP | PCRE2_CASELESS | PCRE2_MULTILINE | PCRE2_DOTALL | PCRE2_EXTENDED | PCRE2_UNGREEDY | PCRE2_DUPNAMES",
+ );
+ expect(generated.source).toContain(
+ "static const PCRE2_UCHAR replacement[] = {",
+ );
+ expect(generated.source).toContain(
+ "0x47, 0x72, 0xc3, 0xbc, 0xc3, 0x9f, 0x65",
+ );
+ expect(generated.source).not.toContain(pattern);
+ expect(generated.source).not.toContain(subject);
+ expect(generated.source).not.toContain(replacement);
+ expect(generated.caveats.join(" ")).toMatch(
+ /checked only after substitution/u,
+ );
+ expect(generated.caveats.join(" ")).toMatch(/wall-clock timeout/u);
+ });
+
+ it("states and implements application-level all-match iteration", () => {
+ const generated = generatePcre2C(
+ request({
+ flags: [],
+ scanAll: true,
+ }),
+ );
+
+ expect(generated.represents).toBe("all-matches");
+ expect(generated.source).toContain("static PCRE2_SIZE advance_utf8");
+ expect(generated.source).toContain(
+ "PCRE2_NOTEMPTY_ATSTART | PCRE2_ANCHORED",
+ );
+ expect(generated.source).toContain(
+ "Regex Tools host-side match/capture result cap reached.",
+ );
+ expect(generated.source).toContain(
+ "pcre2_set_match_limit(context, match_limit)",
+ );
+ expect(generated.source).toContain(
+ "pcre2_set_depth_limit(context, depth_limit)",
+ );
+ expect(generated.source).toContain(
+ "pcre2_set_heap_limit(context, heap_limit_kib)",
+ );
+ });
+
+ it("generates a first-match program only when neither g nor scan-all is selected", () => {
+ const generated = generatePcre2C(request({ flags: ["i"], scanAll: false }));
+
+ expect(generated.represents).toBe("first-match");
+ expect(generated.source).toContain("if (!false)");
+ expect(generated.caveats).toContain(
+ "The program requests only the first match.",
+ );
+ });
+
+ it("fails closed on unsupported mappings and lossy browser Unicode", () => {
+ expect(() => generatePcre2C(request({ flags: ["u"] }))).toThrow(
+ /unsupported/u,
+ );
+ expect(() => generatePcre2C(request({ pattern: "\ud800" }))).toThrow(
+ /unpaired UTF-16 surrogate/u,
+ );
+ expect(() =>
+ generatePcre2C(request({ operation: "replace", replacement: undefined })),
+ ).toThrow(/requires an exact replacement/u);
+ });
+});
diff --git a/src/regex/codegen/pcre2-c.ts b/src/regex/codegen/pcre2-c.ts
new file mode 100644
index 0000000..db81700
--- /dev/null
+++ b/src/regex/codegen/pcre2-c.ts
@@ -0,0 +1,591 @@
+import {
+ AVAILABLE_REGEX_FLAVOURS,
+ parseRegexFlags,
+ parseRegexOptions,
+ resolveRegexFlavourVersion,
+} from "../flavours/flavour-registry";
+import type { RegexEngineOptions } from "../model/flavour";
+import {
+ DEFAULT_REGEX_LIMITS,
+ utf8ByteLength,
+} from "../execution/request-limits";
+
+export type Pcre2COperation = "match" | "replace";
+
+export interface Pcre2CGenerationRequest {
+ readonly flavour: "pcre2";
+ readonly flavourVersion: string;
+ readonly operation: Pcre2COperation;
+ readonly pattern: string;
+ readonly flags: readonly string[];
+ readonly options: RegexEngineOptions;
+ readonly subject: string;
+ readonly scanAll: boolean;
+ readonly replacement?: string;
+ readonly maximumMatches: number;
+ readonly maximumCaptureRows: number;
+ readonly maximumOutputBytes: number;
+}
+
+export interface GeneratedPcre2CProgram {
+ readonly target: "c17-pcre2-8";
+ readonly fileName: string;
+ readonly source: string;
+ readonly compileCommand: string;
+ readonly engineIdentity: string;
+ readonly represents: "first-match" | "all-matches" | "replacement";
+ readonly caveats: readonly string[];
+}
+
+const TEXT_ENCODER = new TextEncoder();
+const MAXIMUM_GENERATED_TEXT_BYTES = 1024 * 1024;
+const PCRE2_VERSION = "PCRE2 10.47 8-bit";
+
+function integer(
+ value: number,
+ label: string,
+ minimum: number,
+ maximum: number,
+): number {
+ if (!Number.isSafeInteger(value) || value < minimum || value > maximum) {
+ throw new RangeError(
+ `${label} must be an integer from ${minimum.toLocaleString()} to ${maximum.toLocaleString()}.`,
+ );
+ }
+ return value;
+}
+
+function assertUnicodeScalarText(value: string, label: string): void {
+ for (let index = 0; index < value.length; index += 1) {
+ const first = value.charCodeAt(index);
+ if (first >= 0xd800 && first <= 0xdbff) {
+ const second = value.charCodeAt(index + 1);
+ if (second >= 0xdc00 && second <= 0xdfff) {
+ index += 1;
+ continue;
+ }
+ throw new Error(
+ `${label} contains an unpaired UTF-16 surrogate and cannot be emitted as exact UTF-8.`,
+ );
+ }
+ if (first >= 0xdc00 && first <= 0xdfff) {
+ throw new Error(
+ `${label} contains an unpaired UTF-16 surrogate and cannot be emitted as exact UTF-8.`,
+ );
+ }
+ }
+}
+
+function byteArray(name: string, value: string): string {
+ const bytes = [...TEXT_ENCODER.encode(value), 0];
+ const rows: string[] = [];
+ for (let index = 0; index < bytes.length; index += 12) {
+ rows.push(
+ ` ${bytes
+ .slice(index, index + 12)
+ .map((byte) => `0x${byte.toString(16).padStart(2, "0")}`)
+ .join(", ")},`,
+ );
+ }
+ return [
+ `static const PCRE2_UCHAR ${name}[] = {`,
+ ...rows,
+ "};",
+ `static const PCRE2_SIZE ${name}_length = sizeof(${name}) - 1;`,
+ ].join("\n");
+}
+
+function compileOptions(flags: readonly string[]): string {
+ const options = [
+ "PCRE2_UTF",
+ "PCRE2_UCP",
+ ...(flags.includes("i") ? ["PCRE2_CASELESS"] : []),
+ ...(flags.includes("m") ? ["PCRE2_MULTILINE"] : []),
+ ...(flags.includes("s") ? ["PCRE2_DOTALL"] : []),
+ ...(flags.includes("x") ? ["PCRE2_EXTENDED"] : []),
+ ...(flags.includes("U") ? ["PCRE2_UNGREEDY"] : []),
+ ...(flags.includes("J") ? ["PCRE2_DUPNAMES"] : []),
+ ];
+ return options.join(" | ");
+}
+
+function commonSource(
+ request: Pcre2CGenerationRequest,
+ options: Readonly<{
+ matchLimit: number;
+ depthLimit: number;
+ heapLimitKib: number;
+ }>,
+): string {
+ return `/*
+ * Generated by add·ideas Regex Tools for the PCRE2 10.47 8-bit C API.
+ * Pattern, subject and replacement are independent exact UTF-8 byte arrays.
+ * PCRE2_UTF and PCRE2_UCP are always enabled to match the browser runtime.
+ *
+ * Build:
+ * cc -std=c17 -Wall -Wextra -Werror regex-tools-pcre2.c \\
+ * $(pkg-config --cflags --libs libpcre2-8) -o regex-tools-pcre2
+ *
+ * There is no native PCRE2 wall-clock timeout. Run this program in a process
+ * that your application can terminate if a hard elapsed-time limit is needed.
+ */
+#define PCRE2_CODE_UNIT_WIDTH 8
+#include
+
+#include
+#include
+#include
+#include
+#include
+#include
+
+#if PCRE2_MAJOR != 10 || PCRE2_MINOR != 47
+#error "This generated program requires PCRE2 10.47 exactly."
+#endif
+
+${byteArray("pattern", request.pattern)}
+
+${byteArray("subject", request.subject)}
+
+static const uint32_t compile_options = ${compileOptions(request.flags)};
+static const uint32_t match_limit = ${options.matchLimit}u;
+static const uint32_t depth_limit = ${options.depthLimit}u;
+static const uint32_t heap_limit_kib = ${options.heapLimitKib}u;
+static const size_t maximum_matches = ${request.maximumMatches}u;
+static const size_t maximum_capture_rows = ${request.maximumCaptureRows}u;
+
+static void print_pcre2_error(const char *phase, int code) {
+ PCRE2_UCHAR message[256];
+ int length = pcre2_get_error_message(code, message, sizeof(message));
+ if (length < 0) {
+ fprintf(stderr, "%s failed with PCRE2 error %d.\\n", phase, code);
+ return;
+ }
+ fprintf(stderr, "%s failed: %.*s (PCRE2 error %d).\\n",
+ phase, length, (const char *)message, code);
+}
+
+static bool verify_runtime_version(void) {
+ char version[64] = {0};
+ uint32_t unicode = 0;
+ int status = pcre2_config(PCRE2_CONFIG_VERSION, version);
+ if (status < 0) {
+ print_pcre2_error("pcre2_config(PCRE2_CONFIG_VERSION)", status);
+ return false;
+ }
+ if (strncmp(version, "10.47 ", 6) != 0) {
+ fprintf(stderr, "Expected PCRE2 10.47, loaded %s.\\n", version);
+ return false;
+ }
+ status = pcre2_config(PCRE2_CONFIG_UNICODE, &unicode);
+ if (status != 0 || unicode != 1) {
+ fputs("The loaded PCRE2 10.47 library has no Unicode support.\\n", stderr);
+ return false;
+ }
+ return true;
+}
+
+static bool configure_limits(pcre2_match_context *context) {
+ int status = pcre2_set_match_limit(context, match_limit);
+ if (status != 0) {
+ print_pcre2_error("pcre2_set_match_limit", status);
+ return false;
+ }
+ status = pcre2_set_depth_limit(context, depth_limit);
+ if (status != 0) {
+ print_pcre2_error("pcre2_set_depth_limit", status);
+ return false;
+ }
+ status = pcre2_set_heap_limit(context, heap_limit_kib);
+ if (status != 0) {
+ print_pcre2_error("pcre2_set_heap_limit", status);
+ return false;
+ }
+ return true;
+}`;
+}
+
+function matchSource(
+ request: Pcre2CGenerationRequest,
+ global: boolean,
+): string {
+ return `${commonSource(request, {
+ matchLimit: Number(request.options.matchLimit),
+ depthLimit: Number(request.options.depthLimit),
+ heapLimitKib: Number(request.options.heapLimitKib),
+ })}
+
+/*
+ * PCRE2's "global" flag is an application loop, not a compile option.
+ * This generated loop implements the empty-match anchored retry described by
+ * the PCRE2 API and advances by one UTF-8 character only after that retry
+ * reports no match.
+ */
+static PCRE2_SIZE advance_utf8(PCRE2_SIZE offset, bool crlf_is_newline) {
+ if (offset >= subject_length) {
+ return subject_length + 1;
+ }
+ if (crlf_is_newline && subject[offset] == 0x0d &&
+ offset + 1 < subject_length &&
+ subject[offset + 1] == 0x0a) {
+ return offset + 2;
+ }
+ PCRE2_SIZE width =
+ (subject[offset] & 0x80u) == 0 ? 1 :
+ (subject[offset] & 0xe0u) == 0xc0u ? 2 :
+ (subject[offset] & 0xf0u) == 0xe0u ? 3 : 4;
+ return offset + width <= subject_length ? offset + width : subject_length + 1;
+}
+
+static const char *capture_name(pcre2_code *code, uint32_t group) {
+ uint32_t name_count = 0;
+ uint32_t entry_size = 0;
+ PCRE2_SPTR table = NULL;
+ if (pcre2_pattern_info(code, PCRE2_INFO_NAMECOUNT, &name_count) != 0 ||
+ name_count == 0 ||
+ pcre2_pattern_info(code, PCRE2_INFO_NAMEENTRYSIZE, &entry_size) != 0 ||
+ pcre2_pattern_info(code, PCRE2_INFO_NAMETABLE, &table) != 0) {
+ return NULL;
+ }
+ for (uint32_t index = 0; index < name_count; index += 1) {
+ PCRE2_SPTR entry = table + index * entry_size;
+ uint32_t number = ((uint32_t)entry[0] << 8) | entry[1];
+ if (number == group) {
+ return (const char *)(entry + 2);
+ }
+ }
+ return NULL;
+}
+
+int main(void) {
+ int exit_code = 1;
+ int error_code = 0;
+ PCRE2_SIZE error_offset = 0;
+ pcre2_code *code = NULL;
+ pcre2_match_data *match_data = NULL;
+ pcre2_match_context *match_context = NULL;
+ size_t match_count = 0;
+ size_t capture_rows = 0;
+ PCRE2_SIZE search_offset = 0;
+ uint32_t match_options = 0;
+ bool crlf_is_newline = false;
+
+ if (!verify_runtime_version()) {
+ return 1;
+ }
+ code = pcre2_compile(pattern, pattern_length, compile_options,
+ &error_code, &error_offset, NULL);
+ if (code == NULL) {
+ fprintf(stderr, "Compile error at pattern UTF-8 byte %" PRIuPTR ": ",
+ (uintptr_t)error_offset);
+ print_pcre2_error("pcre2_compile", error_code);
+ goto cleanup;
+ }
+ match_data = pcre2_match_data_create_from_pattern(code, NULL);
+ match_context = pcre2_match_context_create(NULL);
+ if (match_data == NULL || match_context == NULL) {
+ fputs("Could not allocate PCRE2 match state.\\n", stderr);
+ goto cleanup;
+ }
+ if (!configure_limits(match_context)) {
+ goto cleanup;
+ }
+ uint32_t newline = 0;
+ if (pcre2_pattern_info(code, PCRE2_INFO_NEWLINE, &newline) != 0) {
+ fputs("Could not read the PCRE2 newline convention.\\n", stderr);
+ goto cleanup;
+ }
+ crlf_is_newline =
+ newline == PCRE2_NEWLINE_CRLF ||
+ newline == PCRE2_NEWLINE_ANY ||
+ newline == PCRE2_NEWLINE_ANYCRLF;
+
+ while (search_offset <= subject_length) {
+ int status = pcre2_match(code, subject, subject_length, search_offset,
+ match_options, match_data, match_context);
+ if (status == PCRE2_ERROR_NOMATCH && match_options != 0) {
+ search_offset = advance_utf8(search_offset, crlf_is_newline);
+ match_options = 0;
+ continue;
+ }
+ if (status == PCRE2_ERROR_NOMATCH) {
+ break;
+ }
+ if (status < 0) {
+ print_pcre2_error("pcre2_match", status);
+ goto cleanup;
+ }
+ if (status == 0) {
+ fputs("The match vector was unexpectedly too small.\\n", stderr);
+ goto cleanup;
+ }
+
+ uint32_t capture_count = 0;
+ if (pcre2_pattern_info(code, PCRE2_INFO_CAPTURECOUNT, &capture_count) != 0) {
+ fputs("Could not read the PCRE2 capture count.\\n", stderr);
+ goto cleanup;
+ }
+ size_t required_rows = capture_count == 0 ? 1u : capture_count;
+ if (match_count >= maximum_matches ||
+ required_rows > maximum_capture_rows - capture_rows) {
+ fputs("Regex Tools host-side match/capture result cap reached.\\n", stderr);
+ exit_code = 3;
+ goto cleanup;
+ }
+
+ PCRE2_SIZE *ovector = pcre2_get_ovector_pointer(match_data);
+ printf("match %zu: UTF-8 bytes %" PRIuPTR "..%" PRIuPTR "\\n",
+ match_count + 1, (uintptr_t)ovector[0], (uintptr_t)ovector[1]);
+ for (uint32_t group = 0; group <= capture_count; group += 1) {
+ PCRE2_SIZE start = ovector[group * 2];
+ PCRE2_SIZE end = ovector[group * 2 + 1];
+ const char *name = capture_name(code, group);
+ if (start == PCRE2_UNSET || end == PCRE2_UNSET) {
+ printf(" group %u%s%s: did not participate\\n",
+ group, name == NULL ? "" : " / ", name == NULL ? "" : name);
+ continue;
+ }
+ printf(" group %u%s%s: UTF-8 bytes %" PRIuPTR "..%" PRIuPTR " = ",
+ group, name == NULL ? "" : " / ", name == NULL ? "" : name,
+ (uintptr_t)start, (uintptr_t)end);
+ (void)fwrite(subject + start, 1, (size_t)(end - start), stdout);
+ fputc('\\n', stdout);
+ }
+ match_count += 1;
+ capture_rows += required_rows;
+
+ if (!${global ? "true" : "false"}) {
+ break;
+ }
+ search_offset = ovector[1];
+ match_options =
+ ovector[0] == ovector[1] ? PCRE2_NOTEMPTY_ATSTART | PCRE2_ANCHORED : 0;
+ }
+
+ printf("completed: %zu match(es), %zu capture row(s)\\n",
+ match_count, capture_rows);
+ exit_code = 0;
+
+cleanup:
+ pcre2_match_context_free(match_context);
+ pcre2_match_data_free(match_data);
+ pcre2_code_free(code);
+ return exit_code;
+}
+`;
+}
+
+function replacementSource(
+ request: Pcre2CGenerationRequest,
+ global: boolean,
+): string {
+ const replacement = request.replacement;
+ if (replacement === undefined) {
+ throw new Error("PCRE2 replacement generation requires a replacement.");
+ }
+ return `${commonSource(request, {
+ matchLimit: Number(request.options.matchLimit),
+ depthLimit: Number(request.options.depthLimit),
+ heapLimitKib: Number(request.options.heapLimitKib),
+ })}
+
+${byteArray("replacement", replacement)}
+
+static const size_t maximum_output_bytes = ${request.maximumOutputBytes}u;
+
+int main(void) {
+ int exit_code = 1;
+ int error_code = 0;
+ PCRE2_SIZE error_offset = 0;
+ pcre2_code *code = NULL;
+ pcre2_match_data *match_data = NULL;
+ pcre2_match_context *match_context = NULL;
+ PCRE2_UCHAR *output = NULL;
+
+ if (!verify_runtime_version()) {
+ return 1;
+ }
+ code = pcre2_compile(pattern, pattern_length, compile_options,
+ &error_code, &error_offset, NULL);
+ if (code == NULL) {
+ fprintf(stderr, "Compile error at pattern UTF-8 byte %" PRIuPTR ": ",
+ (uintptr_t)error_offset);
+ print_pcre2_error("pcre2_compile", error_code);
+ goto cleanup;
+ }
+ match_data = pcre2_match_data_create_from_pattern(code, NULL);
+ match_context = pcre2_match_context_create(NULL);
+ output = malloc(maximum_output_bytes + 1);
+ if (match_data == NULL || match_context == NULL || output == NULL) {
+ fputs("Could not allocate PCRE2 replacement state.\\n", stderr);
+ goto cleanup;
+ }
+ if (!configure_limits(match_context)) {
+ goto cleanup;
+ }
+
+ PCRE2_SIZE output_length = maximum_output_bytes;
+ uint32_t substitute_options =
+ PCRE2_SUBSTITUTE_UNSET_EMPTY${global ? " | PCRE2_SUBSTITUTE_GLOBAL" : ""};
+ int substitutions = pcre2_substitute(
+ code, subject, subject_length, 0, substitute_options,
+ match_data, match_context, replacement, replacement_length,
+ output, &output_length);
+ if (substitutions == PCRE2_ERROR_NOMEMORY) {
+ fprintf(stderr,
+ "Replacement exceeds the %zu-byte Regex Tools output cap.\\n",
+ maximum_output_bytes);
+ exit_code = 3;
+ goto cleanup;
+ }
+ if (substitutions < 0) {
+ print_pcre2_error("pcre2_substitute", substitutions);
+ goto cleanup;
+ }
+
+ uint32_t capture_count = 0;
+ if (pcre2_pattern_info(code, PCRE2_INFO_CAPTURECOUNT, &capture_count) != 0) {
+ fputs("Could not read the PCRE2 capture count.\\n", stderr);
+ goto cleanup;
+ }
+ size_t rows_per_match = capture_count == 0 ? 1u : capture_count;
+ if ((size_t)substitutions > maximum_matches ||
+ ((size_t)substitutions != 0 &&
+ rows_per_match > maximum_capture_rows / (size_t)substitutions)) {
+ fputs("Replacement completed but exceeded the Regex Tools result cap; "
+ "output is withheld.\\n", stderr);
+ exit_code = 3;
+ goto cleanup;
+ }
+
+ (void)fwrite(output, 1, (size_t)output_length, stdout);
+ fprintf(stderr, "\\ncompleted: %d substitution(s), %zu output bytes\\n",
+ substitutions, (size_t)output_length);
+ exit_code = 0;
+
+cleanup:
+ free(output);
+ pcre2_match_context_free(match_context);
+ pcre2_match_data_free(match_data);
+ pcre2_code_free(code);
+ return exit_code;
+}
+`;
+}
+
+function validate(request: Pcre2CGenerationRequest): Pcre2CGenerationRequest {
+ if (request.flavour !== "pcre2") {
+ throw new Error("The PCRE2 C generator accepts only PCRE2 snapshots.");
+ }
+ const definition = AVAILABLE_REGEX_FLAVOURS.require("pcre2");
+ const version = resolveRegexFlavourVersion(
+ request.flavourVersion,
+ definition,
+ "PCRE2 code-generation version",
+ ).definition;
+ if (version.value !== "PCRE2 10.47 8-bit WebAssembly") {
+ throw new Error("The C generator is pinned to PCRE2 10.47 8-bit.");
+ }
+ const flags = parseRegexFlags(
+ request.flags,
+ definition,
+ "PCRE2 code-generation flags",
+ );
+ const options = parseRegexOptions(
+ request.options,
+ definition,
+ "PCRE2 code-generation options",
+ );
+ assertUnicodeScalarText(request.pattern, "Pattern");
+ assertUnicodeScalarText(request.subject, "Subject");
+ if (request.replacement !== undefined) {
+ assertUnicodeScalarText(request.replacement, "Replacement");
+ }
+ if (utf8ByteLength(request.pattern) > 1024 * 1024) {
+ throw new RangeError("PCRE2 generated patterns are limited to 1 MiB.");
+ }
+ if (utf8ByteLength(request.subject) > MAXIMUM_GENERATED_TEXT_BYTES) {
+ throw new RangeError(
+ "Generated C embeds at most 1 MiB of subject text; reduce the fixture without changing the pattern.",
+ );
+ }
+ if (
+ request.replacement !== undefined &&
+ utf8ByteLength(request.replacement) > 256 * 1024
+ ) {
+ throw new RangeError(
+ "PCRE2 generated replacements are limited to 256 KiB.",
+ );
+ }
+ if (request.operation !== "match" && request.operation !== "replace") {
+ throw new Error("Unsupported PCRE2 C generation operation.");
+ }
+ if (request.operation === "replace" && request.replacement === undefined) {
+ throw new Error("Replacement generation requires an exact replacement.");
+ }
+ return {
+ ...request,
+ flavourVersion: version.value,
+ flags,
+ options,
+ scanAll: request.scanAll === true,
+ maximumMatches: integer(
+ request.maximumMatches,
+ "Generated match limit",
+ 1,
+ DEFAULT_REGEX_LIMITS.maximumMatches,
+ ),
+ maximumCaptureRows: integer(
+ request.maximumCaptureRows,
+ "Generated capture-row limit",
+ 1,
+ DEFAULT_REGEX_LIMITS.maximumCaptureRows,
+ ),
+ maximumOutputBytes: integer(
+ request.maximumOutputBytes,
+ "Generated output limit",
+ 1,
+ DEFAULT_REGEX_LIMITS.maximumReplacementOutputBytes,
+ ),
+ };
+}
+
+export function generatePcre2C(
+ input: Pcre2CGenerationRequest,
+): GeneratedPcre2CProgram {
+ const request = validate(input);
+ const global = request.scanAll || request.flags.includes("g");
+ const represents =
+ request.operation === "replace"
+ ? "replacement"
+ : global
+ ? "all-matches"
+ : "first-match";
+ const source =
+ request.operation === "replace"
+ ? replacementSource(request, global)
+ : matchSource(request, global);
+ return {
+ target: "c17-pcre2-8",
+ fileName:
+ request.operation === "replace"
+ ? "regex-tools-pcre2-replace.c"
+ : "regex-tools-pcre2-match.c",
+ source,
+ compileCommand:
+ "cc -std=c17 -Wall -Wextra -Werror regex-tools-pcre2.c $(pkg-config --cflags --libs libpcre2-8) -o regex-tools-pcre2",
+ engineIdentity: PCRE2_VERSION,
+ represents,
+ caveats: [
+ "PCRE2_UTF and PCRE2_UCP are mandatory, matching the Regex Tools WebAssembly runtime.",
+ global
+ ? "The selected g/scan-all behavior is implemented by host iteration (or PCRE2_SUBSTITUTE_GLOBAL), not by a PCRE2 compile flag."
+ : "The program requests only the first match.",
+ request.operation === "replace"
+ ? "The output byte cap is a host buffer; PCRE2 has no Regex Tools result-row cap, so replacement row limits are checked only after substitution."
+ : "Match and capture-row caps are host-loop limits, not native PCRE2 options.",
+ "PCRE2 match, depth and heap limits are configured; PCRE2 exposes no native wall-clock timeout, so hard cancellation belongs to the containing process.",
+ "Native ranges printed by the C API are UTF-8 byte offsets; browser editor ranges are normalized separately to UTF-16.",
+ ],
+ };
+}
diff --git a/src/regex/comparison/ComparisonOrchestrator.test.ts b/src/regex/comparison/ComparisonOrchestrator.test.ts
new file mode 100644
index 0000000..94b1ac4
--- /dev/null
+++ b/src/regex/comparison/ComparisonOrchestrator.test.ts
@@ -0,0 +1,335 @@
+import { describe, expect, it, vi } from "vitest";
+import type { RegexEngineInfo } from "../model/flavour";
+import type {
+ RegexExecutionRequest,
+ RegexExecutionResult,
+ RegexReplacementRequest,
+ RegexReplacementResult,
+} from "../model/match";
+import type {
+ RegexSyntaxRequest,
+ RegexSyntaxResult,
+ ReplacementSyntaxResult,
+} from "../model/syntax";
+import { WorkerRequestError } from "../execution/WorkerSupervisor";
+import {
+ ComparisonOrchestrator,
+ type ComparisonEngineRunner,
+ type ComparisonSyntaxRunner,
+} from "./ComparisonOrchestrator";
+import type {
+ ComparisonFlavour,
+ RegexComparisonRequest,
+} from "./comparison.types";
+
+function syntax(request: RegexSyntaxRequest): RegexSyntaxResult {
+ return {
+ accepted: true,
+ root: {
+ id: `${request.flavour}-root`,
+ kind: "pattern",
+ range: { startUtf16: 0, endUtf16: request.pattern.length },
+ raw: request.pattern,
+ explanation: "Fixture syntax",
+ children: [],
+ properties: {
+ zeroWidth: false,
+ nullable: "unknown",
+ consumesInput: "conditional",
+ },
+ support: {
+ flavour: request.flavour,
+ status: "partial",
+ notes: [],
+ },
+ provenance: {
+ provider: `${request.flavour}-fixture`,
+ providerVersion: "1",
+ source: "parsed",
+ },
+ },
+ tokens: [],
+ captures: [
+ {
+ number: 1,
+ name: "word",
+ range: { startUtf16: 0, endUtf16: request.pattern.length },
+ repeated: false,
+ },
+ ],
+ diagnostics: [],
+ provider: { id: `${request.flavour}-fixture`, version: "1" },
+ coverage: {
+ status: "partial",
+ summary: "Fixture",
+ unsupportedConstructs: [],
+ },
+ };
+}
+
+function engineInfo(flavour: ComparisonFlavour): RegexEngineInfo {
+ return {
+ flavour,
+ adapterVersion: "fixture",
+ engineName:
+ flavour === "ecmascript" ? "Native ECMAScript RegExp" : "PCRE2 WASM",
+ engineVersion: flavour === "ecmascript" ? "fixture-browser" : "10.47",
+ offsetUnit: flavour === "ecmascript" ? "utf16" : "utf8-byte",
+ capabilities: {
+ compilation: true,
+ matching: true,
+ replacement: true,
+ namedCaptures: true,
+ captureHistory: false,
+ actualTrace: false,
+ benchmark: false,
+ },
+ };
+}
+
+function execution(request: RegexExecutionRequest): RegexExecutionResult {
+ const nativeEnd = request.flavour === "pcre2" ? 5 : 4;
+ return {
+ accepted: true,
+ engine: engineInfo(request.flavour as ComparisonFlavour),
+ flags: {
+ userFlags: request.flags.join(""),
+ effectiveFlags: request.flags.join(""),
+ internallyAddedIndicesFlag: request.flavour === "ecmascript",
+ internallyAddedGlobalFlag: false,
+ },
+ matches: [
+ {
+ matchNumber: 1,
+ value: "café",
+ valueStatus: "complete",
+ range: { startUtf16: 0, endUtf16: 4 },
+ nativeRange: {
+ start: 0,
+ end: nativeEnd,
+ unit: request.flavour === "pcre2" ? "utf8-byte" : "utf16",
+ },
+ captures: [
+ {
+ groupNumber: 1,
+ groupName: "word",
+ value: "café",
+ status: "participated",
+ range: { startUtf16: 0, endUtf16: 4 },
+ nativeRange: {
+ start: 0,
+ end: nativeEnd,
+ unit: request.flavour === "pcre2" ? "utf8-byte" : "utf16",
+ },
+ },
+ ],
+ },
+ ],
+ diagnostics: [],
+ elapsedMs: 1,
+ truncated: false,
+ };
+}
+
+function request(
+ operation: "match" | "replace" = "match",
+): RegexComparisonRequest {
+ return {
+ schemaVersion: 1,
+ patternModel: "shared",
+ operation,
+ subject: "café",
+ scanAll: true,
+ maximumMatches: 100,
+ maximumCaptureRows: 1_000,
+ maximumOutputBytes: 1_024,
+ timeoutMs: 500,
+ sides: [
+ {
+ flavour: "ecmascript",
+ flavourVersion: "ECMAScript 2025 syntax / current browser runtime",
+ pattern: "(?.+)",
+ flags: ["g", "u"],
+ options: {},
+ ...(operation === "replace" ? { replacement: "$!" } : {}),
+ },
+ {
+ flavour: "pcre2",
+ flavourVersion: "PCRE2 10.47 8-bit WebAssembly",
+ pattern: "(?.+)",
+ flags: ["g"],
+ options: {
+ matchLimit: 1_000_000,
+ depthLimit: 1_000,
+ heapLimitKib: 32_768,
+ },
+ ...(operation === "replace" ? { replacement: "${word}!" } : {}),
+ },
+ ],
+ };
+}
+
+function syntaxRunner(
+ parsePattern = async (value: RegexSyntaxRequest) => syntax(value),
+): ComparisonSyntaxRunner {
+ return {
+ parsePattern,
+ parseReplacement: async (): Promise => ({
+ accepted: true,
+ tokens: [],
+ diagnostics: [],
+ }),
+ cancel: vi.fn(),
+ dispose: vi.fn(),
+ };
+}
+
+function engineRunner(
+ execute = async (value: RegexExecutionRequest) => execution(value),
+): ComparisonEngineRunner {
+ return {
+ execute,
+ replace: async (
+ value: RegexReplacementRequest,
+ ): Promise => ({
+ execution: await execute(value),
+ output: `${value.subject}!`,
+ outputBytes: value.subject.length + 1,
+ outputTruncated: false,
+ truncated: false,
+ }),
+ cancel: vi.fn(),
+ dispose: vi.fn(),
+ };
+}
+
+describe("ComparisonOrchestrator", () => {
+ it("retains exact requests and aligns UTF-16 results while preserving native offsets", async () => {
+ let now = 0;
+ const runner = engineRunner();
+ const orchestrator = new ComparisonOrchestrator({
+ createSyntaxRunner: () => syntaxRunner(),
+ createEngineRunner: () => runner,
+ now: () => (now += 1),
+ isoNow: () => "2026-07-26T12:00:00.000Z",
+ });
+
+ const result = await orchestrator.compare(request());
+
+ expect(result.status).toBe("same-for-current-input");
+ expect(result.notComparable).toEqual([]);
+ expect(result.matchAlignments).toHaveLength(1);
+ expect(result.matchAlignments[0]).toMatchObject({
+ alignment: "normalized-range",
+ equal: true,
+ left: {
+ nativeRange: { end: 4, unit: "utf16" },
+ },
+ right: {
+ nativeRange: { end: 5, unit: "utf8-byte" },
+ },
+ });
+ expect(result.differences.map((difference) => difference.kind)).toEqual([
+ "effective-flags",
+ "offset-model",
+ ]);
+ expect(result.sides[0].executionRequest).toMatchObject({
+ flavour: "ecmascript",
+ pattern: "(?.+)",
+ subject: "café",
+ captureMetadata: [{ number: 1, name: "word" }],
+ });
+ expect(result.sides[1].requestIdentity).toMatch(/^request-[0-9a-f]{8}$/u);
+ expect(result.notices[0]?.message).toMatch(/not a proof/u);
+ orchestrator.dispose();
+ });
+
+ it("records an independent timeout instead of interpreting it as no match", async () => {
+ const runner = engineRunner(async (value) => {
+ if (value.flavour === "pcre2") {
+ throw new WorkerRequestError("timeout", "PCRE2 exceeded 500 ms");
+ }
+ return execution(value);
+ });
+ const orchestrator = new ComparisonOrchestrator({
+ createSyntaxRunner: () => syntaxRunner(),
+ createEngineRunner: () => runner,
+ now: () => 1,
+ isoNow: () => "2026-07-26T12:00:00.000Z",
+ });
+
+ const result = await orchestrator.compare(request());
+
+ expect(result.status).toBe("not-comparable");
+ expect(result.sides[0].runtime.status).toBe("complete");
+ expect(result.sides[1].runtime.status).toBe("timeout");
+ expect(result.notComparable).toContainEqual(
+ expect.objectContaining({
+ code: "execution-timeout",
+ flavour: "pcre2",
+ }),
+ );
+ expect(result.notComparable[0]?.message).toMatch(/not a no-match/u);
+ orchestrator.dispose();
+ });
+
+ it("keeps per-flavour replacement templates exact", async () => {
+ const seen: RegexReplacementRequest[] = [];
+ const runner = engineRunner();
+ runner.replace = async (value) => {
+ seen.push(value);
+ return {
+ execution: execution(value),
+ output:
+ value.flavour === "ecmascript"
+ ? value.replacement
+ : value.replacement,
+ outputBytes: value.replacement.length,
+ outputTruncated: false,
+ truncated: false,
+ };
+ };
+ const orchestrator = new ComparisonOrchestrator({
+ createSyntaxRunner: () => syntaxRunner(),
+ createEngineRunner: () => runner,
+ now: () => 1,
+ isoNow: () => "2026-07-26T12:00:00.000Z",
+ });
+
+ const result = await orchestrator.compare(request("replace"));
+
+ expect(seen.map((value) => value.replacement)).toEqual([
+ "$!",
+ "${word}!",
+ ]);
+ expect(result.status).toBe("different-for-current-input");
+ expect(
+ result.differences.find(
+ (difference) => difference.kind === "replacement-output",
+ ),
+ ).toMatchObject({
+ left: "$!",
+ right: "${word}!",
+ });
+ orchestrator.dispose();
+ });
+
+ it("refuses mismatched shared patterns before constructing a semantic result", async () => {
+ const input = request();
+ const invalid: RegexComparisonRequest = {
+ ...input,
+ sides: [input.sides[0], { ...input.sides[1], pattern: "(?\\w+)" }],
+ };
+ const orchestrator = new ComparisonOrchestrator({
+ createSyntaxRunner: () => syntaxRunner(),
+ createEngineRunner: () => engineRunner(),
+ now: () => 1,
+ isoNow: () => "2026-07-26T12:00:00.000Z",
+ });
+
+ await expect(orchestrator.compare(invalid)).rejects.toThrow(
+ /exact same pattern/u,
+ );
+ orchestrator.dispose();
+ });
+});
diff --git a/src/regex/comparison/ComparisonOrchestrator.ts b/src/regex/comparison/ComparisonOrchestrator.ts
new file mode 100644
index 0000000..a4eabca
--- /dev/null
+++ b/src/regex/comparison/ComparisonOrchestrator.ts
@@ -0,0 +1,536 @@
+import {
+ AVAILABLE_REGEX_FLAVOURS,
+ parseRegexFlags,
+ parseRegexOptions,
+ resolveRegexFlavourVersion,
+} from "../flavours/flavour-registry";
+import type {
+ RegexExecutionRequest,
+ RegexReplacementRequest,
+} from "../model/match";
+import type {
+ RegexSyntaxRequest,
+ RegexSyntaxResult,
+ ReplacementSyntaxResult,
+} from "../model/syntax";
+import {
+ DEFAULT_REGEX_LIMITS,
+ utf8ByteLength,
+} from "../execution/request-limits";
+import { EngineSupervisor } from "../execution/EngineSupervisor";
+import { SyntaxSupervisor } from "../execution/SyntaxSupervisor";
+import { WorkerRequestError } from "../execution/WorkerSupervisor";
+import { compareCompletedResults } from "./compare-results";
+import type {
+ ComparisonExecutionOutcome,
+ ComparisonFlavour,
+ ComparisonSideInput,
+ ComparisonSideOutcome,
+ ComparisonSyntaxOutcome,
+ ComparisonTaskFailure,
+ RegexComparisonRequest,
+ RegexComparisonResult,
+} from "./comparison.types";
+
+export interface ComparisonSyntaxRunner {
+ parsePattern(
+ request: RegexSyntaxRequest,
+ timeoutMs?: number,
+ ): Promise;
+ parseReplacement(
+ request: {
+ readonly flavour: ComparisonFlavour;
+ readonly replacement: string;
+ readonly captureMetadata: RegexSyntaxResult["captures"];
+ },
+ timeoutMs?: number,
+ ): Promise;
+ cancel(): void;
+ dispose(): void;
+}
+
+export interface ComparisonEngineRunner {
+ execute(
+ request: RegexExecutionRequest,
+ timeoutMs: number,
+ ): ReturnType;
+ replace(
+ request: RegexReplacementRequest,
+ timeoutMs: number,
+ ): ReturnType;
+ cancel(): void;
+ dispose(): void;
+}
+
+export interface ComparisonOrchestratorDependencies {
+ readonly createSyntaxRunner: (
+ flavour: ComparisonFlavour,
+ ) => ComparisonSyntaxRunner;
+ readonly createEngineRunner: () => ComparisonEngineRunner;
+ readonly now: () => number;
+ readonly isoNow: () => string;
+}
+
+const DEFAULT_DEPENDENCIES: ComparisonOrchestratorDependencies = {
+ createSyntaxRunner: (flavour) =>
+ new SyntaxSupervisor(
+ `${flavour} comparison syntax`,
+ `regex-tools-comparison-${flavour}-syntax`,
+ ),
+ createEngineRunner: () => new EngineSupervisor(),
+ now: () => performance.now(),
+ isoNow: () => new Date().toISOString(),
+};
+
+interface PreparedSide {
+ readonly input: ComparisonSideInput;
+ readonly syntaxRequest: RegexSyntaxRequest;
+}
+
+function requireBoundedInteger(
+ value: number,
+ label: string,
+ minimum: number,
+ maximum: number,
+): number {
+ if (!Number.isSafeInteger(value) || value < minimum || value > maximum) {
+ throw new RangeError(
+ `${label} must be an integer from ${minimum.toLocaleString()} to ${maximum.toLocaleString()}.`,
+ );
+ }
+ return value;
+}
+
+function validateSide(input: ComparisonSideInput): PreparedSide {
+ const definition = AVAILABLE_REGEX_FLAVOURS.require(input.flavour);
+ const version = resolveRegexFlavourVersion(
+ input.flavourVersion,
+ definition,
+ `${definition.label} comparison version`,
+ ).definition;
+ const flags = parseRegexFlags(
+ input.flags,
+ definition,
+ `${definition.label} comparison flags`,
+ );
+ const options = parseRegexOptions(
+ input.options,
+ definition,
+ `${definition.label} comparison options`,
+ );
+ if (input.pattern.length > DEFAULT_REGEX_LIMITS.patternHardLengthUtf16) {
+ throw new RangeError(
+ `${definition.label} comparison pattern exceeds the ${DEFAULT_REGEX_LIMITS.patternHardLengthUtf16.toLocaleString()} UTF-16 unit limit.`,
+ );
+ }
+ if (
+ input.replacement !== undefined &&
+ input.replacement.length >
+ DEFAULT_REGEX_LIMITS.maximumReplacementTemplateUtf16
+ ) {
+ throw new RangeError(
+ `${definition.label} comparison replacement exceeds the ${DEFAULT_REGEX_LIMITS.maximumReplacementTemplateUtf16.toLocaleString()} UTF-16 unit limit.`,
+ );
+ }
+ const normalized: ComparisonSideInput = {
+ flavour: input.flavour,
+ flavourVersion: version.value,
+ pattern: input.pattern,
+ flags,
+ options,
+ ...(input.replacement === undefined
+ ? {}
+ : { replacement: input.replacement }),
+ };
+ return {
+ input: normalized,
+ syntaxRequest: {
+ flavour: normalized.flavour,
+ flavourVersion: version.syntaxVersion,
+ pattern: normalized.pattern,
+ flags: normalized.flags,
+ options: normalized.options,
+ },
+ };
+}
+
+export function validateRegexComparisonRequest(
+ request: RegexComparisonRequest,
+): {
+ readonly request: RegexComparisonRequest;
+ readonly sides: readonly [PreparedSide, PreparedSide];
+} {
+ if (request.schemaVersion !== 1) {
+ throw new Error("Unsupported comparison request schema.");
+ }
+ if (
+ request.patternModel !== "shared" &&
+ request.patternModel !== "variants"
+ ) {
+ throw new Error("Comparison pattern model is unsupported.");
+ }
+ if (request.operation !== "match" && request.operation !== "replace") {
+ throw new Error("Comparison operation is unsupported.");
+ }
+ const flavours = request.sides.map((side) => side.flavour);
+ if (
+ new Set(flavours).size !== 2 ||
+ !flavours.includes("ecmascript") ||
+ !flavours.includes("pcre2")
+ ) {
+ throw new Error(
+ "This comparison vertical requires one ECMAScript and one PCRE2 side.",
+ );
+ }
+ const preparedInput = request.sides.map(validateSide);
+ const ecmascript = preparedInput.find(
+ (side) => side.input.flavour === "ecmascript",
+ );
+ const pcre2 = preparedInput.find((side) => side.input.flavour === "pcre2");
+ if (!ecmascript || !pcre2) {
+ throw new Error("Comparison sides could not be normalized.");
+ }
+ const prepared = [ecmascript, pcre2] as const;
+ if (
+ request.patternModel === "shared" &&
+ prepared[0].input.pattern !== prepared[1].input.pattern
+ ) {
+ throw new Error(
+ "Shared-pattern comparison must send the exact same pattern to both engines.",
+ );
+ }
+ if (
+ request.operation === "replace" &&
+ prepared.some((side) => side.input.replacement === undefined)
+ ) {
+ throw new Error(
+ "Replacement comparison requires an explicit replacement for each flavour.",
+ );
+ }
+ if (
+ utf8ByteLength(request.subject) >
+ DEFAULT_REGEX_LIMITS.interactiveSubjectHardBytes
+ ) {
+ throw new RangeError(
+ `Comparison subject exceeds the ${DEFAULT_REGEX_LIMITS.interactiveSubjectHardBytes.toLocaleString()} UTF-8 byte limit.`,
+ );
+ }
+ const maximumMatches = requireBoundedInteger(
+ request.maximumMatches,
+ "Comparison match limit",
+ 1,
+ DEFAULT_REGEX_LIMITS.maximumMatches,
+ );
+ const maximumCaptureRows = requireBoundedInteger(
+ request.maximumCaptureRows,
+ "Comparison capture-row limit",
+ 1,
+ DEFAULT_REGEX_LIMITS.maximumCaptureRows,
+ );
+ const maximumOutputBytes = requireBoundedInteger(
+ request.maximumOutputBytes,
+ "Comparison replacement-output limit",
+ 1,
+ DEFAULT_REGEX_LIMITS.maximumReplacementOutputBytes,
+ );
+ const timeoutMs = requireBoundedInteger(
+ request.timeoutMs,
+ "Comparison side timeout",
+ 1,
+ DEFAULT_REGEX_LIMITS.advancedMaximumTimeoutMs,
+ );
+ const normalized: RegexComparisonRequest = {
+ ...request,
+ scanAll: request.scanAll === true,
+ maximumMatches,
+ maximumCaptureRows,
+ maximumOutputBytes,
+ timeoutMs,
+ sides: [prepared[0].input, prepared[1].input],
+ };
+ return { request: normalized, sides: prepared };
+}
+
+function classifyFailure(error: unknown): ComparisonTaskFailure {
+ if (error instanceof WorkerRequestError) {
+ if (error.kind === "timeout") {
+ return { status: "timeout", message: error.message };
+ }
+ if (error.kind === "cancelled") {
+ return { status: "cancelled", message: error.message };
+ }
+ }
+ return {
+ status: "worker-failed",
+ message: error instanceof Error ? error.message : String(error),
+ };
+}
+
+function stableOptions(options: ComparisonSideInput["options"]): string {
+ return JSON.stringify(
+ Object.fromEntries(
+ Object.entries(options).sort(([left], [right]) =>
+ left.localeCompare(right),
+ ),
+ ),
+ );
+}
+
+function requestIdentity(
+ request: RegexExecutionRequest | RegexReplacementRequest,
+): string {
+ const replacement =
+ "replacement" in request ? `\u0000${request.replacement}` : "";
+ const value = [
+ request.flavour,
+ request.flavourVersion ?? "",
+ request.pattern,
+ request.flags.join(""),
+ stableOptions(request.options ?? {}),
+ request.subject,
+ String(request.scanAll),
+ String(request.maximumMatches),
+ String(request.maximumCaptureRows),
+ JSON.stringify(request.captureMetadata),
+ "maximumOutputBytes" in request ? String(request.maximumOutputBytes) : "",
+ replacement,
+ ].join("\u0001");
+ let hash = 0x811c9dc5;
+ for (let index = 0; index < value.length; index += 1) {
+ hash ^= value.charCodeAt(index);
+ hash = Math.imul(hash, 0x01000193);
+ }
+ return `request-${(hash >>> 0).toString(16).padStart(8, "0")}`;
+}
+
+function executionRequest(
+ request: RegexComparisonRequest,
+ prepared: PreparedSide,
+ captures: RegexSyntaxResult["captures"],
+): RegexExecutionRequest {
+ return {
+ flavour: prepared.input.flavour,
+ flavourVersion: prepared.input.flavourVersion,
+ pattern: prepared.input.pattern,
+ flags: prepared.input.flags,
+ options: prepared.input.options,
+ subject: request.subject,
+ captureMetadata: captures,
+ scanAll: request.scanAll,
+ maximumMatches: request.maximumMatches,
+ maximumCaptureRows: request.maximumCaptureRows,
+ };
+}
+
+export class ComparisonOrchestrator {
+ readonly #dependencies: ComparisonOrchestratorDependencies;
+ readonly #syntax: ReadonlyMap;
+ readonly #engine: ComparisonEngineRunner;
+ #generation = 0;
+ #disposed = false;
+
+ constructor(
+ dependencies: ComparisonOrchestratorDependencies = DEFAULT_DEPENDENCIES,
+ ) {
+ this.#dependencies = dependencies;
+ this.#syntax = new Map(
+ (["ecmascript", "pcre2"] as const).map((flavour) => [
+ flavour,
+ dependencies.createSyntaxRunner(flavour),
+ ]),
+ );
+ this.#engine = dependencies.createEngineRunner();
+ }
+
+ async compare(input: RegexComparisonRequest): Promise {
+ if (this.#disposed) {
+ throw new Error("Comparison orchestrator has been disposed.");
+ }
+ this.cancel();
+ const generation = this.#generation;
+ const validated = validateRegexComparisonRequest(input);
+ const startedAt = this.#dependencies.isoNow();
+ const start = this.#dependencies.now();
+ const settled = await Promise.allSettled(
+ validated.sides.map((side) =>
+ this.#runSide(validated.request, side, generation),
+ ),
+ );
+ const outcomes = settled.map((entry, index) => {
+ if (entry.status === "fulfilled") return entry.value;
+ const prepared = validated.sides[index];
+ if (!prepared) throw entry.reason;
+ return this.#failedSide(
+ validated.request,
+ prepared,
+ classifyFailure(entry.reason),
+ );
+ }) as unknown as readonly [ComparisonSideOutcome, ComparisonSideOutcome];
+ return compareCompletedResults(
+ validated.request,
+ startedAt,
+ this.#dependencies.now() - start,
+ outcomes,
+ );
+ }
+
+ cancel(): void {
+ this.#generation += 1;
+ for (const syntax of this.#syntax.values()) syntax.cancel();
+ this.#engine.cancel();
+ }
+
+ dispose(): void {
+ if (this.#disposed) return;
+ this.cancel();
+ this.#disposed = true;
+ for (const syntax of this.#syntax.values()) syntax.dispose();
+ this.#engine.dispose();
+ }
+
+ async #runSide(
+ request: RegexComparisonRequest,
+ prepared: PreparedSide,
+ generation: number,
+ ): Promise {
+ const syntaxRunner = this.#syntax.get(prepared.input.flavour);
+ if (!syntaxRunner) {
+ throw new Error(`Missing ${prepared.input.flavour} syntax runner.`);
+ }
+ const syntaxStart = this.#dependencies.now();
+ let patternSyntax: RegexSyntaxResult | undefined;
+ let replacementSyntax: ReplacementSyntaxResult | undefined;
+ let syntaxFailure: ComparisonTaskFailure | undefined;
+ try {
+ patternSyntax = await syntaxRunner.parsePattern(
+ prepared.syntaxRequest,
+ Math.min(1_000, request.timeoutMs),
+ );
+ if (request.operation === "replace") {
+ replacementSyntax = await syntaxRunner.parseReplacement(
+ {
+ flavour: prepared.input.flavour,
+ replacement: prepared.input.replacement ?? "",
+ captureMetadata: patternSyntax.captures,
+ },
+ Math.min(1_000, request.timeoutMs),
+ );
+ }
+ } catch (error) {
+ syntaxFailure = classifyFailure(error);
+ }
+ const syntax: ComparisonSyntaxOutcome = {
+ status: syntaxFailure?.status ?? "complete",
+ elapsedMs: this.#dependencies.now() - syntaxStart,
+ ...(patternSyntax ? { pattern: patternSyntax } : {}),
+ ...(replacementSyntax ? { replacement: replacementSyntax } : {}),
+ ...(syntaxFailure ? { error: syntaxFailure.message } : {}),
+ };
+ const exactExecution = executionRequest(
+ request,
+ prepared,
+ patternSyntax?.captures ?? [],
+ );
+ const exactReplacement: RegexReplacementRequest | undefined =
+ request.operation === "replace"
+ ? {
+ ...exactExecution,
+ replacement: prepared.input.replacement ?? "",
+ maximumOutputBytes: request.maximumOutputBytes,
+ }
+ : undefined;
+ if (generation !== this.#generation) {
+ const runtime: ComparisonExecutionOutcome = {
+ status: "cancelled",
+ elapsedMs: 0,
+ error: "Comparison was cancelled before engine execution.",
+ };
+ return {
+ requestIdentity: requestIdentity(exactReplacement ?? exactExecution),
+ input: prepared.input,
+ syntaxRequest: prepared.syntaxRequest,
+ executionRequest: exactExecution,
+ ...(exactReplacement ? { replacementRequest: exactReplacement } : {}),
+ syntax,
+ runtime,
+ };
+ }
+
+ const runtimeStart = this.#dependencies.now();
+ let runtime: ComparisonExecutionOutcome;
+ try {
+ if (exactReplacement) {
+ const replacement = await this.#engine.replace(
+ exactReplacement,
+ request.timeoutMs,
+ );
+ runtime = {
+ status: "complete",
+ elapsedMs: this.#dependencies.now() - runtimeStart,
+ replacement,
+ };
+ } else {
+ const execution = await this.#engine.execute(
+ exactExecution,
+ request.timeoutMs,
+ );
+ runtime = {
+ status: "complete",
+ elapsedMs: this.#dependencies.now() - runtimeStart,
+ execution,
+ };
+ }
+ } catch (error) {
+ const failure = classifyFailure(error);
+ runtime = {
+ status: failure.status,
+ elapsedMs: this.#dependencies.now() - runtimeStart,
+ error: failure.message,
+ };
+ }
+ const execution = runtime.replacement?.execution ?? runtime.execution;
+ return {
+ requestIdentity: requestIdentity(exactReplacement ?? exactExecution),
+ input: prepared.input,
+ syntaxRequest: prepared.syntaxRequest,
+ executionRequest: exactExecution,
+ ...(exactReplacement ? { replacementRequest: exactReplacement } : {}),
+ syntax,
+ runtime,
+ ...(execution ? { engine: execution.engine } : {}),
+ };
+ }
+
+ #failedSide(
+ request: RegexComparisonRequest,
+ prepared: PreparedSide,
+ failure: ComparisonTaskFailure,
+ ): ComparisonSideOutcome {
+ const exactExecution = executionRequest(request, prepared, []);
+ const exactReplacement =
+ request.operation === "replace"
+ ? {
+ ...exactExecution,
+ replacement: prepared.input.replacement ?? "",
+ maximumOutputBytes: request.maximumOutputBytes,
+ }
+ : undefined;
+ return {
+ requestIdentity: requestIdentity(exactReplacement ?? exactExecution),
+ input: prepared.input,
+ syntaxRequest: prepared.syntaxRequest,
+ executionRequest: exactExecution,
+ ...(exactReplacement ? { replacementRequest: exactReplacement } : {}),
+ syntax: {
+ status: "worker-failed",
+ elapsedMs: 0,
+ error: "Comparison side failed before syntax completion.",
+ },
+ runtime: {
+ status: failure.status,
+ elapsedMs: 0,
+ error: failure.message,
+ },
+ };
+ }
+}
diff --git a/src/regex/comparison/compare-results.ts b/src/regex/comparison/compare-results.ts
new file mode 100644
index 0000000..d56323b
--- /dev/null
+++ b/src/regex/comparison/compare-results.ts
@@ -0,0 +1,666 @@
+import type {
+ CaptureResult,
+ RegexExecutionResult,
+ RegexMatchResult,
+} from "../model/match";
+import type {
+ CaptureAlignment,
+ ComparisonDifference,
+ ComparisonDifferenceKind,
+ ComparisonFlavour,
+ ComparisonNotComparableReason,
+ ComparisonSideOutcome,
+ MatchAlignment,
+ RegexComparisonRequest,
+ RegexComparisonResult,
+} from "./comparison.types";
+
+const MAXIMUM_RETAINED_DIFFERENCES = 5_000;
+const MAXIMUM_RETAINED_MATCH_ALIGNMENTS = 2_000;
+const MAXIMUM_DIFFERENCE_VALUE_UTF16 = 1_024;
+
+function differenceValue(value: string | undefined): string | undefined {
+ if (value === undefined || value.length <= MAXIMUM_DIFFERENCE_VALUE_UTF16) {
+ return value;
+ }
+ return `${value.slice(0, MAXIMUM_DIFFERENCE_VALUE_UTF16)}… (${value.length.toLocaleString()} UTF-16 units; complete value retained in its side result)`;
+}
+
+class DifferenceCollector {
+ readonly retained: ComparisonDifference[] = [];
+ total = 0;
+ semanticTotal = 0;
+
+ add(
+ kind: ComparisonDifferenceKind,
+ summary: string,
+ left?: string,
+ right?: string,
+ ): void {
+ this.total += 1;
+ if (kind !== "effective-flags" && kind !== "offset-model") {
+ this.semanticTotal += 1;
+ }
+ if (this.retained.length >= MAXIMUM_RETAINED_DIFFERENCES) return;
+ this.retained.push({
+ kind,
+ summary,
+ ...(left === undefined ? {} : { left }),
+ ...(right === undefined ? {} : { right }),
+ });
+ }
+}
+
+function rangeKey(
+ value:
+ | {
+ readonly range?: {
+ readonly startUtf16: number;
+ readonly endUtf16: number;
+ };
+ }
+ | undefined,
+): string | undefined {
+ const range = value?.range;
+ return range ? `${range.startUtf16}:${range.endUtf16}` : undefined;
+}
+
+function displayRange(
+ value:
+ | {
+ readonly range?: {
+ readonly startUtf16: number;
+ readonly endUtf16: number;
+ };
+ }
+ | undefined,
+): string {
+ const range = value?.range;
+ return range ? `${range.startUtf16}–${range.endUtf16}` : "unavailable";
+}
+
+function captureIdentity(capture: CaptureResult): string {
+ return capture.groupName
+ ? `name:${capture.groupName}`
+ : `number:${capture.groupNumber}`;
+}
+
+function takeMatching(
+ values: readonly T[],
+ used: Set,
+ predicate: (value: T) => boolean,
+): { readonly value: T; readonly index: number } | undefined {
+ for (let index = 0; index < values.length; index += 1) {
+ if (used.has(index)) continue;
+ const value = values[index];
+ if (value !== undefined && predicate(value)) return { value, index };
+ }
+ return undefined;
+}
+
+function captureEqual(left: CaptureResult, right: CaptureResult): boolean {
+ return (
+ left.groupNumber === right.groupNumber &&
+ left.groupName === right.groupName &&
+ left.status === right.status &&
+ rangeKey(left) === rangeKey(right) &&
+ left.value === right.value
+ );
+}
+
+function compareCapture(
+ left: CaptureResult | undefined,
+ right: CaptureResult | undefined,
+ alignment: CaptureAlignment["alignment"],
+ differences: DifferenceCollector,
+): CaptureAlignment {
+ if (!left || !right) {
+ differences.add(
+ "capture-presence",
+ "A capture is present on only one side.",
+ left ? captureIdentity(left) : "absent",
+ right ? captureIdentity(right) : "absent",
+ );
+ return {
+ alignment,
+ ...(left ? { left } : {}),
+ ...(right ? { right } : {}),
+ equal: false,
+ };
+ }
+ if (
+ left.groupNumber !== right.groupNumber ||
+ left.groupName !== right.groupName
+ ) {
+ differences.add(
+ "capture-identity",
+ "Aligned captures have different group identities.",
+ captureIdentity(left),
+ captureIdentity(right),
+ );
+ }
+ if (left.status !== right.status) {
+ differences.add(
+ "capture-status",
+ "Aligned captures have different participation states.",
+ left.status,
+ right.status,
+ );
+ }
+ if (rangeKey(left) !== rangeKey(right)) {
+ differences.add(
+ "capture-range",
+ "Aligned captures have different normalized UTF-16 ranges.",
+ displayRange(left),
+ displayRange(right),
+ );
+ }
+ if (left.value !== right.value) {
+ differences.add(
+ "capture-value",
+ "Aligned captures have different retained values.",
+ differenceValue(left.value) ?? "unavailable",
+ differenceValue(right.value) ?? "unavailable",
+ );
+ }
+ return {
+ alignment,
+ left,
+ right,
+ equal: captureEqual(left, right),
+ };
+}
+
+function alignCaptures(
+ left: readonly CaptureResult[],
+ right: readonly CaptureResult[],
+ differences: DifferenceCollector,
+): readonly CaptureAlignment[] {
+ const alignments: CaptureAlignment[] = [];
+ const usedRight = new Set();
+ const deferredLeft: CaptureResult[] = [];
+
+ for (const capture of left) {
+ const key = rangeKey(capture);
+ const sameRange = key
+ ? (takeMatching(
+ right,
+ usedRight,
+ (candidate) =>
+ rangeKey(candidate) === key &&
+ captureIdentity(candidate) === captureIdentity(capture),
+ ) ??
+ takeMatching(
+ right,
+ usedRight,
+ (candidate) => rangeKey(candidate) === key,
+ ))
+ : undefined;
+ if (!sameRange) {
+ deferredLeft.push(capture);
+ continue;
+ }
+ usedRight.add(sameRange.index);
+ alignments.push(
+ compareCapture(capture, sameRange.value, "normalized-range", differences),
+ );
+ }
+
+ const stillDeferred: CaptureResult[] = [];
+ for (const capture of deferredLeft) {
+ const sameIdentity = takeMatching(
+ right,
+ usedRight,
+ (candidate) => captureIdentity(candidate) === captureIdentity(capture),
+ );
+ if (!sameIdentity) {
+ stillDeferred.push(capture);
+ continue;
+ }
+ usedRight.add(sameIdentity.index);
+ alignments.push(
+ compareCapture(
+ capture,
+ sameIdentity.value,
+ "capture-identity",
+ differences,
+ ),
+ );
+ }
+
+ const remainingRight = right
+ .map((value, index) => ({ value, index }))
+ .filter(({ index }) => !usedRight.has(index));
+ const paired = Math.min(stillDeferred.length, remainingRight.length);
+ for (let index = 0; index < paired; index += 1) {
+ const leftCapture = stillDeferred[index];
+ const rightCapture = remainingRight[index];
+ if (!leftCapture || !rightCapture) continue;
+ usedRight.add(rightCapture.index);
+ alignments.push(
+ compareCapture(
+ leftCapture,
+ rightCapture.value,
+ "ordinal-fallback",
+ differences,
+ ),
+ );
+ }
+ for (const capture of stillDeferred.slice(paired)) {
+ alignments.push(
+ compareCapture(capture, undefined, "left-only", differences),
+ );
+ }
+ for (const { value, index } of remainingRight.slice(paired)) {
+ usedRight.add(index);
+ alignments.push(
+ compareCapture(undefined, value, "right-only", differences),
+ );
+ }
+ return alignments;
+}
+
+function matchEqual(
+ left: RegexMatchResult,
+ right: RegexMatchResult,
+ captures: readonly CaptureAlignment[],
+): boolean {
+ return (
+ rangeKey(left) === rangeKey(right) &&
+ left.value === right.value &&
+ left.valueStatus === right.valueStatus &&
+ captures.every((capture) => capture.equal)
+ );
+}
+
+function compareMatch(
+ left: RegexMatchResult | undefined,
+ right: RegexMatchResult | undefined,
+ alignment: MatchAlignment["alignment"],
+ differences: DifferenceCollector,
+): MatchAlignment {
+ if (!left || !right) {
+ differences.add(
+ "match-presence",
+ "A match is present on only one side.",
+ left ? `match ${left.matchNumber} at ${displayRange(left)}` : "absent",
+ right ? `match ${right.matchNumber} at ${displayRange(right)}` : "absent",
+ );
+ return {
+ alignment,
+ ...(left ? { left } : {}),
+ ...(right ? { right } : {}),
+ captures: [],
+ equal: false,
+ };
+ }
+ if (rangeKey(left) !== rangeKey(right)) {
+ differences.add(
+ "match-range",
+ "Aligned matches have different normalized UTF-16 ranges.",
+ displayRange(left),
+ displayRange(right),
+ );
+ }
+ if (left.value !== right.value || left.valueStatus !== right.valueStatus) {
+ differences.add(
+ "match-value",
+ "Aligned matches have different retained values.",
+ differenceValue(left.value),
+ differenceValue(right.value),
+ );
+ }
+ const captures = alignCaptures(left.captures, right.captures, differences);
+ return {
+ alignment,
+ left,
+ right,
+ captures,
+ equal: matchEqual(left, right, captures),
+ };
+}
+
+function alignMatches(
+ left: readonly RegexMatchResult[],
+ right: readonly RegexMatchResult[],
+ differences: DifferenceCollector,
+): readonly MatchAlignment[] {
+ const alignments: MatchAlignment[] = [];
+ const usedRight = new Set();
+ const deferredLeft: RegexMatchResult[] = [];
+
+ for (const match of left) {
+ const key = rangeKey(match);
+ const sameRange = takeMatching(
+ right,
+ usedRight,
+ (candidate) => rangeKey(candidate) === key,
+ );
+ if (!sameRange) {
+ deferredLeft.push(match);
+ continue;
+ }
+ usedRight.add(sameRange.index);
+ alignments.push(
+ compareMatch(match, sameRange.value, "normalized-range", differences),
+ );
+ }
+
+ const remainingRight = right
+ .map((value, index) => ({ value, index }))
+ .filter(({ index }) => !usedRight.has(index));
+ const paired = Math.min(deferredLeft.length, remainingRight.length);
+ for (let index = 0; index < paired; index += 1) {
+ const leftMatch = deferredLeft[index];
+ const rightMatch = remainingRight[index];
+ if (!leftMatch || !rightMatch) continue;
+ usedRight.add(rightMatch.index);
+ alignments.push(
+ compareMatch(
+ leftMatch,
+ rightMatch.value,
+ "ordinal-fallback",
+ differences,
+ ),
+ );
+ }
+ for (const match of deferredLeft.slice(paired)) {
+ alignments.push(compareMatch(match, undefined, "left-only", differences));
+ }
+ for (const { value, index } of remainingRight.slice(paired)) {
+ usedRight.add(index);
+ alignments.push(compareMatch(undefined, value, "right-only", differences));
+ }
+ return alignments;
+}
+
+function executionFor(
+ side: ComparisonSideOutcome,
+): RegexExecutionResult | undefined {
+ return side.runtime.replacement?.execution ?? side.runtime.execution;
+}
+
+function addReason(
+ values: ComparisonNotComparableReason[],
+ reason: ComparisonNotComparableReason,
+): void {
+ if (
+ values.some(
+ (candidate) =>
+ candidate.code === reason.code &&
+ candidate.flavour === reason.flavour &&
+ candidate.message === reason.message,
+ )
+ ) {
+ return;
+ }
+ values.push(reason);
+}
+
+function taskReasons(
+ side: ComparisonSideOutcome,
+ reasons: ComparisonNotComparableReason[],
+): void {
+ const flavour = side.input.flavour;
+ if (side.syntax.status === "timeout") {
+ addReason(reasons, {
+ code: "syntax-timeout",
+ flavour,
+ message: `${flavour} syntax parsing timed out.`,
+ });
+ } else if (side.syntax.status !== "complete") {
+ addReason(reasons, {
+ code: "syntax-worker-failed",
+ flavour,
+ message: `${flavour} syntax parsing did not complete: ${side.syntax.error ?? side.syntax.status}.`,
+ });
+ } else if (
+ side.syntax.pattern?.accepted === false ||
+ side.syntax.replacement?.accepted === false
+ ) {
+ addReason(reasons, {
+ code: "syntax-rejected",
+ flavour,
+ message: `${flavour} syntax provider rejected the exact pattern or replacement snapshot.`,
+ });
+ }
+
+ if (side.runtime.status === "timeout") {
+ addReason(reasons, {
+ code: "execution-timeout",
+ flavour,
+ message: `${flavour} execution timed out; timeout is not a no-match result.`,
+ });
+ return;
+ }
+ if (side.runtime.status === "cancelled") {
+ addReason(reasons, {
+ code: "execution-cancelled",
+ flavour,
+ message: `${flavour} execution was cancelled and has no semantic result.`,
+ });
+ return;
+ }
+ if (side.runtime.status !== "complete") {
+ addReason(reasons, {
+ code: "execution-worker-failed",
+ flavour,
+ message: `${flavour} worker did not complete: ${side.runtime.error ?? side.runtime.status}.`,
+ });
+ return;
+ }
+ const execution = executionFor(side);
+ if (!execution?.accepted) {
+ addReason(reasons, {
+ code: "compile-rejected",
+ flavour,
+ message: `${flavour} engine rejected the exact pattern snapshot.`,
+ });
+ return;
+ }
+ if (execution.truncated) {
+ addReason(reasons, {
+ code: "result-truncated",
+ flavour,
+ message: `${flavour} results reached a configured match or capture limit.`,
+ });
+ }
+ if (
+ execution.matches.some(
+ (match) =>
+ match.valueStatus === "truncated" ||
+ match.captures.some((capture) => capture.status === "truncated"),
+ )
+ ) {
+ addReason(reasons, {
+ code: "value-truncated",
+ flavour,
+ message: `${flavour} retained value previews are incomplete.`,
+ });
+ }
+ if (side.runtime.replacement?.truncated) {
+ addReason(reasons, {
+ code: "replacement-truncated",
+ flavour,
+ message: `${flavour} replacement output or its match collection is incomplete.`,
+ });
+ }
+}
+
+function compareMetadata(
+ left: ComparisonSideOutcome,
+ right: ComparisonSideOutcome,
+ differences: DifferenceCollector,
+): void {
+ const leftSyntax = left.syntax.pattern;
+ const rightSyntax = right.syntax.pattern;
+ if (
+ leftSyntax &&
+ rightSyntax &&
+ leftSyntax.accepted !== rightSyntax.accepted
+ ) {
+ differences.add(
+ "syntax-acceptance",
+ "The syntax providers disagree on acceptance.",
+ leftSyntax.accepted ? "accepted" : "rejected",
+ rightSyntax.accepted ? "accepted" : "rejected",
+ );
+ }
+ const leftExecution = executionFor(left);
+ const rightExecution = executionFor(right);
+ if (
+ leftExecution &&
+ rightExecution &&
+ leftExecution.accepted !== rightExecution.accepted
+ ) {
+ differences.add(
+ "compile-acceptance",
+ "The execution engines disagree on compilation.",
+ leftExecution.accepted ? "accepted" : "rejected",
+ rightExecution.accepted ? "accepted" : "rejected",
+ );
+ }
+ if (
+ leftExecution &&
+ rightExecution &&
+ leftExecution.flags.effectiveFlags !== rightExecution.flags.effectiveFlags
+ ) {
+ differences.add(
+ "effective-flags",
+ "The exact effective engine flags differ.",
+ leftExecution.flags.effectiveFlags || "none",
+ rightExecution.flags.effectiveFlags || "none",
+ );
+ }
+ if (
+ leftExecution &&
+ rightExecution &&
+ leftExecution.engine.offsetUnit !== rightExecution.engine.offsetUnit
+ ) {
+ differences.add(
+ "offset-model",
+ "Native offset units differ; alignments use normalized editor UTF-16 ranges.",
+ leftExecution.engine.offsetUnit,
+ rightExecution.engine.offsetUnit,
+ );
+ }
+}
+
+export function compareCompletedResults(
+ request: RegexComparisonRequest,
+ startedAt: string,
+ elapsedMs: number,
+ sides: readonly [ComparisonSideOutcome, ComparisonSideOutcome],
+): RegexComparisonResult {
+ const [left, right] = sides;
+ const differences = new DifferenceCollector();
+ const notComparable: ComparisonNotComparableReason[] = [];
+ taskReasons(left, notComparable);
+ taskReasons(right, notComparable);
+ compareMetadata(left, right, differences);
+
+ const leftExecution = executionFor(left);
+ const rightExecution = executionFor(right);
+ let allAlignments: readonly MatchAlignment[] = [];
+ if (leftExecution?.accepted && rightExecution?.accepted) {
+ if (leftExecution.matches.length !== rightExecution.matches.length) {
+ differences.add(
+ "match-count",
+ "The engines returned different match counts.",
+ leftExecution.matches.length.toLocaleString(),
+ rightExecution.matches.length.toLocaleString(),
+ );
+ }
+ allAlignments = alignMatches(
+ leftExecution.matches,
+ rightExecution.matches,
+ differences,
+ );
+ const leftReplacement = left.runtime.replacement;
+ const rightReplacement = right.runtime.replacement;
+ if (
+ leftReplacement &&
+ rightReplacement &&
+ leftReplacement.output !== rightReplacement.output
+ ) {
+ differences.add(
+ "replacement-output",
+ "The engine-native replacement outputs differ.",
+ differenceValue(leftReplacement.output),
+ differenceValue(rightReplacement.output),
+ );
+ }
+ }
+
+ const syntaxAcceptanceDiffers =
+ left.syntax.pattern !== undefined &&
+ right.syntax.pattern !== undefined &&
+ left.syntax.pattern.accepted !== right.syntax.pattern.accepted;
+ const compileAcceptanceDiffers =
+ leftExecution !== undefined &&
+ rightExecution !== undefined &&
+ leftExecution.accepted !== rightExecution.accepted;
+ if (syntaxAcceptanceDiffers || compileAcceptanceDiffers) {
+ addReason(notComparable, {
+ code: "flavour-specific-syntax",
+ message:
+ "At least one exact snapshot is accepted on only one side; no runtime equivalence claim is possible.",
+ });
+ }
+
+ const status =
+ notComparable.length > 0
+ ? "not-comparable"
+ : differences.semanticTotal === 0
+ ? "same-for-current-input"
+ : "different-for-current-input";
+ const notices =
+ status === "same-for-current-input"
+ ? [
+ {
+ confidence: "likely-equivalent-for-input" as const,
+ message:
+ "The completed, untruncated results agree for this subject only. This is test evidence, not a proof that the patterns are equivalent.",
+ },
+ ]
+ : status === "different-for-current-input"
+ ? [
+ {
+ confidence: "requires-manual-review" as const,
+ message:
+ "The exact engine results differ for this subject. Review syntax, flags, ranges and captures before porting.",
+ },
+ ]
+ : [
+ {
+ confidence:
+ compileAcceptanceDiffers || syntaxAcceptanceDiffers
+ ? ("no-direct-equivalent" as const)
+ : ("requires-manual-review" as const),
+ message:
+ "The run is not comparable because one or more authoritative results are incomplete or rejected.",
+ },
+ ];
+
+ return {
+ schemaVersion: 1,
+ request,
+ startedAt,
+ elapsedMs,
+ sides,
+ status,
+ notComparable,
+ differences: differences.retained,
+ totalDifferences: differences.total,
+ differencesTruncated: differences.retained.length < differences.total,
+ matchAlignments: allAlignments.slice(0, MAXIMUM_RETAINED_MATCH_ALIGNMENTS),
+ totalMatchAlignments: allAlignments.length,
+ matchAlignmentsTruncated:
+ allAlignments.length > MAXIMUM_RETAINED_MATCH_ALIGNMENTS,
+ notices,
+ };
+}
+
+export function comparisonSideLabel(flavour: ComparisonFlavour): string {
+ return flavour === "ecmascript" ? "ECMAScript" : "PCRE2";
+}
diff --git a/src/regex/comparison/comparison.types.ts b/src/regex/comparison/comparison.types.ts
new file mode 100644
index 0000000..ef25599
--- /dev/null
+++ b/src/regex/comparison/comparison.types.ts
@@ -0,0 +1,174 @@
+import type { RegexEngineInfo, RegexEngineOptions } from "../model/flavour";
+import type {
+ CaptureResult,
+ RegexExecutionRequest,
+ RegexExecutionResult,
+ RegexMatchResult,
+ RegexReplacementRequest,
+ RegexReplacementResult,
+} from "../model/match";
+import type {
+ RegexSyntaxRequest,
+ RegexSyntaxResult,
+ ReplacementSyntaxResult,
+} from "../model/syntax";
+
+export type ComparisonFlavour = "ecmascript" | "pcre2";
+export type ComparisonOperation = "match" | "replace";
+export type ComparisonPatternModel = "shared" | "variants";
+
+export interface ComparisonSideInput {
+ readonly flavour: ComparisonFlavour;
+ readonly flavourVersion: string;
+ readonly pattern: string;
+ readonly flags: readonly string[];
+ readonly options: RegexEngineOptions;
+ readonly replacement?: string;
+}
+
+export interface RegexComparisonRequest {
+ readonly schemaVersion: 1;
+ readonly patternModel: ComparisonPatternModel;
+ readonly operation: ComparisonOperation;
+ readonly subject: string;
+ readonly scanAll: boolean;
+ readonly maximumMatches: number;
+ readonly maximumCaptureRows: number;
+ readonly maximumOutputBytes: number;
+ readonly timeoutMs: number;
+ readonly sides: readonly [ComparisonSideInput, ComparisonSideInput];
+}
+
+export type ComparisonTaskStatus =
+ "complete" | "timeout" | "cancelled" | "worker-failed";
+
+export interface ComparisonTaskFailure {
+ readonly status: Exclude;
+ readonly message: string;
+}
+
+export interface ComparisonSyntaxOutcome {
+ readonly status: ComparisonTaskStatus;
+ readonly elapsedMs: number;
+ readonly pattern?: RegexSyntaxResult;
+ readonly replacement?: ReplacementSyntaxResult;
+ readonly error?: string;
+}
+
+export interface ComparisonExecutionOutcome {
+ readonly status: ComparisonTaskStatus;
+ readonly elapsedMs: number;
+ readonly execution?: RegexExecutionResult;
+ readonly replacement?: RegexReplacementResult;
+ readonly error?: string;
+}
+
+export interface ComparisonSideOutcome {
+ /**
+ * A deterministic display key for the exact request. It is not a
+ * cryptographic digest; the complete request sent to the worker is retained
+ * below and remains authoritative.
+ */
+ readonly requestIdentity: string;
+ readonly input: ComparisonSideInput;
+ readonly syntaxRequest: RegexSyntaxRequest;
+ readonly executionRequest: RegexExecutionRequest;
+ readonly replacementRequest?: RegexReplacementRequest;
+ readonly syntax: ComparisonSyntaxOutcome;
+ readonly runtime: ComparisonExecutionOutcome;
+ readonly engine?: RegexEngineInfo;
+}
+
+export type ComparisonNotComparableCode =
+ | "syntax-timeout"
+ | "syntax-worker-failed"
+ | "syntax-rejected"
+ | "flavour-specific-syntax"
+ | "execution-timeout"
+ | "execution-cancelled"
+ | "execution-worker-failed"
+ | "compile-rejected"
+ | "result-truncated"
+ | "replacement-truncated"
+ | "value-truncated";
+
+export interface ComparisonNotComparableReason {
+ readonly code: ComparisonNotComparableCode;
+ readonly flavour?: ComparisonFlavour;
+ readonly message: string;
+}
+
+export type ComparisonDifferenceKind =
+ | "syntax-acceptance"
+ | "compile-acceptance"
+ | "effective-flags"
+ | "offset-model"
+ | "match-count"
+ | "match-presence"
+ | "match-range"
+ | "match-value"
+ | "capture-presence"
+ | "capture-identity"
+ | "capture-status"
+ | "capture-range"
+ | "capture-value"
+ | "replacement-output";
+
+export interface ComparisonDifference {
+ readonly kind: ComparisonDifferenceKind;
+ readonly summary: string;
+ readonly left?: string;
+ readonly right?: string;
+}
+
+export type MatchAlignmentKind =
+ "normalized-range" | "ordinal-fallback" | "left-only" | "right-only";
+
+export type CaptureAlignmentKind =
+ | "normalized-range"
+ | "capture-identity"
+ | "ordinal-fallback"
+ | "left-only"
+ | "right-only";
+
+export interface CaptureAlignment {
+ readonly alignment: CaptureAlignmentKind;
+ readonly left?: CaptureResult;
+ readonly right?: CaptureResult;
+ readonly equal: boolean;
+}
+
+export interface MatchAlignment {
+ readonly alignment: MatchAlignmentKind;
+ readonly left?: RegexMatchResult;
+ readonly right?: RegexMatchResult;
+ readonly captures: readonly CaptureAlignment[];
+ readonly equal: boolean;
+}
+
+export interface ComparisonNotice {
+ readonly confidence:
+ | "possible-port"
+ | "likely-equivalent-for-input"
+ | "requires-manual-review"
+ | "no-direct-equivalent";
+ readonly message: string;
+}
+
+export interface RegexComparisonResult {
+ readonly schemaVersion: 1;
+ readonly request: RegexComparisonRequest;
+ readonly startedAt: string;
+ readonly elapsedMs: number;
+ readonly sides: readonly [ComparisonSideOutcome, ComparisonSideOutcome];
+ readonly status:
+ "not-comparable" | "same-for-current-input" | "different-for-current-input";
+ readonly notComparable: readonly ComparisonNotComparableReason[];
+ readonly differences: readonly ComparisonDifference[];
+ readonly totalDifferences: number;
+ readonly differencesTruncated: boolean;
+ readonly matchAlignments: readonly MatchAlignment[];
+ readonly totalMatchAlignments: number;
+ readonly matchAlignmentsTruncated: boolean;
+ readonly notices: readonly ComparisonNotice[];
+}
diff --git a/src/regex/corpus/corpus-export.test.ts b/src/regex/corpus/corpus-export.test.ts
new file mode 100644
index 0000000..995c757
--- /dev/null
+++ b/src/regex/corpus/corpus-export.test.ts
@@ -0,0 +1,185 @@
+import { unzipSync } from "fflate";
+import { describe, expect, it } from "vitest";
+import type { CorpusBatchResult } from "./corpus.types";
+import {
+ assertCorpusZipByteLength,
+ canExportAppliedCorpus,
+ corpusAppliedDownloadName,
+ createAppliedCorpusZip,
+ MAXIMUM_CORPUS_REPORT_BYTES,
+ MAXIMUM_CORPUS_ZIP_BYTES,
+ serializeCorpusReport,
+} from "./corpus-export";
+
+const batch: CorpusBatchResult = {
+ operation: "replace",
+ subjectMode: "document",
+ status: "complete",
+ documents: [
+ {
+ documentId: "one",
+ name: "../same.txt",
+ inputBytes: 3,
+ lineCount: 1,
+ status: "complete",
+ matchCount: 1,
+ elapsedMs: 2,
+ output: "ONE",
+ outputBytes: 3,
+ changed: true,
+ captureSummaries: [],
+ captureSummariesTruncated: false,
+ diagnostics: [],
+ },
+ {
+ documentId: "two",
+ name: "..\\same.txt",
+ inputBytes: 3,
+ lineCount: 1,
+ status: "complete",
+ matchCount: 1,
+ elapsedMs: 2,
+ output: "TWO",
+ outputBytes: 3,
+ changed: true,
+ captureSummaries: [],
+ captureSummariesTruncated: false,
+ diagnostics: [],
+ },
+ ],
+ processedDocuments: 2,
+ totalDocuments: 2,
+ totalMatches: 2,
+ totalElapsedMs: 4,
+ retainedOutputBytes: 6,
+ message: "Complete.",
+};
+
+describe("corpus exports", () => {
+ it("exports a content-free, versioned JSON report", () => {
+ const serialized = serializeCorpusReport(batch, {
+ flavour: "ecmascript",
+ flavourVersion: "2025",
+ pattern: "secret-pattern",
+ flags: ["g"],
+ });
+ const parsed = JSON.parse(serialized) as {
+ schemaVersion: number;
+ documents: readonly Record[];
+ };
+
+ expect(parsed.schemaVersion).toBe(1);
+ expect(serialized).not.toContain("ONE");
+ expect(serialized).toContain('"subjectMode": "document"');
+ expect(parsed.documents[0]).not.toHaveProperty("output");
+ });
+
+ it("exports content-free NDJSON and spreadsheet-safe CSV summaries", () => {
+ const formulaBatch: CorpusBatchResult = {
+ ...batch,
+ documents: [
+ {
+ ...batch.documents[0]!,
+ name: '=HYPERLINK("https://invalid.test")',
+ },
+ ],
+ totalDocuments: 1,
+ processedDocuments: 1,
+ totalMatches: 1,
+ };
+ const configuration = {
+ flavour: "ecmascript",
+ flavourVersion: "2025",
+ pattern: "+formula",
+ flags: ["g"],
+ };
+ const ndjson = serializeCorpusReport(formulaBatch, configuration, "ndjson");
+ const lines = ndjson
+ .trimEnd()
+ .split("\n")
+ .map((line) => JSON.parse(line) as unknown);
+ expect(lines).toHaveLength(2);
+ expect(lines[0]).toMatchObject({
+ type: "metadata",
+ subjectMode: "document",
+ });
+ expect(lines[1]).toMatchObject({
+ type: "document",
+ name: '=HYPERLINK("https://invalid.test")',
+ });
+ expect(ndjson).not.toContain("ONE");
+
+ const csv = serializeCorpusReport(formulaBatch, configuration, "csv");
+ expect(csv).toContain(`"'+formula"`);
+ expect(csv).toContain(`"'=HYPERLINK(""https://invalid.test"")"`);
+ expect(csv).not.toContain("ONE");
+ });
+
+ it("sanitizes archive paths and makes collisions unique", async () => {
+ expect(corpusAppliedDownloadName(batch.documents[0]!)).toBe(
+ "same.replaced.txt",
+ );
+ const archive = await createAppliedCorpusZip(batch).promise;
+ const files = unzipSync(archive);
+
+ expect(Object.keys(files).sort()).toEqual([
+ "same.replaced-2.txt",
+ "same.replaced.txt",
+ ]);
+ expect(new TextDecoder().decode(files["same.replaced.txt"])).toBe("ONE");
+ expect(new TextDecoder().decode(files["same.replaced-2.txt"])).toBe("TWO");
+ });
+
+ it("blocks applied exports when any result is incomplete", () => {
+ const partial: CorpusBatchResult = {
+ ...batch,
+ status: "partial",
+ documents: [
+ { ...batch.documents[0]!, status: "partial" },
+ batch.documents[1]!,
+ ],
+ };
+ expect(canExportAppliedCorpus(partial)).toBe(false);
+ expect(() => createAppliedCorpusZip(partial)).toThrow(
+ /completed exactly/iu,
+ );
+ });
+
+ it("enforces an explicit archive byte limit", () => {
+ expect(() =>
+ assertCorpusZipByteLength(MAXIMUM_CORPUS_ZIP_BYTES),
+ ).not.toThrow();
+ expect(() =>
+ assertCorpusZipByteLength(MAXIMUM_CORPUS_ZIP_BYTES + 1),
+ ).toThrow(/68 MiB/iu);
+ expect(() => assertCorpusZipByteLength(Number.NaN)).toThrow(/68 MiB/iu);
+ });
+
+ it("bounds every report format before download", () => {
+ const oversized: CorpusBatchResult = {
+ ...batch,
+ documents: [
+ {
+ ...batch.documents[0]!,
+ message: "x".repeat(MAXIMUM_CORPUS_REPORT_BYTES),
+ },
+ ],
+ totalDocuments: 1,
+ processedDocuments: 1,
+ totalMatches: 1,
+ };
+ for (const format of ["json", "csv", "ndjson"] as const) {
+ expect(() =>
+ serializeCorpusReport(
+ oversized,
+ {
+ flavour: "ecmascript",
+ pattern: "x",
+ flags: [],
+ },
+ format,
+ ),
+ ).toThrow(/4 MiB/iu);
+ }
+ });
+});
diff --git a/src/regex/corpus/corpus-export.ts b/src/regex/corpus/corpus-export.ts
new file mode 100644
index 0000000..46bef8f
--- /dev/null
+++ b/src/regex/corpus/corpus-export.ts
@@ -0,0 +1,313 @@
+import { strToU8, zip } from "fflate";
+import { utf8ByteLength } from "../execution/request-limits";
+import type { CorpusBatchResult, CorpusDocumentResult } from "./corpus.types";
+
+export const CORPUS_REPORT_SCHEMA_VERSION = 1;
+export const MAXIMUM_CORPUS_REPORT_BYTES = 4 * 1024 * 1024;
+export const MAXIMUM_CORPUS_ZIP_BYTES = 68 * 1024 * 1024;
+export type CorpusReportFormat = "json" | "csv" | "ndjson";
+
+export function assertCorpusZipByteLength(byteLength: number): void {
+ if (
+ !Number.isSafeInteger(byteLength) ||
+ byteLength < 0 ||
+ byteLength > MAXIMUM_CORPUS_ZIP_BYTES
+ ) {
+ throw new Error("Applied-output ZIP exceeds the 68 MiB archive limit.");
+ }
+}
+
+function safeFileName(name: string): string {
+ const normalized = Array.from(name.normalize("NFKC").replace(/[/\\]+/gu, "-"))
+ .filter((character) => {
+ const code = character.codePointAt(0) ?? 0;
+ return code >= 0x20 && code !== 0x7f;
+ })
+ .join("")
+ .replace(/^[.\-\s]+/u, "")
+ .trim()
+ .slice(0, 180);
+ return normalized || "document.txt";
+}
+
+function appliedFileName(name: string): string {
+ const safe = safeFileName(name);
+ const dot = safe.lastIndexOf(".");
+ return dot > 0
+ ? `${safe.slice(0, dot)}.replaced${safe.slice(dot)}`
+ : `${safe}.replaced.txt`;
+}
+
+function uniqueAppliedNames(
+ results: readonly CorpusDocumentResult[],
+): ReadonlyMap {
+ const names = new Map();
+ const occupied = new Set();
+ for (const result of results) {
+ const requested = appliedFileName(result.name);
+ const dot = requested.lastIndexOf(".");
+ const stem = dot > 0 ? requested.slice(0, dot) : requested;
+ const extension = dot > 0 ? requested.slice(dot) : "";
+ let candidate = requested;
+ let suffix = 2;
+ while (occupied.has(candidate.toLocaleLowerCase("en-US"))) {
+ candidate = `${stem.slice(0, 160)}-${suffix}${extension}`;
+ suffix += 1;
+ }
+ occupied.add(candidate.toLocaleLowerCase("en-US"));
+ names.set(result.documentId, candidate);
+ }
+ return names;
+}
+
+export function corpusAppliedDownloadName(
+ result: CorpusDocumentResult,
+): string {
+ return appliedFileName(result.name);
+}
+
+export function canExportAppliedCorpus(batch: CorpusBatchResult): boolean {
+ return (
+ batch.operation === "replace" &&
+ batch.status === "complete" &&
+ batch.documents.length > 0 &&
+ batch.documents.every(
+ (result) => result.status === "complete" && result.output !== undefined,
+ )
+ );
+}
+
+interface CorpusReportConfiguration {
+ readonly flavour: string;
+ readonly flavourVersion?: string;
+ readonly pattern: string;
+ readonly flags: readonly string[];
+}
+
+function captureSummaryForReport(result: CorpusDocumentResult) {
+ return result.captureSummaries.map((capture) => ({
+ groupNumber: capture.groupNumber,
+ ...(capture.groupName === undefined
+ ? {}
+ : { groupName: capture.groupName }),
+ participated: capture.participated,
+ didNotParticipate: capture.didNotParticipate,
+ matchedEmpty: capture.matchedEmpty,
+ truncated: capture.truncated,
+ unavailable: capture.unavailable,
+ retainedSampleCount: capture.samples.length,
+ samplesClipped: capture.samplesClipped,
+ }));
+}
+
+function documentForReport(result: CorpusDocumentResult) {
+ return {
+ name: result.name,
+ inputBytes: result.inputBytes,
+ lineCount: result.lineCount,
+ status: result.status,
+ matchCount: result.matchCount,
+ matchCountIsLowerBound: result.status === "partial",
+ elapsedMs: result.elapsedMs,
+ ...(result.outputBytes === undefined
+ ? {}
+ : { outputBytes: result.outputBytes }),
+ ...(result.changed === undefined ? {} : { changed: result.changed }),
+ ...(result.message === undefined ? {} : { message: result.message }),
+ captureSummariesTruncated: result.captureSummariesTruncated,
+ captureSummaries: captureSummaryForReport(result),
+ diagnostics: result.diagnostics.map((diagnostic) => ({
+ severity: diagnostic.severity,
+ code: diagnostic.code,
+ message: diagnostic.message,
+ provenance: diagnostic.provenance,
+ })),
+ };
+}
+
+function reportMetadata(
+ batch: CorpusBatchResult,
+ configuration: CorpusReportConfiguration,
+) {
+ return {
+ schemaVersion: CORPUS_REPORT_SCHEMA_VERSION,
+ application: "de.add-ideas.regex-tools",
+ operation: batch.operation,
+ subjectMode: batch.subjectMode,
+ status: batch.status,
+ engine: {
+ flavour: configuration.flavour,
+ ...(configuration.flavourVersion === undefined
+ ? {}
+ : { version: configuration.flavourVersion }),
+ pattern: configuration.pattern,
+ flags: configuration.flags,
+ scanAll: true,
+ },
+ summary: {
+ processedDocuments: batch.processedDocuments,
+ totalDocuments: batch.totalDocuments,
+ totalMatches: batch.totalMatches,
+ totalElapsedMs: batch.totalElapsedMs,
+ retainedOutputBytes: batch.retainedOutputBytes,
+ message: batch.message,
+ },
+ };
+}
+
+function assertReportSize(serialized: string): string {
+ if (utf8ByteLength(serialized) > MAXIMUM_CORPUS_REPORT_BYTES) {
+ throw new Error("Corpus report exceeds the 4 MiB export limit.");
+ }
+ return serialized;
+}
+
+function csvCell(value: unknown): string {
+ const raw =
+ typeof value === "string"
+ ? value
+ : value === undefined
+ ? ""
+ : JSON.stringify(value);
+ const protectedValue = /^[=+\-@]/u.test(raw.trimStart()) ? `'${raw}` : raw;
+ return `"${protectedValue.replaceAll('"', '""')}"`;
+}
+
+export function serializeCorpusReport(
+ batch: CorpusBatchResult,
+ configuration: CorpusReportConfiguration,
+ format: CorpusReportFormat = "json",
+): string {
+ const metadata = reportMetadata(batch, configuration);
+ const documents = batch.documents.map(documentForReport);
+ if (format === "json") {
+ return assertReportSize(
+ `${JSON.stringify({ ...metadata, documents }, null, 2)}\n`,
+ );
+ }
+ if (format === "ndjson") {
+ return assertReportSize(
+ [
+ JSON.stringify({ type: "metadata", ...metadata }),
+ ...documents.map((document) =>
+ JSON.stringify({ type: "document", ...document }),
+ ),
+ ].join("\n") + "\n",
+ );
+ }
+
+ const headers = [
+ "record_type",
+ "operation",
+ "subject_mode",
+ "batch_status",
+ "flavour",
+ "flavour_version",
+ "pattern",
+ "flags",
+ "name",
+ "status",
+ "input_bytes",
+ "line_count",
+ "match_count",
+ "match_count_is_lower_bound",
+ "output_bytes",
+ "changed",
+ "elapsed_ms",
+ "capture_summaries_truncated",
+ "capture_summaries",
+ "diagnostics",
+ "message",
+ ] as const;
+ const metadataValues: readonly unknown[] = [
+ "metadata",
+ batch.operation,
+ batch.subjectMode,
+ batch.status,
+ configuration.flavour,
+ configuration.flavourVersion,
+ configuration.pattern,
+ configuration.flags.join(""),
+ "",
+ "",
+ "",
+ "",
+ "",
+ "",
+ "",
+ "",
+ "",
+ "",
+ "",
+ "",
+ "",
+ ];
+ const rows = documents.map((document) => {
+ const values: readonly unknown[] = [
+ "document",
+ "",
+ "",
+ "",
+ "",
+ "",
+ "",
+ "",
+ document.name,
+ document.status,
+ document.inputBytes,
+ document.lineCount,
+ document.matchCount,
+ document.matchCountIsLowerBound,
+ document.outputBytes,
+ document.changed,
+ document.elapsedMs,
+ document.captureSummariesTruncated,
+ document.captureSummaries,
+ document.diagnostics,
+ document.message,
+ ];
+ return values.map(csvCell).join(",");
+ });
+ return assertReportSize(
+ `${headers.map(csvCell).join(",")}\r\n${metadataValues.map(csvCell).join(",")}\r\n${rows.join("\r\n")}\r\n`,
+ );
+}
+
+export function createAppliedCorpusZip(batch: CorpusBatchResult): {
+ readonly promise: Promise;
+ readonly cancel: () => void;
+} {
+ if (!canExportAppliedCorpus(batch)) {
+ throw new Error(
+ "Applied output can be exported only when every document completed exactly.",
+ );
+ }
+ const names = uniqueAppliedNames(batch.documents);
+ const files: Record = {};
+ for (const result of batch.documents) {
+ const name = names.get(result.documentId);
+ if (!name || result.output === undefined) {
+ throw new Error("An applied corpus output is unavailable.");
+ }
+ files[name] = strToU8(result.output);
+ }
+ let terminate: (() => void) | undefined;
+ const promise = new Promise((resolve, reject) => {
+ terminate = zip(files, { level: 6 }, (error, data) => {
+ if (error) {
+ reject(error);
+ } else {
+ try {
+ assertCorpusZipByteLength(data.byteLength);
+ resolve(data);
+ } catch (sizeError) {
+ reject(sizeError as Error);
+ }
+ }
+ });
+ });
+ return {
+ promise,
+ cancel: () => terminate?.(),
+ };
+}
diff --git a/src/regex/corpus/corpus-input.test.ts b/src/regex/corpus/corpus-input.test.ts
new file mode 100644
index 0000000..9925aa3
--- /dev/null
+++ b/src/regex/corpus/corpus-input.test.ts
@@ -0,0 +1,122 @@
+import { describe, expect, it } from "vitest";
+import { DEFAULT_REGEX_LIMITS } from "../execution/request-limits";
+import {
+ corpusTotalBytes,
+ createPastedCorpusDocument,
+ readCorpusFiles,
+} from "./corpus-input";
+
+describe("corpus input", () => {
+ it("adds bounded pasted UTF-8 text with unique names", () => {
+ const first = createPastedCorpusDocument("notes.txt", "Grüße", []);
+ const second = createPastedCorpusDocument("notes.txt", "again", [first]);
+
+ expect(first).toMatchObject({
+ name: "notes.txt",
+ bytes: new TextEncoder().encode("Grüße").byteLength,
+ source: "pasted-text",
+ });
+ expect(second.name).toBe("notes (2).txt");
+ expect(corpusTotalBytes([first, second])).toBe(first.bytes + second.bytes);
+ });
+
+ it("rejects binary, invalid UTF-8 and oversized documents before use", async () => {
+ await expect(
+ readCorpusFiles(
+ [
+ new File([new Uint8Array([0, 1, 2])], "sample.bin", {
+ type: "application/octet-stream",
+ }),
+ ],
+ [],
+ ),
+ ).rejects.toThrow(/not a recognized text file/iu);
+
+ await expect(
+ readCorpusFiles(
+ [
+ new File([new Uint8Array([0xc3, 0x28])], "broken.txt", {
+ type: "text/plain",
+ }),
+ ],
+ [],
+ ),
+ ).rejects.toThrow(/not valid UTF-8/iu);
+
+ expect(() =>
+ createPastedCorpusDocument(
+ "huge.txt",
+ "a".repeat(DEFAULT_REGEX_LIMITS.maximumCorpusDocumentBytes + 1),
+ [],
+ ),
+ ).toThrow(/per-document limit/iu);
+ });
+
+ it("honours import cancellation and reports file progress", async () => {
+ const controller = new AbortController();
+ controller.abort();
+ await expect(
+ readCorpusFiles(
+ [new File(["first"], "first.txt", { type: "text/plain" })],
+ [],
+ controller.signal,
+ ),
+ ).rejects.toMatchObject({ name: "AbortError" });
+
+ const progress: string[] = [];
+ const documents = await readCorpusFiles(
+ [
+ new File(["one"], "one.txt", { type: "text/plain" }),
+ new File(["two"], "two.txt", { type: "text/plain" }),
+ ],
+ [],
+ undefined,
+ (completed, total, name) =>
+ progress.push(`${completed}/${total}:${name}`),
+ );
+ expect(documents.map((document) => document.text)).toEqual(["one", "two"]);
+ expect(progress).toEqual(["1/2:one.txt", "2/2:two.txt"]);
+ });
+
+ it("preserves a UTF-8 BOM so no-op apply can round-trip the file", async () => {
+ const [document] = await readCorpusFiles(
+ [
+ new File([new Uint8Array([0xef, 0xbb, 0xbf]), "content"], "bom.txt", {
+ type: "text/plain",
+ }),
+ ],
+ [],
+ );
+ expect(document?.text).toBe("\ufeffcontent");
+ expect(new TextEncoder().encode(document?.text).byteLength).toBe(10);
+ });
+
+ it("preflights aggregate bytes and document count without reading input", () => {
+ const atAggregateLimit = [
+ {
+ id: "existing",
+ name: "existing.txt",
+ text: "",
+ bytes: DEFAULT_REGEX_LIMITS.corpusHardBytes,
+ source: "file" as const,
+ },
+ ];
+ expect(() =>
+ createPastedCorpusDocument("one-more.txt", "x", atAggregateLimit),
+ ).toThrow(/aggregate limit/iu);
+
+ const atDocumentLimit = Array.from(
+ { length: DEFAULT_REGEX_LIMITS.maximumCorpusDocuments },
+ (_, index) => ({
+ id: `existing-${index}`,
+ name: `${index}.txt`,
+ text: "",
+ bytes: 0,
+ source: "file" as const,
+ }),
+ );
+ expect(() =>
+ createPastedCorpusDocument("one-more.txt", "", atDocumentLimit),
+ ).toThrow(/at most 256 documents/iu);
+ });
+});
diff --git a/src/regex/corpus/corpus-input.ts b/src/regex/corpus/corpus-input.ts
new file mode 100644
index 0000000..d19c058
--- /dev/null
+++ b/src/regex/corpus/corpus-input.ts
@@ -0,0 +1,255 @@
+import {
+ DEFAULT_REGEX_LIMITS,
+ utf8ByteLength,
+} from "../execution/request-limits";
+import type { CorpusDocument } from "./corpus.types";
+
+const EXPLICIT_TEXT_MEDIA_TYPES = new Set([
+ "application/csv",
+ "application/json",
+ "application/ld+json",
+ "application/rtf",
+ "application/sql",
+ "application/x-httpd-php",
+ "application/x-ndjson",
+ "application/x-sh",
+ "application/xhtml+xml",
+ "application/xml",
+ "application/yaml",
+ "image/svg+xml",
+]);
+
+const TEXT_EXTENSIONS = new Set([
+ "asc",
+ "c",
+ "cfg",
+ "conf",
+ "cpp",
+ "css",
+ "csv",
+ "go",
+ "h",
+ "hpp",
+ "htm",
+ "html",
+ "ini",
+ "java",
+ "js",
+ "json",
+ "jsonl",
+ "jsx",
+ "log",
+ "md",
+ "ndjson",
+ "php",
+ "properties",
+ "py",
+ "rb",
+ "rs",
+ "rtf",
+ "sh",
+ "sql",
+ "svg",
+ "text",
+ "toml",
+ "ts",
+ "tsv",
+ "tsx",
+ "txt",
+ "xml",
+ "yaml",
+ "yml",
+]);
+
+let fallbackDocumentId = 0;
+
+export function corpusDocumentId(): string {
+ return (
+ globalThis.crypto?.randomUUID?.() ??
+ `corpus-${Date.now()}-${(fallbackDocumentId += 1)}`
+ );
+}
+
+function fileExtension(name: string): string {
+ const match = /\.([^.]+)$/u.exec(name);
+ return match?.[1]?.toLocaleLowerCase("en-US") ?? "";
+}
+
+function supportsTextImport(file: File): boolean {
+ const mediaType = file.type.toLocaleLowerCase("en-US");
+ return (
+ mediaType.length === 0 ||
+ mediaType.startsWith("text/") ||
+ EXPLICIT_TEXT_MEDIA_TYPES.has(mediaType) ||
+ TEXT_EXTENSIONS.has(fileExtension(file.name))
+ );
+}
+
+function uniqueDocumentName(
+ requested: string,
+ existingNames: Set,
+): string {
+ const normalized = Array.from(requested.normalize("NFC"))
+ .map((character) => {
+ const code = character.codePointAt(0) ?? 0;
+ return code < 0x20 || code === 0x7f ? " " : character;
+ })
+ .join("")
+ .replace(/\s+/gu, " ")
+ .trim()
+ .slice(0, 240);
+ const base = normalized || "untitled.txt";
+ if (!existingNames.has(base)) {
+ existingNames.add(base);
+ return base;
+ }
+ const dot = base.lastIndexOf(".");
+ const stem = dot > 0 ? base.slice(0, dot) : base;
+ const extension = dot > 0 ? base.slice(dot) : "";
+ for (let suffix = 2; suffix <= 10_000; suffix += 1) {
+ const candidate = `${stem.slice(0, 220)} (${suffix})${extension}`;
+ if (!existingNames.has(candidate)) {
+ existingNames.add(candidate);
+ return candidate;
+ }
+ }
+ throw new Error("Could not create a unique corpus document name.");
+}
+
+function assertDocumentCount(count: number): void {
+ if (count > DEFAULT_REGEX_LIMITS.maximumCorpusDocuments) {
+ throw new Error(
+ `A corpus can contain at most ${DEFAULT_REGEX_LIMITS.maximumCorpusDocuments.toLocaleString()} documents.`,
+ );
+ }
+}
+
+function assertDocumentBytes(bytes: number, name: string): void {
+ if (bytes > DEFAULT_REGEX_LIMITS.maximumCorpusDocumentBytes) {
+ throw new Error(
+ `${name} exceeds the ${(
+ DEFAULT_REGEX_LIMITS.maximumCorpusDocumentBytes /
+ 1024 /
+ 1024
+ ).toLocaleString()} MiB per-document limit.`,
+ );
+ }
+}
+
+function assertAggregateBytes(bytes: number): void {
+ if (bytes > DEFAULT_REGEX_LIMITS.corpusHardBytes) {
+ throw new Error(
+ `The corpus exceeds the ${(
+ DEFAULT_REGEX_LIMITS.corpusHardBytes /
+ 1024 /
+ 1024
+ ).toLocaleString()} MiB aggregate limit.`,
+ );
+ }
+}
+
+function assertTextContent(text: string, name: string): void {
+ for (const character of text) {
+ const code = character.codePointAt(0) ?? 0;
+ if (
+ code <= 0x08 ||
+ code === 0x0b ||
+ (code >= 0x0e && code <= 0x1f) ||
+ code === 0x7f
+ ) {
+ throw new Error(
+ `${name} contains binary control bytes and does not appear to be text.`,
+ );
+ }
+ }
+}
+
+export function createPastedCorpusDocument(
+ requestedName: string,
+ text: string,
+ existing: readonly CorpusDocument[],
+): CorpusDocument {
+ assertDocumentCount(existing.length + 1);
+ const bytes = utf8ByteLength(text);
+ assertDocumentBytes(bytes, requestedName || "Pasted text");
+ assertAggregateBytes(
+ existing.reduce((sum, document) => sum + document.bytes, 0) + bytes,
+ );
+ assertTextContent(text, requestedName || "Pasted text");
+ const names = new Set(existing.map((document) => document.name));
+ return {
+ id: corpusDocumentId(),
+ name: uniqueDocumentName(requestedName, names),
+ text,
+ bytes,
+ source: "pasted-text",
+ };
+}
+
+export async function readCorpusFiles(
+ files: readonly File[],
+ existing: readonly CorpusDocument[],
+ signal?: AbortSignal,
+ onRead?: (completed: number, total: number, name: string) => void,
+): Promise {
+ assertDocumentCount(existing.length + files.length);
+ const existingBytes = existing.reduce(
+ (sum, document) => sum + document.bytes,
+ 0,
+ );
+ let declaredBytes = existingBytes;
+ for (const file of files) {
+ assertDocumentBytes(file.size, file.name || "Unnamed file");
+ declaredBytes += file.size;
+ assertAggregateBytes(declaredBytes);
+ if (!supportsTextImport(file)) {
+ throw new Error(
+ `${file.name || "Unnamed file"} is not a recognized text file. Binary corpus input is rejected.`,
+ );
+ }
+ }
+
+ const names = new Set(existing.map((document) => document.name));
+ const documents: CorpusDocument[] = [];
+ for (const [index, file] of files.entries()) {
+ if (signal?.aborted)
+ throw new DOMException("Import cancelled.", "AbortError");
+ let buffer: ArrayBuffer;
+ try {
+ buffer = await file.arrayBuffer();
+ } catch (error) {
+ throw new Error(`Could not read ${file.name || "the selected file"}.`, {
+ cause: error,
+ });
+ }
+ if (signal?.aborted)
+ throw new DOMException("Import cancelled.", "AbortError");
+ let text: string;
+ try {
+ // Keeping a UTF-8 BOM in the string lets a no-op apply round-trip it.
+ text = new TextDecoder("utf-8", {
+ fatal: true,
+ ignoreBOM: true,
+ }).decode(buffer);
+ } catch (error) {
+ throw new Error(
+ `${file.name || "The selected file"} is not valid UTF-8 text.`,
+ { cause: error },
+ );
+ }
+ assertTextContent(text, file.name || "The selected file");
+ documents.push({
+ id: corpusDocumentId(),
+ name: uniqueDocumentName(file.name, names),
+ text,
+ bytes: file.size,
+ source: "file",
+ });
+ onRead?.(index + 1, files.length, file.name);
+ }
+ return documents;
+}
+
+export function corpusTotalBytes(documents: readonly CorpusDocument[]): number {
+ return documents.reduce((sum, document) => sum + document.bytes, 0);
+}
diff --git a/src/regex/corpus/corpus-lines.test.ts b/src/regex/corpus/corpus-lines.test.ts
new file mode 100644
index 0000000..045d0f6
--- /dev/null
+++ b/src/regex/corpus/corpus-lines.test.ts
@@ -0,0 +1,25 @@
+import { describe, expect, it } from "vitest";
+import { countCorpusLines, splitCorpusLines } from "./corpus-lines";
+
+describe("corpus line segmentation", () => {
+ it("preserves mixed separators and the trailing empty logical line", () => {
+ const text = "one\r\ntwo\nthree\rfour\u2028five\u2029";
+ const lines = splitCorpusLines(text);
+
+ expect(lines).toEqual([
+ { text: "one", ending: "\r\n" },
+ { text: "two", ending: "\n" },
+ { text: "three", ending: "\r" },
+ { text: "four", ending: "\u2028" },
+ { text: "five", ending: "\u2029" },
+ { text: "", ending: "" },
+ ]);
+ expect(countCorpusLines(text)).toBe(lines.length);
+ expect(lines.map((line) => line.text + line.ending).join("")).toBe(text);
+ });
+
+ it("treats an empty document as one empty line", () => {
+ expect(splitCorpusLines("")).toEqual([{ text: "", ending: "" }]);
+ expect(countCorpusLines("")).toBe(1);
+ });
+});
diff --git a/src/regex/corpus/corpus-lines.ts b/src/regex/corpus/corpus-lines.ts
new file mode 100644
index 0000000..d49e804
--- /dev/null
+++ b/src/regex/corpus/corpus-lines.ts
@@ -0,0 +1,40 @@
+export interface CorpusLine {
+ readonly text: string;
+ readonly ending: string;
+}
+
+function lineEndingLength(text: string, index: number): number {
+ const code = text.charCodeAt(index);
+ if (code === 0x0d) {
+ return text.charCodeAt(index + 1) === 0x0a ? 2 : 1;
+ }
+ return code === 0x0a || code === 0x2028 || code === 0x2029 ? 1 : 0;
+}
+
+export function countCorpusLines(text: string): number {
+ let count = 1;
+ for (let index = 0; index < text.length; index += 1) {
+ const endingLength = lineEndingLength(text, index);
+ if (endingLength === 0) continue;
+ count += 1;
+ index += endingLength - 1;
+ }
+ return count;
+}
+
+export function splitCorpusLines(text: string): readonly CorpusLine[] {
+ const lines: CorpusLine[] = [];
+ let start = 0;
+ for (let index = 0; index < text.length; index += 1) {
+ const endingLength = lineEndingLength(text, index);
+ if (endingLength === 0) continue;
+ lines.push({
+ text: text.slice(start, index),
+ ending: text.slice(index, index + endingLength),
+ });
+ index += endingLength - 1;
+ start = index + 1;
+ }
+ lines.push({ text: text.slice(start), ending: "" });
+ return lines;
+}
diff --git a/src/regex/corpus/corpus-runner.test.ts b/src/regex/corpus/corpus-runner.test.ts
new file mode 100644
index 0000000..dd03eaa
--- /dev/null
+++ b/src/regex/corpus/corpus-runner.test.ts
@@ -0,0 +1,406 @@
+import { describe, expect, it, vi } from "vitest";
+import type { EngineSupervisor } from "../execution/EngineSupervisor";
+import { DEFAULT_REGEX_LIMITS } from "../execution/request-limits";
+import { WorkerRequestError } from "../execution/WorkerSupervisor";
+import type {
+ RegexExecutionRequest,
+ RegexExecutionResult,
+ RegexReplacementRequest,
+} from "../model/match";
+import type { CorpusDocument } from "./corpus.types";
+import { runCorpusBatch, type CorpusRunConfiguration } from "./corpus-runner";
+
+function document(id: string, text: string): CorpusDocument {
+ return {
+ id,
+ name: `${id}.txt`,
+ text,
+ bytes: text.length,
+ source: "file",
+ };
+}
+
+function result(
+ request: RegexExecutionRequest,
+ matchCount: number,
+ truncated = false,
+): RegexExecutionResult {
+ return {
+ accepted: true,
+ engine: {
+ flavour: request.flavour,
+ adapterVersion: "test",
+ engineName: "Test engine",
+ engineVersion: "1",
+ runtimeVersion: "1",
+ offsetUnit: "utf16",
+ capabilities: {
+ compilation: true,
+ matching: true,
+ replacement: true,
+ namedCaptures: true,
+ captureHistory: false,
+ actualTrace: false,
+ benchmark: false,
+ },
+ },
+ flags: {
+ userFlags: request.flags.join(""),
+ effectiveFlags: request.flags.join(""),
+ internallyAddedIndicesFlag: false,
+ internallyAddedGlobalFlag: false,
+ },
+ matches: Array.from({ length: matchCount }, (_, index) => ({
+ matchNumber: index + 1,
+ value: "x",
+ valueStatus: "complete" as const,
+ range: { startUtf16: index, endUtf16: index + 1 },
+ nativeRange: { start: index, end: index + 1, unit: "utf16" as const },
+ captures: [],
+ })),
+ diagnostics: [],
+ elapsedMs: 3,
+ truncated,
+ };
+}
+
+const configuration: CorpusRunConfiguration = {
+ flavour: "ecmascript",
+ flavourVersion: "2025",
+ options: {},
+ pattern: "x",
+ flags: [],
+ captureMetadata: [],
+ replacement: "y",
+ operation: "match",
+ subjectMode: "document",
+ timeoutMs: 1_000,
+};
+
+describe("corpus runner", () => {
+ it("processes documents sequentially with exhaustive matching and progress", async () => {
+ const requests: RegexExecutionRequest[] = [];
+ const engine = {
+ execute: vi.fn((request: RegexExecutionRequest) => {
+ requests.push(request);
+ return Promise.resolve(result(request, request.subject.length));
+ }),
+ replace: vi.fn(),
+ cancel: vi.fn(),
+ } as unknown as EngineSupervisor;
+ const progress: string[] = [];
+
+ const batch = await runCorpusBatch(
+ [document("one", "xx"), document("two", "x")],
+ configuration,
+ {
+ engine,
+ signal: new AbortController().signal,
+ onProgress: (value) =>
+ progress.push(
+ `${value.completedDocuments}/${value.totalDocuments}:${value.totalMatches}`,
+ ),
+ },
+ );
+
+ expect(batch).toMatchObject({
+ status: "complete",
+ processedDocuments: 2,
+ totalMatches: 3,
+ });
+ expect(requests).toHaveLength(2);
+ expect(requests.every((request) => request.scanAll)).toBe(true);
+ expect(progress).toEqual(["0/2:0", "1/2:2", "2/2:3"]);
+ });
+
+ it("retains exact applied outputs and marks truncated output partial", async () => {
+ const engine = {
+ execute: vi.fn(),
+ replace: vi.fn((request: RegexReplacementRequest) => {
+ const execution = result(request, 1, request.subject === "partial");
+ return Promise.resolve({
+ execution,
+ output: request.subject === "partial" ? "p" : "done",
+ outputBytes: request.subject === "partial" ? 1 : 4,
+ outputTruncated: request.subject === "partial",
+ truncated: request.subject === "partial",
+ });
+ }),
+ cancel: vi.fn(),
+ } as unknown as EngineSupervisor;
+
+ const batch = await runCorpusBatch(
+ [document("one", "ok"), document("two", "partial")],
+ { ...configuration, operation: "replace" },
+ { engine, signal: new AbortController().signal },
+ );
+
+ expect(batch.status).toBe("partial");
+ expect(batch.documents[0]).toMatchObject({
+ status: "complete",
+ output: "done",
+ outputBytes: 4,
+ });
+ expect(batch.documents[1]).toMatchObject({
+ status: "partial",
+ output: "p",
+ });
+ });
+
+ it("continues after a document timeout and labels cancellation", async () => {
+ const controller = new AbortController();
+ const engine = {
+ execute: vi
+ .fn()
+ .mockRejectedValueOnce(
+ new WorkerRequestError("timeout", "test engine timed out"),
+ )
+ .mockImplementationOnce((request: RegexExecutionRequest) => {
+ controller.abort();
+ return Promise.resolve(result(request, 1));
+ }),
+ replace: vi.fn(),
+ cancel: vi.fn(),
+ } as unknown as EngineSupervisor;
+
+ const batch = await runCorpusBatch(
+ [
+ document("timeout", "x"),
+ document("complete", "x"),
+ document("cancelled", "x"),
+ ],
+ configuration,
+ { engine, signal: controller.signal },
+ );
+
+ expect(batch.status).toBe("cancelled");
+ expect(batch.documents.map((entry) => entry.status)).toEqual([
+ "error",
+ "complete",
+ "cancelled",
+ ]);
+ expect(batch.documents[0]?.message).toMatch(/worker was terminated/iu);
+ });
+
+ it("distinguishes the aggregate wall-time stop from user cancellation", async () => {
+ const controller = new AbortController();
+ controller.abort("corpus-wall-time");
+ const engine = {
+ execute: vi.fn(),
+ replace: vi.fn(),
+ cancel: vi.fn(),
+ } as unknown as EngineSupervisor;
+
+ const batch = await runCorpusBatch(
+ [document("pending", "x")],
+ configuration,
+ { engine, signal: controller.signal },
+ );
+
+ expect(batch.status).toBe("partial");
+ expect(batch.message).toMatch(/wall-time limit/iu);
+ expect(batch.documents[0]).toMatchObject({ status: "not-run" });
+ });
+
+ it("runs independent lines separately and preserves original separators during apply", async () => {
+ const subjects: string[] = [];
+ const engine = {
+ execute: vi.fn(),
+ replace: vi.fn((request: RegexReplacementRequest) => {
+ subjects.push(request.subject);
+ const execution = result(request, request.subject === "" ? 0 : 1);
+ const output = request.subject.toLocaleUpperCase("en-US");
+ return Promise.resolve({
+ execution,
+ output,
+ outputBytes: new TextEncoder().encode(output).byteLength,
+ outputTruncated: false,
+ truncated: false,
+ });
+ }),
+ cancel: vi.fn(),
+ } as unknown as EngineSupervisor;
+
+ const batch = await runCorpusBatch(
+ [document("mixed-lines", "one\r\ntwo\n")],
+ {
+ ...configuration,
+ operation: "replace",
+ subjectMode: "line",
+ },
+ { engine, signal: new AbortController().signal },
+ );
+
+ expect(subjects).toEqual(["one", "two", ""]);
+ expect(batch.documents[0]).toMatchObject({
+ status: "complete",
+ lineCount: 3,
+ output: "ONE\r\nTWO\n",
+ changed: true,
+ matchCount: 2,
+ });
+ });
+
+ it("aggregates bounded full-match and capture extraction summaries", async () => {
+ const engine = {
+ execute: vi.fn((request: RegexExecutionRequest) => {
+ const base = result(request, 0);
+ return Promise.resolve({
+ ...base,
+ matches: [
+ {
+ matchNumber: 1,
+ value: "42",
+ valueStatus: "complete" as const,
+ range: { startUtf16: 0, endUtf16: 2 },
+ nativeRange: {
+ start: 0,
+ end: 2,
+ unit: "utf16" as const,
+ },
+ captures: [
+ {
+ groupNumber: 1,
+ groupName: "number",
+ value: "42",
+ status: "participated" as const,
+ range: { startUtf16: 0, endUtf16: 2 },
+ nativeRange: {
+ start: 0,
+ end: 2,
+ unit: "utf16" as const,
+ },
+ },
+ {
+ groupNumber: 2,
+ status: "did-not-participate" as const,
+ },
+ ],
+ },
+ ],
+ });
+ }),
+ replace: vi.fn(),
+ cancel: vi.fn(),
+ } as unknown as EngineSupervisor;
+
+ const batch = await runCorpusBatch(
+ [document("captures", "42")],
+ configuration,
+ { engine, signal: new AbortController().signal },
+ );
+
+ expect(batch.documents[0]?.captureSummaries).toEqual([
+ expect.objectContaining({
+ groupNumber: 0,
+ participated: 1,
+ samples: ["42"],
+ }),
+ expect.objectContaining({
+ groupNumber: 1,
+ groupName: "number",
+ participated: 1,
+ samples: ["42"],
+ }),
+ expect.objectContaining({
+ groupNumber: 2,
+ didNotParticipate: 1,
+ samples: [],
+ }),
+ ]);
+ });
+
+ it("preflights document and independent-line limits", async () => {
+ const engine = {
+ execute: vi.fn(),
+ replace: vi.fn(),
+ cancel: vi.fn(),
+ } as unknown as EngineSupervisor;
+ const dependencies = {
+ engine,
+ signal: new AbortController().signal,
+ };
+
+ await expect(
+ runCorpusBatch(
+ Array.from(
+ { length: DEFAULT_REGEX_LIMITS.maximumCorpusDocuments + 1 },
+ (_, index) => document(`document-${index}`, ""),
+ ),
+ configuration,
+ dependencies,
+ ),
+ ).rejects.toThrow(/document limit/iu);
+
+ await expect(
+ runCorpusBatch(
+ [
+ document(
+ "too-many-lines",
+ "\n".repeat(DEFAULT_REGEX_LIMITS.maximumCorpusLines),
+ ),
+ ],
+ { ...configuration, subjectMode: "line" },
+ dependencies,
+ ),
+ ).rejects.toThrow(/logical lines/iu);
+ expect(engine.execute).not.toHaveBeenCalled();
+ });
+
+ it("caps retained capture groups and sample values explicitly", async () => {
+ const oversizedValue =
+ "x".repeat(DEFAULT_REGEX_LIMITS.maximumCorpusCaptureSampleUtf16) + "tail";
+ const engine = {
+ execute: vi.fn((request: RegexExecutionRequest) => {
+ const base = result(request, 0);
+ return Promise.resolve({
+ ...base,
+ matches: [
+ {
+ matchNumber: 1,
+ value: oversizedValue,
+ valueStatus: "complete" as const,
+ range: { startUtf16: 0, endUtf16: oversizedValue.length },
+ nativeRange: {
+ start: 0,
+ end: oversizedValue.length,
+ unit: "utf16" as const,
+ },
+ captures: Array.from(
+ {
+ length:
+ DEFAULT_REGEX_LIMITS.maximumCorpusCaptureSummaries + 5,
+ },
+ (_, index) => ({
+ groupNumber: index + 1,
+ value: String(index),
+ status: "participated" as const,
+ }),
+ ),
+ },
+ ],
+ });
+ }),
+ replace: vi.fn(),
+ cancel: vi.fn(),
+ } as unknown as EngineSupervisor;
+
+ const batch = await runCorpusBatch(
+ [document("many-captures", oversizedValue)],
+ configuration,
+ { engine, signal: new AbortController().signal },
+ );
+
+ expect(batch.documents[0]?.captureSummaries).toHaveLength(
+ DEFAULT_REGEX_LIMITS.maximumCorpusCaptureSummaries,
+ );
+ expect(batch.documents[0]?.captureSummariesTruncated).toBe(true);
+ expect(batch.documents[0]?.captureSummaries[0]).toMatchObject({
+ groupNumber: 0,
+ samplesClipped: true,
+ });
+ expect(batch.documents[0]?.captureSummaries[0]?.samples[0]).toHaveLength(
+ DEFAULT_REGEX_LIMITS.maximumCorpusCaptureSampleUtf16,
+ );
+ });
+});
diff --git a/src/regex/corpus/corpus-runner.ts b/src/regex/corpus/corpus-runner.ts
new file mode 100644
index 0000000..dd0b6e7
--- /dev/null
+++ b/src/regex/corpus/corpus-runner.ts
@@ -0,0 +1,617 @@
+import type { EngineSupervisor } from "../execution/EngineSupervisor";
+import {
+ DEFAULT_REGEX_LIMITS,
+ utf8ByteLength,
+} from "../execution/request-limits";
+import { WorkerRequestError } from "../execution/WorkerSupervisor";
+import type { RegexEngineOptions, RegexFlavourId } from "../model/flavour";
+import type { RegexExecutionResult, RegexMatchResult } from "../model/match";
+import type { CaptureDefinition } from "../model/syntax";
+import { countCorpusLines, splitCorpusLines } from "./corpus-lines";
+import type {
+ CorpusBatchResult,
+ CorpusCaptureSummary,
+ CorpusDocument,
+ CorpusDocumentResult,
+ CorpusOperation,
+ CorpusProgress,
+ CorpusSubjectMode,
+} from "./corpus.types";
+
+const MAXIMUM_DIAGNOSTICS_PER_DOCUMENT = 100;
+
+export interface CorpusRunConfiguration {
+ readonly flavour: RegexFlavourId;
+ readonly flavourVersion?: string;
+ readonly options: RegexEngineOptions;
+ readonly pattern: string;
+ readonly flags: readonly string[];
+ readonly captureMetadata: readonly CaptureDefinition[];
+ readonly replacement: string;
+ readonly operation: CorpusOperation;
+ readonly subjectMode: CorpusSubjectMode;
+ readonly timeoutMs: number;
+}
+
+export interface CorpusRunDependencies {
+ readonly engine: EngineSupervisor;
+ readonly signal: AbortSignal;
+ readonly onProgress?: (progress: CorpusProgress) => void;
+ readonly now?: () => number;
+}
+
+interface MutableCaptureSummary {
+ groupNumber: number;
+ groupName?: string;
+ participated: number;
+ didNotParticipate: number;
+ matchedEmpty: number;
+ truncated: number;
+ unavailable: number;
+ samples: string[];
+ samplesClipped: boolean;
+}
+
+function emptyCaptureSummary(
+ groupNumber: number,
+ groupName?: string,
+): MutableCaptureSummary {
+ return {
+ groupNumber,
+ ...(groupName === undefined ? {} : { groupName }),
+ participated: 0,
+ didNotParticipate: 0,
+ matchedEmpty: 0,
+ truncated: 0,
+ unavailable: 0,
+ samples: [],
+ samplesClipped: false,
+ };
+}
+
+function retainCaptureSample(
+ summary: MutableCaptureSummary,
+ value: string | undefined,
+ engineClipped: boolean,
+): void {
+ if (value === undefined) return;
+ const clipped =
+ engineClipped ||
+ value.length > DEFAULT_REGEX_LIMITS.maximumCorpusCaptureSampleUtf16;
+ summary.samplesClipped ||= clipped;
+ if (
+ summary.samples.length <
+ DEFAULT_REGEX_LIMITS.maximumCorpusCaptureSamplesPerGroup
+ ) {
+ summary.samples.push(
+ value.slice(0, DEFAULT_REGEX_LIMITS.maximumCorpusCaptureSampleUtf16),
+ );
+ }
+}
+
+function addMatchCaptureSummaries(
+ matches: readonly RegexMatchResult[],
+ summaries: Map,
+): boolean {
+ let truncated = false;
+ const summaryFor = (
+ groupNumber: number,
+ groupName?: string,
+ ): MutableCaptureSummary | undefined => {
+ const existing = summaries.get(groupNumber);
+ if (existing) return existing;
+ if (summaries.size >= DEFAULT_REGEX_LIMITS.maximumCorpusCaptureSummaries) {
+ truncated = true;
+ return undefined;
+ }
+ const created = emptyCaptureSummary(groupNumber, groupName);
+ summaries.set(groupNumber, created);
+ return created;
+ };
+
+ for (const match of matches) {
+ const fullMatch = summaryFor(0);
+ if (fullMatch) {
+ if (match.valueStatus === "truncated") {
+ fullMatch.truncated += 1;
+ } else if (match.value.length === 0) {
+ fullMatch.matchedEmpty += 1;
+ } else {
+ fullMatch.participated += 1;
+ }
+ retainCaptureSample(
+ fullMatch,
+ match.value,
+ match.valueStatus === "truncated",
+ );
+ }
+ for (const capture of match.captures) {
+ const summary = summaryFor(capture.groupNumber, capture.groupName);
+ if (!summary) continue;
+ switch (capture.status) {
+ case "participated":
+ summary.participated += 1;
+ break;
+ case "did-not-participate":
+ summary.didNotParticipate += 1;
+ break;
+ case "matched-empty":
+ summary.matchedEmpty += 1;
+ break;
+ case "truncated":
+ summary.truncated += 1;
+ break;
+ case "unavailable":
+ summary.unavailable += 1;
+ break;
+ }
+ retainCaptureSample(
+ summary,
+ capture.value,
+ capture.status === "truncated",
+ );
+ }
+ }
+ return truncated;
+}
+
+function frozenCaptureSummaries(
+ summaries: ReadonlyMap,
+): readonly CorpusCaptureSummary[] {
+ return [...summaries.values()]
+ .sort((left, right) => left.groupNumber - right.groupNumber)
+ .map((summary) => ({
+ ...summary,
+ samples: [...summary.samples],
+ }));
+}
+
+function pendingResult(
+ document: CorpusDocument,
+ lineCount: number,
+ status: "cancelled" | "not-run",
+ message: string,
+): CorpusDocumentResult {
+ return {
+ documentId: document.id,
+ name: document.name,
+ inputBytes: document.bytes,
+ lineCount,
+ status,
+ matchCount: 0,
+ elapsedMs: 0,
+ captureSummaries: [],
+ captureSummariesTruncated: false,
+ diagnostics: [],
+ message,
+ };
+}
+
+function aggregateResult(
+ configuration: CorpusRunConfiguration,
+ documents: readonly CorpusDocument[],
+ results: readonly CorpusDocumentResult[],
+ cancelled: boolean,
+ message?: string,
+): CorpusBatchResult {
+ const processed = results.filter(
+ (result) =>
+ result.status === "complete" ||
+ result.status === "partial" ||
+ result.status === "error",
+ ).length;
+ const partial =
+ results.some((result) => result.status !== "complete") ||
+ processed !== documents.length;
+ return {
+ operation: configuration.operation,
+ subjectMode: configuration.subjectMode,
+ status: cancelled ? "cancelled" : partial ? "partial" : "complete",
+ documents: results,
+ processedDocuments: processed,
+ totalDocuments: documents.length,
+ totalMatches: results.reduce((sum, result) => sum + result.matchCount, 0),
+ totalElapsedMs: results.reduce((sum, result) => sum + result.elapsedMs, 0),
+ retainedOutputBytes: results.reduce(
+ (sum, result) => sum + (result.outputBytes ?? 0),
+ 0,
+ ),
+ message:
+ message ??
+ (partial
+ ? "The batch completed with incomplete documents. Review each status before exporting."
+ : `Processed ${documents.length.toLocaleString()} documents locally.`),
+ };
+}
+
+function wallTimeAbort(signal: AbortSignal): boolean {
+ return signal.reason === "corpus-wall-time";
+}
+
+function errorMessage(error: unknown): string {
+ if (error instanceof WorkerRequestError && error.kind === "timeout") {
+ return `${error.message}. The line/document worker was terminated and recreated for subsequent work.`;
+ }
+ return error instanceof Error ? error.message : String(error);
+}
+
+function addDiagnostics(
+ execution: RegexExecutionResult,
+ diagnostics: RegexExecutionResult["diagnostics"][number][],
+ keys: Set,
+): void {
+ for (const diagnostic of execution.diagnostics) {
+ const key = `${diagnostic.severity}\0${diagnostic.code}\0${diagnostic.message}`;
+ if (keys.has(key)) continue;
+ keys.add(key);
+ if (diagnostics.length < MAXIMUM_DIAGNOSTICS_PER_DOCUMENT) {
+ diagnostics.push(diagnostic);
+ }
+ }
+}
+
+function preflightLineCounts(
+ documents: readonly CorpusDocument[],
+ subjectMode: CorpusSubjectMode,
+): readonly number[] {
+ const counts = documents.map((document) => countCorpusLines(document.text));
+ if (subjectMode === "line") {
+ const total = counts.reduce((sum, count) => sum + count, 0);
+ if (total > DEFAULT_REGEX_LIMITS.maximumCorpusLines) {
+ throw new Error(
+ `Independent-line mode is limited to ${DEFAULT_REGEX_LIMITS.maximumCorpusLines.toLocaleString()} logical lines per batch.`,
+ );
+ }
+ }
+ return counts;
+}
+
+export async function runCorpusBatch(
+ documents: readonly CorpusDocument[],
+ configuration: CorpusRunConfiguration,
+ dependencies: CorpusRunDependencies,
+): Promise {
+ if (documents.length === 0) {
+ throw new Error("Add at least one corpus document before running.");
+ }
+ if (documents.length > DEFAULT_REGEX_LIMITS.maximumCorpusDocuments) {
+ throw new Error("The corpus document limit was exceeded.");
+ }
+ const lineCounts = preflightLineCounts(documents, configuration.subjectMode);
+ const now = dependencies.now ?? (() => performance.now());
+ const startedAt = now();
+ let retainedOutputBytes = 0;
+ let retainedMatches = 0;
+ const results: CorpusDocumentResult[] = [];
+ dependencies.onProgress?.({
+ completedDocuments: 0,
+ totalDocuments: documents.length,
+ currentDocumentName: documents[0]?.name,
+ totalMatches: 0,
+ });
+
+ const finishStoppedBatch = (
+ fromIndex: number,
+ kind: "cancelled" | "wall-time" | "match-limit",
+ currentResult?: CorpusDocumentResult,
+ ): CorpusBatchResult => {
+ const beginning = currentResult
+ ? [...results, currentResult]
+ : [...results];
+ const remainingStart = currentResult ? fromIndex + 1 : fromIndex;
+ const status = kind === "cancelled" ? "cancelled" : "not-run";
+ const reason =
+ kind === "cancelled"
+ ? "Not run because the batch was cancelled."
+ : kind === "wall-time"
+ ? "Not run because the aggregate corpus wall-time limit was reached."
+ : "Not run because the aggregate corpus match limit was reached.";
+ const remaining = documents
+ .slice(remainingStart)
+ .map((document, offset) =>
+ pendingResult(
+ document,
+ lineCounts[remainingStart + offset] ?? 1,
+ status,
+ reason,
+ ),
+ );
+ const message =
+ kind === "cancelled"
+ ? "Corpus processing was cancelled. Completed rows remain labelled and no incomplete applied output can be exported."
+ : kind === "wall-time"
+ ? `Corpus processing stopped at the ${DEFAULT_REGEX_LIMITS.maximumCorpusWallTimeMs / 1_000} second aggregate wall-time limit.`
+ : `Corpus processing stopped at the ${DEFAULT_REGEX_LIMITS.maximumCorpusMatches.toLocaleString()} retained-match limit.`;
+ return aggregateResult(
+ configuration,
+ documents,
+ [...beginning, ...remaining],
+ kind === "cancelled",
+ message,
+ );
+ };
+
+ for (const [documentIndex, document] of documents.entries()) {
+ if (dependencies.signal.aborted) {
+ return finishStoppedBatch(
+ documentIndex,
+ wallTimeAbort(dependencies.signal) ? "wall-time" : "cancelled",
+ );
+ }
+ if (now() - startedAt >= DEFAULT_REGEX_LIMITS.maximumCorpusWallTimeMs) {
+ dependencies.engine.cancel();
+ return finishStoppedBatch(documentIndex, "wall-time");
+ }
+ if (retainedMatches >= DEFAULT_REGEX_LIMITS.maximumCorpusMatches) {
+ return finishStoppedBatch(documentIndex, "match-limit");
+ }
+
+ const segments =
+ configuration.subjectMode === "line"
+ ? splitCorpusLines(document.text)
+ : [{ text: document.text, ending: "" }];
+ const captureSummaries = new Map();
+ const diagnostics: RegexExecutionResult["diagnostics"][number][] = [];
+ const diagnosticKeys = new Set();
+ const outputChunks: string[] = [];
+ const errors: string[] = [];
+ let matchCount = 0;
+ let elapsedMs = 0;
+ let outputBytes = 0;
+ let completedSegments = 0;
+ let successfulSegments = 0;
+ let partial = false;
+ let fatal = false;
+ let captureSummariesTruncated = false;
+ let stoppedByMatchLimit = false;
+
+ for (const segment of segments) {
+ if (dependencies.signal.aborted) {
+ const stopKind = wallTimeAbort(dependencies.signal)
+ ? "wall-time"
+ : "cancelled";
+ const currentResult: CorpusDocumentResult = {
+ documentId: document.id,
+ name: document.name,
+ inputBytes: document.bytes,
+ lineCount: lineCounts[documentIndex] ?? 1,
+ status: stopKind === "cancelled" ? "cancelled" : "partial",
+ matchCount,
+ elapsedMs,
+ ...(configuration.operation === "replace"
+ ? {
+ output: outputChunks.join(""),
+ outputBytes,
+ }
+ : {}),
+ captureSummaries: frozenCaptureSummaries(captureSummaries),
+ captureSummariesTruncated,
+ diagnostics,
+ message:
+ stopKind === "cancelled"
+ ? "Document processing was cancelled before completion."
+ : "Document processing stopped at the aggregate wall-time limit.",
+ };
+ return finishStoppedBatch(documentIndex, stopKind, currentResult);
+ }
+ if (now() - startedAt >= DEFAULT_REGEX_LIMITS.maximumCorpusWallTimeMs) {
+ dependencies.engine.cancel();
+ const currentResult: CorpusDocumentResult = {
+ documentId: document.id,
+ name: document.name,
+ inputBytes: document.bytes,
+ lineCount: lineCounts[documentIndex] ?? 1,
+ status: "partial",
+ matchCount,
+ elapsedMs,
+ ...(configuration.operation === "replace"
+ ? { output: outputChunks.join(""), outputBytes }
+ : {}),
+ captureSummaries: frozenCaptureSummaries(captureSummaries),
+ captureSummariesTruncated,
+ diagnostics,
+ message:
+ "Document processing stopped at the aggregate wall-time limit.",
+ };
+ return finishStoppedBatch(documentIndex, "wall-time", currentResult);
+ }
+
+ const remainingMatches =
+ DEFAULT_REGEX_LIMITS.maximumCorpusMatches - retainedMatches;
+ if (remainingMatches <= 0) {
+ stoppedByMatchLimit = true;
+ partial = true;
+ break;
+ }
+ const commonRequest = {
+ flavour: configuration.flavour,
+ ...(configuration.flavourVersion === undefined
+ ? {}
+ : { flavourVersion: configuration.flavourVersion }),
+ pattern: configuration.pattern,
+ flags: configuration.flags,
+ options: configuration.options,
+ subject: segment.text,
+ captureMetadata: configuration.captureMetadata,
+ // Corpus processing is exhaustive by definition. Each adapter maps
+ // scanAll to its own iteration semantics without changing saved flags.
+ scanAll: true,
+ maximumMatches: Math.min(
+ DEFAULT_REGEX_LIMITS.maximumMatches,
+ remainingMatches,
+ ),
+ maximumCaptureRows: DEFAULT_REGEX_LIMITS.maximumCaptureRows,
+ } as const;
+
+ try {
+ let execution: RegexExecutionResult;
+ if (configuration.operation === "replace") {
+ const endingBytes = utf8ByteLength(segment.ending);
+ const remainingOutput =
+ DEFAULT_REGEX_LIMITS.maximumCorpusOutputBytes -
+ retainedOutputBytes -
+ endingBytes;
+ const replacementResult = await dependencies.engine.replace(
+ {
+ ...commonRequest,
+ replacement: configuration.replacement,
+ maximumOutputBytes: Math.max(0, remainingOutput),
+ },
+ configuration.timeoutMs,
+ );
+ execution = replacementResult.execution;
+ outputChunks.push(replacementResult.output);
+ outputBytes += replacementResult.outputBytes;
+ retainedOutputBytes += replacementResult.outputBytes;
+ if (
+ !replacementResult.outputTruncated &&
+ retainedOutputBytes + endingBytes <=
+ DEFAULT_REGEX_LIMITS.maximumCorpusOutputBytes
+ ) {
+ outputChunks.push(segment.ending);
+ outputBytes += endingBytes;
+ retainedOutputBytes += endingBytes;
+ } else if (segment.ending.length > 0) {
+ partial = true;
+ }
+ partial ||= replacementResult.truncated;
+ } else {
+ execution = await dependencies.engine.execute(
+ commonRequest,
+ configuration.timeoutMs,
+ );
+ }
+
+ addDiagnostics(execution, diagnostics, diagnosticKeys);
+ elapsedMs += execution.elapsedMs;
+ if (!execution.accepted) {
+ fatal = true;
+ errors.push(
+ execution.diagnostics[0]?.message ??
+ "The execution engine rejected the pattern.",
+ );
+ break;
+ }
+ successfulSegments += 1;
+ completedSegments += 1;
+ matchCount += execution.matches.length;
+ retainedMatches += execution.matches.length;
+ partial ||= execution.truncated;
+ captureSummariesTruncated ||= addMatchCaptureSummaries(
+ execution.matches,
+ captureSummaries,
+ );
+ if (
+ retainedMatches >= DEFAULT_REGEX_LIMITS.maximumCorpusMatches &&
+ (completedSegments < segments.length ||
+ documentIndex + 1 < documents.length)
+ ) {
+ stoppedByMatchLimit = true;
+ partial ||= completedSegments < segments.length;
+ break;
+ }
+ } catch (error) {
+ if (
+ dependencies.signal.aborted ||
+ (error instanceof WorkerRequestError && error.kind === "cancelled")
+ ) {
+ const stopKind = wallTimeAbort(dependencies.signal)
+ ? "wall-time"
+ : "cancelled";
+ const currentResult: CorpusDocumentResult = {
+ documentId: document.id,
+ name: document.name,
+ inputBytes: document.bytes,
+ lineCount: lineCounts[documentIndex] ?? 1,
+ status: stopKind === "cancelled" ? "cancelled" : "partial",
+ matchCount,
+ elapsedMs,
+ ...(configuration.operation === "replace"
+ ? { output: outputChunks.join(""), outputBytes }
+ : {}),
+ captureSummaries: frozenCaptureSummaries(captureSummaries),
+ captureSummariesTruncated,
+ diagnostics,
+ message:
+ stopKind === "cancelled"
+ ? "Document processing was cancelled before completion."
+ : "Document processing stopped at the aggregate wall-time limit.",
+ };
+ return finishStoppedBatch(documentIndex, stopKind, currentResult);
+ }
+ errors.push(errorMessage(error));
+ partial = true;
+ completedSegments += 1;
+ if (configuration.operation === "replace") {
+ const original = `${segment.text}${segment.ending}`;
+ const originalBytes = utf8ByteLength(original);
+ if (
+ retainedOutputBytes + originalBytes <=
+ DEFAULT_REGEX_LIMITS.maximumCorpusOutputBytes
+ ) {
+ outputChunks.push(original);
+ outputBytes += originalBytes;
+ retainedOutputBytes += originalBytes;
+ }
+ }
+ }
+ }
+
+ const output =
+ configuration.operation === "replace" ? outputChunks.join("") : undefined;
+ const status =
+ (fatal || errors.length > 0) && successfulSegments === 0
+ ? "error"
+ : partial || completedSegments < segments.length || errors.length > 0
+ ? "partial"
+ : "complete";
+ const messages = [
+ ...errors,
+ ...(captureSummariesTruncated
+ ? [
+ `Capture summaries are limited to ${DEFAULT_REGEX_LIMITS.maximumCorpusCaptureSummaries.toLocaleString()} groups per document.`,
+ ]
+ : []),
+ ...(stoppedByMatchLimit
+ ? ["Processing stopped at the aggregate corpus match limit."]
+ : []),
+ ];
+ results.push({
+ documentId: document.id,
+ name: document.name,
+ inputBytes: document.bytes,
+ lineCount: lineCounts[documentIndex] ?? 1,
+ status,
+ matchCount,
+ elapsedMs,
+ ...(output === undefined
+ ? {}
+ : {
+ output,
+ outputBytes,
+ changed: status === "complete" && output !== document.text,
+ }),
+ captureSummaries: frozenCaptureSummaries(captureSummaries),
+ captureSummariesTruncated,
+ diagnostics,
+ ...(messages.length === 0
+ ? {}
+ : {
+ message: messages.join(" "),
+ }),
+ });
+ dependencies.onProgress?.({
+ completedDocuments: documentIndex + 1,
+ totalDocuments: documents.length,
+ ...(documents[documentIndex + 1]
+ ? { currentDocumentName: documents[documentIndex + 1]?.name }
+ : {}),
+ totalMatches: retainedMatches,
+ });
+ if (stoppedByMatchLimit && documentIndex + 1 < documents.length) {
+ return finishStoppedBatch(documentIndex + 1, "match-limit");
+ }
+ }
+
+ return aggregateResult(configuration, documents, results, false);
+}
diff --git a/src/regex/corpus/corpus.types.ts b/src/regex/corpus/corpus.types.ts
new file mode 100644
index 0000000..3cb06fb
--- /dev/null
+++ b/src/regex/corpus/corpus.types.ts
@@ -0,0 +1,66 @@
+import type { RegexDiagnostic } from "../model/diagnostics";
+
+export type CorpusDocumentSource = "file" | "pasted-text";
+export type CorpusOperation = "match" | "replace";
+export type CorpusSubjectMode = "document" | "line";
+
+export interface CorpusDocument {
+ readonly id: string;
+ readonly name: string;
+ readonly text: string;
+ readonly bytes: number;
+ readonly source: CorpusDocumentSource;
+}
+
+export type CorpusDocumentResultStatus =
+ "complete" | "partial" | "error" | "cancelled" | "not-run";
+
+export interface CorpusDocumentResult {
+ readonly documentId: string;
+ readonly name: string;
+ readonly inputBytes: number;
+ readonly lineCount: number;
+ readonly status: CorpusDocumentResultStatus;
+ /** Exact only when status is complete; otherwise a retained lower bound. */
+ readonly matchCount: number;
+ readonly elapsedMs: number;
+ readonly output?: string;
+ readonly outputBytes?: number;
+ readonly changed?: boolean;
+ readonly captureSummaries: readonly CorpusCaptureSummary[];
+ readonly captureSummariesTruncated: boolean;
+ readonly diagnostics: readonly RegexDiagnostic[];
+ readonly message?: string;
+}
+
+export interface CorpusBatchResult {
+ readonly operation: CorpusOperation;
+ readonly subjectMode: CorpusSubjectMode;
+ readonly status: "complete" | "partial" | "cancelled";
+ readonly documents: readonly CorpusDocumentResult[];
+ readonly processedDocuments: number;
+ readonly totalDocuments: number;
+ readonly totalMatches: number;
+ readonly totalElapsedMs: number;
+ readonly retainedOutputBytes: number;
+ readonly message: string;
+}
+
+export interface CorpusCaptureSummary {
+ readonly groupNumber: number;
+ readonly groupName?: string;
+ readonly participated: number;
+ readonly didNotParticipate: number;
+ readonly matchedEmpty: number;
+ readonly truncated: number;
+ readonly unavailable: number;
+ readonly samples: readonly string[];
+ readonly samplesClipped: boolean;
+}
+
+export interface CorpusProgress {
+ readonly completedDocuments: number;
+ readonly totalDocuments: number;
+ readonly currentDocumentName?: string;
+ readonly totalMatches: number;
+}
diff --git a/src/regex/execution/EngineSupervisor.test.ts b/src/regex/execution/EngineSupervisor.test.ts
new file mode 100644
index 0000000..b9adb5c
--- /dev/null
+++ b/src/regex/execution/EngineSupervisor.test.ts
@@ -0,0 +1,168 @@
+import { describe, expect, it } from "vitest";
+import type {
+ RegexExecutionRequest,
+ RegexExecutionResult,
+} from "../model/match";
+import type { RegexFlavourId } from "../model/flavour";
+import { EngineSupervisor } from "./EngineSupervisor";
+import {
+ ENGINE_WORKER_REGISTRY,
+ EngineWorkerRegistry,
+} from "./engine-registry";
+import type { WorkerLike } from "./WorkerSupervisor";
+import {
+ WORKER_PROTOCOL_VERSION,
+ type EngineWorkerOperation,
+ type EngineWorkerResult,
+ type WorkerRequest,
+ type WorkerResponse,
+} from "./worker-protocol";
+
+function request(flavour: RegexFlavourId): RegexExecutionRequest {
+ return {
+ flavour,
+ flavourVersion: "fixture",
+ pattern: "a",
+ flags: [],
+ options: {},
+ subject: "a",
+ captureMetadata: [],
+ scanAll: false,
+ maximumMatches: 10,
+ maximumCaptureRows: 10,
+ };
+}
+
+function result(flavour: RegexFlavourId): RegexExecutionResult {
+ return {
+ accepted: true,
+ engine: {
+ flavour,
+ adapterVersion: "fixture",
+ engineName: `${flavour} fixture`,
+ engineVersion: "fixture",
+ offsetUnit: "utf16",
+ capabilities: {
+ compilation: true,
+ matching: true,
+ replacement: false,
+ namedCaptures: false,
+ captureHistory: false,
+ actualTrace: false,
+ benchmark: false,
+ },
+ },
+ flags: {
+ userFlags: "",
+ effectiveFlags: "",
+ internallyAddedIndicesFlag: false,
+ internallyAddedGlobalFlag: false,
+ },
+ matches: [],
+ diagnostics: [],
+ elapsedMs: 0,
+ truncated: false,
+ };
+}
+
+class ResponsiveWorker implements WorkerLike {
+ onmessage: ((event: MessageEvent) => void) | null = null;
+ onerror: ((event: ErrorEvent) => void) | null = null;
+ onmessageerror: ((event: MessageEvent) => void) | null = null;
+ readonly operations: EngineWorkerOperation[] = [];
+ terminateCalls = 0;
+
+ postMessage(message: unknown): void {
+ const request = message as WorkerRequest;
+ this.operations.push(request.payload);
+ const payload: EngineWorkerResult =
+ request.payload.kind === "execute"
+ ? {
+ kind: "execute",
+ result: result(request.payload.request.flavour),
+ }
+ : {
+ kind: "replace",
+ result: {
+ execution: result(request.payload.request.flavour),
+ output: "",
+ outputBytes: 0,
+ outputTruncated: false,
+ truncated: false,
+ },
+ };
+ const response: WorkerResponse = {
+ protocolVersion: WORKER_PROTOCOL_VERSION,
+ requestId: request.requestId,
+ generation: request.generation,
+ ok: true,
+ payload,
+ };
+ this.onmessage?.({ data: response } as MessageEvent);
+ }
+
+ terminate(): void {
+ this.terminateCalls += 1;
+ }
+}
+
+describe("engine worker selection", () => {
+ it("registers the shipped ECMAScript and PCRE2 execution workers", () => {
+ expect(
+ ENGINE_WORKER_REGISTRY.registrations.map(
+ (registration) => registration.flavour,
+ ),
+ ).toEqual(["ecmascript", "pcre2"]);
+ expect(ENGINE_WORKER_REGISTRY.require("pcre2").label).toBe("PCRE2 engine");
+ });
+
+ it("selects and reuses a dedicated supervisor per registered flavour", async () => {
+ const workers = new Map();
+ const registration = (flavour: RegexFlavourId) => ({
+ flavour,
+ label: `${flavour} fixture`,
+ createWorker: () => {
+ const worker = new ResponsiveWorker();
+ workers.set(flavour, [...(workers.get(flavour) ?? []), worker]);
+ return worker;
+ },
+ });
+ const supervisor = new EngineSupervisor(
+ new EngineWorkerRegistry([
+ registration("ecmascript"),
+ registration("pcre2"),
+ ]),
+ );
+
+ await expect(
+ supervisor.execute(request("ecmascript"), 100),
+ ).resolves.toEqual(
+ expect.objectContaining({
+ engine: expect.objectContaining({ flavour: "ecmascript" }),
+ }),
+ );
+ await expect(supervisor.execute(request("pcre2"), 100)).resolves.toEqual(
+ expect.objectContaining({
+ engine: expect.objectContaining({ flavour: "pcre2" }),
+ }),
+ );
+ await supervisor.execute(request("ecmascript"), 100);
+
+ expect(workers.get("ecmascript")).toHaveLength(1);
+ expect(workers.get("pcre2")).toHaveLength(1);
+ expect(workers.get("ecmascript")?.[0]?.operations).toHaveLength(2);
+
+ supervisor.dispose();
+ expect(workers.get("ecmascript")?.[0]?.terminateCalls).toBe(1);
+ expect(workers.get("pcre2")?.[0]?.terminateCalls).toBe(1);
+ });
+
+ it("fails closed before constructing a worker for an unavailable flavour", async () => {
+ const supervisor = new EngineSupervisor(new EngineWorkerRegistry([]));
+
+ await expect(supervisor.execute(request("pcre2"), 100)).rejects.toThrow(
+ /No execution worker is registered/u,
+ );
+ supervisor.dispose();
+ });
+});
diff --git a/src/regex/execution/EngineSupervisor.ts b/src/regex/execution/EngineSupervisor.ts
index c671a64..75a3108 100644
--- a/src/regex/execution/EngineSupervisor.ts
+++ b/src/regex/execution/EngineSupervisor.ts
@@ -9,6 +9,11 @@ import type {
EngineWorkerOperation,
EngineWorkerResult,
} from "./worker-protocol";
+import {
+ ENGINE_WORKER_REGISTRY,
+ type EngineWorkerRegistry,
+} from "./engine-registry";
+import type { RegexFlavourId } from "../model/flavour";
export type EngineRuntimeState =
| { readonly status: "unloaded" }
@@ -18,26 +23,39 @@ export type EngineRuntimeState =
| { readonly status: "failed"; readonly error: Error };
export class EngineSupervisor {
- private readonly supervisor = new WorkerSupervisor<
- EngineWorkerOperation,
- EngineWorkerResult
- >(
- "ECMAScript engine",
- () =>
- new Worker(
- new URL("../../workers/ecmascript.worker.ts", import.meta.url),
- {
- type: "module",
- name: "regex-tools-ecmascript",
- },
- ),
- );
+ private readonly registry: EngineWorkerRegistry;
+ private readonly supervisors = new Map<
+ RegexFlavourId,
+ WorkerSupervisor
+ >();
+ private disposed = false;
+
+ constructor(registry: EngineWorkerRegistry = ENGINE_WORKER_REGISTRY) {
+ this.registry = registry;
+ }
+
+ private supervisorFor(
+ flavour: RegexFlavourId,
+ ): WorkerSupervisor {
+ if (this.disposed) {
+ throw new Error("Engine supervisor has been disposed.");
+ }
+ const current = this.supervisors.get(flavour);
+ if (current) return current;
+ const registration = this.registry.require(flavour);
+ const supervisor = new WorkerSupervisor<
+ EngineWorkerOperation,
+ EngineWorkerResult
+ >(registration.label, registration.createWorker);
+ this.supervisors.set(flavour, supervisor);
+ return supervisor;
+ }
async execute(
request: RegexExecutionRequest,
timeoutMs: number,
): Promise {
- const response = await this.supervisor.run(
+ const response = await this.supervisorFor(request.flavour).run(
{ kind: "execute", request },
timeoutMs,
{ supersede: true },
@@ -45,6 +63,11 @@ export class EngineSupervisor {
if (response.kind !== "execute") {
throw new Error("Engine worker returned the wrong response kind");
}
+ if (response.result.engine.flavour !== request.flavour) {
+ throw new Error(
+ `Engine worker for ${request.flavour} returned a ${response.result.engine.flavour} result.`,
+ );
+ }
return response.result;
}
@@ -52,7 +75,7 @@ export class EngineSupervisor {
request: RegexReplacementRequest,
timeoutMs: number,
): Promise {
- const response = await this.supervisor.run(
+ const response = await this.supervisorFor(request.flavour).run(
{ kind: "replace", request },
timeoutMs,
{ supersede: true },
@@ -60,14 +83,22 @@ export class EngineSupervisor {
if (response.kind !== "replace") {
throw new Error("Engine worker returned the wrong response kind");
}
+ if (response.result.execution.engine.flavour !== request.flavour) {
+ throw new Error(
+ `Engine worker for ${request.flavour} returned a ${response.result.execution.engine.flavour} result.`,
+ );
+ }
return response.result;
}
cancel(): void {
- this.supervisor.cancel();
+ for (const supervisor of this.supervisors.values()) supervisor.cancel();
}
dispose(): void {
- this.supervisor.dispose();
+ if (this.disposed) return;
+ this.disposed = true;
+ for (const supervisor of this.supervisors.values()) supervisor.dispose();
+ this.supervisors.clear();
}
}
diff --git a/src/regex/execution/Pcre2TraceSupervisor.ts b/src/regex/execution/Pcre2TraceSupervisor.ts
new file mode 100644
index 0000000..a247304
--- /dev/null
+++ b/src/regex/execution/Pcre2TraceSupervisor.ts
@@ -0,0 +1,51 @@
+import type { Pcre2TraceRequest, Pcre2TraceResult } from "../model/trace";
+import { WorkerSupervisor } from "./WorkerSupervisor";
+import type {
+ TraceWorkerOperation,
+ TraceWorkerResult,
+} from "./worker-protocol";
+
+export class Pcre2TraceSupervisor {
+ readonly #supervisor: WorkerSupervisor<
+ TraceWorkerOperation,
+ TraceWorkerResult
+ >;
+
+ constructor() {
+ this.#supervisor = new WorkerSupervisor<
+ TraceWorkerOperation,
+ TraceWorkerResult
+ >("PCRE2 automatic-callout trace", () => {
+ return new Worker(
+ new URL("../../workers/pcre2-trace.worker.ts", import.meta.url),
+ {
+ type: "module",
+ name: "regex-tools-pcre2-trace",
+ },
+ );
+ });
+ }
+
+ async trace(
+ request: Pcre2TraceRequest,
+ timeoutMs: number,
+ ): Promise {
+ const response = await this.#supervisor.run(
+ { kind: "trace", request },
+ timeoutMs,
+ { supersede: true },
+ );
+ if (response.kind !== "trace") {
+ throw new Error("PCRE2 trace worker returned the wrong response kind.");
+ }
+ return response.result;
+ }
+
+ cancel(): void {
+ this.#supervisor.cancel();
+ }
+
+ dispose(): void {
+ this.#supervisor.dispose();
+ }
+}
diff --git a/src/regex/execution/WorkerSupervisor.ts b/src/regex/execution/WorkerSupervisor.ts
index 471e3d4..edaccdd 100644
--- a/src/regex/execution/WorkerSupervisor.ts
+++ b/src/regex/execution/WorkerSupervisor.ts
@@ -33,7 +33,7 @@ interface ActiveRequest {
readonly generation: number;
readonly resolve: (result: TResult) => void;
readonly reject: (error: Error) => void;
- readonly timeout: number;
+ readonly timeout: ReturnType;
}
export class WorkerSupervisor {
diff --git a/src/regex/execution/adapters/ecmascript/EcmaScriptEngineAdapter.test.ts b/src/regex/execution/adapters/ecmascript/EcmaScriptEngineAdapter.test.ts
index daa0873..7525048 100644
--- a/src/regex/execution/adapters/ecmascript/EcmaScriptEngineAdapter.test.ts
+++ b/src/regex/execution/adapters/ecmascript/EcmaScriptEngineAdapter.test.ts
@@ -36,6 +36,7 @@ describe("native ECMAScript adapter", () => {
expect(result.accepted).toBe(true);
expect(result.engine.engineName).toBe("Native ECMAScript RegExp");
expect(result.engine.offsetUnit).toBe("utf16");
+ expect(result.engine.capabilities.benchmark).toBe(true);
expect(result.matches).toHaveLength(1);
expect(result.matches[0]?.range).toEqual({
startUtf16: 0,
diff --git a/src/regex/execution/adapters/ecmascript/EcmaScriptEngineAdapter.ts b/src/regex/execution/adapters/ecmascript/EcmaScriptEngineAdapter.ts
index 001c753..9b2a364 100644
--- a/src/regex/execution/adapters/ecmascript/EcmaScriptEngineAdapter.ts
+++ b/src/regex/execution/adapters/ecmascript/EcmaScriptEngineAdapter.ts
@@ -1,7 +1,7 @@
import type { RegexEngineAdapter } from "../../RegexEngineAdapter";
import type {
CaptureResult,
- EcmaScriptExecutionFlags,
+ RegexExecutionFlags,
RegexExecutionRequest,
RegexExecutionResult,
RegexMatchResult,
@@ -11,6 +11,7 @@ import type {
import type { RegexEngineInfo } from "../../../model/flavour";
import type { RegexDiagnostic } from "../../../model/diagnostics";
import { DEFAULT_REGEX_LIMITS } from "../../request-limits";
+import { APPLICATION_VERSION } from "../../../../version";
const FLAG_ORDER = "dgimsuvy";
const CAPTURE_VALUE_PREVIEW_UTF16 = 64 * 1024;
@@ -48,7 +49,7 @@ function engineInfo(): RegexEngineInfo {
const identity = runtimeIdentity();
return {
flavour: "ecmascript",
- adapterVersion: "0.1.0",
+ adapterVersion: APPLICATION_VERSION,
engineName: "Native ECMAScript RegExp",
engineVersion: identity,
runtimeVersion: identity,
@@ -60,7 +61,7 @@ function engineInfo(): RegexEngineInfo {
namedCaptures: true,
captureHistory: false,
actualTrace: false,
- benchmark: false,
+ benchmark: true,
},
};
}
@@ -68,7 +69,7 @@ function engineInfo(): RegexEngineInfo {
function normalizedFlags(
userFlags: readonly string[],
scanAll: boolean,
-): EcmaScriptExecutionFlags {
+): RegexExecutionFlags {
const unique = new Set(userFlags);
const user = FLAG_ORDER.split("")
.filter((flag) => unique.has(flag))
@@ -189,7 +190,7 @@ function compileDiagnostic(error: unknown): RegexDiagnostic {
}
function emptyExecution(
- flags: EcmaScriptExecutionFlags,
+ flags: RegexExecutionFlags,
start: number,
error: unknown,
): RegexExecutionResult {
diff --git a/src/regex/execution/adapters/pcre2/Pcre2EngineAdapter.test.ts b/src/regex/execution/adapters/pcre2/Pcre2EngineAdapter.test.ts
new file mode 100644
index 0000000..9d54e49
--- /dev/null
+++ b/src/regex/execution/adapters/pcre2/Pcre2EngineAdapter.test.ts
@@ -0,0 +1,277 @@
+import { describe, expect, it } from "vitest";
+import type {
+ RegexExecutionRequest,
+ RegexReplacementRequest,
+} from "../../../model/match";
+import {
+ Pcre2EngineAdapter,
+ PCRE2_BRIDGE_ABI_VERSION,
+ PCRE2_ENGINE_VERSION,
+ pcre2Limits,
+} from "./Pcre2EngineAdapter";
+import type { Pcre2EmscriptenModule } from "./pcre2-module";
+
+const encoder = new TextEncoder();
+const decoder = new TextDecoder();
+
+class FakePcre2Module implements Pcre2EmscriptenModule {
+ readonly HEAPU8 = new Uint8Array(4 * 1024 * 1024);
+ readonly UTF8ToString = (pointer: number) => {
+ let end = pointer;
+ while (this.HEAPU8[end] !== 0) end += 1;
+ return decoder.decode(this.HEAPU8.slice(pointer, end));
+ };
+ readonly allocated: number[] = [];
+ readonly freed: number[] = [];
+ executeCalls = 0;
+ substituteCalls = 0;
+ #nextPointer = 1_024;
+ readonly #versionPointer: number;
+
+ constructor() {
+ const version = encoder.encode(`${PCRE2_ENGINE_VERSION}\0`);
+ this.#versionPointer = 64;
+ this.HEAPU8.set(version, this.#versionPointer);
+ }
+
+ readonly _malloc = (size: number) => {
+ const pointer = this.#nextPointer;
+ this.#nextPointer += Math.max(1, size) + 16;
+ this.allocated.push(pointer);
+ return pointer;
+ };
+
+ readonly _free = (pointer: number) => {
+ this.freed.push(pointer);
+ };
+
+ readonly _regex_pcre2_bridge_abi_version = () => PCRE2_BRIDGE_ABI_VERSION;
+ readonly _regex_pcre2_config_flags = () => 1;
+ readonly _regex_pcre2_version = () => this.#versionPointer;
+ readonly _regex_pcre2_self_test = () => 0;
+
+ readonly _regex_pcre2_error_message = (
+ errorCode: number,
+ outputPointer: number,
+ outputCapacity: number,
+ ) => {
+ const message = encoder.encode(
+ errorCode === 101 ? "missing closing parenthesis" : "fixture error",
+ );
+ const length = Math.min(message.length, outputCapacity - 1);
+ this.HEAPU8.set(message.slice(0, length), outputPointer);
+ this.HEAPU8[outputPointer + length] = 0;
+ return length;
+ };
+
+ #writeSuccessfulResult(
+ recordsPointer: number,
+ namesPointer: number,
+ nameBytesPointer: number,
+ resultPointer: number,
+ ): void {
+ const view = new DataView(this.HEAPU8.buffer);
+ for (const [index, group] of [0, 1].entries()) {
+ const pointer = recordsPointer + index * 16;
+ view.setUint32(pointer, 1, true);
+ view.setUint32(pointer + 4, group, true);
+ view.setUint32(pointer + 8, 1, true);
+ view.setUint32(pointer + 12, 5, true);
+ }
+ const name = encoder.encode("emoji");
+ view.setUint32(namesPointer, 1, true);
+ view.setUint32(namesPointer + 4, 0, true);
+ view.setUint32(namesPointer + 8, name.length, true);
+ this.HEAPU8.set(name, nameBytesPointer);
+ view.setInt32(resultPointer, 0, true);
+ view.setUint32(resultPointer + 12, 1, true);
+ view.setUint32(resultPointer + 16, 1, true);
+ view.setUint32(resultPointer + 20, 1, true);
+ view.setUint32(resultPointer + 24, 2, true);
+ view.setUint32(resultPointer + 28, 1, true);
+ view.setUint32(resultPointer + 32, name.length, true);
+ }
+
+ readonly _regex_pcre2_execute = (
+ patternPointer: number,
+ patternLength: number,
+ _subjectPointer: number,
+ _subjectLength: number,
+ _applicationFlags: number,
+ _limitsPointer: number,
+ recordsPointer: number,
+ _recordCapacity: number,
+ namesPointer: number,
+ _nameCapacity: number,
+ nameBytesPointer: number,
+ _nameBytesCapacity: number,
+ resultPointer: number,
+ ) => {
+ this.executeCalls += 1;
+ const pattern = decoder.decode(
+ this.HEAPU8.slice(patternPointer, patternPointer + patternLength),
+ );
+ const view = new DataView(this.HEAPU8.buffer);
+ if (pattern.endsWith("(")) {
+ view.setInt32(resultPointer, 101, true);
+ view.setUint32(resultPointer + 4, 2, true);
+ view.setUint32(resultPointer + 8, encoder.encode(pattern).length, true);
+ return 101;
+ }
+ this.#writeSuccessfulResult(
+ recordsPointer,
+ namesPointer,
+ nameBytesPointer,
+ resultPointer,
+ );
+ return 0;
+ };
+
+ readonly _regex_pcre2_substitute = (
+ _patternPointer: number,
+ _patternLength: number,
+ _subjectPointer: number,
+ _subjectLength: number,
+ _replacementPointer: number,
+ _replacementLength: number,
+ _applicationFlags: number,
+ _limitsPointer: number,
+ recordsPointer: number,
+ _recordCapacity: number,
+ namesPointer: number,
+ _nameCapacity: number,
+ nameBytesPointer: number,
+ _nameBytesCapacity: number,
+ outputPointer: number,
+ _outputCapacity: number,
+ resultPointer: number,
+ ) => {
+ this.substituteCalls += 1;
+ this.#writeSuccessfulResult(
+ recordsPointer,
+ namesPointer,
+ nameBytesPointer,
+ resultPointer,
+ );
+ const output = encoder.encode("x[😀]é");
+ this.HEAPU8.set(output, outputPointer);
+ const view = new DataView(this.HEAPU8.buffer);
+ view.setUint32(resultPointer + 48, output.length, true);
+ view.setUint32(resultPointer + 52, 1, true);
+ return 0;
+ };
+
+ readonly _regex_pcre2_trace = () => 0;
+}
+
+function request(
+ overrides: Partial = {},
+): RegexExecutionRequest {
+ return {
+ flavour: "pcre2",
+ flavourVersion: "PCRE2 10.47 8-bit WebAssembly",
+ pattern: "(?😀)",
+ flags: ["g"],
+ options: {
+ matchLimit: 1_000_000,
+ depthLimit: 1_000,
+ heapLimitKib: 32_768,
+ },
+ subject: "x😀é",
+ captureMetadata: [],
+ scanAll: false,
+ maximumMatches: 100,
+ maximumCaptureRows: 1_000,
+ ...overrides,
+ };
+}
+
+describe("PCRE2 WebAssembly adapter", () => {
+ it("normalizes native UTF-8 byte ranges to editor UTF-16 ranges", async () => {
+ const module = new FakePcre2Module();
+ const adapter = new Pcre2EngineAdapter(module, "fixture WebAssembly");
+ await expect(adapter.load()).resolves.toEqual(
+ expect.objectContaining({
+ flavour: "pcre2",
+ engineVersion: PCRE2_ENGINE_VERSION,
+ offsetUnit: "utf8-byte",
+ }),
+ );
+
+ const result = await adapter.execute(request());
+
+ expect(result.accepted).toBe(true);
+ expect(result.matches[0]).toEqual(
+ expect.objectContaining({
+ value: "😀",
+ range: { startUtf16: 1, endUtf16: 3 },
+ nativeRange: { start: 1, end: 5, unit: "utf8-byte" },
+ }),
+ );
+ expect(result.matches[0]?.captures[0]).toEqual(
+ expect.objectContaining({
+ groupNumber: 1,
+ groupName: "emoji",
+ value: "😀",
+ range: { startUtf16: 1, endUtf16: 3 },
+ }),
+ );
+ expect(module.executeCalls).toBe(1);
+ expect(new Set(module.freed)).toEqual(new Set(module.allocated));
+ });
+
+ it("returns bounded native PCRE2 substitution output", async () => {
+ const module = new FakePcre2Module();
+ const adapter = new Pcre2EngineAdapter(module);
+ const replacementRequest: RegexReplacementRequest = {
+ ...request(),
+ replacement: "[$]",
+ maximumOutputBytes: 1_024,
+ };
+
+ const result = await adapter.replace(replacementRequest);
+
+ expect(result.output).toBe("x[😀]é");
+ expect(result.outputBytes).toBe(9);
+ expect(result.execution.matches[0]?.value).toBe("😀");
+ expect(module.substituteCalls).toBe(1);
+ expect(new Set(module.freed)).toEqual(new Set(module.allocated));
+ });
+
+ it("normalizes compile-error byte offsets and rejects lone surrogates", async () => {
+ const module = new FakePcre2Module();
+ const adapter = new Pcre2EngineAdapter(module);
+ const compileError = await adapter.execute(request({ pattern: "é(" }));
+ expect(compileError.accepted).toBe(false);
+ expect(compileError.diagnostics[0]).toEqual(
+ expect.objectContaining({
+ code: "compile-error",
+ range: { startUtf16: 2, endUtf16: 2 },
+ }),
+ );
+
+ const invalidUnicode = await adapter.execute(
+ request({ subject: `x${String.fromCharCode(0xd800)}y` }),
+ );
+ expect(invalidUnicode.accepted).toBe(false);
+ expect(invalidUnicode.diagnostics[0]?.code).toBe("invalid-unicode-input");
+ expect(module.executeCalls).toBe(1);
+ });
+
+ it("validates typed native resource limits before entering WebAssembly", () => {
+ expect(
+ pcre2Limits({
+ matchLimit: 2_000,
+ depthLimit: 20,
+ heapLimitKib: 4_096,
+ }),
+ ).toEqual({
+ matchLimit: 2_000,
+ depthLimit: 20,
+ heapLimitKib: 4_096,
+ });
+ expect(() => pcre2Limits({ matchLimit: 100_000_001 })).toThrow(
+ /matchLimit/u,
+ );
+ });
+});
diff --git a/src/regex/execution/adapters/pcre2/Pcre2EngineAdapter.ts b/src/regex/execution/adapters/pcre2/Pcre2EngineAdapter.ts
new file mode 100644
index 0000000..ae3baac
--- /dev/null
+++ b/src/regex/execution/adapters/pcre2/Pcre2EngineAdapter.ts
@@ -0,0 +1,823 @@
+import type { RegexEngineAdapter } from "../../RegexEngineAdapter";
+import {
+ buildUtf8OffsetMap,
+ utf8ByteToUtf16,
+ type Utf8OffsetMap,
+} from "../../offsets/utf8-to-utf16";
+import { DEFAULT_REGEX_LIMITS } from "../../request-limits";
+import type { RegexDiagnostic } from "../../../model/diagnostics";
+import type {
+ RegexEngineInfo,
+ RegexEngineOptions,
+} from "../../../model/flavour";
+import type {
+ CaptureResult,
+ RegexExecutionFlags,
+ RegexExecutionRequest,
+ RegexExecutionResult,
+ RegexMatchResult,
+ RegexReplacementRequest,
+ RegexReplacementResult,
+} from "../../../model/match";
+import type { Pcre2EmscriptenModule } from "./pcre2-module";
+
+export const PCRE2_ADAPTER_VERSION = "0.3.0";
+export const PCRE2_ENGINE_VERSION = "10.47 2025-10-21";
+export const PCRE2_BRIDGE_ABI_VERSION = 3;
+
+const TEXT_ENCODER = new TextEncoder();
+const TEXT_DECODER = new TextDecoder("utf-8", { fatal: true });
+const RESULT_BYTES = 60;
+const LIMIT_BYTES = 20;
+const MATCH_RECORD_BYTES = 16;
+const NAME_RECORD_BYTES = 12;
+const NAME_BYTES_CAPACITY = 64 * 1024;
+const ERROR_MESSAGE_BYTES = 256;
+const UNSET_OFFSET = 0xffff_ffff;
+const CAPTURE_VALUE_PREVIEW_UTF16 = 64 * 1024;
+const TOTAL_VALUE_PREVIEW_UTF16 = 8 * 1024 * 1024;
+
+const FLAG_BITS = {
+ i: 1 << 0,
+ m: 1 << 1,
+ s: 1 << 2,
+ x: 1 << 3,
+ U: 1 << 4,
+ J: 1 << 7,
+ g: 1 << 8,
+} as const;
+const FLAG_ORDER = "gimsxUJ";
+const UTF_AND_UCP_BITS = (1 << 5) | (1 << 6);
+
+interface Pcre2Limits {
+ readonly matchLimit: number;
+ readonly depthLimit: number;
+ readonly heapLimitKib: number;
+}
+
+interface NativeRunResult {
+ readonly status: number;
+ readonly errorPhase: number;
+ readonly errorOffset: number;
+ readonly captureCount: number;
+ readonly nameCount: number;
+ readonly matchCount: number;
+ readonly recordCount: number;
+ readonly nameRecordCount: number;
+ readonly nameBytesLength: number;
+ readonly effectiveOptions: number;
+ readonly resultsTruncated: boolean;
+ readonly namesTruncated: boolean;
+ readonly outputLength: number;
+ readonly substitutionCount: number;
+ readonly outputTruncated: boolean;
+}
+
+interface NativeBuffers {
+ readonly recordsPointer: number;
+ readonly namesPointer: number;
+ readonly nameBytesPointer: number;
+ readonly outputPointer: number;
+ readonly resultPointer: number;
+ readonly recordCapacity: number;
+}
+
+class WasmAllocations {
+ readonly #module: Pcre2EmscriptenModule;
+ readonly #pointers: number[] = [];
+
+ constructor(module: Pcre2EmscriptenModule) {
+ this.#module = module;
+ }
+
+ allocate(bytes: number): number {
+ const pointer = this.#module._malloc(Math.max(1, bytes));
+ if (!pointer) {
+ throw new Error(
+ `PCRE2 WebAssembly could not allocate ${bytes.toLocaleString()} bytes.`,
+ );
+ }
+ this.#pointers.push(pointer);
+ return pointer;
+ }
+
+ encoded(value: Uint8Array): number {
+ const pointer = this.allocate(value.byteLength);
+ if (value.byteLength > 0) {
+ this.#module.HEAPU8.set(value, pointer);
+ }
+ return pointer;
+ }
+
+ free(): void {
+ for (let index = this.#pointers.length - 1; index >= 0; index -= 1) {
+ this.#module._free(this.#pointers[index] ?? 0);
+ }
+ this.#pointers.length = 0;
+ }
+}
+
+class ValuePreviewBudget {
+ #remaining = TOTAL_VALUE_PREVIEW_UTF16;
+ truncated = false;
+
+ take(value: string): { readonly value: string; readonly truncated: boolean } {
+ const maximum = Math.min(CAPTURE_VALUE_PREVIEW_UTF16, this.#remaining);
+ const preview = value.slice(0, maximum);
+ this.#remaining -= preview.length;
+ const truncated = preview.length !== value.length;
+ this.truncated ||= truncated;
+ return { value: preview, truncated };
+ }
+}
+
+export function pcre2EngineInfo(runtimeVersion?: string): RegexEngineInfo {
+ return {
+ flavour: "pcre2",
+ adapterVersion: PCRE2_ADAPTER_VERSION,
+ engineName: "PCRE2 WebAssembly",
+ engineVersion: PCRE2_ENGINE_VERSION,
+ ...(runtimeVersion ? { runtimeVersion } : {}),
+ offsetUnit: "utf8-byte",
+ capabilities: {
+ compilation: true,
+ matching: true,
+ replacement: true,
+ namedCaptures: true,
+ captureHistory: false,
+ actualTrace: true,
+ benchmark: false,
+ },
+ };
+}
+
+function integerOption(
+ options: RegexEngineOptions | undefined,
+ name: string,
+ fallback: number,
+ minimum: number,
+ maximum: number,
+): number {
+ const value = options?.[name] ?? fallback;
+ if (
+ typeof value !== "number" ||
+ !Number.isSafeInteger(value) ||
+ value < minimum ||
+ value > maximum
+ ) {
+ throw new RangeError(
+ `PCRE2 option ${name} must be an integer from ${minimum.toLocaleString()} to ${maximum.toLocaleString()}.`,
+ );
+ }
+ return value;
+}
+
+export function pcre2Limits(options?: RegexEngineOptions): Pcre2Limits {
+ return {
+ matchLimit: integerOption(options, "matchLimit", 1_000_000, 1, 100_000_000),
+ depthLimit: integerOption(options, "depthLimit", 1_000, 1, 100_000),
+ heapLimitKib: integerOption(options, "heapLimitKib", 32_768, 1, 131_072),
+ };
+}
+
+export function normalizedPcre2Flags(
+ userFlags: readonly string[],
+ scanAll: boolean,
+): RegexExecutionFlags {
+ const selected = new Set(userFlags);
+ const internallyAddedGlobalFlag = scanAll && !selected.has("g");
+ if (internallyAddedGlobalFlag) selected.add("g");
+ return {
+ userFlags: [...FLAG_ORDER]
+ .filter((flag) => userFlags.includes(flag))
+ .join(""),
+ effectiveFlags: [...FLAG_ORDER]
+ .filter((flag) => selected.has(flag))
+ .join(""),
+ internallyAddedIndicesFlag: false,
+ internallyAddedGlobalFlag,
+ };
+}
+
+export function pcre2ApplicationFlags(flags: RegexExecutionFlags): number {
+ let result = UTF_AND_UCP_BITS;
+ for (const flag of flags.effectiveFlags) {
+ result |= FLAG_BITS[flag as keyof typeof FLAG_BITS] ?? 0;
+ }
+ return result;
+}
+
+function validateRequestBounds(
+ request: RegexExecutionRequest,
+ patternBytes: number,
+ subjectBytes: number,
+ replacement?: {
+ readonly bytes: number;
+ readonly maximumOutputBytes: number;
+ },
+): void {
+ if (
+ new Set(request.flags).size !== request.flags.length ||
+ request.flags.some((flag) => !FLAG_ORDER.includes(flag))
+ ) {
+ throw new RangeError("The request contains unsupported PCRE2 flags.");
+ }
+ if (patternBytes > 1_048_576) {
+ throw new RangeError("PCRE2 patterns are limited to 1 MiB of UTF-8.");
+ }
+ if (subjectBytes > DEFAULT_REGEX_LIMITS.interactiveSubjectHardBytes) {
+ throw new RangeError("PCRE2 subjects are limited to 16 MiB of UTF-8.");
+ }
+ if (
+ !Number.isSafeInteger(request.maximumMatches) ||
+ request.maximumMatches < 1 ||
+ request.maximumMatches > DEFAULT_REGEX_LIMITS.maximumMatches ||
+ !Number.isSafeInteger(request.maximumCaptureRows) ||
+ request.maximumCaptureRows < 1 ||
+ request.maximumCaptureRows > DEFAULT_REGEX_LIMITS.maximumCaptureRows
+ ) {
+ throw new RangeError(
+ "PCRE2 result limits are outside the bounded contract.",
+ );
+ }
+ if (
+ replacement &&
+ (replacement.bytes > 262_144 ||
+ !Number.isSafeInteger(replacement.maximumOutputBytes) ||
+ replacement.maximumOutputBytes < 1 ||
+ replacement.maximumOutputBytes >
+ DEFAULT_REGEX_LIMITS.maximumReplacementOutputBytes)
+ ) {
+ throw new RangeError(
+ "PCRE2 replacement or output limits are outside the bounded contract.",
+ );
+ }
+}
+
+function validateNativeBounds(
+ native: NativeRunResult,
+ buffers: NativeBuffers,
+ maximumOutputBytes?: number,
+): void {
+ if (
+ native.captureCount > DEFAULT_REGEX_LIMITS.maximumCaptureGroups ||
+ native.matchCount > DEFAULT_REGEX_LIMITS.maximumMatches ||
+ native.recordCount > buffers.recordCapacity ||
+ native.nameRecordCount > DEFAULT_REGEX_LIMITS.maximumCaptureGroups ||
+ native.nameBytesLength > NAME_BYTES_CAPACITY ||
+ (maximumOutputBytes !== undefined &&
+ native.outputLength > maximumOutputBytes)
+ ) {
+ throw new Error("PCRE2 returned data outside the bounded bridge contract.");
+ }
+}
+
+function bridgeMessage(status: number): string | undefined {
+ switch (status) {
+ case -10_001:
+ return "The PCRE2 bridge received an invalid argument.";
+ case -10_002:
+ return "The request contains unsupported PCRE2 flags.";
+ case -10_003:
+ return "The pattern exceeds the PCRE2 bridge byte limit.";
+ case -10_004:
+ return "The PCRE2 runtime configuration is inconsistent.";
+ case -10_005:
+ return "The subject exceeds the PCRE2 bridge byte limit.";
+ case -10_006:
+ return "The replacement exceeds the PCRE2 bridge byte limit.";
+ case -10_007:
+ return "A PCRE2 resource limit is outside its supported range.";
+ case -10_008:
+ return "The pattern exceeds the 1,000 capture-group execution limit.";
+ case -10_009:
+ return "The PCRE2 output buffer is invalid.";
+ default:
+ return undefined;
+ }
+}
+
+function containsLoneSurrogate(
+ value: string,
+ label: string,
+): RegexDiagnostic | undefined {
+ const map = buildUtf8OffsetMap(value);
+ if (!map.containsLoneSurrogate) return undefined;
+ return {
+ id: `pcre2-${label}-lone-surrogate`,
+ source: "execution-engine",
+ severity: "error",
+ code: "invalid-unicode-input",
+ message: `${label} contains an unpaired UTF-16 surrogate. PCRE2 UTF-8 execution refuses the lossy browser encoding that would otherwise replace it.`,
+ flavour: "pcre2",
+ provenance: "derived",
+ };
+}
+
+function runResult(
+ module: Pcre2EmscriptenModule,
+ pointer: number,
+): NativeRunResult {
+ const view = new DataView(module.HEAPU8.buffer, pointer, RESULT_BYTES);
+ return {
+ status: view.getInt32(0, true),
+ errorPhase: view.getUint32(4, true),
+ errorOffset: view.getUint32(8, true),
+ captureCount: view.getUint32(12, true),
+ nameCount: view.getUint32(16, true),
+ matchCount: view.getUint32(20, true),
+ recordCount: view.getUint32(24, true),
+ nameRecordCount: view.getUint32(28, true),
+ nameBytesLength: view.getUint32(32, true),
+ effectiveOptions: view.getUint32(36, true),
+ resultsTruncated: view.getUint32(40, true) !== 0,
+ namesTruncated: view.getUint32(44, true) !== 0,
+ outputLength: view.getUint32(48, true),
+ substitutionCount: view.getUint32(52, true),
+ outputTruncated: view.getUint32(56, true) !== 0,
+ };
+}
+
+function nativeErrorMessage(
+ module: Pcre2EmscriptenModule,
+ status: number,
+): string {
+ const bridge = bridgeMessage(status);
+ if (bridge) return bridge;
+ const pointer = module._malloc(ERROR_MESSAGE_BYTES);
+ if (!pointer) return `PCRE2 error ${status}`;
+ try {
+ module.HEAPU8.fill(0, pointer, pointer + ERROR_MESSAGE_BYTES);
+ const length = module._regex_pcre2_error_message(
+ status,
+ pointer,
+ ERROR_MESSAGE_BYTES,
+ );
+ return length >= 0
+ ? TEXT_DECODER.decode(module.HEAPU8.slice(pointer, pointer + length))
+ : `PCRE2 error ${status}`;
+ } finally {
+ module._free(pointer);
+ }
+}
+
+function diagnosticForFailure(
+ module: Pcre2EmscriptenModule,
+ native: NativeRunResult,
+ patternMap: Utf8OffsetMap,
+): RegexDiagnostic {
+ const phase =
+ native.errorPhase === 2
+ ? "compile"
+ : native.errorPhase === 3
+ ? "match"
+ : native.errorPhase === 4
+ ? "replacement"
+ : "bridge";
+ const nativeMessage = nativeErrorMessage(module, native.status);
+ let range:
+ { readonly startUtf16: number; readonly endUtf16: number } | undefined;
+ if (native.errorPhase === 2) {
+ try {
+ const offset = utf8ByteToUtf16(patternMap, native.errorOffset);
+ range = { startUtf16: offset, endUtf16: offset };
+ } catch {
+ range = undefined;
+ }
+ }
+ return {
+ id: `pcre2-${phase}-error`,
+ source: "execution-engine",
+ severity: "error",
+ code:
+ native.errorPhase === 2
+ ? "compile-error"
+ : native.errorPhase === 4
+ ? "replacement-error"
+ : native.status === -47
+ ? "match-limit"
+ : "execution-error",
+ message: `PCRE2 ${phase} error: ${nativeMessage}`,
+ ...(range ? { range } : {}),
+ flavour: "pcre2",
+ provenance: "reported",
+ };
+}
+
+function writeLimits(
+ module: Pcre2EmscriptenModule,
+ pointer: number,
+ request: RegexExecutionRequest,
+ limits: Pcre2Limits,
+): void {
+ const view = new DataView(module.HEAPU8.buffer, pointer, LIMIT_BYTES);
+ view.setUint32(0, request.maximumMatches, true);
+ view.setUint32(4, request.maximumCaptureRows, true);
+ view.setUint32(8, limits.matchLimit, true);
+ view.setUint32(12, limits.depthLimit, true);
+ view.setUint32(16, limits.heapLimitKib, true);
+}
+
+function namesByGroup(
+ module: Pcre2EmscriptenModule,
+ buffers: NativeBuffers,
+ native: NativeRunResult,
+ request: RegexExecutionRequest,
+): ReadonlyMap {
+ const names = new Map(
+ request.captureMetadata.flatMap((capture) =>
+ capture.name ? [[capture.number, capture.name] as const] : [],
+ ),
+ );
+ const view = new DataView(module.HEAPU8.buffer);
+ for (let index = 0; index < native.nameRecordCount; index += 1) {
+ const pointer = buffers.namesPointer + index * NAME_RECORD_BYTES;
+ const groupNumber = view.getUint32(pointer, true);
+ const offset = view.getUint32(pointer + 4, true);
+ const length = view.getUint32(pointer + 8, true);
+ if (offset + length > native.nameBytesLength) {
+ throw new Error("PCRE2 returned an invalid capture-name range.");
+ }
+ names.set(
+ groupNumber,
+ TEXT_DECODER.decode(
+ module.HEAPU8.slice(
+ buffers.nameBytesPointer + offset,
+ buffers.nameBytesPointer + offset + length,
+ ),
+ ),
+ );
+ }
+ return names;
+}
+
+function materializeMatches(
+ module: Pcre2EmscriptenModule,
+ buffers: NativeBuffers,
+ native: NativeRunResult,
+ request: RegexExecutionRequest,
+ subjectMap: Utf8OffsetMap,
+): {
+ readonly matches: readonly RegexMatchResult[];
+ readonly valuesTruncated: boolean;
+} {
+ const names = namesByGroup(module, buffers, native, request);
+ const view = new DataView(module.HEAPU8.buffer);
+ const byMatch = new Map();
+ const captures = new Map();
+ const previews = new ValuePreviewBudget();
+ for (let index = 0; index < native.recordCount; index += 1) {
+ const pointer = buffers.recordsPointer + index * MATCH_RECORD_BYTES;
+ const matchNumber = view.getUint32(pointer, true);
+ const groupNumber = view.getUint32(pointer + 4, true);
+ const startByte = view.getUint32(pointer + 8, true);
+ const endByte = view.getUint32(pointer + 12, true);
+ if (groupNumber > native.captureCount || matchNumber > native.matchCount) {
+ throw new Error("PCRE2 returned an invalid match record.");
+ }
+ if (startByte === UNSET_OFFSET && endByte === UNSET_OFFSET) {
+ if (groupNumber === 0) {
+ throw new Error("PCRE2 returned an unset complete-match range.");
+ }
+ const rows = captures.get(matchNumber) ?? [];
+ rows.push({
+ groupNumber,
+ ...(names.get(groupNumber)
+ ? { groupName: names.get(groupNumber) }
+ : {}),
+ status: "did-not-participate",
+ });
+ captures.set(matchNumber, rows);
+ continue;
+ }
+ const startUtf16 = utf8ByteToUtf16(subjectMap, startByte);
+ const endUtf16 = utf8ByteToUtf16(subjectMap, endByte);
+ const value = request.subject.slice(startUtf16, endUtf16);
+ const preview = previews.take(value);
+ const range = { startUtf16, endUtf16 };
+ const nativeRange = {
+ start: startByte,
+ end: endByte,
+ unit: "utf8-byte" as const,
+ };
+ if (groupNumber === 0) {
+ byMatch.set(matchNumber, {
+ matchNumber,
+ value: preview.value,
+ valueStatus: preview.truncated ? "truncated" : "complete",
+ range,
+ nativeRange,
+ captures: [],
+ });
+ } else {
+ const rows = captures.get(matchNumber) ?? [];
+ rows.push({
+ groupNumber,
+ ...(names.get(groupNumber)
+ ? { groupName: names.get(groupNumber) }
+ : {}),
+ value: preview.value,
+ status: preview.truncated
+ ? "truncated"
+ : value.length === 0
+ ? "matched-empty"
+ : "participated",
+ range,
+ nativeRange,
+ });
+ captures.set(matchNumber, rows);
+ }
+ }
+ const matches = [...byMatch.values()]
+ .sort((left, right) => left.matchNumber - right.matchNumber)
+ .map((match) => ({
+ ...match,
+ captures: (captures.get(match.matchNumber) ?? []).sort(
+ (left, right) => left.groupNumber - right.groupNumber,
+ ),
+ }));
+ if (matches.length !== native.matchCount) {
+ throw new Error("PCRE2 returned an incomplete complete-match record set.");
+ }
+ return { matches, valuesTruncated: previews.truncated };
+}
+
+function emptyExecution(
+ info: RegexEngineInfo,
+ flags: RegexExecutionFlags,
+ start: number,
+ diagnostic: RegexDiagnostic,
+): RegexExecutionResult {
+ return {
+ accepted: false,
+ engine: info,
+ flags,
+ matches: [],
+ diagnostics: [diagnostic],
+ elapsedMs: performance.now() - start,
+ truncated: false,
+ };
+}
+
+export class Pcre2EngineAdapter implements RegexEngineAdapter {
+ readonly flavour = "pcre2" as const;
+ readonly #module: Pcre2EmscriptenModule;
+ readonly #info: RegexEngineInfo;
+
+ constructor(module: Pcre2EmscriptenModule, runtimeVersion?: string) {
+ this.#module = module;
+ this.#info = pcre2EngineInfo(runtimeVersion);
+ }
+
+ async load(): Promise {
+ if (
+ this.#module._regex_pcre2_bridge_abi_version() !==
+ PCRE2_BRIDGE_ABI_VERSION ||
+ this.#module._regex_pcre2_config_flags() !== 1 ||
+ this.#module.UTF8ToString(this.#module._regex_pcre2_version()) !==
+ PCRE2_ENGINE_VERSION ||
+ this.#module._regex_pcre2_self_test() !== 0
+ ) {
+ throw new Error(
+ "The bundled PCRE2 engine failed its identity or bridge self-test.",
+ );
+ }
+ return this.#info;
+ }
+
+ async execute(request: RegexExecutionRequest): Promise {
+ const result = this.#run(request);
+ if (!("execution" in result)) return result;
+ throw new Error("PCRE2 execution returned a replacement result.");
+ }
+
+ async replace(
+ request: RegexReplacementRequest,
+ ): Promise {
+ const result = this.#run(request, request);
+ if ("execution" in result) return result;
+ throw new Error("PCRE2 replacement returned an execution-only result.");
+ }
+
+ #run(
+ request: RegexExecutionRequest,
+ replacementRequest?: RegexReplacementRequest,
+ ): RegexExecutionResult | RegexReplacementResult {
+ if (request.flavour !== "pcre2") {
+ throw new Error(`PCRE2 adapter cannot execute ${request.flavour}.`);
+ }
+ const start = performance.now();
+ const flags = normalizedPcre2Flags(request.flags, request.scanAll);
+ const patternMap = buildUtf8OffsetMap(request.pattern);
+ const subjectMap = buildUtf8OffsetMap(request.subject);
+ const invalidUnicode =
+ containsLoneSurrogate(request.pattern, "Pattern") ??
+ containsLoneSurrogate(request.subject, "Subject") ??
+ (replacementRequest
+ ? containsLoneSurrogate(replacementRequest.replacement, "Replacement")
+ : undefined);
+ if (invalidUnicode) {
+ return emptyExecution(this.#info, flags, start, invalidUnicode);
+ }
+ const pattern = TEXT_ENCODER.encode(request.pattern);
+ const subject = TEXT_ENCODER.encode(request.subject);
+ const replacement = replacementRequest
+ ? TEXT_ENCODER.encode(replacementRequest.replacement)
+ : undefined;
+ validateRequestBounds(
+ request,
+ pattern.byteLength,
+ subject.byteLength,
+ replacementRequest && replacement
+ ? {
+ bytes: replacement.byteLength,
+ maximumOutputBytes: replacementRequest.maximumOutputBytes,
+ }
+ : undefined,
+ );
+ if (
+ replacementRequest &&
+ replacementRequest.replacement.length >
+ DEFAULT_REGEX_LIMITS.maximumReplacementTemplateUtf16
+ ) {
+ throw new RangeError(
+ `Replacement template exceeds the ${DEFAULT_REGEX_LIMITS.maximumReplacementTemplateUtf16.toLocaleString()} UTF-16 unit limit.`,
+ );
+ }
+ const limits = pcre2Limits(request.options);
+ const allocations = new WasmAllocations(this.#module);
+ try {
+ const patternPointer = allocations.encoded(pattern);
+ const subjectPointer = allocations.encoded(subject);
+ const replacementPointer = replacement
+ ? allocations.encoded(replacement)
+ : 0;
+ const limitsPointer = allocations.allocate(LIMIT_BYTES);
+ const recordCapacity =
+ request.maximumMatches + request.maximumCaptureRows;
+ const recordsPointer = allocations.allocate(
+ recordCapacity * MATCH_RECORD_BYTES,
+ );
+ const nameCapacity = DEFAULT_REGEX_LIMITS.maximumCaptureGroups;
+ const namesPointer = allocations.allocate(
+ nameCapacity * NAME_RECORD_BYTES,
+ );
+ const nameBytesPointer = allocations.allocate(NAME_BYTES_CAPACITY);
+ const outputPointer = replacementRequest
+ ? allocations.allocate(replacementRequest.maximumOutputBytes + 1)
+ : 0;
+ const resultPointer = allocations.allocate(RESULT_BYTES);
+ this.#module.HEAPU8.fill(0, resultPointer, resultPointer + RESULT_BYTES);
+ writeLimits(this.#module, limitsPointer, request, limits);
+ const buffers: NativeBuffers = {
+ recordsPointer,
+ namesPointer,
+ nameBytesPointer,
+ outputPointer,
+ resultPointer,
+ recordCapacity,
+ };
+ if (replacementRequest && replacement) {
+ this.#module._regex_pcre2_substitute(
+ patternPointer,
+ pattern.byteLength,
+ subjectPointer,
+ subject.byteLength,
+ replacementPointer,
+ replacement.byteLength,
+ pcre2ApplicationFlags(flags),
+ limitsPointer,
+ recordsPointer,
+ recordCapacity,
+ namesPointer,
+ nameCapacity,
+ nameBytesPointer,
+ NAME_BYTES_CAPACITY,
+ outputPointer,
+ replacementRequest.maximumOutputBytes,
+ resultPointer,
+ );
+ } else {
+ this.#module._regex_pcre2_execute(
+ patternPointer,
+ pattern.byteLength,
+ subjectPointer,
+ subject.byteLength,
+ pcre2ApplicationFlags(flags),
+ limitsPointer,
+ recordsPointer,
+ recordCapacity,
+ namesPointer,
+ nameCapacity,
+ nameBytesPointer,
+ NAME_BYTES_CAPACITY,
+ resultPointer,
+ );
+ }
+ const native = runResult(this.#module, resultPointer);
+ validateNativeBounds(
+ native,
+ buffers,
+ replacementRequest?.maximumOutputBytes,
+ );
+ if (native.status !== 0) {
+ return emptyExecution(
+ this.#info,
+ flags,
+ start,
+ diagnosticForFailure(this.#module, native, patternMap),
+ );
+ }
+ const materialized = materializeMatches(
+ this.#module,
+ buffers,
+ native,
+ request,
+ subjectMap,
+ );
+ const diagnostics: RegexDiagnostic[] = [];
+ if (native.resultsTruncated) {
+ diagnostics.push({
+ id: "pcre2-results-truncated",
+ source: "execution-engine",
+ severity: "warning",
+ code: "result-limit",
+ message: `Results reached the configured limit (${request.maximumMatches.toLocaleString()} matches or ${request.maximumCaptureRows.toLocaleString()} capture rows).`,
+ flavour: "pcre2",
+ provenance: "derived",
+ });
+ }
+ if (native.namesTruncated) {
+ diagnostics.push({
+ id: "pcre2-capture-names-truncated",
+ source: "execution-engine",
+ severity: "warning",
+ code: "capture-name-limit",
+ message:
+ "Some PCRE2 capture names exceeded the bounded metadata buffer.",
+ flavour: "pcre2",
+ provenance: "derived",
+ });
+ }
+ if (materialized.valuesTruncated) {
+ diagnostics.push({
+ id: "pcre2-value-previews-truncated",
+ source: "execution-engine",
+ severity: "warning",
+ code: "result-value-preview-limit",
+ message:
+ "Displayed PCRE2 result values were clipped to bounded previews; exact byte and UTF-16 ranges are retained.",
+ flavour: "pcre2",
+ provenance: "derived",
+ });
+ }
+ const execution: RegexExecutionResult = {
+ accepted: true,
+ engine: this.#info,
+ flags,
+ matches: materialized.matches,
+ diagnostics,
+ elapsedMs: performance.now() - start,
+ truncated: native.resultsTruncated,
+ };
+ if (!replacementRequest) return execution;
+ const output = TEXT_DECODER.decode(
+ this.#module.HEAPU8.slice(
+ outputPointer,
+ outputPointer + native.outputLength,
+ ),
+ );
+ const executionWithOutputDiagnostic: RegexExecutionResult =
+ native.outputTruncated
+ ? {
+ ...execution,
+ diagnostics: [
+ ...execution.diagnostics,
+ {
+ id: "pcre2-output-truncated",
+ source: "execution-engine",
+ severity: "warning",
+ code: "replacement-output-limit",
+ message: `Replacement preview was truncated at ${replacementRequest.maximumOutputBytes.toLocaleString()} UTF-8 bytes.`,
+ flavour: "pcre2",
+ provenance: "derived",
+ },
+ ],
+ }
+ : execution;
+ return {
+ execution: executionWithOutputDiagnostic,
+ output,
+ outputBytes: native.outputLength,
+ outputTruncated: native.outputTruncated,
+ truncated: native.outputTruncated || native.resultsTruncated,
+ };
+ } finally {
+ allocations.free();
+ }
+ }
+
+ terminate(): void {
+ // Lifecycle termination is performed by killing the containing worker.
+ }
+}
diff --git a/src/regex/execution/adapters/pcre2/Pcre2TraceAdapter.ts b/src/regex/execution/adapters/pcre2/Pcre2TraceAdapter.ts
new file mode 100644
index 0000000..2efc26a
--- /dev/null
+++ b/src/regex/execution/adapters/pcre2/Pcre2TraceAdapter.ts
@@ -0,0 +1,517 @@
+import type { RegexDiagnostic } from "../../../model/diagnostics";
+import type {
+ Pcre2TraceEvent,
+ Pcre2TraceRequest,
+ Pcre2TraceResult,
+} from "../../../model/trace";
+import {
+ buildUtf8OffsetMap,
+ utf8ByteToUtf16,
+} from "../../offsets/utf8-to-utf16";
+import { DEFAULT_REGEX_LIMITS } from "../../request-limits";
+import {
+ normalizedPcre2Flags,
+ PCRE2_BRIDGE_ABI_VERSION,
+ PCRE2_ENGINE_VERSION,
+ pcre2ApplicationFlags,
+ pcre2EngineInfo,
+ pcre2Limits,
+} from "./Pcre2EngineAdapter";
+import type { Pcre2EmscriptenModule } from "./pcre2-module";
+
+const TEXT_ENCODER = new TextEncoder();
+const TEXT_DECODER = new TextDecoder("utf-8", { fatal: true });
+const TRACE_LIMIT_BYTES = 20;
+const TRACE_EVENT_BYTES = 32;
+const TRACE_RESULT_BYTES = 52;
+const ERROR_MESSAGE_BYTES = 256;
+const FLAG_ORDER = "gimsxUJ";
+const NO_COMPLETE_EVENT = 0xffff_ffff;
+
+interface NativeTraceResult {
+ readonly status: number;
+ readonly errorPhase: number;
+ readonly errorOffset: number;
+ readonly nativeMatchStatus: number;
+ readonly eventCount: number;
+ readonly totalEventCount: number;
+ readonly markBytesLength: number;
+ readonly traceBytesLength: number;
+ readonly eventsTruncated: boolean;
+ readonly marksTruncated: boolean;
+ readonly lastCompleteEvent: number;
+ readonly matched: boolean;
+}
+
+class TraceAllocations {
+ readonly #module: Pcre2EmscriptenModule;
+ readonly #pointers: number[] = [];
+
+ constructor(module: Pcre2EmscriptenModule) {
+ this.#module = module;
+ }
+
+ allocate(bytes: number): number {
+ const pointer = this.#module._malloc(Math.max(1, bytes));
+ if (!pointer) {
+ throw new Error(
+ `PCRE2 trace could not allocate ${bytes.toLocaleString()} bytes.`,
+ );
+ }
+ this.#pointers.push(pointer);
+ return pointer;
+ }
+
+ encoded(value: Uint8Array): number {
+ const pointer = this.allocate(value.byteLength);
+ if (value.byteLength > 0) this.#module.HEAPU8.set(value, pointer);
+ return pointer;
+ }
+
+ free(): void {
+ for (let index = this.#pointers.length - 1; index >= 0; index -= 1) {
+ this.#module._free(this.#pointers[index] ?? 0);
+ }
+ this.#pointers.length = 0;
+ }
+}
+
+function traceFailure(
+ request: Pcre2TraceRequest,
+ message: string,
+ code: string,
+ start: number,
+ range?: { readonly startUtf16: number; readonly endUtf16: number },
+): Pcre2TraceResult {
+ const flags = normalizedPcre2Flags(request.flags, false);
+ return {
+ accepted: false,
+ engine: pcre2EngineInfo(),
+ flags,
+ events: [],
+ diagnostics: [
+ {
+ id: `pcre2-trace-${code}`,
+ source: "execution-engine",
+ severity: "error",
+ code,
+ message,
+ ...(range ? { range } : {}),
+ flavour: "pcre2",
+ provenance: "reported",
+ },
+ ],
+ elapsedMs: performance.now() - start,
+ truncated: false,
+ marksTruncated: false,
+ totalEventCount: 0,
+ traceBytes: 0,
+ nativeMatchStatus: 0,
+ matched: false,
+ };
+}
+
+function validateRequest(
+ request: Pcre2TraceRequest,
+ patternBytes: number,
+ subjectBytes: number,
+): void {
+ if (request.flavour !== "pcre2") {
+ throw new Error(`PCRE2 trace cannot execute ${request.flavour}.`);
+ }
+ if (
+ new Set(request.flags).size !== request.flags.length ||
+ request.flags.some((flag) => !FLAG_ORDER.includes(flag))
+ ) {
+ throw new RangeError("The trace request contains unsupported PCRE2 flags.");
+ }
+ if (patternBytes > 1_048_576) {
+ throw new RangeError("PCRE2 trace patterns are limited to 1 MiB of UTF-8.");
+ }
+ if (subjectBytes > DEFAULT_REGEX_LIMITS.interactiveSubjectHardBytes) {
+ throw new RangeError(
+ "PCRE2 trace subjects are limited to 16 MiB of UTF-8.",
+ );
+ }
+ if (
+ !Number.isSafeInteger(request.maximumTraceEvents) ||
+ request.maximumTraceEvents < 1 ||
+ request.maximumTraceEvents > DEFAULT_REGEX_LIMITS.maximumTraceEvents
+ ) {
+ throw new RangeError(
+ `PCRE2 traces retain from 1 to ${DEFAULT_REGEX_LIMITS.maximumTraceEvents.toLocaleString()} events.`,
+ );
+ }
+ if (
+ !Number.isSafeInteger(request.maximumTraceBytes) ||
+ request.maximumTraceBytes < TRACE_EVENT_BYTES ||
+ request.maximumTraceBytes > DEFAULT_REGEX_LIMITS.maximumTraceBytes
+ ) {
+ throw new RangeError(
+ `PCRE2 traces retain from ${TRACE_EVENT_BYTES} to ${DEFAULT_REGEX_LIMITS.maximumTraceBytes.toLocaleString()} serialized bytes.`,
+ );
+ }
+}
+
+function readNativeResult(
+ module: Pcre2EmscriptenModule,
+ pointer: number,
+): NativeTraceResult {
+ const view = new DataView(module.HEAPU8.buffer, pointer, TRACE_RESULT_BYTES);
+ return {
+ status: view.getInt32(0, true),
+ errorPhase: view.getUint32(4, true),
+ errorOffset: view.getUint32(8, true),
+ nativeMatchStatus: view.getInt32(12, true),
+ eventCount: view.getUint32(20, true),
+ totalEventCount: view.getUint32(24, true),
+ markBytesLength: view.getUint32(28, true),
+ traceBytesLength: view.getUint32(32, true),
+ eventsTruncated: view.getUint32(36, true) !== 0,
+ marksTruncated: view.getUint32(40, true) !== 0,
+ lastCompleteEvent: view.getUint32(44, true),
+ matched: view.getUint32(48, true) !== 0,
+ };
+}
+
+function movement(
+ current: number,
+ previous: number | undefined,
+): Pcre2TraceEvent["movement"] {
+ if (previous === undefined) {
+ return {
+ provenance: "derived",
+ classification: "first-event",
+ subjectDeltaBytes: 0,
+ explanation: "First retained reported callout event.",
+ };
+ }
+ const delta = current - previous;
+ if (delta > 0) {
+ return {
+ provenance: "derived",
+ classification: "forward",
+ subjectDeltaBytes: delta,
+ explanation:
+ "Derived only: the reported subject position moved forward between adjacent retained callouts.",
+ };
+ }
+ if (delta === 0) {
+ return {
+ provenance: "derived",
+ classification: "same-position",
+ subjectDeltaBytes: 0,
+ explanation:
+ "Derived only: adjacent retained callouts report the same subject position.",
+ };
+ }
+ return {
+ provenance: "derived",
+ classification: "apparent-backtrack",
+ subjectDeltaBytes: delta,
+ explanation:
+ "Derived only: the reported subject position moved backward between adjacent retained callouts; this is not a complete internal engine trace.",
+ };
+}
+
+function materializeEvents(
+ module: Pcre2EmscriptenModule,
+ eventsPointer: number,
+ marksPointer: number,
+ native: NativeTraceResult,
+ pattern: string,
+ subject: string,
+): readonly Pcre2TraceEvent[] {
+ const patternMap = buildUtf8OffsetMap(pattern);
+ const subjectMap = buildUtf8OffsetMap(subject);
+ const view = new DataView(module.HEAPU8.buffer);
+ const events: Pcre2TraceEvent[] = [];
+ let previousSubjectPosition: number | undefined;
+ for (let index = 0; index < native.eventCount; index += 1) {
+ const pointer = eventsPointer + index * TRACE_EVENT_BYTES;
+ const calloutNumber = view.getUint32(pointer, true);
+ const patternPositionByte = view.getUint32(pointer + 4, true);
+ const nextItemLengthByte = view.getUint32(pointer + 8, true);
+ const subjectPositionByte = view.getUint32(pointer + 12, true);
+ const captureTop = view.getUint32(pointer + 16, true);
+ const captureLast = view.getUint32(pointer + 20, true);
+ const markOffset = view.getUint32(pointer + 24, true);
+ const markLength = view.getUint32(pointer + 28, true);
+ const nextEndByte = patternPositionByte + nextItemLengthByte;
+ if (
+ nextEndByte > patternMap.byteLength ||
+ subjectPositionByte > subjectMap.byteLength ||
+ markOffset + markLength > native.markBytesLength ||
+ captureTop > DEFAULT_REGEX_LIMITS.maximumCaptureGroups + 1 ||
+ captureLast > DEFAULT_REGEX_LIMITS.maximumCaptureGroups
+ ) {
+ throw new Error("PCRE2 returned an invalid automatic-callout event.");
+ }
+ const patternPositionUtf16 = utf8ByteToUtf16(
+ patternMap,
+ patternPositionByte,
+ );
+ const nextEndUtf16 = utf8ByteToUtf16(patternMap, nextEndByte);
+ const subjectPositionUtf16 = utf8ByteToUtf16(
+ subjectMap,
+ subjectPositionByte,
+ );
+ const mark =
+ markLength === 0
+ ? undefined
+ : TEXT_DECODER.decode(
+ module.HEAPU8.slice(
+ marksPointer + markOffset,
+ marksPointer + markOffset + markLength,
+ ),
+ );
+ events.push({
+ eventNumber: index + 1,
+ reported: {
+ provenance: "reported",
+ calloutNumber,
+ patternPosition: {
+ utf16: patternPositionUtf16,
+ nativeByte: patternPositionByte,
+ },
+ nextPatternItem: {
+ startUtf16: patternPositionUtf16,
+ endUtf16: nextEndUtf16,
+ startNativeByte: patternPositionByte,
+ endNativeByte: nextEndByte,
+ },
+ subjectPosition: {
+ utf16: subjectPositionUtf16,
+ nativeByte: subjectPositionByte,
+ },
+ captureTop,
+ captureLast,
+ ...(mark ? { mark } : {}),
+ },
+ movement: movement(subjectPositionByte, previousSubjectPosition),
+ });
+ previousSubjectPosition = subjectPositionByte;
+ }
+ return events;
+}
+
+function nativeErrorMessage(
+ module: Pcre2EmscriptenModule,
+ status: number,
+): string {
+ const pointer = module._malloc(ERROR_MESSAGE_BYTES);
+ if (!pointer) return `native error ${status}`;
+ try {
+ module.HEAPU8.fill(0, pointer, pointer + ERROR_MESSAGE_BYTES);
+ const length = module._regex_pcre2_error_message(
+ status,
+ pointer,
+ ERROR_MESSAGE_BYTES,
+ );
+ return length >= 0
+ ? TEXT_DECODER.decode(module.HEAPU8.slice(pointer, pointer + length))
+ : `native error ${status}`;
+ } finally {
+ module._free(pointer);
+ }
+}
+
+export class Pcre2TraceAdapter {
+ readonly #module: Pcre2EmscriptenModule;
+ readonly #runtimeVersion?: string;
+
+ constructor(module: Pcre2EmscriptenModule, runtimeVersion?: string) {
+ this.#module = module;
+ this.#runtimeVersion = runtimeVersion;
+ }
+
+ async load(): Promise {
+ if (
+ this.#module._regex_pcre2_bridge_abi_version() !==
+ PCRE2_BRIDGE_ABI_VERSION ||
+ this.#module._regex_pcre2_config_flags() !== 1 ||
+ this.#module.UTF8ToString(this.#module._regex_pcre2_version()) !==
+ PCRE2_ENGINE_VERSION ||
+ this.#module._regex_pcre2_self_test() !== 0
+ ) {
+ throw new Error(
+ "The bundled PCRE2 trace engine failed its identity or bridge self-test.",
+ );
+ }
+ }
+
+ async trace(request: Pcre2TraceRequest): Promise {
+ const start = performance.now();
+ const patternMap = buildUtf8OffsetMap(request.pattern);
+ const subjectMap = buildUtf8OffsetMap(request.subject);
+ if (patternMap.containsLoneSurrogate || subjectMap.containsLoneSurrogate) {
+ return traceFailure(
+ request,
+ "PCRE2 trace refuses an unpaired UTF-16 surrogate because browser UTF-8 encoding would replace it.",
+ "invalid-unicode-input",
+ start,
+ );
+ }
+ const pattern = TEXT_ENCODER.encode(request.pattern);
+ const subject = TEXT_ENCODER.encode(request.subject);
+ validateRequest(request, pattern.byteLength, subject.byteLength);
+ const limits = pcre2Limits(request.options);
+ const flags = normalizedPcre2Flags(request.flags, false);
+ const allocations = new TraceAllocations(this.#module);
+ try {
+ const patternPointer = allocations.encoded(pattern);
+ const subjectPointer = allocations.encoded(subject);
+ const limitsPointer = allocations.allocate(TRACE_LIMIT_BYTES);
+ const eventsPointer = allocations.allocate(
+ request.maximumTraceEvents * TRACE_EVENT_BYTES,
+ );
+ const marksPointer = allocations.allocate(request.maximumTraceBytes);
+ const resultPointer = allocations.allocate(TRACE_RESULT_BYTES);
+ const limitsView = new DataView(
+ this.#module.HEAPU8.buffer,
+ limitsPointer,
+ TRACE_LIMIT_BYTES,
+ );
+ limitsView.setUint32(0, request.maximumTraceEvents, true);
+ limitsView.setUint32(4, request.maximumTraceBytes, true);
+ limitsView.setUint32(8, limits.matchLimit, true);
+ limitsView.setUint32(12, limits.depthLimit, true);
+ limitsView.setUint32(16, limits.heapLimitKib, true);
+ this.#module.HEAPU8.fill(
+ 0,
+ resultPointer,
+ resultPointer + TRACE_RESULT_BYTES,
+ );
+ this.#module._regex_pcre2_trace(
+ patternPointer,
+ pattern.byteLength,
+ subjectPointer,
+ subject.byteLength,
+ pcre2ApplicationFlags(flags),
+ limitsPointer,
+ eventsPointer,
+ request.maximumTraceEvents,
+ marksPointer,
+ request.maximumTraceBytes,
+ resultPointer,
+ );
+ const native = readNativeResult(this.#module, resultPointer);
+ if (
+ native.eventCount > request.maximumTraceEvents ||
+ native.totalEventCount < native.eventCount ||
+ native.markBytesLength > request.maximumTraceBytes ||
+ native.traceBytesLength !==
+ native.eventCount * TRACE_EVENT_BYTES + native.markBytesLength ||
+ native.traceBytesLength > request.maximumTraceBytes ||
+ (native.eventCount === 0 &&
+ native.lastCompleteEvent !== NO_COMPLETE_EVENT) ||
+ (native.eventCount > 0 &&
+ native.lastCompleteEvent !== native.eventCount - 1)
+ ) {
+ throw new Error("PCRE2 returned data outside the trace ABI contract.");
+ }
+ if (native.status !== 0) {
+ let range:
+ | { readonly startUtf16: number; readonly endUtf16: number }
+ | undefined;
+ if (native.errorPhase === 2) {
+ try {
+ const offset = utf8ByteToUtf16(patternMap, native.errorOffset);
+ range = { startUtf16: offset, endUtf16: offset };
+ } catch {
+ range = undefined;
+ }
+ }
+ return traceFailure(
+ request,
+ `PCRE2 trace ${
+ native.errorPhase === 2 ? "compile" : "match"
+ } error: ${nativeErrorMessage(this.#module, native.status)}.`,
+ native.errorPhase === 2
+ ? "compile-error"
+ : native.status === -47
+ ? "match-limit"
+ : "trace-error",
+ start,
+ range,
+ );
+ }
+
+ const events = materializeEvents(
+ this.#module,
+ eventsPointer,
+ marksPointer,
+ native,
+ request.pattern,
+ request.subject,
+ );
+ const diagnostics: RegexDiagnostic[] = [];
+ if (request.flags.includes("g")) {
+ diagnostics.push({
+ id: "pcre2-trace-single-invocation",
+ source: "execution-engine",
+ severity: "information",
+ code: "trace-single-invocation",
+ message:
+ "The application-level g flag is not applied to tracing. This stream reports one exact pcre2_match() invocation.",
+ flavour: "pcre2",
+ provenance: "derived",
+ });
+ }
+ if (native.eventsTruncated) {
+ diagnostics.push({
+ id: "pcre2-trace-events-truncated",
+ source: "execution-engine",
+ severity: "warning",
+ code: "trace-limit",
+ message: `Trace collection stopped after ${native.eventCount.toLocaleString()} complete events at the configured event or serialized-byte cap.`,
+ flavour: "pcre2",
+ provenance: "reported",
+ });
+ }
+ if (native.marksTruncated) {
+ diagnostics.push({
+ id: "pcre2-trace-marks-truncated",
+ source: "execution-engine",
+ severity: "warning",
+ code: "trace-mark-limit",
+ message:
+ "At least one reported PCRE2 mark was clipped at the per-event mark byte limit.",
+ flavour: "pcre2",
+ provenance: "reported",
+ });
+ }
+ if (events.length === 0 && !native.eventsTruncated) {
+ diagnostics.push({
+ id: "pcre2-trace-no-events",
+ source: "execution-engine",
+ severity: "information",
+ code: "trace-empty",
+ message:
+ "PCRE2 reported no automatic callouts. Start optimizations can conclude some matches before the instrumented matcher runs.",
+ flavour: "pcre2",
+ provenance: "reported",
+ });
+ }
+ return {
+ accepted: true,
+ engine: pcre2EngineInfo(this.#runtimeVersion),
+ flags,
+ events,
+ diagnostics,
+ elapsedMs: performance.now() - start,
+ truncated: native.eventsTruncated,
+ marksTruncated: native.marksTruncated,
+ totalEventCount: native.totalEventCount,
+ traceBytes: native.traceBytesLength,
+ nativeMatchStatus: native.nativeMatchStatus,
+ matched: native.matched,
+ ...(native.lastCompleteEvent === NO_COMPLETE_EVENT
+ ? {}
+ : { lastCompleteEvent: native.lastCompleteEvent + 1 }),
+ };
+ } finally {
+ allocations.free();
+ }
+ }
+}
diff --git a/src/regex/execution/adapters/pcre2/pcre2-module.ts b/src/regex/execution/adapters/pcre2/pcre2-module.ts
new file mode 100644
index 0000000..cb0f738
--- /dev/null
+++ b/src/regex/execution/adapters/pcre2/pcre2-module.ts
@@ -0,0 +1,80 @@
+export interface Pcre2EmscriptenModule {
+ readonly HEAPU8: Uint8Array;
+ readonly UTF8ToString: (pointer: number) => string;
+ readonly _malloc: (size: number) => number;
+ readonly _free: (pointer: number) => void;
+ readonly _regex_pcre2_bridge_abi_version: () => number;
+ readonly _regex_pcre2_config_flags: () => number;
+ readonly _regex_pcre2_version: () => number;
+ readonly _regex_pcre2_self_test: () => number;
+ readonly _regex_pcre2_error_message: (
+ errorCode: number,
+ outputPointer: number,
+ outputCapacity: number,
+ ) => number;
+ readonly _regex_pcre2_execute: (
+ patternPointer: number,
+ patternLength: number,
+ subjectPointer: number,
+ subjectLength: number,
+ applicationFlags: number,
+ limitsPointer: number,
+ recordsPointer: number,
+ recordCapacity: number,
+ namesPointer: number,
+ nameCapacity: number,
+ nameBytesPointer: number,
+ nameBytesCapacity: number,
+ resultPointer: number,
+ ) => number;
+ readonly _regex_pcre2_substitute: (
+ patternPointer: number,
+ patternLength: number,
+ subjectPointer: number,
+ subjectLength: number,
+ replacementPointer: number,
+ replacementLength: number,
+ applicationFlags: number,
+ limitsPointer: number,
+ recordsPointer: number,
+ recordCapacity: number,
+ namesPointer: number,
+ nameCapacity: number,
+ nameBytesPointer: number,
+ nameBytesCapacity: number,
+ outputPointer: number,
+ outputCapacity: number,
+ resultPointer: number,
+ ) => number;
+ readonly _regex_pcre2_trace: (
+ patternPointer: number,
+ patternLength: number,
+ subjectPointer: number,
+ subjectLength: number,
+ applicationFlags: number,
+ limitsPointer: number,
+ eventsPointer: number,
+ eventCapacity: number,
+ markBytesPointer: number,
+ markBytesCapacity: number,
+ resultPointer: number,
+ ) => number;
+}
+
+export type Pcre2ModuleFactory = (options: {
+ readonly locateFile: (name: string) => string;
+ readonly print: (line: string) => void;
+ readonly printErr: (line: string) => void;
+}) => Promise;
+
+export async function importPcre2ModuleFactory(
+ moduleUrl: URL,
+): Promise {
+ const imported = (await import(
+ /* @vite-ignore */ moduleUrl.href
+ )) as Partial<{ default: Pcre2ModuleFactory }>;
+ if (typeof imported.default !== "function") {
+ throw new Error("The bundled PCRE2 module has no Emscripten factory.");
+ }
+ return imported.default;
+}
diff --git a/src/regex/execution/engine-registry.ts b/src/regex/execution/engine-registry.ts
new file mode 100644
index 0000000..dc19858
--- /dev/null
+++ b/src/regex/execution/engine-registry.ts
@@ -0,0 +1,66 @@
+import type { RegexFlavourId } from "../model/flavour";
+import type { WorkerFactory } from "./WorkerSupervisor";
+
+export interface EngineWorkerRegistration {
+ readonly flavour: RegexFlavourId;
+ readonly label: string;
+ readonly createWorker: WorkerFactory;
+}
+
+export class EngineWorkerRegistry {
+ readonly registrations: readonly EngineWorkerRegistration[];
+ private readonly byFlavour: ReadonlyMap<
+ RegexFlavourId,
+ EngineWorkerRegistration
+ >;
+
+ constructor(registrations: readonly EngineWorkerRegistration[]) {
+ const byFlavour = new Map();
+ for (const registration of registrations) {
+ if (byFlavour.has(registration.flavour)) {
+ throw new Error(
+ `Execution worker for ${registration.flavour} is registered more than once.`,
+ );
+ }
+ byFlavour.set(registration.flavour, registration);
+ }
+ this.registrations = [...registrations];
+ this.byFlavour = byFlavour;
+ }
+
+ get(flavour: RegexFlavourId): EngineWorkerRegistration | undefined {
+ return this.byFlavour.get(flavour);
+ }
+
+ require(flavour: RegexFlavourId): EngineWorkerRegistration {
+ const registration = this.get(flavour);
+ if (registration) return registration;
+ throw new Error(
+ `No execution worker is registered for regex flavour ${JSON.stringify(flavour)}.`,
+ );
+ }
+}
+
+export const ENGINE_WORKER_REGISTRY = new EngineWorkerRegistry([
+ {
+ flavour: "ecmascript",
+ label: "ECMAScript engine",
+ createWorker: () =>
+ new Worker(
+ new URL("../../workers/ecmascript.worker.ts", import.meta.url),
+ {
+ type: "module",
+ name: "regex-tools-ecmascript",
+ },
+ ),
+ },
+ {
+ flavour: "pcre2",
+ label: "PCRE2 engine",
+ createWorker: () =>
+ new Worker(new URL("../../workers/pcre2.worker.ts", import.meta.url), {
+ type: "module",
+ name: "regex-tools-pcre2",
+ }),
+ },
+]);
diff --git a/src/regex/execution/request-limits.ts b/src/regex/execution/request-limits.ts
index defcb14..cd0210c 100644
--- a/src/regex/execution/request-limits.ts
+++ b/src/regex/execution/request-limits.ts
@@ -10,6 +10,15 @@ export interface RegexResourceLimits {
readonly interactiveSubjectHardBytes: number;
readonly corpusSoftBytes: number;
readonly corpusHardBytes: number;
+ readonly maximumCorpusDocuments: number;
+ readonly maximumCorpusDocumentBytes: number;
+ readonly maximumCorpusLines: number;
+ readonly maximumCorpusMatches: number;
+ readonly maximumCorpusCaptureSummaries: number;
+ readonly maximumCorpusCaptureSamplesPerGroup: number;
+ readonly maximumCorpusCaptureSampleUtf16: number;
+ readonly maximumCorpusOutputBytes: number;
+ readonly maximumCorpusWallTimeMs: number;
readonly maximumReplacementTemplateUtf16: number;
readonly maximumListTemplateUtf16: number;
readonly maximumMatches: number;
@@ -37,6 +46,15 @@ export const DEFAULT_REGEX_LIMITS: RegexResourceLimits = {
interactiveSubjectHardBytes: 16 * 1024 * 1024,
corpusSoftBytes: 64 * 1024 * 1024,
corpusHardBytes: 256 * 1024 * 1024,
+ maximumCorpusDocuments: 256,
+ maximumCorpusDocumentBytes: 16 * 1024 * 1024,
+ maximumCorpusLines: 100_000,
+ maximumCorpusMatches: 100_000,
+ maximumCorpusCaptureSummaries: 32,
+ maximumCorpusCaptureSamplesPerGroup: 3,
+ maximumCorpusCaptureSampleUtf16: 160,
+ maximumCorpusOutputBytes: 64 * 1024 * 1024,
+ maximumCorpusWallTimeMs: 5 * 60 * 1_000,
maximumReplacementTemplateUtf16: 64 * 1024,
maximumListTemplateUtf16: 16 * 1024,
maximumMatches: 10_000,
diff --git a/src/regex/execution/worker-protocol.ts b/src/regex/execution/worker-protocol.ts
index 7ab55e9..6630fee 100644
--- a/src/regex/execution/worker-protocol.ts
+++ b/src/regex/execution/worker-protocol.ts
@@ -10,6 +10,7 @@ import type {
ReplacementSyntaxResult,
} from "../model/syntax";
import type { ReplacementSyntaxRequest } from "../syntax/SyntaxProvider";
+import type { Pcre2TraceRequest, Pcre2TraceResult } from "../model/trace";
export const WORKER_PROTOCOL_VERSION = 1;
@@ -53,6 +54,16 @@ export type EngineWorkerResult =
readonly result: RegexReplacementResult;
};
+export interface TraceWorkerOperation {
+ readonly kind: "trace";
+ readonly request: Pcre2TraceRequest;
+}
+
+export interface TraceWorkerResult {
+ readonly kind: "trace";
+ readonly result: Pcre2TraceResult;
+}
+
export interface WorkerRequest {
readonly protocolVersion: typeof WORKER_PROTOCOL_VERSION;
readonly requestId: number;
diff --git a/src/regex/flavours/flavour-registry.test.ts b/src/regex/flavours/flavour-registry.test.ts
new file mode 100644
index 0000000..63d31a3
--- /dev/null
+++ b/src/regex/flavours/flavour-registry.test.ts
@@ -0,0 +1,158 @@
+import { describe, expect, it } from "vitest";
+import {
+ AVAILABLE_REGEX_FLAVOURS,
+ RegexFlavourRegistry,
+ defaultRegexOptions,
+ parseRegexFlags,
+ parseRegexOptions,
+ resolveRegexFlavourVersion,
+ type RegexFlavourDefinition,
+} from "./flavour-registry";
+
+const fixturePcre2: RegexFlavourDefinition = {
+ id: "pcre2",
+ label: "Fixture PCRE2",
+ versions: [
+ {
+ value: "fixture-1",
+ label: "Fixture 1",
+ syntaxVersion: "fixture-1",
+ },
+ ],
+ defaultVersion: "fixture-1",
+ flags: [
+ { value: "i", label: "Caseless" },
+ { value: "x", label: "Extended" },
+ ],
+ defaultFlags: [],
+ options: [
+ {
+ name: "matchLimit",
+ label: "Match limit",
+ kind: "integer",
+ defaultValue: 100_000,
+ minimum: 1,
+ maximum: 1_000_000,
+ },
+ {
+ name: "ungreedy",
+ label: "Ungreedy",
+ kind: "boolean",
+ defaultValue: false,
+ },
+ ],
+};
+
+describe("available regex flavour contracts", () => {
+ it("publishes the two implemented production flavour boundaries", () => {
+ expect(
+ AVAILABLE_REGEX_FLAVOURS.definitions.map((definition) => definition.id),
+ ).toEqual(["ecmascript", "pcre2"]);
+ expect(
+ AVAILABLE_REGEX_FLAVOURS.require("pcre2").options.map(
+ (option) => option.name,
+ ),
+ ).toEqual(["matchLimit", "depthLimit", "heapLimitKib"]);
+ expect(() =>
+ AVAILABLE_REGEX_FLAVOURS.parse("python", "Project flavour"),
+ ).toThrow(/only supports ECMAScript and PCRE2/u);
+ });
+
+ it("validates ECMAScript flags through the flavour contract", () => {
+ const definition = AVAILABLE_REGEX_FLAVOURS.require("ecmascript");
+
+ expect(parseRegexFlags(["g", "u"], definition, "Flags")).toEqual([
+ "g",
+ "u",
+ ]);
+ expect(() => parseRegexFlags(["u", "v"], definition, "Flags")).toThrow(
+ /cannot be combined/u,
+ );
+ expect(() => parseRegexFlags(["x"], definition, "Flags")).toThrow(
+ /unsupported/u,
+ );
+ expect(
+ resolveRegexFlavourVersion(
+ "ECMAScript 2025",
+ definition,
+ "Flavour version",
+ ),
+ ).toEqual({
+ definition: definition.versions[0],
+ value: "ECMAScript 2025",
+ });
+ expect(
+ resolveRegexFlavourVersion(undefined, definition, "Flavour version")
+ .value,
+ ).toBe("ECMAScript 2025 syntax / current browser runtime");
+ expect(() =>
+ resolveRegexFlavourVersion("2026", definition, "Flavour version"),
+ ).toThrow(/unsupported for ECMAScript/u);
+ });
+
+ it("supports typed future option contracts without registering execution", () => {
+ const registry = new RegexFlavourRegistry([fixturePcre2]);
+ const definition = registry.require("pcre2");
+
+ expect(defaultRegexOptions(definition)).toEqual({
+ matchLimit: 100_000,
+ ungreedy: false,
+ });
+ expect(parseRegexOptions(undefined, definition, "Fixture options")).toEqual(
+ {
+ matchLimit: 100_000,
+ ungreedy: false,
+ },
+ );
+ expect(
+ parseRegexOptions(
+ { matchLimit: 250_000, ungreedy: true },
+ definition,
+ "Fixture options",
+ ),
+ ).toEqual({ matchLimit: 250_000, ungreedy: true });
+ expect(() =>
+ parseRegexOptions({ matchLimit: 0 }, definition, "Fixture options"),
+ ).toThrow(/integer from 1 to 1000000/u);
+ expect(() =>
+ parseRegexOptions({ jit: true }, definition, "Fixture options"),
+ ).toThrow(/unsupported entries: jit/u);
+ });
+
+ it("validates production PCRE2 flags, version and bounded limits", () => {
+ const definition = AVAILABLE_REGEX_FLAVOURS.require("pcre2");
+ expect(parseRegexFlags(["g", "J"], definition, "Flags")).toEqual([
+ "g",
+ "J",
+ ]);
+ expect(defaultRegexOptions(definition)).toEqual({
+ matchLimit: 1_000_000,
+ depthLimit: 1_000,
+ heapLimitKib: 32_768,
+ });
+ expect(
+ parseRegexOptions(
+ { matchLimit: 50_000, depthLimit: 50, heapLimitKib: 1_024 },
+ definition,
+ "PCRE2 options",
+ ),
+ ).toEqual({
+ matchLimit: 50_000,
+ depthLimit: 50,
+ heapLimitKib: 1_024,
+ });
+ expect(() =>
+ parseRegexOptions(
+ { matchLimit: 100_000_001 },
+ definition,
+ "PCRE2 options",
+ ),
+ ).toThrow(/integer from 1 to 100000000/u);
+ });
+
+ it("rejects ambiguous registrations at construction time", () => {
+ expect(
+ () => new RegexFlavourRegistry([fixturePcre2, fixturePcre2]),
+ ).toThrow(/registered more than once/u);
+ });
+});
diff --git a/src/regex/flavours/flavour-registry.ts b/src/regex/flavours/flavour-registry.ts
new file mode 100644
index 0000000..d532152
--- /dev/null
+++ b/src/regex/flavours/flavour-registry.ts
@@ -0,0 +1,389 @@
+import type {
+ RegexEngineOptions,
+ RegexFlavourId,
+ RegexOptionValue,
+} from "../model/flavour";
+
+export interface RegexFlavourVersionDefinition {
+ readonly value: string;
+ readonly label: string;
+ readonly syntaxVersion: string;
+ readonly legacyValues?: readonly string[];
+}
+
+export interface RegexFlagDefinition {
+ readonly value: string;
+ readonly label: string;
+}
+
+export type RegexOptionDefinition =
+ | {
+ readonly name: string;
+ readonly label: string;
+ readonly kind: "boolean";
+ readonly defaultValue: boolean;
+ }
+ | {
+ readonly name: string;
+ readonly label: string;
+ readonly kind: "integer";
+ readonly defaultValue: number;
+ readonly minimum: number;
+ readonly maximum: number;
+ }
+ | {
+ readonly name: string;
+ readonly label: string;
+ readonly kind: "string";
+ readonly defaultValue: string;
+ readonly maximumLength: number;
+ readonly values?: readonly string[];
+ };
+
+export interface RegexFlavourDefinition {
+ readonly id: RegexFlavourId;
+ readonly label: string;
+ readonly versions: readonly RegexFlavourVersionDefinition[];
+ readonly defaultVersion: string;
+ readonly flags: readonly RegexFlagDefinition[];
+ readonly defaultFlags: readonly string[];
+ readonly mutuallyExclusiveFlags?: readonly (readonly string[])[];
+ readonly options: readonly RegexOptionDefinition[];
+}
+
+function duplicate(values: readonly string[]): string | undefined {
+ const seen = new Set();
+ return values.find((value) => {
+ if (seen.has(value)) return true;
+ seen.add(value);
+ return false;
+ });
+}
+
+function validateDefinition(definition: RegexFlavourDefinition): void {
+ if (definition.versions.length === 0) {
+ throw new Error(`${definition.label} must define at least one version.`);
+ }
+ if (
+ !definition.versions.some(
+ (version) => version.value === definition.defaultVersion,
+ )
+ ) {
+ throw new Error(
+ `${definition.label} default version is not present in its version list.`,
+ );
+ }
+ const duplicateVersion = duplicate(
+ definition.versions.map((version) => version.value),
+ );
+ if (duplicateVersion) {
+ throw new Error(
+ `${definition.label} defines duplicate version ${JSON.stringify(duplicateVersion)}.`,
+ );
+ }
+ const flagValues = definition.flags.map((flag) => flag.value);
+ const duplicateFlag = duplicate(flagValues);
+ if (duplicateFlag) {
+ throw new Error(
+ `${definition.label} defines duplicate flag ${JSON.stringify(duplicateFlag)}.`,
+ );
+ }
+ if (definition.defaultFlags.some((flag) => !flagValues.includes(flag))) {
+ throw new Error(`${definition.label} has an unsupported default flag.`);
+ }
+ const duplicateOption = duplicate(
+ definition.options.map((option) => option.name),
+ );
+ if (duplicateOption) {
+ throw new Error(
+ `${definition.label} defines duplicate option ${JSON.stringify(duplicateOption)}.`,
+ );
+ }
+}
+
+export class RegexFlavourRegistry {
+ readonly definitions: readonly RegexFlavourDefinition[];
+ private readonly byId: ReadonlyMap;
+
+ constructor(definitions: readonly RegexFlavourDefinition[]) {
+ const duplicateId = duplicate(
+ definitions.map((definition) => definition.id),
+ );
+ if (duplicateId) {
+ throw new Error(
+ `Regex flavour ${JSON.stringify(duplicateId)} is registered more than once.`,
+ );
+ }
+ definitions.forEach(validateDefinition);
+ this.definitions = [...definitions];
+ this.byId = new Map(
+ definitions.map((definition) => [definition.id, definition]),
+ );
+ }
+
+ get(id: RegexFlavourId): RegexFlavourDefinition | undefined {
+ return this.byId.get(id);
+ }
+
+ require(id: RegexFlavourId): RegexFlavourDefinition {
+ const definition = this.get(id);
+ if (definition) return definition;
+ throw new Error(this.unavailableMessage(id));
+ }
+
+ parse(value: unknown, label: string): RegexFlavourDefinition {
+ if (typeof value !== "string" || value.length > 32) {
+ throw new Error(`${label} must be a supported flavour identifier.`);
+ }
+ const definition = this.definitions.find(
+ (candidate) => candidate.id === value,
+ );
+ if (definition) return definition;
+ throw new Error(this.unavailableMessage(value));
+ }
+
+ private unavailableMessage(value: string): string {
+ const available = this.definitions.map((definition) => definition.label);
+ const support =
+ available.length === 1
+ ? available[0]
+ : available.length === 0
+ ? "no flavours"
+ : `${available.slice(0, -1).join(", ")} and ${available.at(-1)}`;
+ return `Regex flavour ${JSON.stringify(value)} is unavailable; this build only supports ${support}.`;
+ }
+}
+
+function parseFlagArray(value: unknown, label: string): readonly string[] {
+ if (!Array.isArray(value)) throw new Error(`${label} must be an array.`);
+ return value.map((flag) => {
+ if (typeof flag !== "string" || flag.length === 0 || flag.length > 32) {
+ throw new Error(`${label} must contain non-empty flag identifiers.`);
+ }
+ return flag;
+ });
+}
+
+export function parseRegexFlags(
+ value: unknown,
+ definition: RegexFlavourDefinition,
+ label: string,
+): readonly string[] {
+ const flags = parseFlagArray(value, label);
+ const supported = new Set(definition.flags.map((flag) => flag.value));
+ if (
+ new Set(flags).size !== flags.length ||
+ flags.some((flag) => !supported.has(flag))
+ ) {
+ throw new Error(`${label} contain duplicates or unsupported values.`);
+ }
+ for (const group of definition.mutuallyExclusiveFlags ?? []) {
+ const selected = group.filter((flag) => flags.includes(flag));
+ if (selected.length > 1) {
+ throw new Error(
+ `${definition.label} flags ${selected.join(" and ")} cannot be combined.`,
+ );
+ }
+ }
+ return flags;
+}
+
+function optionObject(value: unknown, label: string): Record {
+ if (value === undefined) return {};
+ if (!value || typeof value !== "object" || Array.isArray(value)) {
+ throw new Error(`${label} must be a JSON object.`);
+ }
+ return value as Record;
+}
+
+function parseOptionValue(
+ value: unknown,
+ definition: RegexOptionDefinition,
+ label: string,
+): RegexOptionValue {
+ switch (definition.kind) {
+ case "boolean":
+ if (typeof value !== "boolean") {
+ throw new Error(`${label} must be boolean.`);
+ }
+ return value;
+ case "integer":
+ if (
+ typeof value !== "number" ||
+ !Number.isSafeInteger(value) ||
+ value < definition.minimum ||
+ value > definition.maximum
+ ) {
+ throw new Error(
+ `${label} must be an integer from ${definition.minimum} to ${definition.maximum}.`,
+ );
+ }
+ return value;
+ case "string":
+ if (
+ typeof value !== "string" ||
+ value.length > definition.maximumLength ||
+ (definition.values !== undefined && !definition.values.includes(value))
+ ) {
+ throw new Error(`${label} contains an unsupported text value.`);
+ }
+ return value;
+ }
+}
+
+export function parseRegexOptions(
+ value: unknown,
+ definition: RegexFlavourDefinition,
+ label: string,
+): RegexEngineOptions {
+ const input = optionObject(value, label);
+ const definitions = new Map(
+ definition.options.map((option) => [option.name, option]),
+ );
+ const unsupported = Object.keys(input).filter(
+ (name) => !definitions.has(name),
+ );
+ if (unsupported.length > 0) {
+ if (definition.options.length === 0) {
+ throw new Error(
+ `${label} contains unsupported entries; this build does not implement ${definition.label} engine options.`,
+ );
+ }
+ throw new Error(
+ `${label} contains unsupported entries: ${unsupported.join(", ")}.`,
+ );
+ }
+ const result: Record = {
+ ...defaultRegexOptions(definition),
+ };
+ for (const [name, optionValue] of Object.entries(input)) {
+ const option = definitions.get(name);
+ if (!option) continue;
+ result[name] = parseOptionValue(
+ optionValue,
+ option,
+ `${label} ${JSON.stringify(name)}`,
+ );
+ }
+ return result;
+}
+
+export function defaultRegexOptions(
+ definition: RegexFlavourDefinition,
+): RegexEngineOptions {
+ return Object.fromEntries(
+ definition.options.map((option) => [option.name, option.defaultValue]),
+ );
+}
+
+export function resolveRegexFlavourVersion(
+ value: unknown,
+ definition: RegexFlavourDefinition,
+ label: string,
+): {
+ readonly definition: RegexFlavourVersionDefinition;
+ readonly value: string;
+} {
+ const candidate =
+ value === undefined
+ ? definition.defaultVersion
+ : typeof value === "string" && value.length <= 128
+ ? value
+ : (() => {
+ throw new Error(`${label} must be a supported version identifier.`);
+ })();
+ const version = definition.versions.find(
+ (entry) =>
+ entry.value === candidate || entry.legacyValues?.includes(candidate),
+ );
+ if (!version) {
+ throw new Error(
+ `${label} ${JSON.stringify(candidate)} is unsupported for ${definition.label}.`,
+ );
+ }
+ return { definition: version, value: candidate };
+}
+
+const ECMASCRIPT_VERSION = "ECMAScript 2025 syntax / current browser runtime";
+
+export const ECMASCRIPT_FLAVOUR: RegexFlavourDefinition = {
+ id: "ecmascript",
+ label: "ECMAScript",
+ versions: [
+ {
+ value: ECMASCRIPT_VERSION,
+ label: "2025 syntax · current browser",
+ syntaxVersion: "2025",
+ legacyValues: ["ECMAScript 2025", "2025"],
+ },
+ ],
+ defaultVersion: ECMASCRIPT_VERSION,
+ flags: [
+ { value: "g", label: "Global" },
+ { value: "i", label: "Ignore case" },
+ { value: "m", label: "Multiline" },
+ { value: "s", label: "Dot all" },
+ { value: "u", label: "Unicode" },
+ { value: "v", label: "Unicode sets" },
+ { value: "y", label: "Sticky" },
+ { value: "d", label: "Expose indices" },
+ ],
+ defaultFlags: ["g", "u"],
+ mutuallyExclusiveFlags: [["u", "v"]],
+ options: [],
+};
+
+export const PCRE2_FLAVOUR: RegexFlavourDefinition = {
+ id: "pcre2",
+ label: "PCRE2",
+ versions: [
+ {
+ value: "PCRE2 10.47 8-bit WebAssembly",
+ label: "10.47 · 8-bit WebAssembly",
+ syntaxVersion: "10.47",
+ legacyValues: ["PCRE2 10.47", "10.47"],
+ },
+ ],
+ defaultVersion: "PCRE2 10.47 8-bit WebAssembly",
+ flags: [
+ { value: "g", label: "Global" },
+ { value: "i", label: "Caseless" },
+ { value: "m", label: "Multiline" },
+ { value: "s", label: "Dot all" },
+ { value: "x", label: "Extended" },
+ { value: "U", label: "Ungreedy" },
+ { value: "J", label: "Duplicate names" },
+ ],
+ defaultFlags: ["g"],
+ options: [
+ {
+ name: "matchLimit",
+ label: "Match steps",
+ kind: "integer",
+ defaultValue: 1_000_000,
+ minimum: 1,
+ maximum: 100_000_000,
+ },
+ {
+ name: "depthLimit",
+ label: "Depth",
+ kind: "integer",
+ defaultValue: 1_000,
+ minimum: 1,
+ maximum: 100_000,
+ },
+ {
+ name: "heapLimitKib",
+ label: "Heap KiB",
+ kind: "integer",
+ defaultValue: 32_768,
+ minimum: 1,
+ maximum: 131_072,
+ },
+ ],
+};
+
+export const AVAILABLE_REGEX_FLAVOURS = new RegexFlavourRegistry([
+ ECMASCRIPT_FLAVOUR,
+ PCRE2_FLAVOUR,
+]);
diff --git a/src/regex/formatting/PatternFormatValidator.test.ts b/src/regex/formatting/PatternFormatValidator.test.ts
new file mode 100644
index 0000000..338d714
--- /dev/null
+++ b/src/regex/formatting/PatternFormatValidator.test.ts
@@ -0,0 +1,417 @@
+import { describe, expect, it, vi } from "vitest";
+import {
+ executeEcmaScript,
+ replaceEcmaScript,
+} from "../execution/adapters/ecmascript/EcmaScriptEngineAdapter";
+import type {
+ RegexExecutionRequest,
+ RegexReplacementRequest,
+} from "../model/match";
+import type { RegexSyntaxRequest } from "../model/syntax";
+import { WorkerRequestError } from "../execution/WorkerSupervisor";
+import { EcmaScriptSyntaxProvider } from "../syntax/providers/ecmascript/EcmaScriptSyntaxProvider";
+import type { RegexTestCase } from "../tests/test-case.types";
+import {
+ PatternFormatValidator,
+ type PatternFormatEngineRunner,
+ type PatternFormatSyntaxRunner,
+} from "./PatternFormatValidator";
+import type { PatternFormatValidationInput } from "./formatting.types";
+
+const provider = new EcmaScriptSyntaxProvider();
+
+function syntaxRunner(): PatternFormatSyntaxRunner {
+ return {
+ parsePattern: (request: RegexSyntaxRequest) =>
+ provider.parsePattern(request),
+ cancel: vi.fn(),
+ dispose: vi.fn(),
+ };
+}
+
+function engineRunner(
+ mapPattern: (pattern: string) => string = (pattern) => pattern,
+): PatternFormatEngineRunner {
+ return {
+ execute: vi.fn(async (request: RegexExecutionRequest) =>
+ executeEcmaScript({
+ ...request,
+ pattern: mapPattern(request.pattern),
+ }),
+ ),
+ replace: vi.fn(async (request: RegexReplacementRequest) =>
+ replaceEcmaScript({
+ ...request,
+ pattern: mapPattern(request.pattern),
+ }),
+ ),
+ cancel: vi.fn(),
+ dispose: vi.fn(),
+ };
+}
+
+function testCase(overrides: Partial = {}): RegexTestCase {
+ return {
+ id: "slash-count",
+ name: "Two slash paths",
+ enabled: true,
+ flavour: "ecmascript",
+ pattern: "a/b",
+ flags: ["g", "u"],
+ options: {},
+ scanAll: true,
+ subject: "a/b a/b",
+ expectation: { kind: "match-count", count: 2 },
+ ...overrides,
+ };
+}
+
+function input(
+ overrides: Partial = {},
+): PatternFormatValidationInput {
+ return {
+ flavour: "ecmascript",
+ sourcePattern: "a/b",
+ formattedPattern: "a\\/b",
+ flags: ["g", "u"],
+ options: {},
+ subject: "a/b a/b",
+ replacement: "[$&]",
+ scanAll: true,
+ timeoutMs: 2_000,
+ tests: [
+ testCase(),
+ testCase({
+ id: "other-pattern",
+ name: "Independent fixture",
+ pattern: "other",
+ }),
+ ],
+ ...overrides,
+ };
+}
+
+function validator(
+ formattedPatternMap: (pattern: string) => string = (pattern) => pattern,
+) {
+ return new PatternFormatValidator({
+ createSyntaxRunner: () => syntaxRunner(),
+ createEngineRunner: (label) =>
+ engineRunner(label === "formatted" ? formattedPatternMap : undefined),
+ now: () => performance.now(),
+ });
+}
+
+describe("PatternFormatValidator", () => {
+ it("reparses, recompiles and compares current replacement plus exact applicable tests", async () => {
+ const instance = validator();
+ const progress = vi.fn();
+
+ const result = await instance.validate(input(), progress);
+
+ expect(result.status).toBe("equivalent");
+ expect(result.canApply).toBe(true);
+ expect(result.current.comparison).toBe("same");
+ expect(result.current.before.engineIdentity).toContain(
+ "Native ECMAScript RegExp",
+ );
+ expect(result.applicableTestCount).toBe(1);
+ expect(result.skippedTestCount).toBe(1);
+ expect(result.completedTestCount).toBe(1);
+ expect(result.tests[0]).toEqual(
+ expect.objectContaining({
+ testId: "slash-count",
+ comparison: "same",
+ }),
+ );
+ expect(result.tests[0]?.before.assertionPassed).toBe(true);
+ expect(result.tests[0]?.after.assertionPassed).toBe(true);
+ expect(result.differences).toEqual([]);
+ expect(progress).toHaveBeenLastCalledWith(1, 1);
+ instance.dispose();
+ });
+
+ it("blocks application and reports exact semantic differences", async () => {
+ const instance = validator((pattern) =>
+ pattern === "a\\/b" ? "z+" : pattern,
+ );
+
+ const result = await instance.validate(input());
+
+ expect(result.status).toBe("different");
+ expect(result.canApply).toBe(false);
+ expect(result.current.comparison).toBe("different");
+ expect(result.current.mismatchKinds).toContain("match-count");
+ expect(result.differences).toEqual(
+ expect.arrayContaining([
+ expect.objectContaining({
+ scope: "current-subject",
+ code: "current-behavior-changed",
+ }),
+ expect.objectContaining({
+ scope: "unit-test",
+ code: "test-behavior-changed",
+ testId: "slash-count",
+ }),
+ ]),
+ );
+ instance.dispose();
+ });
+
+ it("does not execute a formatted pattern rejected by the syntax provider", async () => {
+ const sourceEngine = engineRunner();
+ const formattedEngine = engineRunner();
+ const instance = new PatternFormatValidator({
+ createSyntaxRunner: () => syntaxRunner(),
+ createEngineRunner: (label) =>
+ label === "source" ? sourceEngine : formattedEngine,
+ now: () => performance.now(),
+ });
+
+ const result = await instance.validate(input({ formattedPattern: "(" }));
+
+ expect(result.status).toBe("different");
+ expect(result.formattedSyntaxAccepted).toBe(false);
+ expect(result.canApply).toBe(false);
+ expect(formattedEngine.replace).not.toHaveBeenCalled();
+ expect(result.differences).toEqual(
+ expect.arrayContaining([
+ expect.objectContaining({ code: "formatted-syntax-rejected" }),
+ ]),
+ );
+ instance.dispose();
+ });
+
+ it("reports no applicable tests explicitly without weakening the current snapshot gate", async () => {
+ const instance = validator();
+ const result = await instance.validate(
+ input({
+ tests: [
+ testCase({ enabled: false }),
+ testCase({ id: "other", pattern: "other" }),
+ ],
+ }),
+ );
+
+ expect(result.status).toBe("equivalent");
+ expect(result.applicableTestCount).toBe(0);
+ expect(result.notices.join(" ")).toContain(
+ "No enabled unit test has the exact active pattern",
+ );
+ instance.dispose();
+ });
+
+ it("blocks before execution when reparsing changes the capture shape", async () => {
+ const sourceEngine = engineRunner();
+ const formattedEngine = engineRunner();
+ const instance = new PatternFormatValidator({
+ createSyntaxRunner: (label) => {
+ const runner = syntaxRunner();
+ return label === "source"
+ ? runner
+ : {
+ ...runner,
+ parsePattern: async (request: RegexSyntaxRequest) => {
+ const result = await provider.parsePattern(request);
+ return {
+ ...result,
+ captures: [
+ {
+ number: 1,
+ name: "invented",
+ range: { startUtf16: 0, endUtf16: 1 },
+ repeated: false,
+ },
+ ],
+ };
+ },
+ };
+ },
+ createEngineRunner: (label) =>
+ label === "source" ? sourceEngine : formattedEngine,
+ now: () => performance.now(),
+ });
+
+ const result = await instance.validate(input());
+
+ expect(result.status).toBe("different");
+ expect(result.captureShapePreserved).toBe(false);
+ expect(result.differences).toEqual(
+ expect.arrayContaining([
+ expect.objectContaining({ code: "capture-shape-changed" }),
+ ]),
+ );
+ expect(sourceEngine.replace).not.toHaveBeenCalled();
+ expect(formattedEngine.replace).not.toHaveBeenCalled();
+ instance.dispose();
+ });
+
+ it("treats engine-identity drift as inconclusive", async () => {
+ const instance = new PatternFormatValidator({
+ createSyntaxRunner: () => syntaxRunner(),
+ createEngineRunner: (label) => {
+ const runner = engineRunner();
+ if (label === "source") return runner;
+ return {
+ ...runner,
+ replace: vi.fn(async (request: RegexReplacementRequest) => {
+ const replacement = await replaceEcmaScript(request);
+ return {
+ ...replacement,
+ execution: {
+ ...replacement.execution,
+ engine: {
+ ...replacement.execution.engine,
+ engineVersion: "different-runtime",
+ },
+ },
+ };
+ }),
+ };
+ },
+ now: () => performance.now(),
+ });
+
+ const result = await instance.validate(input({ tests: [] }));
+
+ expect(result.status).toBe("inconclusive");
+ expect(result.canApply).toBe(false);
+ expect(result.current.comparison).toBe("inconclusive");
+ expect(result.current.mismatchKinds).toEqual(["engine-identity"]);
+ instance.dispose();
+ });
+
+ it("treats truncated semantic output as inconclusive", async () => {
+ const instance = new PatternFormatValidator({
+ createSyntaxRunner: () => syntaxRunner(),
+ createEngineRunner: (label) => {
+ const runner = engineRunner();
+ if (label === "source") return runner;
+ return {
+ ...runner,
+ replace: vi.fn(async (request: RegexReplacementRequest) => {
+ const replacement = await replaceEcmaScript(request);
+ return {
+ ...replacement,
+ execution: {
+ ...replacement.execution,
+ truncated: true,
+ },
+ truncated: true,
+ };
+ }),
+ };
+ },
+ now: () => performance.now(),
+ });
+
+ const result = await instance.validate(input({ tests: [] }));
+
+ expect(result.status).toBe("inconclusive");
+ expect(result.canApply).toBe(false);
+ expect(result.current.comparison).toBe("inconclusive");
+ expect(result.current.summary).toContain("incomplete data");
+ instance.dispose();
+ });
+
+ it("accepts matching fixed unit-test timeouts but not two current-snapshot timeouts", async () => {
+ const timedTest = testCase({
+ subject: "slow",
+ expectation: { kind: "must-time-out" },
+ });
+ const createRunner = (): PatternFormatEngineRunner => {
+ const runner = engineRunner();
+ return {
+ ...runner,
+ execute: vi.fn(async (request: RegexExecutionRequest) => {
+ if (request.subject === "slow") {
+ throw new WorkerRequestError("timeout", "Fixture timeout.");
+ }
+ return executeEcmaScript(request);
+ }),
+ };
+ };
+ const instance = new PatternFormatValidator({
+ createSyntaxRunner: () => syntaxRunner(),
+ createEngineRunner: () => createRunner(),
+ now: () => performance.now(),
+ });
+
+ const result = await instance.validate(input({ tests: [timedTest] }));
+
+ expect(result.status).toBe("equivalent");
+ expect(result.tests[0]).toEqual(
+ expect.objectContaining({
+ comparison: "same",
+ before: expect.objectContaining({
+ status: "timeout",
+ assertionPassed: true,
+ }),
+ after: expect.objectContaining({
+ status: "timeout",
+ assertionPassed: true,
+ }),
+ }),
+ );
+ instance.dispose();
+
+ const currentTimeout = new PatternFormatValidator({
+ createSyntaxRunner: () => syntaxRunner(),
+ createEngineRunner: () => ({
+ ...engineRunner(),
+ replace: vi.fn(async () => {
+ throw new WorkerRequestError("timeout", "Fixture timeout.");
+ }),
+ }),
+ now: () => performance.now(),
+ });
+ const currentResult = await currentTimeout.validate(input({ tests: [] }));
+ expect(currentResult.status).toBe("inconclusive");
+ expect(currentResult.current.comparison).toBe("inconclusive");
+ currentTimeout.dispose();
+ });
+
+ it("does not start a test whose full timeout no longer fits the aggregate wall budget", async () => {
+ let clock = 0;
+ const sourceEngine = engineRunner();
+ const formattedEngine = engineRunner();
+ const sourceReplace = sourceEngine.replace.bind(sourceEngine);
+ const formattedReplace = formattedEngine.replace.bind(formattedEngine);
+ sourceEngine.replace = vi.fn(async (request, timeoutMs) => {
+ const result = await sourceReplace(request, timeoutMs);
+ clock = 59_000;
+ return result;
+ });
+ formattedEngine.replace = vi.fn(async (request, timeoutMs) => {
+ const result = await formattedReplace(request, timeoutMs);
+ clock = 59_000;
+ return result;
+ });
+ const instance = new PatternFormatValidator({
+ createSyntaxRunner: () => syntaxRunner(),
+ createEngineRunner: (label) =>
+ label === "source" ? sourceEngine : formattedEngine,
+ now: () => clock,
+ });
+
+ const result = await instance.validate(
+ input({
+ tests: [
+ testCase({
+ resourceLimits: { manualExecutionTimeoutMs: 2_000 },
+ }),
+ ],
+ }),
+ );
+
+ expect(result.status).toBe("inconclusive");
+ expect(result.applicableTestCount).toBe(1);
+ expect(result.completedTestCount).toBe(0);
+ expect(sourceEngine.execute).not.toHaveBeenCalled();
+ expect(formattedEngine.execute).not.toHaveBeenCalled();
+ expect(result.notices.join(" ")).toContain(
+ "full 2,000 ms worker timeout no longer fit",
+ );
+ instance.dispose();
+ });
+});
diff --git a/src/regex/formatting/PatternFormatValidator.ts b/src/regex/formatting/PatternFormatValidator.ts
new file mode 100644
index 0000000..b425848
--- /dev/null
+++ b/src/regex/formatting/PatternFormatValidator.ts
@@ -0,0 +1,777 @@
+import {
+ DEFAULT_REGEX_LIMITS,
+ utf8ByteLength,
+} from "../execution/request-limits";
+import { EngineSupervisor } from "../execution/EngineSupervisor";
+import { SyntaxSupervisor } from "../execution/SyntaxSupervisor";
+import { WorkerRequestError } from "../execution/WorkerSupervisor";
+import {
+ AVAILABLE_REGEX_FLAVOURS,
+ parseRegexFlags,
+ parseRegexOptions,
+ resolveRegexFlavourVersion,
+} from "../flavours/flavour-registry";
+import type {
+ RegexExecutionRequest,
+ RegexExecutionResult,
+ RegexReplacementRequest,
+ RegexReplacementResult,
+} from "../model/match";
+import type {
+ CaptureDefinition,
+ RegexSyntaxRequest,
+ RegexSyntaxResult,
+} from "../model/syntax";
+import {
+ compareSemanticResults,
+ engineIdentity,
+ evaluateUnitTestFailure,
+ type SemanticSideResult,
+} from "../minimization/oracles";
+import {
+ MAXIMUM_REGEX_TESTS,
+ type RegexTestCase,
+} from "../tests/test-case.types";
+import type {
+ PatternFormatDifference,
+ PatternFormatRuntimeObservation,
+ PatternFormatSnapshotCheck,
+ PatternFormatTestCheck,
+ PatternFormatValidationInput,
+ PatternFormatValidationResult,
+} from "./formatting.types";
+
+const MAXIMUM_VALIDATION_WALL_TIME_MS = 60_000;
+const MAXIMUM_RETAINED_DIFFERENCES = 1_000;
+
+export interface PatternFormatSyntaxRunner {
+ parsePattern(
+ request: RegexSyntaxRequest,
+ timeoutMs?: number,
+ ): Promise;
+ cancel(): void;
+ dispose(): void;
+}
+
+export interface PatternFormatEngineRunner {
+ execute(
+ request: RegexExecutionRequest,
+ timeoutMs: number,
+ ): Promise;
+ replace(
+ request: RegexReplacementRequest,
+ timeoutMs: number,
+ ): Promise;
+ cancel(): void;
+ dispose(): void;
+}
+
+export interface PatternFormatValidatorDependencies {
+ readonly createSyntaxRunner: (
+ label: "source" | "formatted",
+ ) => PatternFormatSyntaxRunner;
+ readonly createEngineRunner: (
+ label: "source" | "formatted",
+ ) => PatternFormatEngineRunner;
+ readonly now: () => number;
+}
+
+const DEFAULT_DEPENDENCIES: PatternFormatValidatorDependencies = {
+ createSyntaxRunner: (label) =>
+ new SyntaxSupervisor(
+ `${label} formatter syntax`,
+ `regex-tools-formatter-${label}-syntax`,
+ ),
+ createEngineRunner: () => new EngineSupervisor(),
+ now: () => performance.now(),
+};
+
+interface CompleteRuntime {
+ readonly status: "complete";
+ readonly semantic: SemanticSideResult;
+ readonly assertionPassed?: boolean;
+ readonly assertionMessage?: string;
+}
+
+interface FailedRuntime {
+ readonly status: "timeout" | "cancelled" | "crash" | "worker-error";
+ readonly message: string;
+ readonly assertionPassed?: boolean;
+}
+
+type Runtime = CompleteRuntime | FailedRuntime;
+
+function stableValue(value: unknown): unknown {
+ if (Array.isArray(value)) return value.map(stableValue);
+ if (value && typeof value === "object") {
+ return Object.fromEntries(
+ Object.entries(value)
+ .sort(([left], [right]) => left.localeCompare(right))
+ .map(([key, entry]) => [key, stableValue(entry)]),
+ );
+ }
+ return value;
+}
+
+function stableJson(value: unknown): string {
+ return JSON.stringify(stableValue(value));
+}
+
+function boundedInteger(
+ value: number,
+ label: string,
+ minimum: number,
+ maximum: number,
+): number {
+ if (!Number.isSafeInteger(value) || value < minimum || value > maximum) {
+ throw new RangeError(
+ `${label} must be an integer from ${minimum.toLocaleString()} to ${maximum.toLocaleString()}.`,
+ );
+ }
+ return value;
+}
+
+function normalizedVersion(value: string | undefined): string {
+ const definition = AVAILABLE_REGEX_FLAVOURS.require("ecmascript");
+ return resolveRegexFlavourVersion(
+ value,
+ definition,
+ "ECMAScript formatter version",
+ ).definition.value;
+}
+
+function validateInput(
+ input: PatternFormatValidationInput,
+): PatternFormatValidationInput {
+ const definition = AVAILABLE_REGEX_FLAVOURS.require("ecmascript");
+ if (input.flavour !== "ecmascript") {
+ throw new Error("Only ECMAScript formatting is implemented.");
+ }
+ if (input.sourcePattern === input.formattedPattern) {
+ throw new Error("The formatter candidate does not change the pattern.");
+ }
+ if (
+ input.sourcePattern.length > DEFAULT_REGEX_LIMITS.patternHardLengthUtf16 ||
+ input.formattedPattern.length > DEFAULT_REGEX_LIMITS.patternHardLengthUtf16
+ ) {
+ throw new RangeError(
+ `Both patterns must fit the ${DEFAULT_REGEX_LIMITS.patternHardLengthUtf16.toLocaleString()} UTF-16 unit limit.`,
+ );
+ }
+ if (
+ utf8ByteLength(input.subject) >
+ DEFAULT_REGEX_LIMITS.interactiveSubjectHardBytes
+ ) {
+ throw new RangeError(
+ `Formatter validation subject exceeds the ${DEFAULT_REGEX_LIMITS.interactiveSubjectHardBytes.toLocaleString()} UTF-8 byte limit.`,
+ );
+ }
+ if (
+ input.replacement.length >
+ DEFAULT_REGEX_LIMITS.maximumReplacementTemplateUtf16
+ ) {
+ throw new RangeError(
+ `Replacement exceeds the ${DEFAULT_REGEX_LIMITS.maximumReplacementTemplateUtf16.toLocaleString()} UTF-16 unit limit.`,
+ );
+ }
+ if (input.tests.length > MAXIMUM_REGEX_TESTS) {
+ throw new RangeError(
+ `Formatter validation accepts at most ${MAXIMUM_REGEX_TESTS.toLocaleString()} unit tests.`,
+ );
+ }
+ const flags = parseRegexFlags(
+ input.flags,
+ definition,
+ "ECMAScript formatter flags",
+ );
+ const options = parseRegexOptions(
+ input.options,
+ definition,
+ "ECMAScript formatter options",
+ );
+ const timeoutMs = boundedInteger(
+ input.timeoutMs,
+ "Formatter worker timeout",
+ 1,
+ DEFAULT_REGEX_LIMITS.advancedMaximumTimeoutMs,
+ );
+ return {
+ ...input,
+ flavourVersion: normalizedVersion(input.flavourVersion),
+ flags,
+ options,
+ timeoutMs,
+ };
+}
+
+function syntaxRequest(
+ input: PatternFormatValidationInput,
+ pattern: string,
+): RegexSyntaxRequest {
+ const definition = AVAILABLE_REGEX_FLAVOURS.require("ecmascript");
+ const version = resolveRegexFlavourVersion(
+ input.flavourVersion,
+ definition,
+ "ECMAScript formatter version",
+ ).definition;
+ return {
+ flavour: "ecmascript",
+ flavourVersion: version.syntaxVersion,
+ pattern,
+ flags: input.flags,
+ options: input.options,
+ };
+}
+
+function captureShape(
+ captures: readonly CaptureDefinition[],
+): readonly string[] {
+ return captures.map((capture) =>
+ [
+ capture.number,
+ capture.name ?? "",
+ capture.repeated ? "repeated" : "single",
+ capture.parentCaptureNumber ?? "",
+ ].join(":"),
+ );
+}
+
+function sameCaptureShape(
+ left: readonly CaptureDefinition[],
+ right: readonly CaptureDefinition[],
+): boolean {
+ const leftShape = captureShape(left);
+ const rightShape = captureShape(right);
+ return (
+ leftShape.length === rightShape.length &&
+ leftShape.every((value, index) => value === rightShape[index])
+ );
+}
+
+function classifyFailure(error: unknown): FailedRuntime {
+ if (error instanceof WorkerRequestError) {
+ return { status: error.kind, message: error.message };
+ }
+ return {
+ status: "worker-error",
+ message: error instanceof Error ? error.message : String(error),
+ };
+}
+
+function runtimeObservation(runtime: Runtime): PatternFormatRuntimeObservation {
+ if (runtime.status !== "complete") {
+ return {
+ status: runtime.status,
+ ...(runtime.assertionPassed === undefined
+ ? {}
+ : { assertionPassed: runtime.assertionPassed }),
+ message: runtime.message,
+ };
+ }
+ return {
+ status: "complete",
+ engineIdentity: engineIdentity(runtime.semantic.execution),
+ ...(runtime.assertionPassed === undefined
+ ? {}
+ : { assertionPassed: runtime.assertionPassed }),
+ message:
+ runtime.assertionMessage ??
+ "The actual ECMAScript engine completed with a bounded result.",
+ };
+}
+
+function executionRequest(
+ input: PatternFormatValidationInput,
+ pattern: string,
+ subject: string,
+ captures: readonly CaptureDefinition[],
+ scanAll: boolean,
+): RegexExecutionRequest {
+ return {
+ flavour: "ecmascript",
+ ...(input.flavourVersion === undefined
+ ? {}
+ : { flavourVersion: input.flavourVersion }),
+ pattern,
+ flags: input.flags,
+ options: input.options,
+ subject,
+ captureMetadata: captures,
+ scanAll,
+ maximumMatches: DEFAULT_REGEX_LIMITS.maximumMatches,
+ maximumCaptureRows: DEFAULT_REGEX_LIMITS.maximumCaptureRows,
+ };
+}
+
+async function runCurrent(
+ runner: PatternFormatEngineRunner,
+ input: PatternFormatValidationInput,
+ pattern: string,
+ captures: readonly CaptureDefinition[],
+): Promise {
+ try {
+ const replacement = await runner.replace(
+ {
+ ...executionRequest(
+ input,
+ pattern,
+ input.subject,
+ captures,
+ input.scanAll,
+ ),
+ replacement: input.replacement,
+ maximumOutputBytes: DEFAULT_REGEX_LIMITS.maximumReplacementOutputBytes,
+ },
+ input.timeoutMs,
+ );
+ return {
+ status: "complete",
+ semantic: { execution: replacement.execution, replacement },
+ };
+ } catch (error) {
+ return classifyFailure(error);
+ }
+}
+
+async function runTest(
+ runner: PatternFormatEngineRunner,
+ input: PatternFormatValidationInput,
+ pattern: string,
+ captures: readonly CaptureDefinition[],
+ test: RegexTestCase,
+ timeoutMs: number,
+): Promise {
+ const request = executionRequest(
+ input,
+ pattern,
+ test.subject,
+ captures,
+ test.scanAll ?? false,
+ );
+ try {
+ let semantic: SemanticSideResult;
+ if (test.expectation.kind === "replacement") {
+ const replacement = await runner.replace(
+ {
+ ...request,
+ replacement: test.replacement ?? "",
+ maximumOutputBytes:
+ DEFAULT_REGEX_LIMITS.maximumReplacementOutputBytes,
+ },
+ timeoutMs,
+ );
+ semantic = { execution: replacement.execution, replacement };
+ } else {
+ semantic = {
+ execution: await runner.execute(request, timeoutMs),
+ };
+ }
+ const assertion = evaluateUnitTestFailure(
+ test.expectation,
+ semantic.execution,
+ semantic.replacement,
+ test.subject,
+ );
+ return {
+ status: "complete",
+ semantic,
+ ...(assertion.kind === "passes"
+ ? { assertionPassed: true }
+ : assertion.kind === "failure"
+ ? { assertionPassed: false }
+ : {}),
+ assertionMessage: assertion.summary,
+ };
+ } catch (error) {
+ const failure = classifyFailure(error);
+ return {
+ ...failure,
+ ...(failure.status === "timeout"
+ ? { assertionPassed: test.expectation.kind === "must-time-out" }
+ : {}),
+ };
+ }
+}
+
+function compareRuntimes(
+ before: Runtime,
+ after: Runtime,
+ allowMatchingTimeout: boolean,
+): PatternFormatSnapshotCheck {
+ const observations = {
+ before: runtimeObservation(before),
+ after: runtimeObservation(after),
+ };
+ if (before.status === "complete" && after.status === "complete") {
+ const beforeIdentity = engineIdentity(before.semantic.execution);
+ const afterIdentity = engineIdentity(after.semantic.execution);
+ if (beforeIdentity !== afterIdentity) {
+ return {
+ comparison: "inconclusive",
+ ...observations,
+ summary: `Engine identity changed from ${beforeIdentity} to ${afterIdentity}.`,
+ mismatchKinds: ["engine-identity"],
+ };
+ }
+ const semantic = compareSemanticResults(before.semantic, after.semantic);
+ if (semantic.kind !== "same") {
+ return {
+ comparison: semantic.kind === "mismatch" ? "different" : "inconclusive",
+ ...observations,
+ summary: semantic.summary,
+ mismatchKinds: semantic.mismatchKinds,
+ };
+ }
+ if (before.assertionPassed !== after.assertionPassed) {
+ return {
+ comparison: "different",
+ ...observations,
+ summary: "The unit-test assertion outcome changed after formatting.",
+ mismatchKinds: ["unit-test-outcome"],
+ };
+ }
+ return {
+ comparison: "same",
+ ...observations,
+ summary:
+ before.assertionPassed === undefined
+ ? semantic.summary
+ : `Exact semantic output is unchanged; the assertion ${
+ before.assertionPassed ? "passes" : "fails"
+ } on both snapshots.`,
+ mismatchKinds: [],
+ };
+ }
+ if (
+ allowMatchingTimeout &&
+ before.status === "timeout" &&
+ after.status === "timeout" &&
+ before.assertionPassed === after.assertionPassed
+ ) {
+ return {
+ comparison: "same",
+ ...observations,
+ summary:
+ "Both exact test snapshots reached the same fixed worker timeout and retained the same assertion outcome.",
+ mismatchKinds: [],
+ };
+ }
+ if (
+ (before.status === "complete" && after.status === "timeout") ||
+ (before.status === "timeout" && after.status === "complete")
+ ) {
+ return {
+ comparison: "different",
+ ...observations,
+ summary:
+ "One exact snapshot completed while the other reached the fixed worker timeout.",
+ mismatchKinds: ["timeout-behavior"],
+ };
+ }
+ return {
+ comparison: "inconclusive",
+ ...observations,
+ summary: `Validation did not produce two authoritative results (${before.status} versus ${after.status}).`,
+ mismatchKinds: ["runtime-inconclusive"],
+ };
+}
+
+function testIsApplicable(
+ input: PatternFormatValidationInput,
+ test: RegexTestCase,
+): boolean {
+ return (
+ test.enabled &&
+ test.flavour === "ecmascript" &&
+ normalizedVersion(test.flavourVersion) ===
+ normalizedVersion(input.flavourVersion) &&
+ test.pattern === input.sourcePattern &&
+ test.flags.join("") === input.flags.join("") &&
+ stableJson(test.options) === stableJson(input.options)
+ );
+}
+
+function unavailableCheck(summary: string): PatternFormatSnapshotCheck {
+ const observation: PatternFormatRuntimeObservation = {
+ status: "worker-error",
+ message: "Engine validation was not started.",
+ };
+ return {
+ comparison: "inconclusive",
+ before: observation,
+ after: observation,
+ summary,
+ mismatchKinds: ["validation-not-run"],
+ };
+}
+
+export class PatternFormatValidator {
+ readonly #dependencies: PatternFormatValidatorDependencies;
+ readonly #sourceSyntax: PatternFormatSyntaxRunner;
+ readonly #formattedSyntax: PatternFormatSyntaxRunner;
+ readonly #sourceEngine: PatternFormatEngineRunner;
+ readonly #formattedEngine: PatternFormatEngineRunner;
+ #generation = 0;
+ #disposed = false;
+
+ constructor(
+ dependencies: PatternFormatValidatorDependencies = DEFAULT_DEPENDENCIES,
+ ) {
+ this.#dependencies = dependencies;
+ this.#sourceSyntax = dependencies.createSyntaxRunner("source");
+ this.#formattedSyntax = dependencies.createSyntaxRunner("formatted");
+ this.#sourceEngine = dependencies.createEngineRunner("source");
+ this.#formattedEngine = dependencies.createEngineRunner("formatted");
+ }
+
+ async validate(
+ rawInput: PatternFormatValidationInput,
+ onProgress?: (completed: number, total: number) => void,
+ ): Promise {
+ if (this.#disposed) {
+ throw new Error("Pattern formatter validator has been disposed.");
+ }
+ this.cancel();
+ const generation = this.#generation;
+ const input = validateInput(rawInput);
+ const started = this.#dependencies.now();
+ const elapsed = () => Math.max(0, this.#dependencies.now() - started);
+ const assertCurrent = () => {
+ if (generation !== this.#generation || this.#disposed) {
+ throw new WorkerRequestError(
+ "cancelled",
+ "Pattern formatter validation was cancelled.",
+ );
+ }
+ };
+ const differences: PatternFormatDifference[] = [];
+ const notices: string[] = [
+ "Validation executes exact source and formatted snapshots in separate actual-engine workers. It is bounded evidence for the selected subject and applicable tests, not a proof over every possible input.",
+ ];
+ const addDifference = (difference: PatternFormatDifference) => {
+ if (differences.length < MAXIMUM_RETAINED_DIFFERENCES) {
+ differences.push(difference);
+ }
+ };
+
+ const syntaxTimeout = Math.min(1_000, input.timeoutMs);
+ const [sourceSyntax, formattedSyntax] = await Promise.all([
+ this.#sourceSyntax.parsePattern(
+ syntaxRequest(input, input.sourcePattern),
+ syntaxTimeout,
+ ),
+ this.#formattedSyntax.parsePattern(
+ syntaxRequest(input, input.formattedPattern),
+ syntaxTimeout,
+ ),
+ ]);
+ assertCurrent();
+ const captureShapePreserved =
+ sourceSyntax.accepted &&
+ formattedSyntax.accepted &&
+ sameCaptureShape(sourceSyntax.captures, formattedSyntax.captures);
+ if (!sourceSyntax.accepted) {
+ addDifference({
+ scope: "syntax",
+ code: "source-syntax-rejected",
+ summary:
+ sourceSyntax.diagnostics[0]?.message ??
+ "The syntax provider rejected the source pattern.",
+ });
+ }
+ if (!formattedSyntax.accepted) {
+ addDifference({
+ scope: "syntax",
+ code: "formatted-syntax-rejected",
+ summary:
+ formattedSyntax.diagnostics[0]?.message ??
+ "The syntax provider rejected the formatted pattern.",
+ });
+ }
+ if (
+ sourceSyntax.accepted &&
+ formattedSyntax.accepted &&
+ !captureShapePreserved
+ ) {
+ addDifference({
+ scope: "syntax",
+ code: "capture-shape-changed",
+ summary:
+ "Capture numbering, naming, repetition or parent structure changed after formatting.",
+ });
+ }
+ if (
+ !sourceSyntax.accepted ||
+ !formattedSyntax.accepted ||
+ !captureShapePreserved
+ ) {
+ const current = unavailableCheck(
+ "Engine validation was not started because the syntax gate did not preserve an accepted capture structure.",
+ );
+ return {
+ status:
+ sourceSyntax.accepted !== formattedSyntax.accepted ||
+ !captureShapePreserved
+ ? "different"
+ : "inconclusive",
+ canApply: false,
+ sourceSyntaxAccepted: sourceSyntax.accepted,
+ formattedSyntaxAccepted: formattedSyntax.accepted,
+ captureShapePreserved,
+ current,
+ tests: [],
+ applicableTestCount: 0,
+ skippedTestCount: input.tests.length,
+ completedTestCount: 0,
+ differences,
+ elapsedMs: elapsed(),
+ notices,
+ };
+ }
+
+ const applicable = input.tests.filter((test) =>
+ testIsApplicable(input, test),
+ );
+ const skippedTestCount = input.tests.length - applicable.length;
+ onProgress?.(0, applicable.length);
+ const [sourceCurrent, formattedCurrent] = await Promise.all([
+ runCurrent(
+ this.#sourceEngine,
+ input,
+ input.sourcePattern,
+ sourceSyntax.captures,
+ ),
+ runCurrent(
+ this.#formattedEngine,
+ input,
+ input.formattedPattern,
+ formattedSyntax.captures,
+ ),
+ ]);
+ assertCurrent();
+ const current = compareRuntimes(sourceCurrent, formattedCurrent, false);
+ if (current.comparison !== "same") {
+ addDifference({
+ scope: "current-subject",
+ code:
+ current.comparison === "different"
+ ? "current-behavior-changed"
+ : "current-behavior-inconclusive",
+ summary: current.summary,
+ });
+ }
+
+ const testChecks: PatternFormatTestCheck[] = [];
+ let suiteIncomplete = false;
+ for (const test of applicable) {
+ assertCurrent();
+ const testTimeout = boundedInteger(
+ test.resourceLimits?.manualExecutionTimeoutMs ?? input.timeoutMs,
+ `Unit-test timeout for ${test.name}`,
+ 1,
+ DEFAULT_REGEX_LIMITS.advancedMaximumTimeoutMs,
+ );
+ if (elapsed() + testTimeout > MAXIMUM_VALIDATION_WALL_TIME_MS) {
+ suiteIncomplete = true;
+ notices.push(
+ `Unit-test validation stopped before ${test.name} because its full ${testTimeout.toLocaleString()} ms worker timeout no longer fit the ${MAXIMUM_VALIDATION_WALL_TIME_MS / 1_000} second aggregate wall budget.`,
+ );
+ break;
+ }
+ const [sourceTest, formattedTest] = await Promise.all([
+ runTest(
+ this.#sourceEngine,
+ input,
+ input.sourcePattern,
+ sourceSyntax.captures,
+ test,
+ testTimeout,
+ ),
+ runTest(
+ this.#formattedEngine,
+ input,
+ input.formattedPattern,
+ formattedSyntax.captures,
+ test,
+ testTimeout,
+ ),
+ ]);
+ assertCurrent();
+ const comparison = compareRuntimes(sourceTest, formattedTest, true);
+ const check: PatternFormatTestCheck = {
+ testId: test.id,
+ name: test.name,
+ ...comparison,
+ };
+ testChecks.push(check);
+ if (check.comparison !== "same") {
+ addDifference({
+ scope: "unit-test",
+ code:
+ check.comparison === "different"
+ ? "test-behavior-changed"
+ : "test-behavior-inconclusive",
+ summary: `${test.name}: ${check.summary}`,
+ testId: test.id,
+ });
+ }
+ onProgress?.(testChecks.length, applicable.length);
+ }
+ if (testChecks.length < applicable.length) suiteIncomplete = true;
+ if (applicable.length === 0) {
+ notices.push(
+ "No enabled unit test has the exact active pattern, version, flags and options; only the current subject/replacement snapshot was compared.",
+ );
+ } else {
+ notices.push(
+ `${testChecks.length.toLocaleString()} of ${applicable.length.toLocaleString()} applicable enabled unit tests completed on both exact snapshots.`,
+ );
+ }
+
+ const hasDifferent =
+ current.comparison === "different" ||
+ testChecks.some((test) => test.comparison === "different");
+ const hasInconclusive =
+ current.comparison === "inconclusive" ||
+ suiteIncomplete ||
+ testChecks.some((test) => test.comparison === "inconclusive");
+ const status = hasDifferent
+ ? "different"
+ : hasInconclusive
+ ? "inconclusive"
+ : "equivalent";
+ return {
+ status,
+ canApply: status === "equivalent",
+ sourceSyntaxAccepted: true,
+ formattedSyntaxAccepted: true,
+ captureShapePreserved: true,
+ current,
+ tests: testChecks,
+ applicableTestCount: applicable.length,
+ skippedTestCount,
+ completedTestCount: testChecks.length,
+ differences,
+ elapsedMs: elapsed(),
+ notices,
+ };
+ }
+
+ cancel(): void {
+ this.#generation += 1;
+ this.#sourceSyntax.cancel();
+ this.#formattedSyntax.cancel();
+ this.#sourceEngine.cancel();
+ this.#formattedEngine.cancel();
+ }
+
+ dispose(): void {
+ if (this.#disposed) return;
+ this.cancel();
+ this.#disposed = true;
+ this.#sourceSyntax.dispose();
+ this.#formattedSyntax.dispose();
+ this.#sourceEngine.dispose();
+ this.#formattedEngine.dispose();
+ }
+}
diff --git a/src/regex/formatting/ecmascript-format.test.ts b/src/regex/formatting/ecmascript-format.test.ts
new file mode 100644
index 0000000..d9e8d6f
--- /dev/null
+++ b/src/regex/formatting/ecmascript-format.test.ts
@@ -0,0 +1,125 @@
+import { describe, expect, it } from "vitest";
+import { EcmaScriptSyntaxProvider } from "../syntax/providers/ecmascript/EcmaScriptSyntaxProvider";
+import { formatEcmaScriptPattern } from "./ecmascript-format";
+
+const provider = new EcmaScriptSyntaxProvider();
+
+async function format(pattern: string, flags: readonly string[] = []) {
+ const syntax = await provider.parsePattern({
+ flavour: "ecmascript",
+ pattern,
+ flags,
+ options: {},
+ });
+ expect(syntax.accepted).toBe(true);
+ return formatEcmaScriptPattern(pattern, syntax);
+}
+
+describe("formatEcmaScriptPattern", () => {
+ it("canonicalizes literal slashes and raw controls without changing structure", async () => {
+ const source =
+ "^(a/b)\u0000\t\n\r\f\v\u0008\u001f\u007f\u2028\u2029(? z )$";
+ const preview = await format(source, ["u"]);
+
+ expect(preview.formattedPattern).toBe(
+ "^(a\\/b)\\x00\\t\\n\\r\\f\\v\\x08\\x1f\\x7f\\u2028\\u2029(? z )$",
+ );
+ expect(preview.totalChanges).toBe(12);
+ expect(preview.changes.map((change) => change.kind)).toEqual(
+ expect.arrayContaining([
+ "literal-slash",
+ "raw-control",
+ "raw-line-terminator",
+ ]),
+ );
+ expect(preview.withinPatternLimit).toBe(true);
+ });
+
+ it("preserves existing escapes and is idempotent", async () => {
+ const source = String.raw`a\/b\\\/c\t\n\x00\u2028[ / ]`;
+ const first = await format(source, ["u"]);
+ const secondSyntax = await provider.parsePattern({
+ flavour: "ecmascript",
+ pattern: first.formattedPattern,
+ flags: ["u"],
+ options: {},
+ });
+ const second = formatEcmaScriptPattern(
+ first.formattedPattern,
+ secondSyntax,
+ );
+
+ expect(first.formattedPattern).toBe(
+ String.raw`a\/b\\\/c\t\n\x00\u2028[ \/ ]`,
+ );
+ expect(second.formattedPattern).toBe(first.formattedPattern);
+ expect(second.changed).toBe(false);
+ });
+
+ it("canonicalizes legacy escaped raw controls without adding a backslash", async () => {
+ const source = "\\\t[\\\u0000]";
+ const preview = await format(source);
+
+ expect(preview.formattedPattern).toBe("\\t[\\x00]");
+ expect(preview.changes).toEqual([
+ expect.objectContaining({
+ before: "\\\t",
+ after: "\\t",
+ startUtf16: 0,
+ endUtf16: 2,
+ }),
+ expect.objectContaining({
+ before: "\\\u0000",
+ after: "\\x00",
+ startUtf16: 3,
+ endUtf16: 5,
+ }),
+ ]);
+ for (const subject of ["\t\u0000", "\t", "\\t\u0000"]) {
+ expect(new RegExp(preview.formattedPattern).exec(subject)).toEqual(
+ new RegExp(source).exec(subject),
+ );
+ }
+ });
+
+ it.each(["", "u", "v"])(
+ "retains native match behavior under %j flags",
+ async (flagText) => {
+ const flags = flagText ? [flagText] : [];
+ const source = "^(?a/b)\t[\u0000-\u001f]\u2028$";
+ const preview = await format(source, flags);
+ const before = new RegExp(source, flagText);
+ const after = new RegExp(preview.formattedPattern, flagText);
+ for (const subject of [
+ "a/b\t\u0000\u2028",
+ "a/b\t\u001f\u2028",
+ "a/b x",
+ "a\\/b\t\u0000\u2028",
+ ]) {
+ expect(after.exec(subject)).toEqual(before.exec(subject));
+ }
+ },
+ );
+
+ it("rejects recovered or stale syntax snapshots", async () => {
+ const rejected = await provider.parsePattern({
+ flavour: "ecmascript",
+ pattern: "(",
+ flags: [],
+ options: {},
+ });
+ expect(() => formatEcmaScriptPattern("(", rejected)).toThrow(
+ "accepted, non-recovered",
+ );
+
+ const accepted = await provider.parsePattern({
+ flavour: "ecmascript",
+ pattern: "a",
+ flags: [],
+ options: {},
+ });
+ expect(() => formatEcmaScriptPattern("b", accepted)).toThrow(
+ "accepted, non-recovered",
+ );
+ });
+});
diff --git a/src/regex/formatting/ecmascript-format.ts b/src/regex/formatting/ecmascript-format.ts
new file mode 100644
index 0000000..5fb1cc9
--- /dev/null
+++ b/src/regex/formatting/ecmascript-format.ts
@@ -0,0 +1,190 @@
+import { DEFAULT_REGEX_LIMITS } from "../execution/request-limits";
+import type { RegexSyntaxResult } from "../model/syntax";
+import {
+ ECMASCRIPT_FORMATTER_ID,
+ ECMASCRIPT_FORMATTER_VERSION,
+ type PatternFormatChange,
+ type PatternFormatChangeKind,
+ type PatternFormatPreview,
+} from "./formatting.types";
+
+const MAXIMUM_RETAINED_CHANGES = 10_000;
+
+interface EscapeReplacement {
+ readonly kind: PatternFormatChangeKind;
+ readonly after: string;
+ readonly explanation: string;
+}
+
+const CONTROL_ESCAPES: Readonly> = {
+ 0x00: "\\x00",
+ 0x08: "\\x08",
+ 0x09: "\\t",
+ 0x0a: "\\n",
+ 0x0b: "\\v",
+ 0x0c: "\\f",
+ 0x0d: "\\r",
+} as const;
+
+function hexEscape(codeUnit: number): string {
+ return `\\x${codeUnit.toString(16).padStart(2, "0")}`;
+}
+
+function literalReplacement(raw: string): EscapeReplacement | undefined {
+ const lastIndex = raw.length - 1;
+ const codeUnit = raw.charCodeAt(lastIndex);
+ if (codeUnit === 0x2f) {
+ if (raw !== "/") return undefined;
+ return {
+ kind: "literal-slash",
+ after: "\\/",
+ explanation:
+ "Escaped a literal slash so the pattern is safe to copy into an ECMAScript regex literal.",
+ };
+ }
+ if (codeUnit === 0x2028 || codeUnit === 0x2029) {
+ return {
+ kind: "raw-line-terminator",
+ after: `\\u${codeUnit.toString(16).padStart(4, "0")}`,
+ explanation:
+ "Replaced a raw ECMAScript line-separator code point with its equivalent Unicode escape.",
+ };
+ }
+ if (codeUnit <= 0x1f || codeUnit === 0x7f) {
+ return {
+ kind: "raw-control",
+ after: CONTROL_ESCAPES[codeUnit] ?? hexEscape(codeUnit),
+ explanation:
+ "Replaced a raw control code unit with its equivalent explicit ECMAScript escape.",
+ };
+ }
+ return undefined;
+}
+
+interface LocatedReplacement extends EscapeReplacement {
+ readonly startUtf16: number;
+ readonly endUtf16: number;
+ readonly before: string;
+}
+
+function replacementsFromSyntax(
+ pattern: string,
+ syntax: RegexSyntaxResult,
+): readonly LocatedReplacement[] {
+ return syntax.tokens
+ .filter(
+ (token) => token.kind === "literal" || token.kind === "escaped-literal",
+ )
+ .flatMap((token): readonly LocatedReplacement[] => {
+ const before = pattern.slice(
+ token.range.startUtf16,
+ token.range.endUtf16,
+ );
+ if (before !== token.raw) return [];
+ const replacement = literalReplacement(before);
+ return replacement
+ ? [
+ {
+ ...replacement,
+ startUtf16: token.range.startUtf16,
+ endUtf16: token.range.endUtf16,
+ before,
+ },
+ ]
+ : [];
+ })
+ .sort(
+ (left, right) =>
+ left.startUtf16 - right.startUtf16 || left.endUtf16 - right.endUtf16,
+ );
+}
+
+function assertEligibleSyntax(
+ pattern: string,
+ syntax: RegexSyntaxResult,
+): void {
+ if (
+ !syntax.accepted ||
+ syntax.root.support.flavour !== "ecmascript" ||
+ syntax.root.provenance.source === "recovered" ||
+ syntax.root.raw !== pattern
+ ) {
+ throw new Error(
+ "ECMAScript formatting requires the current accepted, non-recovered syntax-provider snapshot.",
+ );
+ }
+ if (syntax.provider.id !== "regexpp-community") {
+ throw new Error(
+ `The active syntax provider ${syntax.provider.id} does not advertise this ECMAScript formatter.`,
+ );
+ }
+}
+
+/**
+ * Canonicalizes only raw characters whose escape is context-independent.
+ * Existing explicit escape spelling and every structural token remain byte-for-byte
+ * untouched. Reparse/engine/test validation is a separate mandatory gate.
+ */
+export function formatEcmaScriptPattern(
+ pattern: string,
+ syntax: RegexSyntaxResult,
+): PatternFormatPreview {
+ assertEligibleSyntax(pattern, syntax);
+ const output: string[] = [];
+ const changes: PatternFormatChange[] = [];
+ const replacements = replacementsFromSyntax(pattern, syntax);
+ let totalChanges = 0;
+ let cursor = 0;
+ for (const replacement of replacements) {
+ if (replacement.startUtf16 < cursor) continue;
+ output.push(pattern.slice(cursor, replacement.startUtf16));
+ output.push(replacement.after);
+ cursor = replacement.endUtf16;
+ totalChanges += 1;
+ if (changes.length < MAXIMUM_RETAINED_CHANGES) {
+ changes.push({
+ kind: replacement.kind,
+ startUtf16: replacement.startUtf16,
+ endUtf16: replacement.endUtf16,
+ before: replacement.before,
+ after: replacement.after,
+ explanation: replacement.explanation,
+ });
+ }
+ }
+ output.push(pattern.slice(cursor));
+ const formattedPattern = output.join("");
+ const withinPatternLimit =
+ formattedPattern.length <= DEFAULT_REGEX_LIMITS.patternHardLengthUtf16;
+ const warnings: string[] = [
+ "Formatting preserves existing explicit escape spelling, grouping, alternation, quantifiers, inline flags and literal whitespace. It does not pretty-print by inserting whitespace or comments.",
+ "Successful bounded validation is evidence for the checked subject and tests, not a proof of equivalence for every possible subject.",
+ ];
+ if (!withinPatternLimit) {
+ warnings.push(
+ `The formatted candidate is ${formattedPattern.length.toLocaleString()} UTF-16 units and exceeds the ${DEFAULT_REGEX_LIMITS.patternHardLengthUtf16.toLocaleString()} unit pattern limit.`,
+ );
+ }
+ if (totalChanges > changes.length) {
+ warnings.push(
+ `Change details retain the first ${changes.length.toLocaleString()} of ${totalChanges.toLocaleString()} transformations.`,
+ );
+ }
+ return {
+ formatter: {
+ id: ECMASCRIPT_FORMATTER_ID,
+ version: ECMASCRIPT_FORMATTER_VERSION,
+ flavour: "ecmascript",
+ syntaxProvider: syntax.provider.id,
+ syntaxProviderVersion: syntax.provider.version,
+ },
+ sourcePattern: pattern,
+ formattedPattern,
+ changed: totalChanges > 0,
+ changes,
+ totalChanges,
+ changesTruncated: totalChanges > changes.length,
+ withinPatternLimit,
+ warnings,
+ };
+}
diff --git a/src/regex/formatting/formatting.types.ts b/src/regex/formatting/formatting.types.ts
new file mode 100644
index 0000000..407ce03
--- /dev/null
+++ b/src/regex/formatting/formatting.types.ts
@@ -0,0 +1,98 @@
+import type { RegexEngineOptions } from "../model/flavour";
+import type { RegexTestCase } from "../tests/test-case.types";
+
+export const ECMASCRIPT_FORMATTER_ID = "regex-tools-ecmascript-literal-escapes";
+export const ECMASCRIPT_FORMATTER_VERSION = "1";
+
+export type PatternFormatChangeKind =
+ "literal-slash" | "raw-control" | "raw-line-terminator";
+
+export interface PatternFormatChange {
+ readonly kind: PatternFormatChangeKind;
+ readonly startUtf16: number;
+ readonly endUtf16: number;
+ readonly before: string;
+ readonly after: string;
+ readonly explanation: string;
+}
+
+export interface PatternFormatPreview {
+ readonly formatter: {
+ readonly id: typeof ECMASCRIPT_FORMATTER_ID;
+ readonly version: typeof ECMASCRIPT_FORMATTER_VERSION;
+ readonly flavour: "ecmascript";
+ readonly syntaxProvider: string;
+ readonly syntaxProviderVersion: string;
+ };
+ readonly sourcePattern: string;
+ readonly formattedPattern: string;
+ readonly changed: boolean;
+ readonly changes: readonly PatternFormatChange[];
+ readonly totalChanges: number;
+ readonly changesTruncated: boolean;
+ readonly withinPatternLimit: boolean;
+ readonly warnings: readonly string[];
+}
+
+export interface PatternFormatValidationInput {
+ readonly flavour: "ecmascript";
+ readonly flavourVersion?: string;
+ readonly sourcePattern: string;
+ readonly formattedPattern: string;
+ readonly flags: readonly string[];
+ readonly options: RegexEngineOptions;
+ readonly subject: string;
+ readonly replacement: string;
+ readonly scanAll: boolean;
+ readonly timeoutMs: number;
+ readonly tests: readonly RegexTestCase[];
+}
+
+export type PatternFormatRuntimeStatus =
+ "complete" | "timeout" | "cancelled" | "crash" | "worker-error";
+
+export interface PatternFormatRuntimeObservation {
+ readonly status: PatternFormatRuntimeStatus;
+ readonly engineIdentity?: string;
+ readonly assertionPassed?: boolean;
+ readonly message: string;
+}
+
+export type PatternFormatComparisonStatus =
+ "same" | "different" | "inconclusive";
+
+export interface PatternFormatSnapshotCheck {
+ readonly comparison: PatternFormatComparisonStatus;
+ readonly before: PatternFormatRuntimeObservation;
+ readonly after: PatternFormatRuntimeObservation;
+ readonly summary: string;
+ readonly mismatchKinds: readonly string[];
+}
+
+export interface PatternFormatTestCheck extends PatternFormatSnapshotCheck {
+ readonly testId: string;
+ readonly name: string;
+}
+
+export interface PatternFormatDifference {
+ readonly scope: "syntax" | "current-subject" | "unit-test";
+ readonly code: string;
+ readonly summary: string;
+ readonly testId?: string;
+}
+
+export interface PatternFormatValidationResult {
+ readonly status: "equivalent" | "different" | "inconclusive";
+ readonly canApply: boolean;
+ readonly sourceSyntaxAccepted: boolean;
+ readonly formattedSyntaxAccepted: boolean;
+ readonly captureShapePreserved: boolean;
+ readonly current: PatternFormatSnapshotCheck;
+ readonly tests: readonly PatternFormatTestCheck[];
+ readonly applicableTestCount: number;
+ readonly skippedTestCount: number;
+ readonly completedTestCount: number;
+ readonly differences: readonly PatternFormatDifference[];
+ readonly elapsedMs: number;
+ readonly notices: readonly string[];
+}
diff --git a/src/regex/generation/CaseGenerationOrchestrator.test.ts b/src/regex/generation/CaseGenerationOrchestrator.test.ts
new file mode 100644
index 0000000..555dc38
--- /dev/null
+++ b/src/regex/generation/CaseGenerationOrchestrator.test.ts
@@ -0,0 +1,455 @@
+import { describe, expect, it } from "vitest";
+import { WorkerRequestError } from "../execution/WorkerSupervisor";
+import { EcmaScriptEngineAdapter } from "../execution/adapters/ecmascript/EcmaScriptEngineAdapter";
+import type { RegexEngineInfo } from "../model/flavour";
+import type {
+ RegexExecutionRequest,
+ RegexExecutionResult,
+} from "../model/match";
+import type { NormalizedRegexNode } from "../model/syntax";
+import { CaseGenerationOrchestrator } from "./CaseGenerationOrchestrator";
+import { DEFAULT_GENERATION_SETTINGS } from "./generation-limits";
+import { generateCandidateSubjects } from "./generate-candidates";
+import type {
+ CandidateGenerationReport,
+ CaseGenerationInput,
+ GeneratedSubjectCandidate,
+} from "./generation.types";
+import type { GenerationWorkerClient } from "./GenerationSupervisor";
+import { EcmaScriptSyntaxProvider } from "../syntax/providers/ecmascript/EcmaScriptSyntaxProvider";
+
+const ENGINE: RegexEngineInfo = {
+ flavour: "ecmascript",
+ adapterVersion: "fixture-1",
+ engineName: "Native ECMAScript RegExp",
+ engineVersion: "fixture",
+ runtimeVersion: "test",
+ offsetUnit: "utf16",
+ capabilities: {
+ compilation: true,
+ matching: true,
+ replacement: true,
+ namedCaptures: true,
+ captureHistory: false,
+ actualTrace: false,
+ benchmark: true,
+ },
+};
+
+function root(): NormalizedRegexNode {
+ return {
+ id: "root",
+ kind: "pattern",
+ range: { startUtf16: 0, endUtf16: 3 },
+ raw: "yes",
+ explanation: "fixture",
+ children: [],
+ properties: {
+ zeroWidth: false,
+ nullable: false,
+ minimumLength: 3,
+ maximumLength: 3,
+ consumesInput: true,
+ },
+ support: { flavour: "ecmascript", status: "supported", notes: [] },
+ provenance: {
+ provider: "fixture",
+ providerVersion: "1",
+ source: "parsed",
+ },
+ };
+}
+
+function candidate(
+ id: string,
+ subject: string,
+ intendedOutcome: "match" | "no-match",
+): GeneratedSubjectCandidate {
+ return {
+ id,
+ intendedOutcome,
+ subject,
+ subjectBytes: new TextEncoder().encode(subject).byteLength,
+ label: `candidate ${id}`,
+ features: [intendedOutcome === "match" ? "shortest" : "likely-near-miss"],
+ notes: [],
+ };
+}
+
+function report(
+ candidates: readonly GeneratedSubjectCandidate[],
+): CandidateGenerationReport {
+ return {
+ generator: { id: "regex-tools-ast-cases", version: "1" },
+ flavour: "ecmascript",
+ seed: "orchestrator-seed",
+ seedHash: "12345678",
+ candidates,
+ coverage: [
+ {
+ feature: "shortest",
+ status: "covered",
+ detail: "fixture",
+ relevantNodes: 1,
+ },
+ ],
+ unsupportedConstructs: [],
+ visitedAstNodes: 1,
+ candidateBytes: candidates.reduce(
+ (total, item) => total + item.subjectBytes,
+ 0,
+ ),
+ candidateAttempts: candidates.length,
+ duplicateCandidates: 0,
+ candidatesTruncated: false,
+ traversalTruncated: false,
+ warnings: [],
+ };
+}
+
+class FakeGenerator implements GenerationWorkerClient {
+ cancelled = false;
+ disposed = false;
+ readonly value:
+ CandidateGenerationReport | Promise;
+
+ constructor(
+ value: CandidateGenerationReport | Promise,
+ ) {
+ this.value = value;
+ }
+
+ generate(): Promise {
+ return Promise.resolve(this.value);
+ }
+
+ cancel(): void {
+ this.cancelled = true;
+ }
+
+ dispose(): void {
+ this.disposed = true;
+ }
+}
+
+function execution(
+ request: RegexExecutionRequest,
+ {
+ accepted = true,
+ matched = request.subject.includes("yes"),
+ }: { readonly accepted?: boolean; readonly matched?: boolean } = {},
+): RegexExecutionResult {
+ return {
+ accepted,
+ engine: ENGINE,
+ flags: {
+ userFlags: request.flags.join(""),
+ effectiveFlags: request.flags.join(""),
+ internallyAddedIndicesFlag: false,
+ internallyAddedGlobalFlag: false,
+ },
+ matches: matched
+ ? [
+ {
+ matchNumber: 1,
+ value: "yes",
+ valueStatus: "complete",
+ range: { startUtf16: 0, endUtf16: 3 },
+ nativeRange: { start: 0, end: 3, unit: "utf16" },
+ captures: [],
+ },
+ ]
+ : [],
+ diagnostics: accepted
+ ? []
+ : [
+ {
+ id: "compile",
+ source: "execution-engine",
+ severity: "error",
+ code: "compile-error",
+ message: "fixture compile rejection",
+ flavour: "ecmascript",
+ provenance: "reported",
+ },
+ ],
+ elapsedMs: 0.25,
+ truncated: false,
+ };
+}
+
+function input(
+ settings: Partial = {},
+): CaseGenerationInput {
+ return {
+ flavour: "ecmascript",
+ flavourVersion: "2025",
+ pattern: "yes",
+ flags: [],
+ options: {},
+ scanAll: false,
+ root: root(),
+ captureMetadata: [],
+ syntaxAccepted: true,
+ seed: "orchestrator-seed",
+ settings: { ...DEFAULT_GENERATION_SETTINGS, ...settings },
+ };
+}
+
+describe("CaseGenerationOrchestrator", () => {
+ it("retains only candidates confirmed by the selected actual engine", async () => {
+ const generator = new FakeGenerator(
+ report([
+ candidate("positive", "yes", "match"),
+ candidate("negative", "no", "no-match"),
+ candidate("false-positive", "no", "match"),
+ candidate("false-negative", "yes", "no-match"),
+ ]),
+ );
+ const requests: RegexExecutionRequest[] = [];
+ const engine = {
+ execute: async (request: RegexExecutionRequest) => {
+ requests.push(request);
+ return execution(request);
+ },
+ cancel: () => undefined,
+ dispose: () => undefined,
+ };
+ const progress: string[] = [];
+ const orchestrator = new CaseGenerationOrchestrator({
+ generator,
+ engine,
+ });
+
+ const result = await orchestrator.run(input(), (value) =>
+ progress.push(value.phase),
+ );
+
+ expect(result.cases.map((item) => item.expectation)).toEqual([
+ "should-match",
+ "should-not-match",
+ ]);
+ expect(result.cases[0]?.provenance).toEqual(
+ expect.objectContaining({
+ kind: "generated",
+ seed: "orchestrator-seed",
+ candidateId: "positive",
+ }),
+ );
+ expect(result.discarded.map((item) => item.reason)).toEqual([
+ "unexpected-no-match",
+ "unexpected-match",
+ ]);
+ expect(result.engine).toEqual(ENGINE);
+ expect(result.engineExecutions).toBe(4);
+ expect(requests.every((request) => request.flavour === "ecmascript")).toBe(
+ true,
+ );
+ expect(progress).toContain("synthesizing");
+ expect(progress).toContain("verifying");
+ });
+
+ it("keeps timeout, crash and compile rejection distinct", async () => {
+ const candidates = [
+ candidate("timeout", "timeout", "match"),
+ candidate("crash", "crash", "match"),
+ candidate("compile", "yes", "match"),
+ ];
+ const engine = {
+ execute: async (request: RegexExecutionRequest) => {
+ if (request.subject === "timeout") {
+ throw new WorkerRequestError("timeout", "fixture timeout");
+ }
+ if (request.subject === "crash") {
+ throw new WorkerRequestError("crash", "fixture crash");
+ }
+ return execution(request, { accepted: false });
+ },
+ cancel: () => undefined,
+ dispose: () => undefined,
+ };
+ const orchestrator = new CaseGenerationOrchestrator({
+ generator: new FakeGenerator(report(candidates)),
+ engine,
+ });
+
+ const result = await orchestrator.run(input());
+
+ expect(result.status).toBe("compile-rejected");
+ expect(result.discarded.map((item) => item.reason)).toEqual([
+ "timeout",
+ "worker-crash",
+ "compile-rejected",
+ ]);
+ expect(result.stoppedReason).toContain("fixture compile rejection");
+ });
+
+ it("honours the retained-case cap and aggregate wall clock", async () => {
+ const candidates = [
+ candidate("one", "yes", "match"),
+ candidate("two", "yes yes", "match"),
+ ];
+ const engine = {
+ execute: async (request: RegexExecutionRequest) => execution(request),
+ cancel: () => undefined,
+ dispose: () => undefined,
+ };
+ const retained = new CaseGenerationOrchestrator({
+ generator: new FakeGenerator(report(candidates)),
+ engine,
+ });
+ const retainedResult = await retained.run(
+ input({ maximumCases: 1, maximumCandidateAttempts: 4 }),
+ );
+ expect(retainedResult.status).toBe("partial");
+ expect(retainedResult.cases).toHaveLength(1);
+ expect(retainedResult.stoppedReason).toContain("maximum of 1");
+
+ let now = 0;
+ const wall = new CaseGenerationOrchestrator({
+ generator: new FakeGenerator(report(candidates)),
+ engine,
+ now: () => {
+ now += 6;
+ return now;
+ },
+ });
+ const wallResult = await wall.run(
+ input({
+ maximumWallTimeMs: 10,
+ maximumCases: 2,
+ maximumCandidateAttempts: 4,
+ }),
+ );
+ expect(wallResult.status).toBe("wall-time-limit");
+ expect(wallResult.stoppedReason).toContain("aggregate wall-time");
+ });
+
+ it("cancels both synthesis and engine workers and rejects stale work", async () => {
+ let rejectGeneration: ((error: Error) => void) | undefined;
+ const pending = new Promise((_, reject) => {
+ rejectGeneration = reject;
+ });
+ const generator = new FakeGenerator(pending);
+ const engine = {
+ execute: async (request: RegexExecutionRequest) => execution(request),
+ cancelled: false,
+ disposed: false,
+ cancel() {
+ this.cancelled = true;
+ },
+ dispose() {
+ this.disposed = true;
+ },
+ };
+ const orchestrator = new CaseGenerationOrchestrator({
+ generator,
+ engine,
+ });
+ const run = orchestrator.run(input());
+ orchestrator.cancel();
+ rejectGeneration?.(
+ new WorkerRequestError("cancelled", "fixture cancellation"),
+ );
+
+ await expect(run).rejects.toMatchObject({ kind: "cancelled" });
+ expect(generator.cancelled).toBe(true);
+ expect(engine.cancelled).toBe(true);
+ orchestrator.dispose();
+ expect(generator.disposed).toBe(true);
+ expect(engine.disposed).toBe(true);
+ });
+
+ it("does not invoke an engine for unsupported or rejected syntax", async () => {
+ const empty = report([]);
+ const generator = new FakeGenerator(empty);
+ let executions = 0;
+ const engine = {
+ execute: async (request: RegexExecutionRequest) => {
+ executions += 1;
+ return execution(request);
+ },
+ cancel: () => undefined,
+ dispose: () => undefined,
+ };
+ const orchestrator = new CaseGenerationOrchestrator({
+ generator,
+ engine,
+ });
+
+ const unsupported = await orchestrator.run(input());
+ expect(unsupported.status).toBe("unsupported");
+ expect(executions).toBe(0);
+
+ const rejected = await orchestrator.run({
+ ...input(),
+ syntaxAccepted: false,
+ });
+ expect(rejected.status).toBe("unsupported");
+ expect(executions).toBe(0);
+ });
+
+ it("verifies generated positives and negatives through the real ECMAScript adapter", async () => {
+ const pattern = "^(cat|dog)s{0,2}\\p{Letter}$";
+ const flags = ["u"];
+ const syntax = await new EcmaScriptSyntaxProvider().parsePattern({
+ flavour: "ecmascript",
+ flavourVersion: "2025",
+ pattern,
+ flags,
+ options: {},
+ });
+ const adapter = new EcmaScriptEngineAdapter();
+ const orchestrator = new CaseGenerationOrchestrator({
+ generator: {
+ generate: async (candidateRequest) =>
+ generateCandidateSubjects(candidateRequest),
+ cancel: () => undefined,
+ dispose: () => undefined,
+ },
+ engine: {
+ execute: async (executionRequest) => adapter.execute(executionRequest),
+ cancel: () => undefined,
+ dispose: () => undefined,
+ },
+ });
+
+ const result = await orchestrator.run({
+ flavour: "ecmascript",
+ flavourVersion: "2025",
+ pattern,
+ flags,
+ options: {},
+ scanAll: false,
+ root: syntax.root,
+ captureMetadata: syntax.captures,
+ syntaxAccepted: syntax.accepted,
+ seed: "real-adapter-seed",
+ settings: {
+ ...DEFAULT_GENERATION_SETTINGS,
+ maximumCases: 2,
+ maximumCandidateAttempts: 96,
+ randomVariantCount: 8,
+ },
+ });
+
+ expect(result.engine?.engineName).toBe("Native ECMAScript RegExp");
+ expect(
+ result.cases.some(
+ (candidate) => candidate.expectation === "should-match",
+ ),
+ ).toBe(true);
+ expect(
+ result.cases.some(
+ (candidate) => candidate.expectation === "should-not-match",
+ ),
+ ).toBe(true);
+ expect(result.cases).toHaveLength(2);
+ const expression = new RegExp(pattern, "u");
+ for (const candidate of result.cases) {
+ expect(expression.test(candidate.subject)).toBe(
+ candidate.expectation === "should-match",
+ );
+ }
+ });
+});
diff --git a/src/regex/generation/CaseGenerationOrchestrator.ts b/src/regex/generation/CaseGenerationOrchestrator.ts
new file mode 100644
index 0000000..b3549ea
--- /dev/null
+++ b/src/regex/generation/CaseGenerationOrchestrator.ts
@@ -0,0 +1,429 @@
+import { DEFAULT_REGEX_LIMITS } from "../execution/request-limits";
+import { EngineSupervisor } from "../execution/EngineSupervisor";
+import { WorkerRequestError } from "../execution/WorkerSupervisor";
+import type {
+ RegexExecutionRequest,
+ RegexExecutionResult,
+} from "../model/match";
+import { GENERATION_LIMITS } from "./generation-limits";
+import {
+ GENERATED_CASE_GENERATOR_ID,
+ GENERATED_CASE_GENERATOR_VERSION,
+ type CaseGenerationInput,
+ type CaseGenerationProgress,
+ type CaseGenerationResult,
+ type DiscardedGeneratedCandidate,
+ type GeneratedSubjectCandidate,
+ type GenerationDiscardReason,
+ type VerifiedGeneratedCase,
+} from "./generation.types";
+import {
+ GenerationSupervisor,
+ type GenerationWorkerClient,
+} from "./GenerationSupervisor";
+import { generationSeedHash } from "./seeded-random";
+
+export interface GenerationEngineClient {
+ execute(
+ request: RegexExecutionRequest,
+ timeoutMs: number,
+ ): Promise;
+ cancel(): void;
+ dispose(): void;
+}
+
+export interface CaseGenerationDependencies {
+ readonly generator?: GenerationWorkerClient;
+ readonly engine?: GenerationEngineClient;
+ readonly now?: () => number;
+}
+
+export interface CaseGenerationClient {
+ run(
+ input: CaseGenerationInput,
+ onProgress?: (progress: CaseGenerationProgress) => void,
+ ): Promise;
+ cancel(): void;
+ dispose(): void;
+}
+
+function resultPreview(value: string): string {
+ const maximum = GENERATION_LIMITS.maximumDiagnosticPreviewUtf16;
+ return value.length <= maximum ? value : `${value.slice(0, maximum)}…`;
+}
+
+function executionDetail(result: RegexExecutionResult): string {
+ return (
+ result.diagnostics.find((diagnostic) => diagnostic.severity === "error")
+ ?.message ?? "The selected engine rejected the pattern."
+ );
+}
+
+function discardedReason(error: WorkerRequestError): {
+ readonly reason: GenerationDiscardReason;
+ readonly detail: string;
+} {
+ switch (error.kind) {
+ case "timeout":
+ return {
+ reason: "timeout",
+ detail:
+ "The actual engine did not complete within the configured per-case timeout.",
+ };
+ case "crash":
+ return {
+ reason: "worker-crash",
+ detail: "The actual engine worker crashed while verifying this case.",
+ };
+ case "worker-error":
+ return { reason: "execution-error", detail: error.message };
+ case "cancelled":
+ throw error;
+ }
+}
+
+function generatedName(
+ candidate: GeneratedSubjectCandidate,
+ index: number,
+): string {
+ const outcome =
+ candidate.intendedOutcome === "match" ? "positive" : "negative";
+ return `Generated ${outcome} ${index + 1} · ${candidate.label}`;
+}
+
+function assertInput(input: CaseGenerationInput): void {
+ if (input.pattern.length > DEFAULT_REGEX_LIMITS.patternHardLengthUtf16) {
+ throw new RangeError(
+ `Pattern exceeds the ${DEFAULT_REGEX_LIMITS.patternHardLengthUtf16.toLocaleString()} UTF-16 unit limit.`,
+ );
+ }
+ if (
+ input.captureMetadata.length > DEFAULT_REGEX_LIMITS.maximumCaptureGroups
+ ) {
+ throw new RangeError(
+ `Pattern defines more than ${DEFAULT_REGEX_LIMITS.maximumCaptureGroups.toLocaleString()} capture groups.`,
+ );
+ }
+ if (input.root.support.flavour !== input.flavour) {
+ throw new Error(
+ `The normalized AST belongs to ${input.root.support.flavour}, not ${input.flavour}.`,
+ );
+ }
+}
+
+export class CaseGenerationOrchestrator implements CaseGenerationClient {
+ readonly #generator: GenerationWorkerClient;
+ readonly #engine: GenerationEngineClient;
+ readonly #now: () => number;
+ #revision = 0;
+ #disposed = false;
+
+ constructor(dependencies: CaseGenerationDependencies = {}) {
+ this.#generator = dependencies.generator ?? new GenerationSupervisor();
+ this.#engine = dependencies.engine ?? new EngineSupervisor();
+ this.#now = dependencies.now ?? (() => performance.now());
+ }
+
+ async run(
+ input: CaseGenerationInput,
+ onProgress?: (progress: CaseGenerationProgress) => void,
+ ): Promise {
+ if (this.#disposed) {
+ throw new Error("Generated-case orchestrator has been disposed.");
+ }
+ assertInput(input);
+ const revision = ++this.#revision;
+ const startedAt = this.#now();
+ const elapsed = () => Math.max(0, this.#now() - startedAt);
+ const remaining = () =>
+ Math.max(0, input.settings.maximumWallTimeMs - elapsed());
+ const assertCurrent = () => {
+ if (revision !== this.#revision) {
+ throw new WorkerRequestError(
+ "cancelled",
+ "Generated-case run was cancelled.",
+ );
+ }
+ };
+
+ onProgress?.({
+ phase: "synthesizing",
+ completed: 0,
+ total: 0,
+ retained: 0,
+ message: "Synthesizing bounded candidates from the normalized AST.",
+ });
+
+ if (!input.syntaxAccepted) {
+ const reason =
+ "The syntax provider rejected the pattern; candidate generation was not started.";
+ return {
+ status: "unsupported",
+ seed: input.seed,
+ seedHash: generationSeedHash(input.seed),
+ flavour: input.flavour,
+ cases: [],
+ discarded: [],
+ coverage: [],
+ unsupportedConstructs: [],
+ candidateAttempts: 0,
+ engineExecutions: 0,
+ generatedCandidateBytes: 0,
+ verifiedSubjectBytes: 0,
+ wallTimeMs: elapsed(),
+ stoppedReason: reason,
+ warnings: [reason],
+ };
+ }
+
+ const generation = await this.#generator.generate(
+ {
+ flavour: input.flavour,
+ root: input.root,
+ flags: input.flags,
+ seed: input.seed,
+ settings: input.settings,
+ },
+ Math.max(1, remaining()),
+ );
+ assertCurrent();
+
+ if (generation.candidates.length === 0) {
+ return {
+ status: "unsupported",
+ seed: input.seed,
+ seedHash: generation.seedHash,
+ flavour: input.flavour,
+ cases: [],
+ discarded: [],
+ coverage: generation.coverage,
+ unsupportedConstructs: generation.unsupportedConstructs,
+ candidateAttempts: generation.candidateAttempts,
+ engineExecutions: 0,
+ generatedCandidateBytes: generation.candidateBytes,
+ verifiedSubjectBytes: 0,
+ wallTimeMs: elapsed(),
+ ...(generation.stoppedReason
+ ? { stoppedReason: generation.stoppedReason }
+ : {}),
+ warnings: generation.warnings,
+ };
+ }
+
+ const cases: VerifiedGeneratedCase[] = [];
+ const discarded: DiscardedGeneratedCandidate[] = [];
+ const warnings = [...generation.warnings];
+ let engine: RegexExecutionResult["engine"] | undefined;
+ let engineExecutions = 0;
+ let status: CaseGenerationResult["status"] = "complete";
+ let stoppedReason = generation.stoppedReason;
+
+ const discard = (
+ candidate: GeneratedSubjectCandidate,
+ reason: GenerationDiscardReason,
+ detail: string,
+ ) => {
+ if (discarded.length < GENERATION_LIMITS.maximumDiscardedDiagnostics) {
+ discarded.push({
+ candidateId: candidate.id,
+ intendedOutcome: candidate.intendedOutcome,
+ reason,
+ detail,
+ subjectBytes: candidate.subjectBytes,
+ subjectPreview: resultPreview(candidate.subject),
+ });
+ } else if (
+ !warnings.some((warning) => warning.includes("discard diagnostics"))
+ ) {
+ warnings.push(
+ `Advanced discard diagnostics stopped at ${GENERATION_LIMITS.maximumDiscardedDiagnostics.toLocaleString()} entries.`,
+ );
+ }
+ };
+
+ for (let index = 0; index < generation.candidates.length; index += 1) {
+ assertCurrent();
+ if (cases.length >= input.settings.maximumCases) {
+ status = "partial";
+ stoppedReason = `Verification retained the configured maximum of ${input.settings.maximumCases.toLocaleString()} cases.`;
+ break;
+ }
+ const wallTimeRemaining = remaining();
+ if (wallTimeRemaining <= 0) {
+ status = "wall-time-limit";
+ stoppedReason = `Generation stopped at the ${input.settings.maximumWallTimeMs.toLocaleString()} ms aggregate wall-time limit.`;
+ const candidate = generation.candidates[
+ index
+ ] as GeneratedSubjectCandidate;
+ discard(candidate, "wall-time-limit", stoppedReason);
+ break;
+ }
+ const candidate = generation.candidates[
+ index
+ ] as GeneratedSubjectCandidate;
+ onProgress?.({
+ phase: "verifying",
+ completed: index,
+ total: generation.candidates.length,
+ retained: cases.length,
+ message: `Verifying candidate ${index + 1} of ${generation.candidates.length} through the selected actual engine.`,
+ });
+ const request: RegexExecutionRequest = {
+ flavour: input.flavour,
+ ...(input.flavourVersion === undefined
+ ? {}
+ : { flavourVersion: input.flavourVersion }),
+ pattern: input.pattern,
+ flags: input.flags,
+ options: input.options,
+ subject: candidate.subject,
+ captureMetadata: input.captureMetadata,
+ scanAll: input.scanAll,
+ maximumMatches: 1,
+ maximumCaptureRows: Math.max(
+ 1,
+ Math.min(
+ DEFAULT_REGEX_LIMITS.maximumCaptureRows,
+ input.captureMetadata.length,
+ ),
+ ),
+ };
+ let execution: RegexExecutionResult;
+ try {
+ engineExecutions += 1;
+ execution = await this.#engine.execute(
+ request,
+ Math.max(
+ 1,
+ Math.min(input.settings.perCaseTimeoutMs, wallTimeRemaining),
+ ),
+ );
+ } catch (error) {
+ assertCurrent();
+ if (error instanceof WorkerRequestError) {
+ const failure = discardedReason(error);
+ discard(candidate, failure.reason, failure.detail);
+ if (remaining() <= 0) {
+ status = "wall-time-limit";
+ stoppedReason = `Generation stopped at the ${input.settings.maximumWallTimeMs.toLocaleString()} ms aggregate wall-time limit.`;
+ break;
+ }
+ continue;
+ }
+ discard(
+ candidate,
+ "execution-error",
+ error instanceof Error ? error.message : String(error),
+ );
+ continue;
+ }
+ assertCurrent();
+ engine ??= execution.engine;
+ if (!execution.accepted) {
+ status = "compile-rejected";
+ stoppedReason = `The selected actual engine rejected the pattern: ${executionDetail(execution)}`;
+ discard(candidate, "compile-rejected", executionDetail(execution));
+ break;
+ }
+ const matched = execution.matches.length > 0;
+ const expectedMatch = candidate.intendedOutcome === "match";
+ if (matched !== expectedMatch) {
+ discard(
+ candidate,
+ matched ? "unexpected-match" : "unexpected-no-match",
+ matched
+ ? "The actual engine matched this intended negative candidate, so it was not presented as a negative test."
+ : "The actual engine did not match this intended positive candidate, so it was not presented as a positive test.",
+ );
+ continue;
+ }
+ cases.push({
+ id: candidate.id,
+ name: generatedName(candidate, cases.length),
+ subject: candidate.subject,
+ subjectBytes: candidate.subjectBytes,
+ expectation: matched ? "should-match" : "should-not-match",
+ features: candidate.features,
+ notes: candidate.notes,
+ elapsedMs: execution.elapsedMs,
+ matched,
+ matchCount: execution.matches.length,
+ provenance: {
+ kind: "generated",
+ generatorId: GENERATED_CASE_GENERATOR_ID,
+ generatorVersion: GENERATED_CASE_GENERATOR_VERSION,
+ seed: input.seed,
+ candidateId: candidate.id,
+ intendedOutcome: candidate.intendedOutcome,
+ },
+ });
+ }
+
+ onProgress?.({
+ phase: "verifying",
+ completed: Math.min(engineExecutions, generation.candidates.length),
+ total: generation.candidates.length,
+ retained: cases.length,
+ message: `Retained ${cases.length.toLocaleString()} actual-engine-verified cases.`,
+ });
+
+ const positives = cases.filter(
+ (candidate) => candidate.expectation === "should-match",
+ ).length;
+ const negatives = cases.length - positives;
+ if (positives === 0) {
+ warnings.push(
+ "No intended positive candidate was confirmed by the selected actual engine.",
+ );
+ }
+ if (negatives === 0) {
+ warnings.push(
+ "No intended negative candidate was confirmed by the selected actual engine.",
+ );
+ }
+ if (
+ status === "complete" &&
+ (generation.candidatesTruncated ||
+ generation.unsupportedConstructs.length > 0 ||
+ generation.coverage.some((entry) => entry.status === "partial"))
+ ) {
+ status = "partial";
+ }
+
+ return {
+ status,
+ seed: input.seed,
+ seedHash: generation.seedHash,
+ flavour: input.flavour,
+ cases,
+ discarded,
+ coverage: generation.coverage,
+ unsupportedConstructs: generation.unsupportedConstructs,
+ ...(engine ? { engine } : {}),
+ candidateAttempts: generation.candidateAttempts,
+ engineExecutions,
+ generatedCandidateBytes: generation.candidateBytes,
+ verifiedSubjectBytes: cases.reduce(
+ (total, candidate) => total + candidate.subjectBytes,
+ 0,
+ ),
+ wallTimeMs: elapsed(),
+ ...(stoppedReason ? { stoppedReason } : {}),
+ warnings,
+ };
+ }
+
+ cancel(): void {
+ this.#revision += 1;
+ this.#generator.cancel();
+ this.#engine.cancel();
+ }
+
+ dispose(): void {
+ if (this.#disposed) return;
+ this.#disposed = true;
+ this.#revision += 1;
+ this.#generator.dispose();
+ this.#engine.dispose();
+ }
+}
diff --git a/src/regex/generation/GenerationSupervisor.test.ts b/src/regex/generation/GenerationSupervisor.test.ts
new file mode 100644
index 0000000..2e8584e
--- /dev/null
+++ b/src/regex/generation/GenerationSupervisor.test.ts
@@ -0,0 +1,108 @@
+import { describe, expect, it } from "vitest";
+import {
+ WORKER_PROTOCOL_VERSION,
+ type WorkerRequest,
+} from "../execution/worker-protocol";
+import type { WorkerLike } from "../execution/WorkerSupervisor";
+import type { NormalizedRegexNode } from "../model/syntax";
+import { DEFAULT_GENERATION_SETTINGS } from "./generation-limits";
+import type {
+ GenerationWorkerOperation,
+ GenerationWorkerResult,
+} from "./generation.types";
+import { GenerationSupervisor } from "./GenerationSupervisor";
+
+function root(): NormalizedRegexNode {
+ return {
+ id: "root",
+ kind: "pattern",
+ range: { startUtf16: 0, endUtf16: 1 },
+ raw: "a",
+ explanation: "pattern",
+ children: [],
+ properties: {
+ zeroWidth: false,
+ nullable: false,
+ minimumLength: 1,
+ maximumLength: 1,
+ consumesInput: true,
+ },
+ support: { flavour: "ecmascript", status: "supported", notes: [] },
+ provenance: {
+ provider: "fixture",
+ providerVersion: "1",
+ source: "parsed",
+ },
+ };
+}
+
+class ImmediateWorker implements WorkerLike {
+ onmessage: ((event: MessageEvent) => void) | null = null;
+ onerror: ((event: ErrorEvent) => void) | null = null;
+ onmessageerror: ((event: MessageEvent) => void) | null = null;
+ terminated = false;
+
+ postMessage(value: unknown): void {
+ const request = value as WorkerRequest;
+ const payload: GenerationWorkerResult = {
+ kind: "generate",
+ result: {
+ generator: {
+ id: "regex-tools-ast-cases",
+ version: "1",
+ },
+ flavour: "ecmascript",
+ seed: request.payload.request.seed,
+ seedHash: "12345678",
+ candidates: [],
+ coverage: [],
+ unsupportedConstructs: [],
+ visitedAstNodes: 1,
+ candidateBytes: 0,
+ candidateAttempts: 0,
+ duplicateCandidates: 0,
+ candidatesTruncated: false,
+ traversalTruncated: false,
+ warnings: [],
+ },
+ };
+ queueMicrotask(() =>
+ this.onmessage?.(
+ new MessageEvent("message", {
+ data: {
+ protocolVersion: WORKER_PROTOCOL_VERSION,
+ requestId: request.requestId,
+ generation: request.generation,
+ ok: true,
+ payload,
+ },
+ }),
+ ),
+ );
+ }
+
+ terminate(): void {
+ this.terminated = true;
+ }
+}
+
+describe("GenerationSupervisor", () => {
+ it("routes the bounded request and returns the matching result", async () => {
+ const worker = new ImmediateWorker();
+ const supervisor = new GenerationSupervisor(() => worker);
+ await expect(
+ supervisor.generate(
+ {
+ flavour: "ecmascript",
+ root: root(),
+ flags: ["u"],
+ seed: "supervisor-seed",
+ settings: DEFAULT_GENERATION_SETTINGS,
+ },
+ 100,
+ ),
+ ).resolves.toEqual(expect.objectContaining({ seed: "supervisor-seed" }));
+ supervisor.dispose();
+ expect(worker.terminated).toBe(true);
+ });
+});
diff --git a/src/regex/generation/GenerationSupervisor.ts b/src/regex/generation/GenerationSupervisor.ts
new file mode 100644
index 0000000..6180f18
--- /dev/null
+++ b/src/regex/generation/GenerationSupervisor.ts
@@ -0,0 +1,64 @@
+import {
+ WorkerSupervisor,
+ type WorkerFactory,
+} from "../execution/WorkerSupervisor";
+import type {
+ CandidateGenerationReport,
+ CandidateGenerationRequest,
+ GenerationWorkerOperation,
+ GenerationWorkerResult,
+} from "./generation.types";
+
+export interface GenerationWorkerClient {
+ generate(
+ request: CandidateGenerationRequest,
+ timeoutMs: number,
+ ): Promise;
+ cancel(): void;
+ dispose(): void;
+}
+
+const createGenerationWorker: WorkerFactory = () =>
+ new Worker(new URL("../../workers/generation.worker.ts", import.meta.url), {
+ type: "module",
+ name: "regex-tools-case-generation",
+ });
+
+export class GenerationSupervisor implements GenerationWorkerClient {
+ readonly #supervisor: WorkerSupervisor<
+ GenerationWorkerOperation,
+ GenerationWorkerResult
+ >;
+
+ constructor(workerFactory: WorkerFactory = createGenerationWorker) {
+ this.#supervisor = new WorkerSupervisor(
+ "Generated-case synthesis worker",
+ workerFactory,
+ );
+ }
+
+ async generate(
+ request: CandidateGenerationRequest,
+ timeoutMs: number,
+ ): Promise {
+ const response = await this.#supervisor.run(
+ { kind: "generate", request },
+ timeoutMs,
+ { supersede: true },
+ );
+ if (response.kind !== "generate") {
+ throw new Error(
+ "Generated-case worker returned the wrong response kind.",
+ );
+ }
+ return response.result;
+ }
+
+ cancel(): void {
+ this.#supervisor.cancel();
+ }
+
+ dispose(): void {
+ this.#supervisor.dispose();
+ }
+}
diff --git a/src/regex/generation/generate-candidates.test.ts b/src/regex/generation/generate-candidates.test.ts
new file mode 100644
index 0000000..4e387bf
--- /dev/null
+++ b/src/regex/generation/generate-candidates.test.ts
@@ -0,0 +1,252 @@
+import { describe, expect, it } from "vitest";
+import type { NormalizedRegexNode } from "../model/syntax";
+import { EcmaScriptSyntaxProvider } from "../syntax/providers/ecmascript/EcmaScriptSyntaxProvider";
+import { DEFAULT_GENERATION_SETTINGS } from "./generation-limits";
+import { generateCandidateSubjects } from "./generate-candidates";
+import type {
+ CandidateGenerationRequest,
+ CandidateGenerationSettings,
+} from "./generation.types";
+
+const provider = new EcmaScriptSyntaxProvider();
+
+async function request(
+ pattern: string,
+ {
+ seed = "fixture-seed",
+ flags = ["u"],
+ settings = {},
+ }: {
+ readonly seed?: string;
+ readonly flags?: readonly string[];
+ readonly settings?: Partial;
+ } = {},
+): Promise {
+ const syntax = await provider.parsePattern({
+ flavour: "ecmascript",
+ pattern,
+ flags,
+ options: {},
+ });
+ expect(syntax.accepted).toBe(true);
+ return {
+ flavour: "ecmascript",
+ root: syntax.root,
+ flags,
+ seed,
+ settings: { ...DEFAULT_GENERATION_SETTINGS, ...settings },
+ };
+}
+
+describe("generateCandidateSubjects", () => {
+ it("covers alternatives, optional boundaries, classes and Unicode representatives", async () => {
+ const report = generateCandidateSubjects(
+ await request("(cat|dog)s{0,2}\\p{Letter}", {
+ settings: { randomVariantCount: 4 },
+ }),
+ );
+
+ expect(
+ report.candidates.some((candidate) => candidate.subject === "catA"),
+ ).toBe(true);
+ expect(
+ report.candidates.some((candidate) => candidate.subject === "dogA"),
+ ).toBe(true);
+ expect(
+ report.candidates.some(
+ (candidate) =>
+ candidate.features.includes("quantifier-boundary") &&
+ candidate.subject.includes("ss"),
+ ),
+ ).toBe(true);
+ expect(
+ report.candidates.some((candidate) => candidate.subject.includes("é")),
+ ).toBe(true);
+ expect(
+ report.coverage.find((entry) => entry.feature === "alternative-coverage")
+ ?.status,
+ ).toBe("covered");
+ expect(
+ report.coverage.find((entry) => entry.feature === "optional-presence")
+ ?.status,
+ ).toBe("covered");
+ expect(report.seed).toBe("fixture-seed");
+ expect(report.seedHash).toMatch(/^[\da-f]{8}$/u);
+ expect(
+ report.candidates
+ .slice(0, 2)
+ .map((candidate) => candidate.intendedOutcome),
+ ).toEqual(["match", "no-match"]);
+ });
+
+ it("is byte-for-byte deterministic for a seed and changes seeded variants for another seed", async () => {
+ const source = "(a|b|c)(x|y|z){1,4}[0-9]";
+ const first = generateCandidateSubjects(
+ await request(source, {
+ seed: "alpha",
+ settings: { randomVariantCount: 12 },
+ }),
+ );
+ const replay = generateCandidateSubjects(
+ await request(source, {
+ seed: "alpha",
+ settings: { randomVariantCount: 12 },
+ }),
+ );
+ const second = generateCandidateSubjects(
+ await request(source, {
+ seed: "bravo",
+ settings: { randomVariantCount: 12 },
+ }),
+ );
+
+ expect(replay).toEqual(first);
+ expect(
+ first.candidates
+ .filter((candidate) => candidate.features.includes("seeded-random"))
+ .map((candidate) => candidate.subject),
+ ).not.toEqual(
+ second.candidates
+ .filter((candidate) => candidate.features.includes("seeded-random"))
+ .map((candidate) => candidate.subject),
+ );
+ });
+
+ it("reports unsupported stateful constructs without claiming complete coverage", async () => {
+ const report = generateCandidateSubjects(
+ await request("(a)\\1(?=x)x", { flags: [] }),
+ );
+
+ expect(report.unsupportedConstructs).toEqual(
+ expect.arrayContaining([
+ expect.objectContaining({ kind: "backreference" }),
+ expect.objectContaining({ kind: "lookahead" }),
+ ]),
+ );
+ expect(
+ report.coverage.find((entry) => entry.feature === "shortest")?.status,
+ ).toBe("partial");
+ });
+
+ it("does not claim structural enumeration for Unicode properties of strings", async () => {
+ const report = generateCandidateSubjects(
+ await request("\\p{RGI_Emoji}", { flags: ["v"] }),
+ );
+
+ expect(report.unsupportedConstructs).toEqual(
+ expect.arrayContaining([
+ expect.objectContaining({
+ kind: "unicode-property",
+ reason: expect.stringContaining("multiple code points"),
+ }),
+ ]),
+ );
+ expect(
+ report.coverage.find(
+ (entry) => entry.feature === "unicode-representative",
+ )?.status,
+ ).toBe("partial");
+ });
+
+ it("refuses to reinterpret a partial PCRE2 AST as ECMAScript", () => {
+ const root: NormalizedRegexNode = {
+ id: "pcre-root",
+ kind: "pattern",
+ range: { startUtf16: 0, endUtf16: 6 },
+ raw: "(?R)",
+ explanation: "PCRE2 pattern",
+ children: [],
+ properties: {
+ zeroWidth: false,
+ nullable: "unknown",
+ consumesInput: "conditional",
+ },
+ support: {
+ flavour: "pcre2",
+ status: "partial",
+ notes: [],
+ },
+ provenance: {
+ provider: "pcre2-lexical",
+ providerVersion: "1",
+ source: "parsed",
+ },
+ };
+ const report = generateCandidateSubjects({
+ flavour: "pcre2",
+ root,
+ flags: [],
+ seed: "pcre-seed",
+ settings: DEFAULT_GENERATION_SETTINGS,
+ });
+
+ expect(report.candidates).toHaveLength(0);
+ expect(
+ report.coverage.every((entry) => entry.status === "unsupported"),
+ ).toBe(true);
+ expect(report.stoppedReason).toContain("not reinterpreted");
+ });
+
+ it("enforces traversal, attempt and byte caps before verification", async () => {
+ const attemptLimited = generateCandidateSubjects(
+ await request("(a|b|c|d)", {
+ settings: {
+ maximumCases: 3,
+ maximumCandidateAttempts: 3,
+ randomVariantCount: 0,
+ },
+ }),
+ );
+ expect(attemptLimited.candidateAttempts).toBe(3);
+ expect(attemptLimited.candidatesTruncated).toBe(true);
+ expect(attemptLimited.stoppedReason).toContain("attempt limit");
+
+ const byteLimited = generateCandidateSubjects(
+ await request("abcdef", {
+ settings: {
+ maximumSubjectBytes: 3,
+ maximumTotalSubjectBytes: 3,
+ randomVariantCount: 0,
+ },
+ }),
+ );
+ expect(byteLimited.candidates).toHaveLength(0);
+ expect(byteLimited.warnings.join(" ")).toContain("per-subject limit");
+
+ const nodeLimited = generateCandidateSubjects(
+ await request("abcdefgh", {
+ settings: {
+ maximumAstNodes: 2,
+ randomVariantCount: 0,
+ },
+ }),
+ );
+ expect(nodeLimited.traversalTruncated).toBe(true);
+ expect(nodeLimited.candidates).toHaveLength(0);
+ });
+
+ it("validates every user-controlled bound and the explicit seed", async () => {
+ const base = await request("a");
+ expect(() => generateCandidateSubjects({ ...base, seed: "" })).toThrow(
+ "Generation seed",
+ );
+ expect(() =>
+ generateCandidateSubjects({
+ ...base,
+ settings: {
+ ...base.settings,
+ maximumCandidateAttempts: base.settings.maximumCases - 1,
+ },
+ }),
+ ).toThrow("Maximum candidate attempts");
+ expect(() =>
+ generateCandidateSubjects({
+ ...base,
+ settings: {
+ ...base.settings,
+ maximumSubjectBytes: base.settings.maximumTotalSubjectBytes + 1,
+ },
+ }),
+ ).toThrow("per-subject byte limit");
+ });
+});
diff --git a/src/regex/generation/generate-candidates.ts b/src/regex/generation/generate-candidates.ts
new file mode 100644
index 0000000..e035f13
--- /dev/null
+++ b/src/regex/generation/generate-candidates.ts
@@ -0,0 +1,1222 @@
+import {
+ DEFAULT_REGEX_LIMITS,
+ utf8ByteLength,
+} from "../execution/request-limits";
+import type { NormalizedRegexNode } from "../model/syntax";
+import { GENERATION_LIMITS } from "./generation-limits";
+import {
+ GENERATED_CASE_GENERATOR_ID,
+ GENERATED_CASE_GENERATOR_VERSION,
+ type CandidateGenerationReport,
+ type CandidateGenerationRequest,
+ type GeneratedSubjectCandidate,
+ type GenerationCoverageEntry,
+ type GenerationCoverageStatus,
+ type GenerationFeature,
+ type UnsupportedGenerationConstruct,
+} from "./generation.types";
+import {
+ createDeterministicRandom,
+ generationSeedHash,
+ type DeterministicRandom,
+} from "./seeded-random";
+
+const MAXIMUM_WARNINGS = 100;
+const MAXIMUM_UNSUPPORTED_CONSTRUCTS = 250;
+const MAXIMUM_RAW_PREVIEW_UTF16 = 80;
+
+interface AstRecord {
+ readonly node: NormalizedRegexNode;
+ readonly depth: number;
+ readonly parent?: NormalizedRegexNode;
+}
+
+interface Fragment {
+ readonly text: string;
+ readonly bytes: number;
+}
+
+interface RenderOverrides {
+ readonly choices?: ReadonlyMap;
+ readonly repetitions?: ReadonlyMap;
+ readonly classChoices?: ReadonlyMap;
+ readonly random?: DeterministicRandom;
+}
+
+class RenderLimitError extends Error {
+ constructor(message: string) {
+ super(message);
+ this.name = "RenderLimitError";
+ }
+}
+
+function boundedInteger(
+ value: number,
+ label: string,
+ minimum: number,
+ maximum: number,
+): void {
+ if (!Number.isSafeInteger(value) || value < minimum || value > maximum) {
+ throw new RangeError(
+ `${label} must be a whole number from ${minimum.toLocaleString()} to ${maximum.toLocaleString()}.`,
+ );
+ }
+}
+
+export function validateGenerationRequest(
+ request: CandidateGenerationRequest,
+): void {
+ if (request.root.raw.length > DEFAULT_REGEX_LIMITS.patternHardLengthUtf16) {
+ throw new RangeError(
+ `Normalized pattern exceeds the ${DEFAULT_REGEX_LIMITS.patternHardLengthUtf16.toLocaleString()} UTF-16 unit limit.`,
+ );
+ }
+ if (
+ request.seed.length === 0 ||
+ request.seed.length > GENERATION_LIMITS.maximumSeedUtf16
+ ) {
+ throw new RangeError(
+ `Generation seed must contain 1 to ${GENERATION_LIMITS.maximumSeedUtf16} UTF-16 units.`,
+ );
+ }
+ const { settings } = request;
+ boundedInteger(
+ settings.maximumCases,
+ "Maximum generated cases",
+ 1,
+ DEFAULT_REGEX_LIMITS.maximumGeneratedCases,
+ );
+ boundedInteger(
+ settings.maximumCandidateAttempts,
+ "Maximum candidate attempts",
+ settings.maximumCases,
+ GENERATION_LIMITS.maximumCandidateAttempts,
+ );
+ boundedInteger(
+ settings.maximumTotalSubjectBytes,
+ "Maximum generated subject bytes",
+ 1,
+ GENERATION_LIMITS.maximumTotalSubjectBytes,
+ );
+ boundedInteger(
+ settings.maximumSubjectBytes,
+ "Maximum bytes per generated subject",
+ 1,
+ GENERATION_LIMITS.maximumSubjectBytes,
+ );
+ if (settings.maximumSubjectBytes > settings.maximumTotalSubjectBytes) {
+ throw new RangeError(
+ "The per-subject byte limit cannot exceed the aggregate generated-subject limit.",
+ );
+ }
+ boundedInteger(
+ settings.maximumAstNodes,
+ "Maximum AST nodes",
+ 1,
+ GENERATION_LIMITS.maximumAstNodes,
+ );
+ boundedInteger(
+ settings.maximumAstDepth,
+ "Maximum AST depth",
+ 1,
+ GENERATION_LIMITS.maximumAstDepth,
+ );
+ boundedInteger(
+ settings.maximumQuantifierRepetitions,
+ "Maximum synthesized repetitions",
+ 1,
+ GENERATION_LIMITS.maximumQuantifierRepetitions,
+ );
+ boundedInteger(
+ settings.randomVariantCount,
+ "Random variants",
+ 0,
+ DEFAULT_REGEX_LIMITS.maximumGeneratedCases,
+ );
+ boundedInteger(
+ settings.perCaseTimeoutMs,
+ "Per-case timeout",
+ GENERATION_LIMITS.minimumPerCaseTimeoutMs,
+ GENERATION_LIMITS.maximumPerCaseTimeoutMs,
+ );
+ boundedInteger(
+ settings.maximumWallTimeMs,
+ "Generation wall time",
+ 1,
+ GENERATION_LIMITS.maximumWallTimeMs,
+ );
+ if (request.root.support.flavour !== request.flavour) {
+ throw new Error(
+ `The normalized AST belongs to ${request.root.support.flavour}, not ${request.flavour}.`,
+ );
+ }
+}
+
+function preview(value: string, maximum = MAXIMUM_RAW_PREVIEW_UTF16): string {
+ return value.length <= maximum ? value : `${value.slice(0, maximum)}…`;
+}
+
+function recordsFor(
+ root: NormalizedRegexNode,
+ maximumNodes: number,
+ maximumDepth: number,
+): {
+ readonly records: readonly AstRecord[];
+ readonly traversalTruncated: boolean;
+ readonly depthTruncated: boolean;
+} {
+ const records: AstRecord[] = [];
+ const stack: AstRecord[] = [{ node: root, depth: 1 }];
+ let traversalTruncated = false;
+ let depthTruncated = false;
+ while (stack.length > 0) {
+ const record = stack.pop() as AstRecord;
+ if (record.depth > maximumDepth) {
+ depthTruncated = true;
+ continue;
+ }
+ if (records.length >= maximumNodes) {
+ traversalTruncated = true;
+ break;
+ }
+ records.push(record);
+ for (let index = record.node.children.length - 1; index >= 0; index -= 1) {
+ stack.push({
+ node: record.node.children[index] as NormalizedRegexNode,
+ depth: record.depth + 1,
+ parent: record.node,
+ });
+ }
+ }
+ return { records, traversalTruncated, depthTruncated };
+}
+
+function decodeEscapedAtom(raw: string): string | undefined {
+ if (!raw.startsWith("\\")) return raw;
+ const simple: Readonly> = {
+ "\\0": "\0",
+ "\\b": "\b",
+ "\\f": "\f",
+ "\\n": "\n",
+ "\\r": "\r",
+ "\\t": "\t",
+ "\\v": "\v",
+ };
+ if (raw in simple) return simple[raw];
+ const hex = /^\\x([\da-f]{2})$/iu.exec(raw);
+ if (hex?.[1]) return String.fromCharCode(Number.parseInt(hex[1], 16));
+ const unicode = /^\\u([\da-f]{4})$/iu.exec(raw);
+ if (unicode?.[1]) {
+ return String.fromCharCode(Number.parseInt(unicode[1], 16));
+ }
+ const unicodePoint = /^\\u\{([\da-f]{1,6})\}$/iu.exec(raw);
+ if (unicodePoint?.[1]) {
+ const value = Number.parseInt(unicodePoint[1], 16);
+ if (value <= 0x10ffff) return String.fromCodePoint(value);
+ return undefined;
+ }
+ const control = /^\\c([a-z])$/iu.exec(raw);
+ if (control?.[1]) {
+ return String.fromCharCode(control[1].toUpperCase().charCodeAt(0) % 32);
+ }
+ if (/^\\[^dDsSwWpPkK]$/u.test(raw)) return raw.slice(1);
+ return undefined;
+}
+
+function unique(values: readonly string[]): readonly string[] {
+ return [...new Set(values)];
+}
+
+function unicodePropertyRepresentatives(raw: string): readonly string[] {
+ const lower = raw.toLocaleLowerCase("en-US");
+ const negated = raw.startsWith("\\P");
+ if (negated) {
+ if (/letter|\\p\{l\}/iu.test(raw)) return ["0", "_", " "];
+ if (/number|decimal_number|\\p\{n[d]?\}/iu.test(raw)) {
+ return ["A", "_", " "];
+ }
+ if (/white_space|pattern_white_space/iu.test(raw)) return ["A", "0", "_"];
+ return ["A", "0", " ", "😀"];
+ }
+ if (lower.includes("letter") || /^\\p\{l[ultmo]?\}$/iu.test(raw)) {
+ return ["A", "é", "Ж", "Ω"];
+ }
+ if (
+ lower.includes("number") ||
+ lower.includes("decimal_number") ||
+ /^\\p\{n[dlo]?\}$/iu.test(raw)
+ ) {
+ return ["0", "٥", "Ⅷ"];
+ }
+ if (lower.includes("greek")) return ["Ω", "β"];
+ if (lower.includes("cyrillic")) return ["Ж", "я"];
+ if (lower.includes("latin")) return ["A", "é"];
+ if (lower.includes("emoji")) return ["😀", "©"];
+ if (lower.includes("white_space")) return [" ", "\t", "\n"];
+ if (lower.includes("punctuation") || /^\\p\{p\}$/iu.test(raw)) {
+ return [".", "—", "!"];
+ }
+ if (lower.includes("symbol") || /^\\p\{s\}$/iu.test(raw)) {
+ return ["+", "€", "😀"];
+ }
+ return ["A", "0", "é", "Ω", "Ж", "😀", " "];
+}
+
+function unicodePropertyCoverageGap(raw: string): string | undefined {
+ const lower = raw.toLocaleLowerCase("en-US");
+ if (
+ lower.includes("rgi_emoji") ||
+ lower.includes("basic_emoji") ||
+ lower.includes("emoji_keycap_sequence") ||
+ lower.includes("emoji_flag_sequence") ||
+ lower.includes("emoji_tag_sequence") ||
+ lower.includes("emoji_zwj_sequence")
+ ) {
+ return "Unicode properties of strings can consume multiple code points and are not structurally enumerated.";
+ }
+ if (
+ lower.includes("letter") ||
+ /^\\[pP]\{l[ultmo]?\}$/u.test(raw) ||
+ lower.includes("number") ||
+ lower.includes("decimal_number") ||
+ /^\\[pP]\{n[dlo]?\}$/u.test(raw) ||
+ lower.includes("greek") ||
+ lower.includes("cyrillic") ||
+ lower.includes("latin") ||
+ lower.includes("emoji") ||
+ lower.includes("white_space") ||
+ lower.includes("punctuation") ||
+ /^\\[pP]\{p\}$/u.test(raw) ||
+ lower.includes("symbol") ||
+ /^\\[pP]\{s\}$/u.test(raw)
+ ) {
+ return undefined;
+ }
+ return "This Unicode property uses only the generic bounded representative pool; its value space is not claimed as covered.";
+}
+
+function leafClassRepresentatives(raw: string): readonly string[] {
+ switch (raw) {
+ case "\\d":
+ return ["0", "5", "9"];
+ case "\\D":
+ return ["A", "_", "é", " "];
+ case "\\w":
+ return ["A", "0", "_"];
+ case "\\W":
+ return ["-", " ", "é"];
+ case "\\s":
+ return [" ", "\t", "\n", "\u00a0"];
+ case "\\S":
+ return ["A", "0", "_", "é"];
+ default:
+ return raw.startsWith("\\p") || raw.startsWith("\\P")
+ ? unicodePropertyRepresentatives(raw)
+ : [];
+ }
+}
+
+function descendants(
+ node: NormalizedRegexNode,
+ maximum = 256,
+): readonly NormalizedRegexNode[] {
+ const output: NormalizedRegexNode[] = [];
+ const stack = [...node.children].reverse();
+ while (stack.length > 0 && output.length < maximum) {
+ const current = stack.pop() as NormalizedRegexNode;
+ output.push(current);
+ for (let index = current.children.length - 1; index >= 0; index -= 1) {
+ stack.push(current.children[index] as NormalizedRegexNode);
+ }
+ }
+ return output;
+}
+
+function classRepresentatives(node: NormalizedRegexNode): readonly string[] {
+ if (node.kind === "dot") return ["a", "0", "é", "😀"];
+ if (node.kind === "unicode-property") {
+ return unicodePropertyRepresentatives(node.raw);
+ }
+ const leaf = leafClassRepresentatives(node.raw);
+ if (leaf.length > 0) return leaf;
+ if (node.kind === "character-class-range") {
+ const endpoints = node.children
+ .map((child) => decodeEscapedAtom(child.raw))
+ .filter((value): value is string => value !== undefined);
+ if (endpoints.length >= 2) {
+ const start = endpoints[0]?.codePointAt(0);
+ const end = endpoints[1]?.codePointAt(0);
+ const middle =
+ start !== undefined && end !== undefined && end >= start
+ ? String.fromCodePoint(start + Math.floor((end - start) / 2))
+ : undefined;
+ return unique([
+ endpoints[0] as string,
+ ...(middle ? [middle] : []),
+ endpoints[1] as string,
+ ]);
+ }
+ }
+ if (!node.raw.startsWith("[")) {
+ const literal = decodeEscapedAtom(node.raw);
+ return literal === undefined ? [] : [literal];
+ }
+ if (node.raw.startsWith("[^")) {
+ return ["A", "0", "_", " ", "-", "é", "Ω", "Ж", "😀", "\n"];
+ }
+ const values: string[] = [];
+ for (const child of descendants(node)) {
+ if (child.kind === "literal" || child.kind === "escaped-literal") {
+ const decoded = decodeEscapedAtom(child.raw);
+ if (decoded !== undefined) values.push(decoded);
+ } else if (
+ child.kind === "character-class-range" ||
+ child.kind === "unicode-property"
+ ) {
+ values.push(...classRepresentatives(child));
+ } else if (child.kind === "character-class" && !child.raw.startsWith("[")) {
+ values.push(...leafClassRepresentatives(child.raw));
+ }
+ }
+ return unique(values);
+}
+
+function choiceChildren(
+ node: NormalizedRegexNode,
+): readonly NormalizedRegexNode[] {
+ switch (node.kind) {
+ case "pattern":
+ case "disjunction":
+ case "capture-group":
+ case "named-capture-group":
+ case "noncapture-group":
+ case "inline-flags":
+ return node.children;
+ default:
+ return [];
+ }
+}
+
+function shortestChoice(children: readonly NormalizedRegexNode[]): number {
+ let selected = 0;
+ let length = Number.POSITIVE_INFINITY;
+ children.forEach((child, index) => {
+ const candidate =
+ child.properties.minimumLength ?? Number.POSITIVE_INFINITY;
+ if (candidate < length) {
+ selected = index;
+ length = candidate;
+ }
+ });
+ return selected;
+}
+
+function appendFragments(
+ fragments: readonly Fragment[],
+ maximumBytes: number,
+): Fragment {
+ let bytes = 0;
+ for (const fragment of fragments) {
+ bytes += fragment.bytes;
+ if (bytes > maximumBytes) {
+ throw new RenderLimitError(
+ `Candidate exceeds the ${maximumBytes.toLocaleString()} byte per-subject limit.`,
+ );
+ }
+ }
+ return { text: fragments.map((fragment) => fragment.text).join(""), bytes };
+}
+
+function repeatFragment(
+ fragment: Fragment,
+ repetitions: number,
+ maximumBytes: number,
+): Fragment {
+ if (
+ fragment.bytes > 0 &&
+ repetitions > Math.floor(maximumBytes / fragment.bytes)
+ ) {
+ throw new RenderLimitError(
+ `Quantifier expansion exceeds the ${maximumBytes.toLocaleString()} byte per-subject limit.`,
+ );
+ }
+ const bytes = fragment.bytes * repetitions;
+ if (bytes > maximumBytes) {
+ throw new RenderLimitError(
+ `Quantifier expansion exceeds the ${maximumBytes.toLocaleString()} byte per-subject limit.`,
+ );
+ }
+ return { text: fragment.text.repeat(repetitions), bytes };
+}
+
+function renderNode(
+ node: NormalizedRegexNode,
+ overrides: RenderOverrides,
+ maximumBytes: number,
+ maximumDepth: number,
+ maximumRepetitions: number,
+ depth = 1,
+): Fragment {
+ if (depth > maximumDepth) {
+ throw new RenderLimitError(
+ `Candidate rendering exceeds the ${maximumDepth.toLocaleString()} level AST-depth limit.`,
+ );
+ }
+ switch (node.kind) {
+ case "literal":
+ case "escaped-literal": {
+ const decoded = decodeEscapedAtom(node.raw);
+ if (decoded === undefined) return { text: "", bytes: 0 };
+ return { text: decoded, bytes: utf8ByteLength(decoded) };
+ }
+ case "dot": {
+ const representatives = classRepresentatives(node);
+ const selected =
+ overrides.classChoices?.get(node.id) ??
+ (overrides.random
+ ? overrides.random.integer(representatives.length)
+ : 0);
+ const text = representatives[selected % representatives.length] ?? "a";
+ return { text, bytes: utf8ByteLength(text) };
+ }
+ case "character-class":
+ case "character-class-range":
+ case "unicode-property": {
+ const representatives = classRepresentatives(node);
+ if (representatives.length === 0) return { text: "", bytes: 0 };
+ const selected =
+ overrides.classChoices?.get(node.id) ??
+ (overrides.random
+ ? overrides.random.integer(representatives.length)
+ : 0);
+ const text =
+ representatives[selected % representatives.length] ??
+ (representatives[0] as string);
+ return { text, bytes: utf8ByteLength(text) };
+ }
+ case "quantifier": {
+ const child = node.children[0];
+ if (!child) return { text: "", bytes: 0 };
+ const configured = overrides.repetitions?.get(node.id);
+ const minimum = node.quantifier?.minimum ?? 0;
+ const maximum = node.quantifier?.maximum;
+ let repetitions =
+ configured ??
+ (overrides.random
+ ? minimum +
+ overrides.random.integer(
+ Math.max(
+ 1,
+ Math.min(
+ 4,
+ maximum === null || maximum === undefined
+ ? 4
+ : maximum - minimum + 1,
+ ),
+ ),
+ )
+ : minimum);
+ if (maximum !== null && maximum !== undefined) {
+ repetitions = Math.min(repetitions, maximum);
+ }
+ if (repetitions > maximumRepetitions) {
+ throw new RenderLimitError(
+ `Quantifier expansion exceeds the ${maximumRepetitions.toLocaleString()} repetition synthesis limit.`,
+ );
+ }
+ const fragment = renderNode(
+ child,
+ overrides,
+ maximumBytes,
+ maximumDepth,
+ maximumRepetitions,
+ depth + 1,
+ );
+ return repeatFragment(fragment, repetitions, maximumBytes);
+ }
+ case "anchor":
+ case "word-boundary":
+ case "lookahead":
+ case "negative-lookahead":
+ case "lookbehind":
+ case "negative-lookbehind":
+ case "backreference":
+ case "subroutine-call":
+ case "recursion":
+ case "conditional":
+ case "comment":
+ case "control-verb":
+ case "callout":
+ case "unsupported":
+ case "error-recovery":
+ return { text: "", bytes: 0 };
+ default: {
+ const choices = choiceChildren(node);
+ if (choices.length > 0) {
+ const configured = overrides.choices?.get(node.id);
+ const selected =
+ configured ??
+ (overrides.random
+ ? overrides.random.integer(choices.length)
+ : shortestChoice(choices));
+ return renderNode(
+ choices[selected % choices.length] as NormalizedRegexNode,
+ overrides,
+ maximumBytes,
+ maximumDepth,
+ maximumRepetitions,
+ depth + 1,
+ );
+ }
+ return appendFragments(
+ node.children.map((child) =>
+ renderNode(
+ child,
+ overrides,
+ maximumBytes,
+ maximumDepth,
+ maximumRepetitions,
+ depth + 1,
+ ),
+ ),
+ maximumBytes,
+ );
+ }
+ }
+}
+
+function quantifierCounts(
+ node: NormalizedRegexNode,
+ maximumRepetitions: number,
+): {
+ readonly counts: readonly number[];
+ readonly truncated: boolean;
+} {
+ const minimum = node.quantifier?.minimum ?? 0;
+ const maximum = node.quantifier?.maximum;
+ const candidates = new Set([minimum]);
+ if (minimum < maximumRepetitions) candidates.add(minimum + 1);
+ if (minimum === 0 && maximumRepetitions >= 2) candidates.add(2);
+ let truncated = minimum > maximumRepetitions;
+ if (maximum === null || maximum === undefined) {
+ candidates.add(Math.min(maximumRepetitions, Math.max(minimum, 3)));
+ } else {
+ const boundedMaximum = Math.min(maximum, maximumRepetitions);
+ candidates.add(boundedMaximum);
+ if (boundedMaximum > minimum) candidates.add(boundedMaximum - 1);
+ if (maximum > maximumRepetitions) truncated = true;
+ }
+ return {
+ counts: [...candidates]
+ .filter((count) => count >= 0 && count <= maximumRepetitions)
+ .sort((left, right) => left - right),
+ truncated,
+ };
+}
+
+function unsupportedReason(node: NormalizedRegexNode): string | undefined {
+ switch (node.kind) {
+ case "backreference":
+ return "Backreference text depends on capture participation and is not synthesized.";
+ case "subroutine-call":
+ return "Subroutine calls are not expanded by the generator.";
+ case "recursion":
+ return "Recursive patterns are not expanded by the generator.";
+ case "conditional":
+ return "Conditional branches depend on runtime capture or assertion state.";
+ case "branch-reset-group":
+ return "Branch-reset numbering needs a complete flavour-specific structural provider.";
+ case "atomic-group":
+ return "Atomic grouping is not structurally generated in version 1.";
+ case "control-verb":
+ return "Engine control verbs are not interpreted by the generator.";
+ case "callout":
+ return "Engine callouts are not interpreted by the generator.";
+ case "lookahead":
+ case "negative-lookahead":
+ case "lookbehind":
+ case "negative-lookbehind":
+ return "Lookaround conditions are not synthesized; engine verification may still accept surrounding candidates.";
+ case "word-boundary":
+ return "Word-boundary context is sampled but is not solved symbolically.";
+ case "error-recovery":
+ return "Recovered syntax is not valid input for generation.";
+ case "unsupported":
+ return "The active syntax provider marked this construct unsupported.";
+ case "character-class":
+ if (
+ node.raw.startsWith("[") &&
+ (node.raw.includes("&&") ||
+ node.raw.includes("--") ||
+ node.raw.includes("\\q{"))
+ ) {
+ return "Complex Unicode-set algebra and string properties are not synthesized.";
+ }
+ return classRepresentatives(node).length === 0
+ ? "No bounded representative is available for this character class."
+ : undefined;
+ case "unicode-property":
+ return unicodePropertyCoverageGap(node.raw);
+ case "escaped-literal":
+ return decodeEscapedAtom(node.raw) === undefined
+ ? "This escaped atom is not decoded by the generator."
+ : undefined;
+ default:
+ return undefined;
+ }
+}
+
+function coverageEntry(
+ feature: GenerationFeature,
+ status: GenerationCoverageStatus,
+ detail: string,
+ relevantNodes: number,
+): GenerationCoverageEntry {
+ return { feature, status, detail, relevantNodes };
+}
+
+function nearMisses(subject: string): readonly string[] {
+ if (subject.length === 0) return ["x", "\n", " "];
+ const middle = Math.floor(subject.length / 2);
+ const replacement = subject[middle] === "!" ? "x" : "!";
+ return unique([
+ "",
+ subject.slice(1),
+ subject.slice(0, -1),
+ `${subject.slice(0, middle)}${replacement}${subject.slice(middle + 1)}`,
+ `x${subject}`,
+ `${subject}x`,
+ ]);
+}
+
+export function generateCandidateSubjects(
+ request: CandidateGenerationRequest,
+): CandidateGenerationReport {
+ validateGenerationRequest(request);
+ const { records, traversalTruncated, depthTruncated } = recordsFor(
+ request.root,
+ request.settings.maximumAstNodes,
+ request.settings.maximumAstDepth,
+ );
+ const seedHash = generationSeedHash(request.seed);
+ const warnings: string[] = [];
+ const warn = (message: string) => {
+ if (warnings.length < MAXIMUM_WARNINGS && !warnings.includes(message)) {
+ warnings.push(message);
+ }
+ };
+
+ const featureNodes = {
+ alternatives: records.filter(({ node }) => choiceChildren(node).length > 1),
+ quantifiers: records.filter(({ node }) => node.kind === "quantifier"),
+ optionals: records.filter(
+ ({ node }) =>
+ node.kind === "quantifier" && node.quantifier?.minimum === 0,
+ ),
+ classes: records.filter(
+ ({ node, parent }) =>
+ ["dot", "character-class", "unicode-property"].includes(node.kind) &&
+ !parent?.kind.startsWith("character-class"),
+ ),
+ unicode: records.filter(
+ ({ node, parent }) =>
+ !parent?.kind.startsWith("character-class") &&
+ (node.kind === "unicode-property" ||
+ node.raw.includes("\\p") ||
+ node.raw.includes("\\P")),
+ ),
+ anchors: records.filter(
+ ({ node }) => node.kind === "anchor" || node.kind === "word-boundary",
+ ),
+ };
+
+ if (request.flavour !== "ecmascript") {
+ const detail =
+ "Version 1 generation requires the complete ECMAScript normalized AST. The selected flavour’s partial provider is not reinterpreted as ECMAScript.";
+ return {
+ generator: {
+ id: GENERATED_CASE_GENERATOR_ID,
+ version: GENERATED_CASE_GENERATOR_VERSION,
+ },
+ flavour: request.flavour,
+ seed: request.seed,
+ seedHash,
+ candidates: [],
+ coverage: (
+ [
+ "shortest",
+ "alternative-coverage",
+ "quantifier-boundary",
+ "optional-presence",
+ "character-class",
+ "unicode-representative",
+ "likely-near-miss",
+ "anchor-boundary",
+ "seeded-random",
+ ] as const
+ ).map((feature) => coverageEntry(feature, "unsupported", detail, 0)),
+ unsupportedConstructs: [
+ {
+ id: "unsupported-flavour",
+ kind: request.root.kind,
+ range: request.root.range,
+ rawPreview: preview(request.root.raw),
+ reason: detail,
+ },
+ ],
+ visitedAstNodes: records.length,
+ candidateBytes: 0,
+ candidateAttempts: 0,
+ duplicateCandidates: 0,
+ candidatesTruncated: false,
+ traversalTruncated: traversalTruncated || depthTruncated,
+ stoppedReason: detail,
+ warnings: [detail],
+ };
+ }
+
+ if (traversalTruncated || depthTruncated) {
+ const detail = traversalTruncated
+ ? `Generation stopped at the ${request.settings.maximumAstNodes.toLocaleString()} node AST traversal limit.`
+ : `Generation stopped at the ${request.settings.maximumAstDepth.toLocaleString()} level AST-depth limit.`;
+ return {
+ generator: {
+ id: GENERATED_CASE_GENERATOR_ID,
+ version: GENERATED_CASE_GENERATOR_VERSION,
+ },
+ flavour: request.flavour,
+ seed: request.seed,
+ seedHash,
+ candidates: [],
+ coverage: (
+ [
+ "shortest",
+ "alternative-coverage",
+ "quantifier-boundary",
+ "optional-presence",
+ "character-class",
+ "unicode-representative",
+ "likely-near-miss",
+ "anchor-boundary",
+ "seeded-random",
+ ] as const
+ ).map((feature) => coverageEntry(feature, "unsupported", detail, 0)),
+ unsupportedConstructs: [],
+ visitedAstNodes: records.length,
+ candidateBytes: 0,
+ candidateAttempts: 0,
+ duplicateCandidates: 0,
+ candidatesTruncated: false,
+ traversalTruncated: true,
+ stoppedReason: detail,
+ warnings: [detail],
+ };
+ }
+
+ const unsupportedConstructs: UnsupportedGenerationConstruct[] = [];
+ for (const { node } of records) {
+ const reason = unsupportedReason(node);
+ if (!reason) continue;
+ if (unsupportedConstructs.length < MAXIMUM_UNSUPPORTED_CONSTRUCTS) {
+ unsupportedConstructs.push({
+ id: `unsupported-${unsupportedConstructs.length + 1}`,
+ kind: node.kind,
+ range: node.range,
+ rawPreview: preview(node.raw),
+ reason,
+ });
+ } else {
+ warn(
+ `Unsupported-construct reporting stopped at ${MAXIMUM_UNSUPPORTED_CONSTRUCTS.toLocaleString()} entries.`,
+ );
+ break;
+ }
+ }
+
+ const candidates: GeneratedSubjectCandidate[] = [];
+ const candidateKeys = new Set();
+ let candidateAttempts = 0;
+ let duplicateCandidates = 0;
+ let candidateBytes = 0;
+ let candidatesTruncated = false;
+ let stoppedReason: string | undefined;
+
+ const addCandidate = (
+ intendedOutcome: GeneratedSubjectCandidate["intendedOutcome"],
+ subject: string,
+ label: string,
+ features: readonly GenerationFeature[],
+ notes: readonly string[] = [],
+ ): boolean => {
+ if (candidateAttempts >= request.settings.maximumCandidateAttempts) {
+ candidatesTruncated = true;
+ stoppedReason ??= `Candidate synthesis reached the ${request.settings.maximumCandidateAttempts.toLocaleString()} attempt limit.`;
+ return false;
+ }
+ candidateAttempts += 1;
+ const bytes = utf8ByteLength(subject);
+ if (bytes > request.settings.maximumSubjectBytes) {
+ warn(
+ `At least one candidate exceeded the ${request.settings.maximumSubjectBytes.toLocaleString()} byte per-subject limit and was discarded.`,
+ );
+ return true;
+ }
+ const key = `${intendedOutcome}\0${subject}`;
+ if (candidateKeys.has(key)) {
+ duplicateCandidates += 1;
+ return true;
+ }
+ if (candidateBytes + bytes > request.settings.maximumTotalSubjectBytes) {
+ candidatesTruncated = true;
+ stoppedReason ??= `Candidate synthesis reached the ${request.settings.maximumTotalSubjectBytes.toLocaleString()} byte aggregate subject limit.`;
+ return false;
+ }
+ candidateKeys.add(key);
+ candidateBytes += bytes;
+ candidates.push({
+ id: `generated-${seedHash}-${candidates.length + 1}`,
+ intendedOutcome,
+ subject,
+ subjectBytes: bytes,
+ label,
+ features: unique(features) as readonly GenerationFeature[],
+ notes,
+ });
+ return true;
+ };
+
+ const render = (
+ overrides: RenderOverrides,
+ label: string,
+ features: readonly GenerationFeature[],
+ notes?: readonly string[],
+ ) => {
+ try {
+ const fragment = renderNode(
+ request.root,
+ overrides,
+ request.settings.maximumSubjectBytes,
+ request.settings.maximumAstDepth,
+ request.settings.maximumQuantifierRepetitions,
+ );
+ addCandidate("match", fragment.text, label, features, notes);
+ } catch (error) {
+ if (error instanceof RenderLimitError) warn(error.message);
+ else throw error;
+ }
+ };
+
+ render(
+ {},
+ "Shortest structural candidate",
+ ["shortest"],
+ unsupportedConstructs.length > 0
+ ? [
+ "Some pattern constructs are outside generator coverage; actual-engine verification is authoritative.",
+ ]
+ : [],
+ );
+
+ for (const { node } of featureNodes.alternatives) {
+ const alternatives = choiceChildren(node);
+ for (let index = 0; index < alternatives.length; index += 1) {
+ if (candidatesTruncated) break;
+ render(
+ { choices: new Map([[node.id, index]]) },
+ `Alternative ${index + 1} at ${node.range.startUtf16}`,
+ ["alternative-coverage"],
+ );
+ }
+ }
+
+ let quantifierCoverageTruncated = false;
+ for (const { node } of featureNodes.quantifiers) {
+ const counts = quantifierCounts(
+ node,
+ request.settings.maximumQuantifierRepetitions,
+ );
+ quantifierCoverageTruncated ||= counts.truncated;
+ for (const count of counts.counts) {
+ if (candidatesTruncated) break;
+ render(
+ { repetitions: new Map([[node.id, count]]) },
+ `Quantifier boundary ${count} at ${node.range.startUtf16}`,
+ [
+ "quantifier-boundary",
+ ...(node.quantifier?.minimum === 0
+ ? (["optional-presence"] as const)
+ : []),
+ ],
+ counts.truncated
+ ? [
+ `The exact upper boundary exceeds the ${request.settings.maximumQuantifierRepetitions.toLocaleString()} synthesis cap.`,
+ ]
+ : [],
+ );
+ }
+ }
+
+ for (const { node } of featureNodes.classes) {
+ const representatives = classRepresentatives(node);
+ representatives.slice(0, 8).forEach((_, index) => {
+ if (candidatesTruncated) return;
+ render(
+ { classChoices: new Map([[node.id, index]]) },
+ `Character-class representative ${index + 1} at ${node.range.startUtf16}`,
+ [
+ "character-class",
+ ...(node.kind === "unicode-property" ||
+ node.raw.startsWith("\\p") ||
+ node.raw.startsWith("\\P")
+ ? (["unicode-representative"] as const)
+ : []),
+ ],
+ );
+ });
+ }
+ for (const { node } of featureNodes.unicode) {
+ const representatives = classRepresentatives(node);
+ representatives.slice(0, 8).forEach((_, index) => {
+ if (candidatesTruncated) return;
+ render(
+ { classChoices: new Map([[node.id, index]]) },
+ `Unicode representative ${index + 1} at ${node.range.startUtf16}`,
+ ["unicode-representative"],
+ );
+ });
+ }
+
+ const random = createDeterministicRandom(request.seed);
+ for (
+ let index = 0;
+ index < request.settings.randomVariantCount && !candidatesTruncated;
+ index += 1
+ ) {
+ render(
+ { random },
+ `Seeded variant ${index + 1}`,
+ ["seeded-random"],
+ [`Seed ${JSON.stringify(request.seed)} · hash ${seedHash}`],
+ );
+ }
+
+ const positiveSubjects = candidates
+ .filter((candidate) => candidate.intendedOutcome === "match")
+ .map((candidate) => candidate.subject)
+ .slice(0, 32);
+
+ if (featureNodes.anchors.length > 0 && positiveSubjects[0] !== undefined) {
+ const subject = positiveSubjects[0];
+ const multiline = request.flags.includes("m");
+ addCandidate(
+ multiline ? "match" : "no-match",
+ `x\n${subject}`,
+ "Start-anchor boundary",
+ ["anchor-boundary"],
+ [
+ multiline
+ ? "Multiline mode makes a post-newline start a likely match boundary."
+ : "Without multiline mode, a non-empty prefix is a likely near miss.",
+ ],
+ );
+ addCandidate(
+ multiline ? "match" : "no-match",
+ `${subject}\nx`,
+ "End-anchor boundary",
+ ["anchor-boundary"],
+ [
+ multiline
+ ? "Multiline mode makes a pre-newline end a likely match boundary."
+ : "Without multiline mode, a non-newline suffix is a likely near miss.",
+ ],
+ );
+ }
+
+ for (const subject of positiveSubjects) {
+ if (candidatesTruncated) break;
+ for (const candidate of nearMisses(subject)) {
+ if (
+ !addCandidate(
+ "no-match",
+ candidate,
+ "Likely near miss",
+ ["likely-near-miss"],
+ [
+ "This mutation is retained only if the selected actual engine reports no match.",
+ ],
+ )
+ ) {
+ break;
+ }
+ }
+ }
+
+ if (quantifierCoverageTruncated) {
+ warn(
+ `At least one quantifier boundary exceeded the ${request.settings.maximumQuantifierRepetitions.toLocaleString()} repetition synthesis cap.`,
+ );
+ }
+
+ const partialFromUnsupported = unsupportedConstructs.length > 0;
+ const limited = candidatesTruncated || quantifierCoverageTruncated;
+ const applicable = (
+ relevant: number,
+ completeDetail: string,
+ unavailableDetail: string,
+ ): {
+ readonly status: GenerationCoverageStatus;
+ readonly detail: string;
+ } =>
+ relevant === 0
+ ? { status: "not-applicable", detail: unavailableDetail }
+ : partialFromUnsupported || limited
+ ? {
+ status: "partial",
+ detail: `${completeDetail} Coverage is bounded and some requested or structural paths were omitted.`,
+ }
+ : { status: "covered", detail: completeDetail };
+
+ const shortestStatus: GenerationCoverageStatus = partialFromUnsupported
+ ? "partial"
+ : "covered";
+ const alternative = applicable(
+ featureNodes.alternatives.length,
+ "Each structurally visible alternative was requested at least once.",
+ "The normalized AST contains no multi-way alternative.",
+ );
+ const quantifier = applicable(
+ featureNodes.quantifiers.length,
+ "Minimum, adjacent and bounded upper repetition samples were requested.",
+ "The normalized AST contains no quantifier.",
+ );
+ const optional = applicable(
+ featureNodes.optionals.length,
+ "Optional-absent and optional-present boundaries were requested.",
+ "The normalized AST contains no optional quantifier.",
+ );
+ const characterClass = applicable(
+ featureNodes.classes.length,
+ "Bounded representatives were requested for structurally visible classes.",
+ "The normalized AST contains no character class or dot.",
+ );
+ const unicode = applicable(
+ featureNodes.unicode.length,
+ "A bounded, documented representative set was requested for visible Unicode properties.",
+ "The normalized AST contains no Unicode property.",
+ );
+ const anchor = applicable(
+ featureNodes.anchors.length,
+ "Newline and non-boundary context candidates were requested.",
+ "The normalized AST contains no anchor or word-boundary assertion.",
+ );
+
+ const coverage: GenerationCoverageEntry[] = [
+ coverageEntry(
+ "shortest",
+ shortestStatus,
+ partialFromUnsupported
+ ? "A shortest structural candidate was requested, but unsupported zero-width or stateful constructs were not solved."
+ : "A candidate using shortest alternatives and minimum repetitions was requested.",
+ 1,
+ ),
+ coverageEntry(
+ "alternative-coverage",
+ alternative.status,
+ alternative.detail,
+ featureNodes.alternatives.length,
+ ),
+ coverageEntry(
+ "quantifier-boundary",
+ quantifier.status,
+ quantifier.detail,
+ featureNodes.quantifiers.length,
+ ),
+ coverageEntry(
+ "optional-presence",
+ optional.status,
+ optional.detail,
+ featureNodes.optionals.length,
+ ),
+ coverageEntry(
+ "character-class",
+ characterClass.status,
+ characterClass.detail,
+ featureNodes.classes.length,
+ ),
+ coverageEntry(
+ "unicode-representative",
+ unicode.status,
+ unicode.detail,
+ featureNodes.unicode.length,
+ ),
+ coverageEntry(
+ "likely-near-miss",
+ positiveSubjects.length > 0
+ ? candidatesTruncated
+ ? "partial"
+ : "covered"
+ : "unsupported",
+ positiveSubjects.length > 0
+ ? "Deletion, replacement and boundary mutations were requested; only actual-engine non-matches may be retained."
+ : "No positive structural subject was available to mutate.",
+ positiveSubjects.length,
+ ),
+ coverageEntry(
+ "anchor-boundary",
+ anchor.status,
+ anchor.detail,
+ featureNodes.anchors.length,
+ ),
+ coverageEntry(
+ "seeded-random",
+ request.settings.randomVariantCount === 0
+ ? "not-applicable"
+ : candidatesTruncated
+ ? "partial"
+ : "covered",
+ request.settings.randomVariantCount === 0
+ ? "Random variants were disabled."
+ : `${request.settings.randomVariantCount.toLocaleString()} deterministic variant requests used seed hash ${seedHash}.`,
+ request.settings.randomVariantCount,
+ ),
+ ];
+
+ const positiveCandidates = candidates.filter(
+ (candidate) => candidate.intendedOutcome === "match",
+ );
+ const negativeCandidates = candidates.filter(
+ (candidate) => candidate.intendedOutcome === "no-match",
+ );
+ const orderedCandidates: GeneratedSubjectCandidate[] = [];
+ const orderedLength = Math.max(
+ positiveCandidates.length,
+ negativeCandidates.length,
+ );
+ for (let index = 0; index < orderedLength; index += 1) {
+ const positive = positiveCandidates[index];
+ const negative = negativeCandidates[index];
+ if (positive) orderedCandidates.push(positive);
+ if (negative) orderedCandidates.push(negative);
+ }
+
+ return {
+ generator: {
+ id: GENERATED_CASE_GENERATOR_ID,
+ version: GENERATED_CASE_GENERATOR_VERSION,
+ },
+ flavour: request.flavour,
+ seed: request.seed,
+ seedHash,
+ candidates: orderedCandidates,
+ coverage,
+ unsupportedConstructs,
+ visitedAstNodes: records.length,
+ candidateBytes,
+ candidateAttempts,
+ duplicateCandidates,
+ candidatesTruncated,
+ traversalTruncated: false,
+ ...(stoppedReason ? { stoppedReason } : {}),
+ warnings,
+ };
+}
diff --git a/src/regex/generation/generation-limits.ts b/src/regex/generation/generation-limits.ts
new file mode 100644
index 0000000..ee3e021
--- /dev/null
+++ b/src/regex/generation/generation-limits.ts
@@ -0,0 +1,30 @@
+import { DEFAULT_REGEX_LIMITS } from "../execution/request-limits";
+import type { CandidateGenerationSettings } from "./generation.types";
+
+export const GENERATION_LIMITS = {
+ maximumSeedUtf16: 128,
+ maximumCandidateAttempts: 4_000,
+ maximumAstNodes: 20_000,
+ maximumAstDepth: 512,
+ maximumQuantifierRepetitions: 1_024,
+ maximumDiscardedDiagnostics: 250,
+ maximumDiagnosticPreviewUtf16: 160,
+ minimumPerCaseTimeoutMs: 25,
+ maximumPerCaseTimeoutMs: DEFAULT_REGEX_LIMITS.advancedMaximumTimeoutMs,
+ maximumWallTimeMs: DEFAULT_REGEX_LIMITS.maximumBenchmarkWallTimeMs,
+ maximumTotalSubjectBytes: DEFAULT_REGEX_LIMITS.interactiveSubjectHardBytes,
+ maximumSubjectBytes: DEFAULT_REGEX_LIMITS.interactiveSubjectHardBytes,
+} as const;
+
+export const DEFAULT_GENERATION_SETTINGS: CandidateGenerationSettings = {
+ maximumCases: 48,
+ maximumCandidateAttempts: 192,
+ maximumTotalSubjectBytes: 1024 * 1024,
+ maximumSubjectBytes: 64 * 1024,
+ maximumAstNodes: GENERATION_LIMITS.maximumAstNodes,
+ maximumAstDepth: GENERATION_LIMITS.maximumAstDepth,
+ maximumQuantifierRepetitions: 32,
+ randomVariantCount: 16,
+ perCaseTimeoutMs: 500,
+ maximumWallTimeMs: 15_000,
+};
diff --git a/src/regex/generation/generation-provenance.test.ts b/src/regex/generation/generation-provenance.test.ts
new file mode 100644
index 0000000..073abc3
--- /dev/null
+++ b/src/regex/generation/generation-provenance.test.ts
@@ -0,0 +1,29 @@
+import { describe, expect, it } from "vitest";
+import { parseGeneratedCaseProvenance } from "./generation-provenance";
+
+const fixture = {
+ kind: "generated",
+ generatorId: "regex-tools-ast-cases",
+ generatorVersion: "1",
+ seed: "release-seed",
+ candidateId: "generated-12345678-1",
+ intendedOutcome: "match",
+} as const;
+
+describe("generated-case provenance", () => {
+ it("round-trips the exact bounded generator identity and seed", () => {
+ expect(parseGeneratedCaseProvenance(fixture)).toEqual(fixture);
+ });
+
+ it.each([
+ [{ ...fixture, kind: "manual" }, "kind"],
+ [{ ...fixture, generatorId: "other" }, "generatorId"],
+ [{ ...fixture, generatorVersion: "2" }, "generatorVersion"],
+ [{ ...fixture, seed: "" }, "seed"],
+ [{ ...fixture, candidateId: "" }, "candidateId"],
+ [{ ...fixture, intendedOutcome: "maybe" }, "intendedOutcome"],
+ [{ ...fixture, unexpected: true }, "unsupported entries"],
+ ])("rejects unsupported imported provenance %#", (value, message) => {
+ expect(() => parseGeneratedCaseProvenance(value)).toThrow(message);
+ });
+});
diff --git a/src/regex/generation/generation-provenance.ts b/src/regex/generation/generation-provenance.ts
new file mode 100644
index 0000000..14ba4b4
--- /dev/null
+++ b/src/regex/generation/generation-provenance.ts
@@ -0,0 +1,79 @@
+import { GENERATION_LIMITS } from "./generation-limits";
+import {
+ GENERATED_CASE_GENERATOR_ID,
+ GENERATED_CASE_GENERATOR_VERSION,
+ type GeneratedCaseProvenance,
+} from "./generation.types";
+
+export const MAXIMUM_GENERATED_CANDIDATE_ID_UTF16 = 128;
+
+function record(value: unknown, label: string): Record {
+ if (!value || typeof value !== "object" || Array.isArray(value)) {
+ throw new Error(`${label} must be a JSON object.`);
+ }
+ return value as Record;
+}
+
+function text(value: unknown, label: string, maximum: number): string {
+ if (typeof value !== "string") {
+ throw new Error(`${label} must be text.`);
+ }
+ if (value.length === 0 || value.length > maximum) {
+ throw new Error(
+ `${label} must contain 1 to ${maximum.toLocaleString()} UTF-16 units.`,
+ );
+ }
+ return value;
+}
+
+export function parseGeneratedCaseProvenance(
+ value: unknown,
+ label = "Generated-case provenance",
+): GeneratedCaseProvenance {
+ const input = record(value, label);
+ const supported = new Set([
+ "kind",
+ "generatorId",
+ "generatorVersion",
+ "seed",
+ "candidateId",
+ "intendedOutcome",
+ ]);
+ const unknown = Object.keys(input).filter((key) => !supported.has(key));
+ if (unknown.length > 0) {
+ throw new Error(
+ `${label} contains unsupported entries: ${unknown.join(", ")}.`,
+ );
+ }
+ if (input.kind !== "generated") {
+ throw new Error(`${label} kind must be "generated".`);
+ }
+ if (input.generatorId !== GENERATED_CASE_GENERATOR_ID) {
+ throw new Error(
+ `${label} generatorId must be ${JSON.stringify(GENERATED_CASE_GENERATOR_ID)}.`,
+ );
+ }
+ if (input.generatorVersion !== GENERATED_CASE_GENERATOR_VERSION) {
+ throw new Error(
+ `${label} generatorVersion must be ${JSON.stringify(GENERATED_CASE_GENERATOR_VERSION)}.`,
+ );
+ }
+ if (
+ input.intendedOutcome !== "match" &&
+ input.intendedOutcome !== "no-match"
+ ) {
+ throw new Error(`${label} intendedOutcome must be "match" or "no-match".`);
+ }
+ return {
+ kind: "generated",
+ generatorId: GENERATED_CASE_GENERATOR_ID,
+ generatorVersion: GENERATED_CASE_GENERATOR_VERSION,
+ seed: text(input.seed, `${label} seed`, GENERATION_LIMITS.maximumSeedUtf16),
+ candidateId: text(
+ input.candidateId,
+ `${label} candidateId`,
+ MAXIMUM_GENERATED_CANDIDATE_ID_UTF16,
+ ),
+ intendedOutcome: input.intendedOutcome,
+ };
+}
diff --git a/src/regex/generation/generation.types.ts b/src/regex/generation/generation.types.ts
new file mode 100644
index 0000000..fae16a8
--- /dev/null
+++ b/src/regex/generation/generation.types.ts
@@ -0,0 +1,195 @@
+import type {
+ RegexEngineInfo,
+ RegexEngineOptions,
+ RegexFlavourId,
+} from "../model/flavour";
+import type { NormalizedRegexNode, SourceRange } from "../model/syntax";
+
+export const GENERATED_CASE_GENERATOR_ID = "regex-tools-ast-cases";
+export const GENERATED_CASE_GENERATOR_VERSION = "1";
+
+export type GenerationFeature =
+ | "shortest"
+ | "alternative-coverage"
+ | "quantifier-boundary"
+ | "optional-presence"
+ | "character-class"
+ | "unicode-representative"
+ | "likely-near-miss"
+ | "anchor-boundary"
+ | "seeded-random";
+
+export type GenerationCoverageStatus =
+ "covered" | "partial" | "unsupported" | "not-applicable";
+
+export interface GenerationCoverageEntry {
+ readonly feature: GenerationFeature;
+ readonly status: GenerationCoverageStatus;
+ readonly detail: string;
+ readonly relevantNodes: number;
+}
+
+export interface UnsupportedGenerationConstruct {
+ readonly id: string;
+ readonly kind: NormalizedRegexNode["kind"];
+ readonly range: SourceRange;
+ readonly rawPreview: string;
+ readonly reason: string;
+}
+
+export interface CandidateGenerationSettings {
+ readonly maximumCases: number;
+ readonly maximumCandidateAttempts: number;
+ readonly maximumTotalSubjectBytes: number;
+ readonly maximumSubjectBytes: number;
+ readonly maximumAstNodes: number;
+ readonly maximumAstDepth: number;
+ readonly maximumQuantifierRepetitions: number;
+ readonly randomVariantCount: number;
+ readonly perCaseTimeoutMs: number;
+ readonly maximumWallTimeMs: number;
+}
+
+export interface CandidateGenerationRequest {
+ readonly flavour: RegexFlavourId;
+ readonly root: NormalizedRegexNode;
+ readonly flags: readonly string[];
+ readonly seed: string;
+ readonly settings: CandidateGenerationSettings;
+}
+
+export interface GeneratedSubjectCandidate {
+ readonly id: string;
+ readonly intendedOutcome: "match" | "no-match";
+ readonly subject: string;
+ readonly subjectBytes: number;
+ readonly label: string;
+ readonly features: readonly GenerationFeature[];
+ readonly notes: readonly string[];
+}
+
+export interface CandidateGenerationReport {
+ readonly generator: {
+ readonly id: typeof GENERATED_CASE_GENERATOR_ID;
+ readonly version: typeof GENERATED_CASE_GENERATOR_VERSION;
+ };
+ readonly flavour: RegexFlavourId;
+ readonly seed: string;
+ readonly seedHash: string;
+ readonly candidates: readonly GeneratedSubjectCandidate[];
+ readonly coverage: readonly GenerationCoverageEntry[];
+ readonly unsupportedConstructs: readonly UnsupportedGenerationConstruct[];
+ readonly visitedAstNodes: number;
+ readonly candidateBytes: number;
+ readonly candidateAttempts: number;
+ readonly duplicateCandidates: number;
+ readonly candidatesTruncated: boolean;
+ readonly traversalTruncated: boolean;
+ readonly stoppedReason?: string;
+ readonly warnings: readonly string[];
+}
+
+export type GenerationDiscardReason =
+ | "unexpected-match"
+ | "unexpected-no-match"
+ | "compile-rejected"
+ | "timeout"
+ | "worker-crash"
+ | "execution-error"
+ | "wall-time-limit";
+
+export interface DiscardedGeneratedCandidate {
+ readonly candidateId: string;
+ readonly intendedOutcome: GeneratedSubjectCandidate["intendedOutcome"];
+ readonly reason: GenerationDiscardReason;
+ readonly detail: string;
+ readonly subjectBytes: number;
+ readonly subjectPreview: string;
+}
+
+export interface GeneratedCaseProvenance {
+ readonly kind: "generated";
+ readonly generatorId: typeof GENERATED_CASE_GENERATOR_ID;
+ readonly generatorVersion: typeof GENERATED_CASE_GENERATOR_VERSION;
+ readonly seed: string;
+ readonly candidateId: string;
+ readonly intendedOutcome: GeneratedSubjectCandidate["intendedOutcome"];
+}
+
+export interface VerifiedGeneratedCase {
+ readonly id: string;
+ readonly name: string;
+ readonly subject: string;
+ readonly subjectBytes: number;
+ readonly expectation: "should-match" | "should-not-match";
+ readonly features: readonly GenerationFeature[];
+ readonly notes: readonly string[];
+ readonly elapsedMs: number;
+ readonly matched: boolean;
+ readonly matchCount: number;
+ readonly provenance: GeneratedCaseProvenance;
+}
+
+export type CaseGenerationStatus =
+ | "complete"
+ | "partial"
+ | "unsupported"
+ | "compile-rejected"
+ | "wall-time-limit";
+
+export interface CaseGenerationResult {
+ readonly status: CaseGenerationStatus;
+ readonly seed: string;
+ readonly seedHash: string;
+ readonly flavour: RegexFlavourId;
+ readonly cases: readonly VerifiedGeneratedCase[];
+ readonly discarded: readonly DiscardedGeneratedCandidate[];
+ readonly coverage: readonly GenerationCoverageEntry[];
+ readonly unsupportedConstructs: readonly UnsupportedGenerationConstruct[];
+ readonly engine?: RegexEngineInfo;
+ readonly candidateAttempts: number;
+ readonly engineExecutions: number;
+ readonly generatedCandidateBytes: number;
+ readonly verifiedSubjectBytes: number;
+ readonly wallTimeMs: number;
+ readonly stoppedReason?: string;
+ readonly warnings: readonly string[];
+}
+
+export interface CaseGenerationInput {
+ readonly flavour: RegexFlavourId;
+ readonly flavourVersion?: string;
+ readonly pattern: string;
+ readonly flags: readonly string[];
+ readonly options: RegexEngineOptions;
+ readonly scanAll: boolean;
+ readonly root: NormalizedRegexNode;
+ readonly captureMetadata: readonly {
+ readonly number: number;
+ readonly name?: string;
+ readonly range: SourceRange;
+ readonly repeated: boolean;
+ readonly parentCaptureNumber?: number;
+ }[];
+ readonly syntaxAccepted: boolean;
+ readonly seed: string;
+ readonly settings: CandidateGenerationSettings;
+}
+
+export interface CaseGenerationProgress {
+ readonly phase: "synthesizing" | "verifying";
+ readonly completed: number;
+ readonly total: number;
+ readonly retained: number;
+ readonly message: string;
+}
+
+export type GenerationWorkerOperation = {
+ readonly kind: "generate";
+ readonly request: CandidateGenerationRequest;
+};
+
+export type GenerationWorkerResult = {
+ readonly kind: "generate";
+ readonly result: CandidateGenerationReport;
+};
diff --git a/src/regex/generation/seeded-random.test.ts b/src/regex/generation/seeded-random.test.ts
new file mode 100644
index 0000000..63edfd5
--- /dev/null
+++ b/src/regex/generation/seeded-random.test.ts
@@ -0,0 +1,34 @@
+import { describe, expect, it } from "vitest";
+import {
+ createDeterministicRandom,
+ generationSeedHash,
+ hashGenerationSeed,
+} from "./seeded-random";
+
+describe("deterministic generation random", () => {
+ it("replays the same sequence and exposes a stable seed hash", () => {
+ const first = createDeterministicRandom("release-seed-17");
+ const second = createDeterministicRandom("release-seed-17");
+ const expected = [
+ 997_101_319, 2_253_165_009, 3_652_172_136, 3_704_011_065, 218_024_818,
+ 4_086_101_089, 1_433_907_551, 2_457_124_617,
+ ];
+
+ expect(Array.from({ length: 8 }, () => first.nextUint32())).toEqual(
+ expected,
+ );
+ expect(Array.from({ length: 8 }, () => second.nextUint32())).toEqual(
+ expected,
+ );
+ expect(generationSeedHash("release-seed-17")).toBe("24eea7d5");
+ expect(hashGenerationSeed("release-seed-17")).not.toBe(
+ hashGenerationSeed("release-seed-18"),
+ );
+ });
+
+ it("rejects invalid selection bounds", () => {
+ const random = createDeterministicRandom("seed");
+ expect(() => random.integer(0)).toThrow("positive bound");
+ expect(() => random.pick([])).toThrow("empty collection");
+ });
+});
diff --git a/src/regex/generation/seeded-random.ts b/src/regex/generation/seeded-random.ts
new file mode 100644
index 0000000..187eb25
--- /dev/null
+++ b/src/regex/generation/seeded-random.ts
@@ -0,0 +1,51 @@
+/**
+ * Stable FNV-1a over UTF-16 code units. The explicit code-unit model makes the
+ * same seed reproducible in every JavaScript runtime without locale or encoder
+ * dependencies.
+ */
+export function hashGenerationSeed(seed: string): number {
+ let hash = 0x811c9dc5;
+ for (let index = 0; index < seed.length; index += 1) {
+ const unit = seed.charCodeAt(index);
+ hash ^= unit & 0xff;
+ hash = Math.imul(hash, 0x01000193);
+ hash ^= unit >>> 8;
+ hash = Math.imul(hash, 0x01000193);
+ }
+ return hash >>> 0;
+}
+
+export function generationSeedHash(seed: string): string {
+ return hashGenerationSeed(seed).toString(16).padStart(8, "0");
+}
+
+export interface DeterministicRandom {
+ nextUint32(): number;
+ integer(maximumExclusive: number): number;
+ pick(values: readonly T[]): T;
+}
+
+export function createDeterministicRandom(seed: string): DeterministicRandom {
+ let state = hashGenerationSeed(seed);
+ return {
+ nextUint32(): number {
+ state = (state + 0x6d2b79f5) >>> 0;
+ let value = state;
+ value = Math.imul(value ^ (value >>> 15), value | 1);
+ value ^= value + Math.imul(value ^ (value >>> 7), value | 61);
+ return (value ^ (value >>> 14)) >>> 0;
+ },
+ integer(maximumExclusive: number): number {
+ if (!Number.isSafeInteger(maximumExclusive) || maximumExclusive < 1) {
+ throw new RangeError("Random selection needs a positive bound.");
+ }
+ return this.nextUint32() % maximumExclusive;
+ },
+ pick(values: readonly T[]): T {
+ if (values.length === 0) {
+ throw new RangeError("Cannot select from an empty collection.");
+ }
+ return values[this.integer(values.length)] as T;
+ },
+ };
+}
diff --git a/src/regex/minimization/SubjectMinimizer.test.ts b/src/regex/minimization/SubjectMinimizer.test.ts
new file mode 100644
index 0000000..b270096
--- /dev/null
+++ b/src/regex/minimization/SubjectMinimizer.test.ts
@@ -0,0 +1,471 @@
+import { describe, expect, it, vi } from "vitest";
+import type { RegexEngineInfo } from "../model/flavour";
+import type {
+ RegexExecutionRequest,
+ RegexExecutionResult,
+ RegexReplacementRequest,
+ RegexReplacementResult,
+} from "../model/match";
+import type { RegexSyntaxRequest, RegexSyntaxResult } from "../model/syntax";
+import { DEFAULT_REGEX_LIMITS } from "../execution/request-limits";
+import { WorkerRequestError } from "../execution/WorkerSupervisor";
+import {
+ SubjectMinimizer,
+ validateSubjectMinimizationRequest,
+ type MinimizerEngineRunner,
+ type MinimizerSyntaxRunner,
+} from "./SubjectMinimizer";
+import {
+ MINIMIZER_MAXIMUM_SUBJECT_BYTES,
+ type SubjectMinimizationRequest,
+} from "./minimization.types";
+
+const ECMASCRIPT_VERSION = "ECMAScript 2025 syntax / current browser runtime";
+const PCRE2_VERSION = "PCRE2 10.47 8-bit WebAssembly";
+
+function engineInfo(flavour: "ecmascript" | "pcre2"): RegexEngineInfo {
+ return {
+ flavour,
+ adapterVersion: "fixture",
+ engineName: flavour === "ecmascript" ? "Fixture JS" : "Fixture PCRE2",
+ engineVersion: flavour === "ecmascript" ? "browser-1" : "10.47",
+ offsetUnit: flavour === "ecmascript" ? "utf16" : "utf8-byte",
+ capabilities: {
+ compilation: true,
+ matching: true,
+ replacement: true,
+ namedCaptures: true,
+ captureHistory: false,
+ actualTrace: false,
+ benchmark: false,
+ },
+ };
+}
+
+function syntax(request: RegexSyntaxRequest): RegexSyntaxResult {
+ return {
+ accepted: true,
+ root: {
+ id: `${request.flavour}-root`,
+ kind: "pattern",
+ range: { startUtf16: 0, endUtf16: request.pattern.length },
+ raw: request.pattern,
+ explanation: "Fixture",
+ children: [],
+ properties: {
+ zeroWidth: false,
+ nullable: "unknown",
+ consumesInput: "conditional",
+ },
+ support: {
+ flavour: request.flavour,
+ status: "supported",
+ notes: [],
+ },
+ provenance: {
+ provider: "fixture",
+ providerVersion: "1",
+ source: "parsed",
+ },
+ },
+ tokens: [],
+ captures: [],
+ diagnostics: [],
+ provider: { id: "fixture", version: "1" },
+ coverage: {
+ status: "full-tested",
+ summary: "Fixture",
+ unsupportedConstructs: [],
+ },
+ };
+}
+
+function syntaxRunner(): MinimizerSyntaxRunner {
+ return {
+ parsePattern: async (request) => syntax(request),
+ cancel: vi.fn(),
+ dispose: vi.fn(),
+ };
+}
+
+function execution(
+ request: RegexExecutionRequest,
+ value?: string,
+): RegexExecutionResult {
+ const start = request.subject.indexOf("X");
+ const hasMatch = value !== undefined;
+ const rangeStart = Math.max(0, start);
+ return {
+ accepted: true,
+ engine: engineInfo(request.flavour as "ecmascript" | "pcre2"),
+ flags: {
+ userFlags: request.flags.join(""),
+ effectiveFlags: request.flags.join(""),
+ internallyAddedIndicesFlag: request.flavour === "ecmascript",
+ internallyAddedGlobalFlag: false,
+ },
+ matches: hasMatch
+ ? [
+ {
+ matchNumber: 1,
+ value,
+ valueStatus: "complete",
+ range: { startUtf16: rangeStart, endUtf16: rangeStart + 1 },
+ nativeRange: {
+ start: rangeStart,
+ end: rangeStart + 1,
+ unit: request.flavour === "pcre2" ? "utf8-byte" : "utf16",
+ },
+ captures: [],
+ },
+ ]
+ : [],
+ diagnostics: [],
+ elapsedMs: 1,
+ truncated: false,
+ };
+}
+
+function engineRunner(
+ execute: (
+ request: RegexExecutionRequest,
+ timeoutMs: number,
+ ) => Promise,
+): MinimizerEngineRunner {
+ return {
+ execute,
+ replace: async (
+ request: RegexReplacementRequest,
+ timeoutMs: number,
+ ): Promise => {
+ const result = await execute(request, timeoutMs);
+ return {
+ execution: result,
+ output: request.subject.replaceAll("X", request.replacement),
+ outputBytes: request.subject.length,
+ outputTruncated: false,
+ truncated: false,
+ };
+ },
+ cancel: vi.fn(),
+ dispose: vi.fn(),
+ };
+}
+
+function side(flavour: "ecmascript" | "pcre2") {
+ return {
+ flavour,
+ flavourVersion:
+ flavour === "ecmascript" ? ECMASCRIPT_VERSION : PCRE2_VERSION,
+ pattern: "X",
+ flags: ["g"],
+ options: {},
+ } as const;
+}
+
+function baseRequest(
+ target: SubjectMinimizationRequest["target"],
+ subject = "abXcd",
+): SubjectMinimizationRequest {
+ return {
+ schemaVersion: 1,
+ subject,
+ target,
+ budgets: {
+ maximumEvaluations: 200,
+ maximumWallTimeMs: 10_000,
+ candidateTimeoutMs: 100,
+ },
+ maximumMatches: 100,
+ maximumCaptureRows: 1_000,
+ maximumOutputBytes: 4_096,
+ };
+}
+
+function minimizer(runner: MinimizerEngineRunner): SubjectMinimizer {
+ return new SubjectMinimizer({
+ createSyntaxRunner: () => syntaxRunner(),
+ createEngineRunner: () => runner,
+ now: () => 0,
+ });
+}
+
+describe("SubjectMinimizer real-run orchestration", () => {
+ it("minimizes a failing unit-test subject against the selected engine", async () => {
+ const runner = engineRunner(async (request) =>
+ execution(request, request.subject.includes("X") ? "X" : undefined),
+ );
+ const instance = minimizer(runner);
+ const progress: string[] = [];
+
+ const result = await instance.minimize(
+ baseRequest({
+ kind: "engine",
+ side: side("ecmascript"),
+ scanAll: true,
+ oracle: {
+ kind: "unit-test-failure",
+ expectation: { kind: "should-not-match" },
+ },
+ }),
+ (value) => progress.push(value.phase),
+ );
+
+ expect(result).toMatchObject({
+ status: "complete",
+ stopReason: "locally-minimal",
+ minimizedSubject: "X",
+ baseline: {
+ reproduced: true,
+ fingerprint: ["expected-no-match:present"],
+ },
+ final: {
+ reproduced: true,
+ fingerprint: ["expected-no-match:present"],
+ },
+ });
+ expect(result.baseline.sides?.[0]?.engineIdentity).toMatch(/Fixture JS/u);
+ expect(progress).toContain("syntax-setup");
+ expect(progress).toContain("local-sweep");
+ instance.dispose();
+ });
+
+ it("preserves an exact engine timeout and never relabels a crash as one", async () => {
+ const timeoutRunner = engineRunner(async (request) => {
+ if (request.subject.includes("X")) {
+ throw new WorkerRequestError("timeout", "fixture deadline");
+ }
+ return execution(request);
+ });
+ const timeoutMinimizer = minimizer(timeoutRunner);
+ const timeoutResult = await timeoutMinimizer.minimize(
+ baseRequest({
+ kind: "engine",
+ side: side("ecmascript"),
+ scanAll: false,
+ oracle: { kind: "engine-timeout" },
+ }),
+ );
+
+ expect(timeoutResult).toMatchObject({
+ status: "complete",
+ minimizedSubject: "X",
+ baseline: { status: "timeout", reproduced: true },
+ final: { status: "timeout", reproduced: true },
+ });
+ timeoutMinimizer.dispose();
+
+ const crashRunner = engineRunner(async () => {
+ throw new WorkerRequestError("crash", "fixture worker crashed");
+ });
+ const crashMinimizer = minimizer(crashRunner);
+ const crashResult = await crashMinimizer.minimize(
+ baseRequest({
+ kind: "engine",
+ side: side("ecmascript"),
+ scanAll: false,
+ oracle: { kind: "engine-timeout" },
+ }),
+ );
+
+ expect(crashResult).toMatchObject({
+ status: "unsupported",
+ stopReason: "engine-failure",
+ locallyMinimal: false,
+ baseline: { status: "crash", reproduced: false },
+ });
+ expect(crashResult.baseline.summary).toMatch(/not treated as a timeout/u);
+ crashMinimizer.dispose();
+ });
+
+ it("preserves the exact semantic comparison mismatch category", async () => {
+ const runner = engineRunner(async (request) => {
+ if (!request.subject) return execution(request);
+ if (request.flavour === "ecmascript") {
+ return execution(request, request.subject.includes("X") ? "X" : "a");
+ }
+ return request.subject.includes("X")
+ ? execution(request, "!")
+ : execution(request);
+ });
+ const instance = minimizer(runner);
+
+ const result = await instance.minimize(
+ baseRequest(
+ {
+ kind: "comparison-mismatch",
+ operation: "match",
+ scanAll: true,
+ sides: [side("ecmascript"), side("pcre2")],
+ },
+ "aX",
+ ),
+ );
+
+ expect(result.minimizedSubject).toBe("X");
+ expect(result.baseline).toMatchObject({
+ reproduced: true,
+ mismatchKinds: ["match-value"],
+ fingerprint: ["comparison:semantic-mismatch", "match-value"],
+ });
+ expect(result.final.mismatchKinds).toEqual(["match-value"]);
+ expect(result.locallyMinimal).toBe(true);
+ instance.dispose();
+ });
+
+ it("preserves which comparison side timed out while its peer completes authoritatively", async () => {
+ const runner = engineRunner(async (request) => {
+ if (request.flavour === "pcre2" && request.subject.includes("X")) {
+ throw new WorkerRequestError("timeout", "PCRE2 fixture deadline");
+ }
+ return execution(request);
+ });
+ const instance = minimizer(runner);
+
+ const result = await instance.minimize(
+ baseRequest({
+ kind: "comparison-mismatch",
+ operation: "match",
+ scanAll: true,
+ sides: [side("ecmascript"), side("pcre2")],
+ }),
+ );
+
+ expect(result).toMatchObject({
+ status: "complete",
+ minimizedSubject: "X",
+ baseline: {
+ status: "timeout",
+ reproduced: true,
+ fingerprint: ["comparison:timeout:pcre2"],
+ },
+ final: {
+ status: "timeout",
+ reproduced: true,
+ fingerprint: ["comparison:timeout:pcre2"],
+ },
+ });
+ instance.dispose();
+ });
+
+ it("terminates active workers and returns cancellation progress", async () => {
+ let rejectActive: ((reason: WorkerRequestError) => void) | undefined;
+ let signalStarted: (() => void) | undefined;
+ const started = new Promise((resolve) => {
+ signalStarted = resolve;
+ });
+ const runner = engineRunner(
+ async () =>
+ new Promise((_resolve, reject) => {
+ rejectActive = reject;
+ signalStarted?.();
+ }),
+ );
+ runner.cancel = vi.fn(() => {
+ rejectActive?.(
+ new WorkerRequestError("cancelled", "fixture request cancelled"),
+ );
+ });
+ const instance = minimizer(runner);
+ const running = instance.minimize(
+ baseRequest({
+ kind: "engine",
+ side: side("ecmascript"),
+ scanAll: false,
+ oracle: { kind: "engine-timeout" },
+ }),
+ );
+
+ await started;
+ instance.cancel();
+ const result = await running;
+
+ expect(runner.cancel).toHaveBeenCalled();
+ expect(result).toMatchObject({
+ status: "cancelled",
+ stopReason: "cancelled",
+ locallyMinimal: false,
+ });
+ instance.dispose();
+ });
+
+ it("fails closed when the baseline result is truncated", async () => {
+ const runner = engineRunner(async (request) => ({
+ ...execution(request, "X"),
+ truncated: true,
+ }));
+ const instance = minimizer(runner);
+
+ const result = await instance.minimize(
+ baseRequest({
+ kind: "engine",
+ side: side("ecmascript"),
+ scanAll: true,
+ oracle: {
+ kind: "unit-test-failure",
+ expectation: { kind: "should-not-match" },
+ },
+ }),
+ );
+
+ expect(result).toMatchObject({
+ status: "unsupported",
+ stopReason: "baseline-not-reproduced",
+ evaluations: 1,
+ baseline: {
+ status: "inconclusive",
+ reproduced: false,
+ fingerprint: ["truncated-result"],
+ },
+ });
+ instance.dispose();
+ });
+});
+
+describe("subject-minimization request bounds", () => {
+ const request = () =>
+ baseRequest({
+ kind: "engine",
+ side: side("ecmascript"),
+ scanAll: false,
+ oracle: { kind: "engine-timeout" },
+ });
+
+ it("enforces evaluation, aggregate wall and fixed candidate budgets", () => {
+ expect(() =>
+ validateSubjectMinimizationRequest({
+ ...request(),
+ budgets: {
+ maximumEvaluations: DEFAULT_REGEX_LIMITS.maximumMinimizerRuns + 1,
+ maximumWallTimeMs: 1_000,
+ candidateTimeoutMs: 100,
+ },
+ }),
+ ).toThrow(/evaluation budget/u);
+ expect(() =>
+ validateSubjectMinimizationRequest({
+ ...request(),
+ budgets: {
+ maximumEvaluations: 10,
+ maximumWallTimeMs: 100,
+ candidateTimeoutMs: 101,
+ },
+ }),
+ ).toThrow(/must fit inside/u);
+ });
+
+ it("rejects oversized or malformed Unicode subjects before worker setup", () => {
+ expect(() =>
+ validateSubjectMinimizationRequest({
+ ...request(),
+ subject: "a".repeat(MINIMIZER_MAXIMUM_SUBJECT_BYTES + 1),
+ }),
+ ).toThrow(/1,048,576 UTF-8 bytes/u);
+ expect(() =>
+ validateSubjectMinimizationRequest({
+ ...request(),
+ subject: "\ud800",
+ }),
+ ).toThrow(/lone surrogate/u);
+ });
+});
diff --git a/src/regex/minimization/SubjectMinimizer.ts b/src/regex/minimization/SubjectMinimizer.ts
new file mode 100644
index 0000000..e038d46
--- /dev/null
+++ b/src/regex/minimization/SubjectMinimizer.ts
@@ -0,0 +1,1052 @@
+import type {
+ RegexExecutionRequest,
+ RegexExecutionResult,
+ RegexReplacementRequest,
+ RegexReplacementResult,
+} from "../model/match";
+import type { RegexSyntaxRequest, RegexSyntaxResult } from "../model/syntax";
+import {
+ AVAILABLE_REGEX_FLAVOURS,
+ parseRegexFlags,
+ parseRegexOptions,
+ resolveRegexFlavourVersion,
+} from "../flavours/flavour-registry";
+import {
+ DEFAULT_REGEX_LIMITS,
+ utf8ByteLength,
+} from "../execution/request-limits";
+import { EngineSupervisor } from "../execution/EngineSupervisor";
+import { SyntaxSupervisor } from "../execution/SyntaxSupervisor";
+import { WorkerRequestError } from "../execution/WorkerSupervisor";
+import type {
+ ComparisonFlavour,
+ ComparisonSideInput,
+} from "../comparison/comparison.types";
+import type { RegexTestExpectation } from "../tests/test-case.types";
+import {
+ compareSemanticResults,
+ engineIdentity,
+ evaluateCaptureRangeMismatch,
+ evaluateUnitTestFailure,
+ type PredicateEvaluation,
+ type SemanticSideResult,
+} from "./oracles";
+import {
+ MINIMIZER_MAXIMUM_SUBJECT_BYTES,
+ MINIMIZER_MAXIMUM_WALL_TIME_MS,
+ type CandidateSideObservation,
+ type MinimizationCandidateObservation,
+ type SubjectMinimizationProgress,
+ type SubjectMinimizationRequest,
+ type SubjectMinimizationResult,
+} from "./minimization.types";
+import { containsLoneSurrogate, reduceSubject } from "./subject-transforms";
+
+export interface MinimizerSyntaxRunner {
+ parsePattern(
+ request: RegexSyntaxRequest,
+ timeoutMs?: number,
+ ): Promise;
+ cancel(): void;
+ dispose(): void;
+}
+
+export interface MinimizerEngineRunner {
+ execute(
+ request: RegexExecutionRequest,
+ timeoutMs: number,
+ ): Promise;
+ replace(
+ request: RegexReplacementRequest,
+ timeoutMs: number,
+ ): Promise;
+ cancel(): void;
+ dispose(): void;
+}
+
+export interface SubjectMinimizerDependencies {
+ readonly createSyntaxRunner: (
+ flavour: ComparisonFlavour,
+ ) => MinimizerSyntaxRunner;
+ readonly createEngineRunner: () => MinimizerEngineRunner;
+ readonly now: () => number;
+}
+
+const DEFAULT_DEPENDENCIES: SubjectMinimizerDependencies = {
+ createSyntaxRunner: (flavour) =>
+ new SyntaxSupervisor(
+ `${flavour} minimizer syntax`,
+ `regex-tools-minimizer-${flavour}-syntax`,
+ ),
+ createEngineRunner: () => new EngineSupervisor(),
+ now: () => performance.now(),
+};
+
+interface PreparedSide {
+ readonly input: ComparisonSideInput;
+ readonly syntaxRequest: RegexSyntaxRequest;
+ readonly captures: readonly RegexSyntaxResult["captures"][number][];
+}
+
+interface SideRuntimeComplete {
+ readonly status: "complete";
+ readonly execution: RegexExecutionResult;
+ readonly replacement?: RegexReplacementResult;
+}
+
+interface SideRuntimeFailure {
+ readonly status: "timeout" | "cancelled" | "crash" | "worker-error";
+ readonly message: string;
+}
+
+type SideRuntime = SideRuntimeComplete | SideRuntimeFailure;
+
+interface ValidatedRequest {
+ readonly request: SubjectMinimizationRequest;
+ readonly sides: readonly ComparisonSideInput[];
+ readonly targetIdentity: string;
+}
+
+function boundedInteger(
+ value: number,
+ label: string,
+ minimum: number,
+ maximum: number,
+): number {
+ if (!Number.isSafeInteger(value) || value < minimum || value > maximum) {
+ throw new RangeError(
+ `${label} must be an integer from ${minimum.toLocaleString()} to ${maximum.toLocaleString()}.`,
+ );
+ }
+ return value;
+}
+
+function normalizedSide(input: ComparisonSideInput): ComparisonSideInput {
+ const definition = AVAILABLE_REGEX_FLAVOURS.require(input.flavour);
+ const version = resolveRegexFlavourVersion(
+ input.flavourVersion,
+ definition,
+ `${definition.label} minimizer version`,
+ ).definition;
+ const flags = parseRegexFlags(
+ input.flags,
+ definition,
+ `${definition.label} minimizer flags`,
+ );
+ const options = parseRegexOptions(
+ input.options,
+ definition,
+ `${definition.label} minimizer options`,
+ );
+ if (input.pattern.length > DEFAULT_REGEX_LIMITS.patternHardLengthUtf16) {
+ throw new RangeError(
+ `${definition.label} minimizer pattern exceeds the ${DEFAULT_REGEX_LIMITS.patternHardLengthUtf16.toLocaleString()} UTF-16 unit limit.`,
+ );
+ }
+ if (
+ input.replacement !== undefined &&
+ input.replacement.length >
+ DEFAULT_REGEX_LIMITS.maximumReplacementTemplateUtf16
+ ) {
+ throw new RangeError(
+ `${definition.label} replacement exceeds the ${DEFAULT_REGEX_LIMITS.maximumReplacementTemplateUtf16.toLocaleString()} UTF-16 unit limit.`,
+ );
+ }
+ return {
+ flavour: input.flavour,
+ flavourVersion: version.value,
+ pattern: input.pattern,
+ flags,
+ options,
+ ...(input.replacement === undefined
+ ? {}
+ : { replacement: input.replacement }),
+ };
+}
+
+function stableValue(value: unknown): unknown {
+ if (Array.isArray(value)) return value.map(stableValue);
+ if (value && typeof value === "object") {
+ return Object.fromEntries(
+ Object.entries(value)
+ .sort(([left], [right]) => left.localeCompare(right))
+ .map(([key, entry]) => [key, stableValue(entry)]),
+ );
+ }
+ return value;
+}
+
+function identity(value: unknown): string {
+ const source = JSON.stringify(stableValue(value));
+ let hash = 0x811c9dc5;
+ for (let index = 0; index < source.length; index += 1) {
+ hash ^= source.charCodeAt(index);
+ hash = Math.imul(hash, 0x01000193);
+ }
+ return `minimization-${(hash >>> 0).toString(16).padStart(8, "0")}`;
+}
+
+function validateCaptureSelector(
+ value: {
+ readonly matchIndex: number;
+ readonly groupNumber?: number;
+ readonly groupName?: string;
+ },
+ label: string,
+): void {
+ boundedInteger(value.matchIndex, `${label} match index`, 0, 1_000_000);
+ if (value.groupNumber === undefined && value.groupName === undefined) {
+ throw new Error(`${label} requires a capture number or name.`);
+ }
+ if (value.groupNumber !== undefined) {
+ boundedInteger(
+ value.groupNumber,
+ `${label} group number`,
+ 1,
+ DEFAULT_REGEX_LIMITS.maximumCaptureGroups,
+ );
+ }
+ if (
+ value.groupName !== undefined &&
+ (value.groupName.length === 0 || value.groupName.length > 256)
+ ) {
+ throw new Error(`${label} group name must contain 1–256 UTF-16 units.`);
+ }
+}
+
+function validateExpectation(expectation: RegexTestExpectation): void {
+ switch (expectation.kind) {
+ case "match-count":
+ boundedInteger(
+ expectation.count,
+ "Expected match count",
+ 0,
+ DEFAULT_REGEX_LIMITS.maximumMatches,
+ );
+ return;
+ case "full-match":
+ boundedInteger(
+ expectation.matchIndex,
+ "Expected match index",
+ 0,
+ DEFAULT_REGEX_LIMITS.maximumMatches - 1,
+ );
+ return;
+ case "capture":
+ validateCaptureSelector(expectation, "Capture expectation");
+ return;
+ case "must-complete-within":
+ boundedInteger(
+ expectation.milliseconds,
+ "Completion assertion",
+ 1,
+ DEFAULT_REGEX_LIMITS.advancedMaximumTimeoutMs,
+ );
+ return;
+ default:
+ return;
+ }
+}
+
+export function validateSubjectMinimizationRequest(
+ input: SubjectMinimizationRequest,
+): ValidatedRequest {
+ if (input.schemaVersion !== 1) {
+ throw new Error("Unsupported subject-minimization request schema.");
+ }
+ if (containsLoneSurrogate(input.subject)) {
+ throw new Error(
+ "Subject minimization requires well-formed Unicode and rejects lone surrogate code units.",
+ );
+ }
+ if (utf8ByteLength(input.subject) > MINIMIZER_MAXIMUM_SUBJECT_BYTES) {
+ throw new RangeError(
+ `Subject minimization is limited to ${MINIMIZER_MAXIMUM_SUBJECT_BYTES.toLocaleString()} UTF-8 bytes.`,
+ );
+ }
+ const maximumEvaluations = boundedInteger(
+ input.budgets.maximumEvaluations,
+ "Minimizer evaluation budget",
+ 1,
+ DEFAULT_REGEX_LIMITS.maximumMinimizerRuns,
+ );
+ const maximumWallTimeMs = boundedInteger(
+ input.budgets.maximumWallTimeMs,
+ "Minimizer wall-time budget",
+ 1,
+ MINIMIZER_MAXIMUM_WALL_TIME_MS,
+ );
+ const candidateTimeoutMs = boundedInteger(
+ input.budgets.candidateTimeoutMs,
+ "Minimizer candidate timeout",
+ 1,
+ DEFAULT_REGEX_LIMITS.advancedMaximumTimeoutMs,
+ );
+ if (candidateTimeoutMs > maximumWallTimeMs) {
+ throw new RangeError(
+ "The fixed candidate timeout must fit inside the aggregate wall-time budget.",
+ );
+ }
+ const maximumMatches = boundedInteger(
+ input.maximumMatches,
+ "Minimizer match limit",
+ 1,
+ DEFAULT_REGEX_LIMITS.maximumMatches,
+ );
+ const maximumCaptureRows = boundedInteger(
+ input.maximumCaptureRows,
+ "Minimizer capture-row limit",
+ 1,
+ DEFAULT_REGEX_LIMITS.maximumCaptureRows,
+ );
+ const maximumOutputBytes = boundedInteger(
+ input.maximumOutputBytes,
+ "Minimizer replacement-output limit",
+ 1,
+ DEFAULT_REGEX_LIMITS.maximumReplacementOutputBytes,
+ );
+
+ let sides: readonly ComparisonSideInput[];
+ let target: SubjectMinimizationRequest["target"];
+ if (input.target.kind === "engine") {
+ const side = normalizedSide(input.target.side);
+ if (input.target.oracle.kind === "unit-test-failure") {
+ validateExpectation(input.target.oracle.expectation);
+ if (
+ input.target.oracle.expectation.kind === "replacement" &&
+ side.replacement === undefined
+ ) {
+ throw new Error(
+ "A replacement unit-test failure requires an explicit replacement template.",
+ );
+ }
+ } else if (input.target.oracle.kind === "capture-range-mismatch") {
+ validateCaptureSelector(input.target.oracle, "Capture-range mismatch");
+ const range = input.target.oracle.expectedRange;
+ if (
+ !Number.isSafeInteger(range.startUtf16) ||
+ !Number.isSafeInteger(range.endUtf16) ||
+ range.startUtf16 < 0 ||
+ range.endUtf16 < range.startUtf16 ||
+ range.endUtf16 > input.subject.length
+ ) {
+ throw new RangeError(
+ "Expected capture range must be a valid UTF-16 interval in the original subject.",
+ );
+ }
+ }
+ target = { ...input.target, side };
+ sides = [side];
+ } else {
+ const normalized = input.target.sides.map(normalizedSide);
+ const ecmascript = normalized.find((side) => side.flavour === "ecmascript");
+ const pcre2 = normalized.find((side) => side.flavour === "pcre2");
+ if (
+ normalized.length !== 2 ||
+ !ecmascript ||
+ !pcre2 ||
+ ecmascript === pcre2
+ ) {
+ throw new Error(
+ "Comparison minimization requires exactly one ECMAScript side and one PCRE2 side.",
+ );
+ }
+ if (
+ input.target.operation === "replace" &&
+ normalized.some((side) => side.replacement === undefined)
+ ) {
+ throw new Error(
+ "Replacement comparison minimization requires both replacement templates.",
+ );
+ }
+ const pair = [ecmascript, pcre2] as const;
+ target = { ...input.target, sides: pair };
+ sides = pair;
+ }
+ const request: SubjectMinimizationRequest = {
+ schemaVersion: 1,
+ subject: input.subject,
+ target,
+ budgets: {
+ maximumEvaluations,
+ maximumWallTimeMs,
+ candidateTimeoutMs,
+ },
+ maximumMatches,
+ maximumCaptureRows,
+ maximumOutputBytes,
+ };
+ return {
+ request,
+ sides,
+ targetIdentity: identity({
+ target,
+ maximumMatches,
+ maximumCaptureRows,
+ maximumOutputBytes,
+ candidateTimeoutMs,
+ }),
+ };
+}
+
+function classifyWorkerFailure(error: unknown): SideRuntimeFailure {
+ const kind =
+ error instanceof WorkerRequestError
+ ? error.kind
+ : error &&
+ typeof error === "object" &&
+ "kind" in error &&
+ ["timeout", "cancelled", "crash", "worker-error"].includes(
+ String((error as { readonly kind?: unknown }).kind),
+ )
+ ? (String(
+ (error as { readonly kind?: unknown }).kind,
+ ) as SideRuntimeFailure["status"])
+ : "worker-error";
+ return {
+ status: kind,
+ message: error instanceof Error ? error.message : String(error),
+ };
+}
+
+function sideObservation(
+ flavour: ComparisonFlavour,
+ runtime: SideRuntime,
+): CandidateSideObservation {
+ return {
+ flavour,
+ status: runtime.status,
+ ...(runtime.status === "complete"
+ ? { engineIdentity: engineIdentity(runtime.execution) }
+ : {}),
+ };
+}
+
+function sameFingerprint(
+ left: readonly string[] | undefined,
+ right: readonly string[],
+): boolean {
+ return (
+ left !== undefined &&
+ left.length === right.length &&
+ left.every((value, index) => value === right[index])
+ );
+}
+
+function fatalObservation(
+ runtime: SideRuntimeFailure,
+ sides: readonly CandidateSideObservation[],
+): MinimizationCandidateObservation {
+ return {
+ status: runtime.status,
+ reproduced: false,
+ fingerprint: [`worker:${runtime.status}`],
+ summary:
+ runtime.status === "crash"
+ ? `A worker crashed. Crash minimization is unsupported and was not treated as a timeout: ${runtime.message}`
+ : runtime.message,
+ sides,
+ };
+}
+
+function predicateObservation(
+ evaluation: PredicateEvaluation,
+ reproduced: boolean,
+ side: CandidateSideObservation,
+): MinimizationCandidateObservation {
+ return {
+ status: evaluation.kind === "inconclusive" ? "inconclusive" : "complete",
+ reproduced,
+ fingerprint: evaluation.fingerprint,
+ summary: evaluation.summary,
+ sides: [side],
+ };
+}
+
+export class SubjectMinimizer {
+ readonly #dependencies: SubjectMinimizerDependencies;
+ readonly #syntax: ReadonlyMap;
+ readonly #engine: MinimizerEngineRunner;
+ #generation = 0;
+ #disposed = false;
+
+ constructor(
+ dependencies: SubjectMinimizerDependencies = DEFAULT_DEPENDENCIES,
+ ) {
+ this.#dependencies = dependencies;
+ this.#syntax = new Map(
+ (["ecmascript", "pcre2"] as const).map((flavour) => [
+ flavour,
+ dependencies.createSyntaxRunner(flavour),
+ ]),
+ );
+ this.#engine = dependencies.createEngineRunner();
+ }
+
+ async minimize(
+ input: SubjectMinimizationRequest,
+ onProgress?: (progress: SubjectMinimizationProgress) => void,
+ ): Promise {
+ if (this.#disposed) {
+ throw new Error("Subject minimizer has been disposed.");
+ }
+ this.cancel();
+ const generation = this.#generation;
+ const startedAtMs = this.#dependencies.now();
+ const validated = validateSubjectMinimizationRequest(input);
+ onProgress?.({
+ phase: "syntax-setup",
+ evaluations: 0,
+ maximumEvaluations: validated.request.budgets.maximumEvaluations,
+ acceptedReductions: 0,
+ currentScalars: Array.from(validated.request.subject).length,
+ currentUtf8Bytes: utf8ByteLength(validated.request.subject),
+ elapsedMs: Math.max(0, this.#dependencies.now() - startedAtMs),
+ maximumWallTimeMs: validated.request.budgets.maximumWallTimeMs,
+ });
+
+ let prepared: readonly PreparedSide[] = [];
+ let preparationFailure: MinimizationCandidateObservation | undefined;
+ const syntaxTimeoutMs = Math.min(
+ 1_000,
+ validated.request.budgets.candidateTimeoutMs,
+ );
+ try {
+ if (
+ this.#dependencies.now() - startedAtMs + syntaxTimeoutMs <=
+ validated.request.budgets.maximumWallTimeMs
+ ) {
+ prepared = await Promise.all(
+ validated.sides.map((side) =>
+ this.#prepareSide(side, syntaxTimeoutMs),
+ ),
+ );
+ }
+ } catch (error) {
+ const failure = classifyWorkerFailure(error);
+ preparationFailure = {
+ status: failure.status === "timeout" ? "worker-error" : failure.status,
+ reproduced: false,
+ fingerprint: [`syntax:${failure.status}`],
+ summary:
+ failure.status === "timeout"
+ ? `Syntax setup timed out; this was not counted as an engine timeout: ${failure.message}`
+ : failure.message,
+ };
+ }
+
+ const isCancelled = () => generation !== this.#generation || this.#disposed;
+ const evaluate = preparationFailure
+ ? async () => preparationFailure
+ : validated.request.target.kind === "engine"
+ ? this.#engineEvaluator(validated.request, prepared[0], isCancelled)
+ : this.#comparisonEvaluator(
+ validated.request,
+ prepared as readonly [PreparedSide, PreparedSide],
+ isCancelled,
+ );
+ return reduceSubject({
+ subject: validated.request.subject,
+ targetIdentity: validated.targetIdentity,
+ budgets: validated.request.budgets,
+ evaluate,
+ now: this.#dependencies.now,
+ startedAtMs,
+ isCancelled,
+ onProgress,
+ });
+ }
+
+ cancel(): void {
+ this.#generation += 1;
+ for (const runner of this.#syntax.values()) runner.cancel();
+ this.#engine.cancel();
+ }
+
+ dispose(): void {
+ if (this.#disposed) return;
+ this.cancel();
+ this.#disposed = true;
+ for (const runner of this.#syntax.values()) runner.dispose();
+ this.#engine.dispose();
+ }
+
+ async #prepareSide(
+ input: ComparisonSideInput,
+ timeoutMs: number,
+ ): Promise {
+ const definition = AVAILABLE_REGEX_FLAVOURS.require(input.flavour);
+ const version = resolveRegexFlavourVersion(
+ input.flavourVersion,
+ definition,
+ `${definition.label} minimizer version`,
+ ).definition;
+ const syntaxRequest: RegexSyntaxRequest = {
+ flavour: input.flavour,
+ flavourVersion: version.syntaxVersion,
+ pattern: input.pattern,
+ flags: input.flags,
+ options: input.options,
+ };
+ const runner = this.#syntax.get(input.flavour);
+ if (!runner) throw new Error(`Missing ${input.flavour} syntax runner.`);
+ const syntax = await runner.parsePattern(syntaxRequest, timeoutMs);
+ if (!syntax.accepted) {
+ throw new WorkerRequestError(
+ "worker-error",
+ `${input.flavour} syntax parsing rejected the fixed pattern.`,
+ );
+ }
+ if (syntax.captures.length > DEFAULT_REGEX_LIMITS.maximumCaptureGroups) {
+ throw new WorkerRequestError(
+ "worker-error",
+ `${input.flavour} pattern exceeds the capture-group limit.`,
+ );
+ }
+ return {
+ input,
+ syntaxRequest,
+ captures: syntax.captures,
+ };
+ }
+
+ #executionRequest(
+ request: SubjectMinimizationRequest,
+ side: PreparedSide,
+ subject: string,
+ ): RegexExecutionRequest {
+ return {
+ flavour: side.input.flavour,
+ flavourVersion: side.input.flavourVersion,
+ pattern: side.input.pattern,
+ flags: side.input.flags,
+ options: side.input.options,
+ subject,
+ captureMetadata: side.captures,
+ scanAll: request.target.scanAll,
+ maximumMatches: request.maximumMatches,
+ maximumCaptureRows: request.maximumCaptureRows,
+ };
+ }
+
+ async #runSide(
+ request: SubjectMinimizationRequest,
+ side: PreparedSide,
+ subject: string,
+ replace: boolean,
+ ): Promise {
+ const execution = this.#executionRequest(request, side, subject);
+ try {
+ if (replace) {
+ const replacement = await this.#engine.replace(
+ {
+ ...execution,
+ replacement: side.input.replacement ?? "",
+ maximumOutputBytes: request.maximumOutputBytes,
+ },
+ request.budgets.candidateTimeoutMs,
+ );
+ return {
+ status: "complete",
+ execution: replacement.execution,
+ replacement,
+ };
+ }
+ return {
+ status: "complete",
+ execution: await this.#engine.execute(
+ execution,
+ request.budgets.candidateTimeoutMs,
+ ),
+ };
+ } catch (error) {
+ return classifyWorkerFailure(error);
+ }
+ }
+
+ #engineEvaluator(
+ request: SubjectMinimizationRequest,
+ side: PreparedSide | undefined,
+ isCancelled: () => boolean,
+ ): (subject: string) => Promise {
+ if (!side || request.target.kind !== "engine") {
+ return async () => ({
+ status: "worker-error",
+ reproduced: false,
+ fingerprint: ["setup:missing-side"],
+ summary: "The selected engine side was not prepared.",
+ });
+ }
+ const target = request.target;
+ let baselineFingerprint: readonly string[] | undefined;
+ let baselineEngineIdentity: string | undefined;
+ let first = true;
+ const replace =
+ target.oracle.kind === "unit-test-failure" &&
+ target.oracle.expectation.kind === "replacement";
+
+ return async (subject) => {
+ if (isCancelled()) {
+ return {
+ status: "cancelled",
+ reproduced: false,
+ fingerprint: ["cancelled"],
+ summary: "Subject minimization was cancelled.",
+ };
+ }
+ const baselineRun = first;
+ first = false;
+ const runtime = await this.#runSide(request, side, subject, replace);
+ const observedSide = sideObservation(side.input.flavour, runtime);
+ if (
+ runtime.status === "crash" ||
+ runtime.status === "worker-error" ||
+ runtime.status === "cancelled"
+ ) {
+ return fatalObservation(runtime, [observedSide]);
+ }
+ if (runtime.status === "timeout") {
+ let fingerprint: readonly string[];
+ let qualifying = false;
+ if (target.oracle.kind === "engine-timeout") {
+ fingerprint = ["engine:timeout"];
+ qualifying = true;
+ } else if (
+ target.oracle.kind === "unit-test-failure" &&
+ target.oracle.expectation.kind !== "must-time-out"
+ ) {
+ fingerprint = ["unit-test:unexpected-timeout"];
+ qualifying = true;
+ } else {
+ fingerprint = ["engine:timeout"];
+ }
+ if (baselineRun && qualifying) baselineFingerprint = fingerprint;
+ return {
+ status: "timeout",
+ reproduced:
+ qualifying &&
+ (baselineRun || sameFingerprint(baselineFingerprint, fingerprint)),
+ fingerprint,
+ summary: qualifying
+ ? "The selected engine hit the exact worker timeout."
+ : "The engine timed out, so the selected non-timeout predicate was not reproduced.",
+ sides: [observedSide],
+ };
+ }
+ if (runtime.status !== "complete") {
+ return fatalObservation(runtime, [observedSide]);
+ }
+
+ const currentIdentity = engineIdentity(runtime.execution);
+ if (
+ !baselineRun &&
+ baselineEngineIdentity !== undefined &&
+ baselineEngineIdentity !== currentIdentity
+ ) {
+ return {
+ status: "worker-error",
+ reproduced: false,
+ fingerprint: ["engine:identity-changed"],
+ summary: `Engine identity changed from ${baselineEngineIdentity} to ${currentIdentity}; reduction stopped.`,
+ sides: [observedSide],
+ };
+ }
+ if (baselineRun) baselineEngineIdentity = currentIdentity;
+ if (target.oracle.kind === "engine-timeout") {
+ return {
+ status: "complete",
+ reproduced: false,
+ fingerprint: ["engine:completed"],
+ summary: "The selected engine completed instead of timing out.",
+ sides: [observedSide],
+ };
+ }
+
+ const evaluation =
+ target.oracle.kind === "capture-range-mismatch"
+ ? evaluateCaptureRangeMismatch(
+ target.oracle,
+ runtime.execution,
+ subject,
+ )
+ : evaluateUnitTestFailure(
+ target.oracle.expectation,
+ runtime.execution,
+ runtime.replacement,
+ subject,
+ );
+ if (baselineRun && evaluation.kind === "failure") {
+ baselineFingerprint = evaluation.fingerprint;
+ }
+ return predicateObservation(
+ evaluation,
+ evaluation.kind === "failure" &&
+ (baselineRun ||
+ sameFingerprint(baselineFingerprint, evaluation.fingerprint)),
+ observedSide,
+ );
+ };
+ }
+
+ #comparisonEvaluator(
+ request: SubjectMinimizationRequest,
+ sides: readonly [PreparedSide, PreparedSide],
+ isCancelled: () => boolean,
+ ): (subject: string) => Promise {
+ const target = request.target;
+ if (target.kind !== "comparison-mismatch") {
+ return async () => ({
+ status: "worker-error",
+ reproduced: false,
+ fingerprint: ["setup:wrong-target"],
+ summary: "The comparison target was not prepared.",
+ });
+ }
+ let first = true;
+ let baselineMode: "semantic" | "timeout" | undefined;
+ let timeoutFlavour: ComparisonFlavour | undefined;
+ let baselineMismatchKinds: readonly string[] | undefined;
+ const baselineIdentities = new Map();
+
+ return async (subject) => {
+ if (isCancelled()) {
+ return {
+ status: "cancelled",
+ reproduced: false,
+ fingerprint: ["cancelled"],
+ summary: "Subject minimization was cancelled.",
+ };
+ }
+ const baselineRun = first;
+ first = false;
+ const runtimes = await Promise.all(
+ sides.map((side) =>
+ this.#runSide(request, side, subject, target.operation === "replace"),
+ ),
+ );
+ const observedSides = sides.map((side, index) =>
+ sideObservation(
+ side.input.flavour,
+ runtimes[index] ??
+ ({
+ status: "worker-error",
+ message: "Missing comparison side result.",
+ } as const),
+ ),
+ );
+ const fatal = runtimes.find(
+ (runtime): runtime is SideRuntimeFailure =>
+ runtime?.status === "crash" ||
+ runtime?.status === "worker-error" ||
+ runtime?.status === "cancelled",
+ );
+ if (fatal) return fatalObservation(fatal, observedSides);
+
+ const timeouts = runtimes
+ .map((runtime, index) => ({ runtime, side: sides[index] }))
+ .filter(
+ (
+ value,
+ ): value is {
+ readonly runtime: SideRuntimeFailure & {
+ readonly status: "timeout";
+ };
+ readonly side: PreparedSide;
+ } => value.runtime?.status === "timeout" && value.side !== undefined,
+ );
+ const completes = runtimes
+ .map((runtime, index) => ({ runtime, side: sides[index] }))
+ .filter(
+ (
+ value,
+ ): value is {
+ readonly runtime: SideRuntimeComplete;
+ readonly side: PreparedSide;
+ } => value.runtime?.status === "complete" && value.side !== undefined,
+ );
+
+ if (baselineRun) {
+ if (timeouts.length === 1 && completes.length === 1) {
+ const complete = completes[0];
+ const timeout = timeouts[0];
+ if (!complete || !timeout) {
+ throw new Error("Incomplete comparison baseline.");
+ }
+ const authoritative = compareSemanticResults(
+ {
+ execution: complete.runtime.execution,
+ ...(complete.runtime.replacement
+ ? { replacement: complete.runtime.replacement }
+ : {}),
+ },
+ {
+ execution: complete.runtime.execution,
+ ...(complete.runtime.replacement
+ ? { replacement: complete.runtime.replacement }
+ : {}),
+ },
+ );
+ if (authoritative.kind !== "same") {
+ return {
+ status: "inconclusive",
+ reproduced: false,
+ fingerprint: ["comparison:timeout-with-incomplete-peer"],
+ summary:
+ "One engine timed out, but the peer result was not complete and authoritative.",
+ sides: observedSides,
+ };
+ }
+ baselineMode = "timeout";
+ timeoutFlavour = timeout.side.input.flavour;
+ baselineIdentities.set(
+ complete.side.input.flavour,
+ engineIdentity(complete.runtime.execution),
+ );
+ return {
+ status: "timeout",
+ reproduced: true,
+ fingerprint: [`comparison:timeout:${timeoutFlavour}`],
+ summary: `${timeoutFlavour} timed out while the other selected engine completed authoritatively.`,
+ sides: observedSides,
+ };
+ }
+ if (timeouts.length > 0) {
+ return {
+ status: "timeout",
+ reproduced: false,
+ fingerprint: ["comparison:both-timeout"],
+ summary:
+ "Both engines timed out; this is not a one-sided comparison mismatch.",
+ sides: observedSides,
+ };
+ }
+ }
+
+ if (baselineMode === "timeout") {
+ if (timeouts.length !== 1 || completes.length !== 1) {
+ return {
+ status: timeouts.length > 0 ? "timeout" : "complete",
+ reproduced: false,
+ fingerprint: ["comparison:timeout-shape-changed"],
+ summary:
+ "The candidate did not preserve the same one-sided timeout.",
+ sides: observedSides,
+ };
+ }
+ const timeout = timeouts[0];
+ const complete = completes[0];
+ if (!timeout || !complete) {
+ throw new Error("Incomplete comparison timeout candidate.");
+ }
+ const authoritative = compareSemanticResults(
+ {
+ execution: complete.runtime.execution,
+ ...(complete.runtime.replacement
+ ? { replacement: complete.runtime.replacement }
+ : {}),
+ },
+ {
+ execution: complete.runtime.execution,
+ ...(complete.runtime.replacement
+ ? { replacement: complete.runtime.replacement }
+ : {}),
+ },
+ );
+ const expectedIdentity = baselineIdentities.get(
+ complete.side.input.flavour,
+ );
+ const actualIdentity = engineIdentity(complete.runtime.execution);
+ if (
+ expectedIdentity !== undefined &&
+ expectedIdentity !== actualIdentity
+ ) {
+ return {
+ status: "worker-error",
+ reproduced: false,
+ fingerprint: ["engine:identity-changed"],
+ summary: `Comparison peer identity changed from ${expectedIdentity} to ${actualIdentity}.`,
+ sides: observedSides,
+ };
+ }
+ const reproduced =
+ timeout.side.input.flavour === timeoutFlavour &&
+ authoritative.kind === "same";
+ return {
+ status: "timeout",
+ reproduced,
+ fingerprint: [`comparison:timeout:${timeout.side.input.flavour}`],
+ summary: reproduced
+ ? "The exact one-sided timeout mismatch was preserved."
+ : "The candidate did not preserve an authoritative peer completion.",
+ sides: observedSides,
+ };
+ }
+
+ if (timeouts.length > 0 || completes.length !== 2) {
+ return {
+ status: timeouts.length > 0 ? "timeout" : "inconclusive",
+ reproduced: false,
+ fingerprint: ["comparison:not-complete"],
+ summary:
+ "Both selected engines must complete for a semantic mismatch candidate.",
+ sides: observedSides,
+ };
+ }
+ const left = completes[0];
+ const right = completes[1];
+ if (!left || !right)
+ throw new Error("Missing completed comparison side.");
+ for (const complete of completes) {
+ const current = engineIdentity(complete.runtime.execution);
+ const flavour = complete.side.input.flavour;
+ const expected = baselineIdentities.get(flavour);
+ if (!baselineRun && expected !== undefined && expected !== current) {
+ return {
+ status: "worker-error",
+ reproduced: false,
+ fingerprint: ["engine:identity-changed"],
+ summary: `${flavour} engine identity changed from ${expected} to ${current}.`,
+ sides: observedSides,
+ };
+ }
+ if (baselineRun) baselineIdentities.set(flavour, current);
+ }
+ const semantic = compareSemanticResults(
+ {
+ execution: left.runtime.execution,
+ ...(left.runtime.replacement
+ ? { replacement: left.runtime.replacement }
+ : {}),
+ } satisfies SemanticSideResult,
+ {
+ execution: right.runtime.execution,
+ ...(right.runtime.replacement
+ ? { replacement: right.runtime.replacement }
+ : {}),
+ } satisfies SemanticSideResult,
+ );
+ if (baselineRun && semantic.kind === "mismatch") {
+ baselineMode = "semantic";
+ baselineMismatchKinds = semantic.mismatchKinds;
+ }
+ const exactMismatch =
+ semantic.kind === "mismatch" &&
+ (baselineRun ||
+ sameFingerprint(baselineMismatchKinds, semantic.mismatchKinds));
+ return {
+ status: semantic.kind === "inconclusive" ? "inconclusive" : "complete",
+ reproduced: exactMismatch,
+ fingerprint:
+ semantic.kind === "mismatch"
+ ? ["comparison:semantic-mismatch", ...semantic.mismatchKinds]
+ : [`comparison:${semantic.kind}`],
+ summary:
+ semantic.kind === "mismatch" && !exactMismatch
+ ? `${semantic.summary} The mismatch class changed, so this candidate was rejected.`
+ : semantic.summary,
+ mismatchKinds: semantic.mismatchKinds,
+ sides: observedSides,
+ };
+ };
+ }
+}
diff --git a/src/regex/minimization/minimization.types.ts b/src/regex/minimization/minimization.types.ts
new file mode 100644
index 0000000..c7f2bd9
--- /dev/null
+++ b/src/regex/minimization/minimization.types.ts
@@ -0,0 +1,140 @@
+import type { ComparisonSideInput } from "../comparison/comparison.types";
+import type { RegexFlavourId } from "../model/flavour";
+import type { SourceRange } from "../model/syntax";
+import type { RegexTestExpectation } from "../tests/test-case.types";
+
+export const MINIMIZER_MAXIMUM_SUBJECT_BYTES = 1024 * 1024;
+export const MINIMIZER_MAXIMUM_WALL_TIME_MS = 60_000;
+export const MINIMIZER_MAXIMUM_HISTORY_ENTRIES = 200;
+
+export interface SubjectMinimizationBudgets {
+ /** Includes the baseline run and every candidate run. */
+ readonly maximumEvaluations: number;
+ /** Covers syntax setup and all candidate runs. */
+ readonly maximumWallTimeMs: number;
+ /** Fixed worker deadline used for every engine candidate. */
+ readonly candidateTimeoutMs: number;
+}
+
+export type EngineFailureOracle =
+ | {
+ readonly kind: "unit-test-failure";
+ readonly expectation: RegexTestExpectation;
+ }
+ | {
+ readonly kind: "capture-range-mismatch";
+ readonly matchIndex: number;
+ readonly groupNumber?: number;
+ readonly groupName?: string;
+ readonly expectedRange: SourceRange;
+ }
+ | { readonly kind: "engine-timeout" };
+
+export interface EngineMinimizationTarget {
+ readonly kind: "engine";
+ readonly side: ComparisonSideInput;
+ readonly scanAll: boolean;
+ readonly oracle: EngineFailureOracle;
+}
+
+export interface ComparisonMinimizationTarget {
+ readonly kind: "comparison-mismatch";
+ readonly operation: "match" | "replace";
+ readonly scanAll: boolean;
+ readonly sides: readonly [ComparisonSideInput, ComparisonSideInput];
+}
+
+export interface SubjectMinimizationRequest {
+ readonly schemaVersion: 1;
+ readonly subject: string;
+ readonly target: EngineMinimizationTarget | ComparisonMinimizationTarget;
+ readonly budgets: SubjectMinimizationBudgets;
+ readonly maximumMatches: number;
+ readonly maximumCaptureRows: number;
+ readonly maximumOutputBytes: number;
+}
+
+export type CandidateRuntimeStatus =
+ | "complete"
+ | "timeout"
+ | "cancelled"
+ | "crash"
+ | "worker-error"
+ | "inconclusive";
+
+export interface CandidateSideObservation {
+ readonly flavour: RegexFlavourId;
+ readonly status: Exclude;
+ readonly engineIdentity?: string;
+}
+
+export interface MinimizationCandidateObservation {
+ readonly status: CandidateRuntimeStatus;
+ readonly reproduced: boolean;
+ readonly fingerprint: readonly string[];
+ readonly summary: string;
+ readonly mismatchKinds?: readonly string[];
+ readonly sides?: readonly CandidateSideObservation[];
+}
+
+export type MinimizationPhase =
+ "syntax-setup" | "baseline" | "chunk-deletion" | "local-sweep" | "complete";
+
+export interface SubjectMinimizationProgress {
+ readonly phase: MinimizationPhase;
+ readonly evaluations: number;
+ readonly maximumEvaluations: number;
+ readonly acceptedReductions: number;
+ readonly currentScalars: number;
+ readonly currentUtf8Bytes: number;
+ readonly elapsedMs: number;
+ readonly maximumWallTimeMs: number;
+ readonly lastObservation?: MinimizationCandidateObservation;
+}
+
+export type AcceptedTransformKind =
+ "chunk-deletion" | "single-scalar-deletion" | "scalar-simplification";
+
+export interface AcceptedSubjectTransform {
+ readonly kind: AcceptedTransformKind;
+ readonly detail: string;
+ readonly beforeScalars: number;
+ readonly afterScalars: number;
+ readonly beforeUtf8Bytes: number;
+ readonly afterUtf8Bytes: number;
+}
+
+export type SubjectMinimizationStopReason =
+ | "locally-minimal"
+ | "evaluation-budget-exhausted"
+ | "wall-time-budget-exhausted"
+ | "inconclusive-candidate"
+ | "baseline-not-reproduced"
+ | "cancelled"
+ | "engine-failure";
+
+export interface SubjectMinimizationResult {
+ readonly schemaVersion: 1;
+ readonly targetIdentity: string;
+ readonly status: "complete" | "partial" | "cancelled" | "unsupported";
+ readonly stopReason: SubjectMinimizationStopReason;
+ /**
+ * True only after a complete fixed-point sweep of the documented
+ * single-scalar deletion and canonical scalar-replacement transforms.
+ */
+ readonly locallyMinimal: boolean;
+ readonly originalSubject: string;
+ readonly minimizedSubject: string;
+ readonly originalScalars: number;
+ readonly minimizedScalars: number;
+ readonly originalUtf8Bytes: number;
+ readonly minimizedUtf8Bytes: number;
+ readonly evaluations: number;
+ readonly acceptedReductions: number;
+ readonly elapsedMs: number;
+ readonly budgets: SubjectMinimizationBudgets;
+ readonly baseline: MinimizationCandidateObservation;
+ readonly final: MinimizationCandidateObservation;
+ readonly retainedHistory: readonly AcceptedSubjectTransform[];
+ readonly historyTruncated: boolean;
+}
diff --git a/src/regex/minimization/oracles.test.ts b/src/regex/minimization/oracles.test.ts
new file mode 100644
index 0000000..f19dd54
--- /dev/null
+++ b/src/regex/minimization/oracles.test.ts
@@ -0,0 +1,275 @@
+import { describe, expect, it } from "vitest";
+import type { RegexEngineInfo } from "../model/flavour";
+import type {
+ CaptureResult,
+ RegexExecutionResult,
+ RegexMatchResult,
+ RegexReplacementResult,
+} from "../model/match";
+import type { RegexTestExpectation } from "../tests/test-case.types";
+import {
+ compareSemanticResults,
+ evaluateCaptureRangeMismatch,
+ evaluateUnitTestFailure,
+} from "./oracles";
+
+const ENGINE: RegexEngineInfo = {
+ flavour: "ecmascript",
+ adapterVersion: "fixture",
+ engineName: "Fixture",
+ engineVersion: "1",
+ offsetUnit: "utf16",
+ capabilities: {
+ compilation: true,
+ matching: true,
+ replacement: true,
+ namedCaptures: true,
+ captureHistory: false,
+ actualTrace: false,
+ benchmark: false,
+ },
+};
+
+function capture(overrides: Partial = {}): CaptureResult {
+ return {
+ groupNumber: 1,
+ groupName: "word",
+ value: "a",
+ status: "participated",
+ range: { startUtf16: 0, endUtf16: 1 },
+ nativeRange: { start: 0, end: 1, unit: "utf16" },
+ ...overrides,
+ };
+}
+
+function match(overrides: Partial = {}): RegexMatchResult {
+ return {
+ matchNumber: 1,
+ value: "a",
+ valueStatus: "complete",
+ range: { startUtf16: 0, endUtf16: 1 },
+ nativeRange: { start: 0, end: 1, unit: "utf16" },
+ captures: [capture()],
+ ...overrides,
+ };
+}
+
+function execution(
+ matches: readonly RegexMatchResult[] = [match()],
+ overrides: Partial = {},
+): RegexExecutionResult {
+ return {
+ accepted: true,
+ engine: ENGINE,
+ flags: {
+ userFlags: "g",
+ effectiveFlags: "gd",
+ internallyAddedIndicesFlag: true,
+ internallyAddedGlobalFlag: false,
+ },
+ matches,
+ diagnostics: [],
+ elapsedMs: 2,
+ truncated: false,
+ ...overrides,
+ };
+}
+
+function replacement(
+ output: string,
+ result = execution(),
+): RegexReplacementResult {
+ return {
+ execution: result,
+ output,
+ outputBytes: output.length,
+ outputTruncated: false,
+ truncated: false,
+ };
+}
+
+describe("unit-test failure oracle", () => {
+ const cases: readonly {
+ expectation: RegexTestExpectation;
+ result: RegexExecutionResult;
+ replacement?: RegexReplacementResult;
+ fingerprint: string;
+ }[] = [
+ {
+ expectation: { kind: "should-match" },
+ result: execution([]),
+ fingerprint: "expected-match:absent",
+ },
+ {
+ expectation: { kind: "should-not-match" },
+ result: execution(),
+ fingerprint: "expected-no-match:present",
+ },
+ {
+ expectation: { kind: "match-count", count: 2 },
+ result: execution(),
+ fingerprint: "match-count:mismatch",
+ },
+ {
+ expectation: { kind: "full-match", matchIndex: 2, value: "a" },
+ result: execution(),
+ fingerprint: "full-match:missing",
+ },
+ {
+ expectation: {
+ kind: "capture",
+ matchIndex: 0,
+ groupName: "word",
+ status: "matched-empty",
+ value: "different",
+ },
+ result: execution(),
+ fingerprint: "capture:status-mismatch",
+ },
+ {
+ expectation: { kind: "replacement", expected: "expected" },
+ result: execution(),
+ replacement: replacement("actual"),
+ fingerprint: "replacement:value-mismatch",
+ },
+ {
+ expectation: { kind: "must-complete-within", milliseconds: 1 },
+ result: execution(),
+ fingerprint: "duration:above-limit",
+ },
+ {
+ expectation: { kind: "must-time-out" },
+ result: execution(),
+ fingerprint: "timeout:unexpected-completion",
+ },
+ ];
+
+ it.each(cases)(
+ "dispatches $expectation.kind without weakening the failure class",
+ ({ expectation, result, replacement: replacementResult, fingerprint }) => {
+ const evaluated = evaluateUnitTestFailure(
+ expectation,
+ result,
+ replacementResult,
+ "a",
+ );
+ expect(evaluated.kind).toBe("failure");
+ expect(evaluated.fingerprint).toContain(fingerprint);
+ },
+ );
+
+ it("records capture participation and value failures independently", () => {
+ const evaluated = evaluateUnitTestFailure(
+ {
+ kind: "capture",
+ matchIndex: 0,
+ groupNumber: 1,
+ status: "matched-empty",
+ value: "different",
+ },
+ execution(),
+ undefined,
+ "a",
+ );
+
+ expect(evaluated.fingerprint).toEqual([
+ "capture:status-mismatch",
+ "capture:value-mismatch",
+ ]);
+ });
+
+ it("fails closed on truncated output", () => {
+ const evaluated = evaluateUnitTestFailure(
+ { kind: "should-not-match" },
+ execution([], { truncated: true }),
+ undefined,
+ "",
+ );
+
+ expect(evaluated).toMatchObject({
+ kind: "inconclusive",
+ fingerprint: ["truncated-result"],
+ });
+ });
+});
+
+describe("capture-range and semantic comparison oracles", () => {
+ it("preserves a wrong range by capture identity", () => {
+ expect(
+ evaluateCaptureRangeMismatch(
+ {
+ matchIndex: 0,
+ groupName: "word",
+ expectedRange: { startUtf16: 1, endUtf16: 1 },
+ },
+ execution(),
+ "a",
+ ),
+ ).toMatchObject({
+ kind: "failure",
+ fingerprint: ["capture:range-mismatch"],
+ });
+ });
+
+ it("compares normalized ranges, values and capture semantics but not native offsets or timing", () => {
+ const right = execution(
+ [
+ match({
+ nativeRange: { start: 0, end: 99, unit: "utf8-byte" },
+ captures: [
+ capture({
+ nativeRange: { start: 0, end: 99, unit: "utf8-byte" },
+ }),
+ ],
+ }),
+ ],
+ {
+ engine: {
+ ...ENGINE,
+ flavour: "pcre2",
+ engineName: "Another engine",
+ offsetUnit: "utf8-byte",
+ },
+ elapsedMs: 999,
+ },
+ );
+
+ expect(
+ compareSemanticResults({ execution: execution() }, { execution: right }),
+ ).toMatchObject({ kind: "same", mismatchKinds: [] });
+ });
+
+ it("reports exact mismatch categories including replacement output", () => {
+ const rightExecution = execution([
+ match({
+ value: "b",
+ range: { startUtf16: 0, endUtf16: 2 },
+ captures: [
+ capture({
+ status: "matched-empty",
+ range: { startUtf16: 1, endUtf16: 1 },
+ value: "",
+ }),
+ ],
+ }),
+ ]);
+
+ const compared = compareSemanticResults(
+ { execution: execution(), replacement: replacement("left") },
+ {
+ execution: rightExecution,
+ replacement: replacement("right", rightExecution),
+ },
+ );
+
+ expect(compared.kind).toBe("mismatch");
+ expect(compared.mismatchKinds).toEqual([
+ "capture-range",
+ "capture-status",
+ "capture-value",
+ "match-range",
+ "match-value",
+ "replacement-output",
+ ]);
+ });
+});
diff --git a/src/regex/minimization/oracles.ts b/src/regex/minimization/oracles.ts
new file mode 100644
index 0000000..a669195
--- /dev/null
+++ b/src/regex/minimization/oracles.ts
@@ -0,0 +1,406 @@
+import type {
+ CaptureResult,
+ RegexExecutionResult,
+ RegexReplacementResult,
+} from "../model/match";
+import type { SourceRange } from "../model/syntax";
+import type { RegexTestExpectation } from "../tests/test-case.types";
+
+export type PredicateEvaluationKind =
+ "failure" | "passes" | "inconclusive" | "unsupported";
+
+export interface PredicateEvaluation {
+ readonly kind: PredicateEvaluationKind;
+ readonly fingerprint: readonly string[];
+ readonly summary: string;
+}
+
+export interface CaptureRangeExpectation {
+ readonly matchIndex: number;
+ readonly groupNumber?: number;
+ readonly groupName?: string;
+ readonly expectedRange: SourceRange;
+}
+
+export interface SemanticSideResult {
+ readonly execution: RegexExecutionResult;
+ readonly replacement?: RegexReplacementResult;
+}
+
+export interface SemanticComparisonEvaluation {
+ readonly kind: "mismatch" | "same" | "inconclusive" | "unsupported";
+ readonly mismatchKinds: readonly string[];
+ readonly summary: string;
+}
+
+function rangesEqual(
+ left: SourceRange | undefined,
+ right: SourceRange | undefined,
+): boolean {
+ return (
+ left?.startUtf16 === right?.startUtf16 && left?.endUtf16 === right?.endUtf16
+ );
+}
+
+function sortedFingerprint(values: Iterable): readonly string[] {
+ return [...new Set(values)].sort();
+}
+
+function selectedCapture(
+ execution: RegexExecutionResult,
+ expectation: {
+ readonly matchIndex: number;
+ readonly groupNumber?: number;
+ readonly groupName?: string;
+ },
+): CaptureResult | undefined {
+ const match = execution.matches[expectation.matchIndex];
+ const named = expectation.groupName
+ ? match?.captures.filter(
+ (capture) => capture.groupName === expectation.groupName,
+ )
+ : undefined;
+ return (
+ named?.find((capture) => capture.status !== "did-not-participate") ??
+ named?.[0] ??
+ match?.captures.find(
+ (capture) => capture.groupNumber === expectation.groupNumber,
+ )
+ );
+}
+
+function captureValue(
+ capture: CaptureResult,
+ subject: string,
+): string | undefined {
+ return capture.range
+ ? subject.slice(capture.range.startUtf16, capture.range.endUtf16)
+ : capture.value;
+}
+
+function completeResult(
+ execution: RegexExecutionResult,
+ replacement: RegexReplacementResult | undefined,
+): PredicateEvaluation | undefined {
+ if (!execution.accepted) {
+ return {
+ kind: "unsupported",
+ fingerprint: ["compile-rejected"],
+ summary:
+ "The selected engine rejected the fixed pattern; a subject reducer cannot repair or preserve that compile failure.",
+ };
+ }
+ if (
+ execution.truncated ||
+ execution.matches.some(
+ (match) =>
+ match.valueStatus === "truncated" ||
+ match.captures.some(
+ (capture) =>
+ capture.status === "truncated" || capture.status === "unavailable",
+ ),
+ ) ||
+ replacement?.truncated ||
+ replacement?.outputTruncated
+ ) {
+ return {
+ kind: "inconclusive",
+ fingerprint: ["truncated-result"],
+ summary:
+ "The engine result was truncated or unavailable, so it cannot establish the exact failure predicate.",
+ };
+ }
+ return undefined;
+}
+
+/**
+ * Evaluates the inverse of a saved unit-test assertion. The returned
+ * fingerprint records the exact class of assertion failure so a reducer
+ * cannot silently turn, for example, a wrong capture value into a missing
+ * capture.
+ */
+export function evaluateUnitTestFailure(
+ expectation: RegexTestExpectation,
+ execution: RegexExecutionResult,
+ replacement: RegexReplacementResult | undefined,
+ subject: string,
+): PredicateEvaluation {
+ const incomplete = completeResult(execution, replacement);
+ if (incomplete) return incomplete;
+
+ const matches = execution.matches;
+ switch (expectation.kind) {
+ case "should-match":
+ return matches.length === 0
+ ? {
+ kind: "failure",
+ fingerprint: ["expected-match:absent"],
+ summary: "The expected match is still absent.",
+ }
+ : {
+ kind: "passes",
+ fingerprint: [],
+ summary: "The subject now matches.",
+ };
+ case "should-not-match":
+ return matches.length > 0
+ ? {
+ kind: "failure",
+ fingerprint: ["expected-no-match:present"],
+ summary: "At least one forbidden match is still present.",
+ }
+ : {
+ kind: "passes",
+ fingerprint: [],
+ summary: "The subject no longer matches.",
+ };
+ case "match-count":
+ return matches.length !== expectation.count
+ ? {
+ kind: "failure",
+ fingerprint: ["match-count:mismatch"],
+ summary: `The match count is ${matches.length}, not ${expectation.count}.`,
+ }
+ : {
+ kind: "passes",
+ fingerprint: [],
+ summary: "The expected match count is now produced.",
+ };
+ case "full-match": {
+ const match = matches[expectation.matchIndex];
+ if (!match) {
+ return {
+ kind: "failure",
+ fingerprint: ["full-match:missing"],
+ summary: `Match ${expectation.matchIndex} is still missing.`,
+ };
+ }
+ const actual = subject.slice(
+ match.range.startUtf16,
+ match.range.endUtf16,
+ );
+ return actual !== expectation.value
+ ? {
+ kind: "failure",
+ fingerprint: ["full-match:value-mismatch"],
+ summary: "The selected full-match value is still wrong.",
+ }
+ : {
+ kind: "passes",
+ fingerprint: [],
+ summary:
+ "The selected full-match value now equals the expectation.",
+ };
+ }
+ case "capture": {
+ const capture = selectedCapture(execution, expectation);
+ if (!capture) {
+ return {
+ kind: "failure",
+ fingerprint: ["capture:missing"],
+ summary: "The selected capture is still missing.",
+ };
+ }
+ const failures = new Set();
+ if (capture.status !== expectation.status) {
+ failures.add("capture:status-mismatch");
+ }
+ if (
+ expectation.value !== undefined &&
+ captureValue(capture, subject) !== expectation.value
+ ) {
+ failures.add("capture:value-mismatch");
+ }
+ return failures.size > 0
+ ? {
+ kind: "failure",
+ fingerprint: sortedFingerprint(failures),
+ summary: `The selected capture still fails: ${[...failures]
+ .map((value) => value.replace("capture:", ""))
+ .join(", ")}.`,
+ }
+ : {
+ kind: "passes",
+ fingerprint: [],
+ summary: "The selected capture now satisfies the expectation.",
+ };
+ }
+ case "replacement":
+ return replacement?.output !== expectation.expected
+ ? {
+ kind: "failure",
+ fingerprint: ["replacement:value-mismatch"],
+ summary: "The replacement output is still different.",
+ }
+ : {
+ kind: "passes",
+ fingerprint: [],
+ summary: "The replacement output now equals the expectation.",
+ };
+ case "must-complete-within":
+ return execution.elapsedMs > expectation.milliseconds
+ ? {
+ kind: "failure",
+ fingerprint: ["duration:above-limit"],
+ summary: `The engine still exceeds the ${expectation.milliseconds} ms assertion.`,
+ }
+ : {
+ kind: "passes",
+ fingerprint: [],
+ summary: "The engine now completes within the asserted duration.",
+ };
+ case "must-time-out":
+ return {
+ kind: "failure",
+ fingerprint: ["timeout:unexpected-completion"],
+ summary:
+ "The engine still completes, which fails the must-time-out assertion.",
+ };
+ }
+}
+
+export function evaluateCaptureRangeMismatch(
+ expectation: CaptureRangeExpectation,
+ execution: RegexExecutionResult,
+ subject: string,
+): PredicateEvaluation {
+ const incomplete = completeResult(execution, undefined);
+ if (incomplete) return incomplete;
+ const capture = selectedCapture(execution, expectation);
+ if (!capture) {
+ return {
+ kind: "unsupported",
+ fingerprint: ["capture:missing"],
+ summary:
+ "The selected capture is absent; that is not a reproducible capture-range mismatch.",
+ };
+ }
+ if (!capture.range) {
+ return {
+ kind: "unsupported",
+ fingerprint: ["capture:range-unavailable"],
+ summary: "The selected capture has no complete UTF-16 range to compare.",
+ };
+ }
+ // Read the slice as well as the range so malformed engine ranges cannot be
+ // accepted merely because their numeric coordinates differ.
+ void subject.slice(capture.range.startUtf16, capture.range.endUtf16);
+ return rangesEqual(capture.range, expectation.expectedRange)
+ ? {
+ kind: "passes",
+ fingerprint: [],
+ summary: "The selected capture range now equals the expected range.",
+ }
+ : {
+ kind: "failure",
+ fingerprint: ["capture:range-mismatch"],
+ summary: `The selected capture is still at ${capture.range.startUtf16}–${capture.range.endUtf16}, not ${expectation.expectedRange.startUtf16}–${expectation.expectedRange.endUtf16}.`,
+ };
+}
+
+function collectCaptureDifferences(
+ left: readonly CaptureResult[],
+ right: readonly CaptureResult[],
+ kinds: Set,
+): void {
+ if (left.length !== right.length) kinds.add("capture-presence");
+ const length = Math.min(left.length, right.length);
+ for (let index = 0; index < length; index += 1) {
+ const leftCapture = left[index];
+ const rightCapture = right[index];
+ if (!leftCapture || !rightCapture) continue;
+ if (
+ leftCapture.groupNumber !== rightCapture.groupNumber ||
+ leftCapture.groupName !== rightCapture.groupName
+ ) {
+ kinds.add("capture-identity");
+ }
+ if (leftCapture.status !== rightCapture.status) {
+ kinds.add("capture-status");
+ }
+ if (!rangesEqual(leftCapture.range, rightCapture.range)) {
+ kinds.add("capture-range");
+ }
+ if (leftCapture.value !== rightCapture.value) {
+ kinds.add("capture-value");
+ }
+ }
+}
+
+/**
+ * Compares only normalized semantic output: ordered UTF-16 ranges, retained
+ * values, capture identity/participation/ranges/values, and replacement text.
+ * Engine metadata, native offsets, timing, and effective implementation flags
+ * are deliberately excluded.
+ */
+export function compareSemanticResults(
+ left: SemanticSideResult,
+ right: SemanticSideResult,
+): SemanticComparisonEvaluation {
+ const leftIncomplete = completeResult(left.execution, left.replacement);
+ const rightIncomplete = completeResult(right.execution, right.replacement);
+ if (leftIncomplete?.kind === "unsupported") {
+ return {
+ kind: "unsupported",
+ mismatchKinds: [],
+ summary: `Left side: ${leftIncomplete.summary}`,
+ };
+ }
+ if (rightIncomplete?.kind === "unsupported") {
+ return {
+ kind: "unsupported",
+ mismatchKinds: [],
+ summary: `Right side: ${rightIncomplete.summary}`,
+ };
+ }
+ if (leftIncomplete || rightIncomplete) {
+ return {
+ kind: "inconclusive",
+ mismatchKinds: [],
+ summary:
+ "At least one side returned incomplete data, so semantic equality cannot be decided.",
+ };
+ }
+
+ const kinds = new Set();
+ const leftMatches = left.execution.matches;
+ const rightMatches = right.execution.matches;
+ if (leftMatches.length !== rightMatches.length) kinds.add("match-count");
+ const matchCount = Math.min(leftMatches.length, rightMatches.length);
+ for (let index = 0; index < matchCount; index += 1) {
+ const leftMatch = leftMatches[index];
+ const rightMatch = rightMatches[index];
+ if (!leftMatch || !rightMatch) continue;
+ if (!rangesEqual(leftMatch.range, rightMatch.range)) {
+ kinds.add("match-range");
+ }
+ if (leftMatch.value !== rightMatch.value) kinds.add("match-value");
+ collectCaptureDifferences(leftMatch.captures, rightMatch.captures, kinds);
+ }
+ if (left.replacement !== undefined || right.replacement !== undefined) {
+ if (left.replacement?.output !== right.replacement?.output) {
+ kinds.add("replacement-output");
+ }
+ }
+ const mismatchKinds = [...kinds].sort();
+ return mismatchKinds.length > 0
+ ? {
+ kind: "mismatch",
+ mismatchKinds,
+ summary: `Semantic mismatch: ${mismatchKinds.join(", ")}.`,
+ }
+ : {
+ kind: "same",
+ mismatchKinds: [],
+ summary: "Both engines produced the same normalized semantic result.",
+ };
+}
+
+export function engineIdentity(execution: RegexExecutionResult): string {
+ return [
+ execution.engine.flavour,
+ execution.engine.engineName,
+ execution.engine.engineVersion,
+ execution.engine.offsetUnit,
+ ].join(" · ");
+}
diff --git a/src/regex/minimization/subject-transforms.test.ts b/src/regex/minimization/subject-transforms.test.ts
new file mode 100644
index 0000000..1b0b44a
--- /dev/null
+++ b/src/regex/minimization/subject-transforms.test.ts
@@ -0,0 +1,242 @@
+import { describe, expect, it } from "vitest";
+import type { MinimizationCandidateObservation } from "./minimization.types";
+import {
+ containsLoneSurrogate,
+ reduceSubject,
+ unicodeScalars,
+} from "./subject-transforms";
+
+function observation(
+ reproduced: boolean,
+ overrides: Partial = {},
+): MinimizationCandidateObservation {
+ return {
+ status: "complete",
+ reproduced,
+ fingerprint: reproduced ? ["fixture:failure"] : [],
+ summary: reproduced ? "Fixture failure reproduced." : "Fixture passed.",
+ ...overrides,
+ };
+}
+
+const BUDGETS = {
+ maximumEvaluations: 200,
+ maximumWallTimeMs: 10_000,
+ candidateTimeoutMs: 100,
+} as const;
+
+describe("deterministic subject transforms", () => {
+ it("reduces left-to-right and reports only transform-local minimality", async () => {
+ const candidates: string[] = [];
+ const result = await reduceSubject({
+ subject: "abXcd",
+ targetIdentity: "fixture",
+ budgets: BUDGETS,
+ evaluate: async (candidate) => {
+ candidates.push(candidate);
+ return observation(candidate.includes("X"));
+ },
+ now: () => 0,
+ });
+
+ expect(result.minimizedSubject).toBe("X");
+ expect(result).toMatchObject({
+ status: "complete",
+ stopReason: "locally-minimal",
+ locallyMinimal: true,
+ baseline: { reproduced: true },
+ final: { reproduced: true },
+ });
+ expect(result.retainedHistory[0]?.kind).toBe("chunk-deletion");
+
+ const repeated: string[] = [];
+ await reduceSubject({
+ subject: "abXcd",
+ targetIdentity: "fixture",
+ budgets: BUDGETS,
+ evaluate: async (candidate) => {
+ repeated.push(candidate);
+ return observation(candidate.includes("X"));
+ },
+ now: () => 0,
+ });
+ expect(repeated).toEqual(candidates);
+ });
+
+ it("never splits supplementary Unicode scalars", async () => {
+ const seen: string[] = [];
+ const result = await reduceSubject({
+ subject: "a😀b",
+ targetIdentity: "unicode",
+ budgets: BUDGETS,
+ evaluate: async (candidate) => {
+ seen.push(candidate);
+ return observation(candidate.includes("😀"));
+ },
+ now: () => 0,
+ });
+
+ expect(result.minimizedSubject).toBe("😀");
+ expect(seen.every((candidate) => !containsLoneSurrogate(candidate))).toBe(
+ true,
+ );
+ expect(unicodeScalars(result.minimizedSubject)).toEqual(["😀"]);
+ });
+
+ it("uses only lower-rank canonical scalar simplifications", async () => {
+ const result = await reduceSubject({
+ subject: "Z",
+ targetIdentity: "canonical",
+ budgets: BUDGETS,
+ evaluate: async (candidate) => observation(candidate.length > 0),
+ now: () => 0,
+ });
+
+ expect(result.minimizedSubject).toBe("a");
+ expect(result.retainedHistory).toContainEqual(
+ expect.objectContaining({
+ kind: "scalar-simplification",
+ detail: 'Replaced scalar 0 with "a".',
+ }),
+ );
+ });
+
+ it("stops without a minimality claim when the evaluation budget is exhausted", async () => {
+ const result = await reduceSubject({
+ subject: "XX",
+ targetIdentity: "budget",
+ budgets: { ...BUDGETS, maximumEvaluations: 2 },
+ evaluate: async (candidate) => observation(candidate.includes("X")),
+ now: () => 0,
+ });
+
+ expect(result).toMatchObject({
+ status: "partial",
+ stopReason: "evaluation-budget-exhausted",
+ locallyMinimal: false,
+ evaluations: 2,
+ });
+ });
+
+ it("reserves a full fixed candidate timeout inside the wall budget", async () => {
+ let clock = 0;
+ const result = await reduceSubject({
+ subject: "XX",
+ targetIdentity: "wall",
+ budgets: {
+ maximumEvaluations: 20,
+ maximumWallTimeMs: 20,
+ candidateTimeoutMs: 6,
+ },
+ evaluate: async (candidate) => {
+ clock += 10;
+ return observation(candidate.includes("X"));
+ },
+ now: () => clock,
+ });
+
+ expect(result).toMatchObject({
+ status: "partial",
+ stopReason: "wall-time-budget-exhausted",
+ locallyMinimal: false,
+ evaluations: 2,
+ });
+ });
+
+ it("stops on a crash instead of treating it as a timeout or a reproduced failure", async () => {
+ let calls = 0;
+ const result = await reduceSubject({
+ subject: "XX",
+ targetIdentity: "crash",
+ budgets: BUDGETS,
+ evaluate: async () => {
+ calls += 1;
+ return calls === 1
+ ? observation(true)
+ : observation(false, {
+ status: "crash",
+ fingerprint: ["worker:crash"],
+ summary: "Fixture worker crashed.",
+ });
+ },
+ now: () => 0,
+ });
+
+ expect(result).toMatchObject({
+ status: "unsupported",
+ stopReason: "engine-failure",
+ locallyMinimal: false,
+ minimizedSubject: "XX",
+ final: { status: "complete", reproduced: true },
+ });
+ });
+
+ it("does not claim local minimality after an inconclusive candidate", async () => {
+ let calls = 0;
+ const result = await reduceSubject({
+ subject: "XX",
+ targetIdentity: "inconclusive",
+ budgets: BUDGETS,
+ evaluate: async () => {
+ calls += 1;
+ return calls === 1
+ ? observation(true)
+ : observation(false, {
+ status: "inconclusive",
+ fingerprint: ["truncated-result"],
+ summary: "Fixture output was truncated.",
+ });
+ },
+ now: () => 0,
+ });
+
+ expect(result).toMatchObject({
+ status: "partial",
+ stopReason: "inconclusive-candidate",
+ locallyMinimal: false,
+ evaluations: 2,
+ minimizedSubject: "XX",
+ });
+ });
+
+ it("returns a structured cancellation result", async () => {
+ let cancelled = false;
+ const progress: string[] = [];
+ const result = await reduceSubject({
+ subject: "XX",
+ targetIdentity: "cancel",
+ budgets: BUDGETS,
+ evaluate: async () => {
+ cancelled = true;
+ return observation(true);
+ },
+ isCancelled: () => cancelled,
+ onProgress: (value) => progress.push(value.phase),
+ now: () => 0,
+ });
+
+ expect(result).toMatchObject({
+ status: "cancelled",
+ stopReason: "cancelled",
+ locallyMinimal: false,
+ });
+ expect(progress).toContain("complete");
+ });
+
+ it("rejects lone surrogates before evaluating any candidate", async () => {
+ let called = false;
+ await expect(
+ reduceSubject({
+ subject: "\ud800",
+ targetIdentity: "invalid-unicode",
+ budgets: BUDGETS,
+ evaluate: async () => {
+ called = true;
+ return observation(true);
+ },
+ now: () => 0,
+ }),
+ ).rejects.toThrow(/lone surrogate/u);
+ expect(called).toBe(false);
+ });
+});
diff --git a/src/regex/minimization/subject-transforms.ts b/src/regex/minimization/subject-transforms.ts
new file mode 100644
index 0000000..779210b
--- /dev/null
+++ b/src/regex/minimization/subject-transforms.ts
@@ -0,0 +1,370 @@
+import { utf8ByteLength } from "../execution/request-limits";
+import {
+ MINIMIZER_MAXIMUM_HISTORY_ENTRIES,
+ type AcceptedSubjectTransform,
+ type AcceptedTransformKind,
+ type MinimizationCandidateObservation,
+ type MinimizationPhase,
+ type SubjectMinimizationBudgets,
+ type SubjectMinimizationProgress,
+ type SubjectMinimizationResult,
+ type SubjectMinimizationStopReason,
+} from "./minimization.types";
+
+const CANONICAL_SCALARS = ["a", "0", " ", "\n", "!"] as const;
+
+export type CandidateEvaluator = (
+ subject: string,
+) => Promise;
+
+export interface SubjectReducerOptions {
+ readonly subject: string;
+ readonly targetIdentity: string;
+ readonly budgets: SubjectMinimizationBudgets;
+ readonly evaluate: CandidateEvaluator;
+ readonly now: () => number;
+ readonly startedAtMs?: number;
+ readonly isCancelled?: () => boolean;
+ readonly onProgress?: (progress: SubjectMinimizationProgress) => void;
+}
+
+type EvaluationAttempt =
+ | {
+ readonly kind: "observation";
+ readonly observation: MinimizationCandidateObservation;
+ }
+ | {
+ readonly kind: "stop";
+ readonly reason: SubjectMinimizationStopReason;
+ readonly observation?: MinimizationCandidateObservation;
+ };
+
+const NOT_RUN: MinimizationCandidateObservation = {
+ status: "inconclusive",
+ reproduced: false,
+ fingerprint: ["not-run"],
+ summary: "No engine candidate could start within the configured budgets.",
+};
+
+export function containsLoneSurrogate(value: string): boolean {
+ for (let index = 0; index < value.length; index += 1) {
+ const first = value.charCodeAt(index);
+ if (first >= 0xd800 && first <= 0xdbff) {
+ const second = value.charCodeAt(index + 1);
+ if (!(second >= 0xdc00 && second <= 0xdfff)) return true;
+ index += 1;
+ } else if (first >= 0xdc00 && first <= 0xdfff) {
+ return true;
+ }
+ }
+ return false;
+}
+
+export function unicodeScalars(value: string): readonly string[] {
+ if (containsLoneSurrogate(value)) {
+ throw new Error(
+ "Subject minimization requires well-formed Unicode and will not create or preserve lone surrogate code units.",
+ );
+ }
+ return Array.from(value);
+}
+
+function resultStatus(
+ reason: SubjectMinimizationStopReason,
+): SubjectMinimizationResult["status"] {
+ if (reason === "locally-minimal") return "complete";
+ if (reason === "cancelled") return "cancelled";
+ if (reason === "baseline-not-reproduced" || reason === "engine-failure") {
+ return "unsupported";
+ }
+ return "partial";
+}
+
+function stopForObservation(
+ observation: MinimizationCandidateObservation,
+): SubjectMinimizationStopReason | undefined {
+ if (observation.status === "cancelled") return "cancelled";
+ if (observation.status === "crash" || observation.status === "worker-error") {
+ return "engine-failure";
+ }
+ return undefined;
+}
+
+function canonicalRank(value: string): number {
+ const rank = CANONICAL_SCALARS.indexOf(
+ value as (typeof CANONICAL_SCALARS)[number],
+ );
+ return rank < 0 ? Number.POSITIVE_INFINITY : rank;
+}
+
+/**
+ * Deterministic subject reducer.
+ *
+ * It first performs contiguous, left-to-right ddmin chunk deletion. It then
+ * reaches a fixed point over single-scalar deletion followed by lower-rank
+ * canonical scalar replacement. No claim is made beyond local minimality for
+ * that documented transform set.
+ */
+export async function reduceSubject(
+ options: SubjectReducerOptions,
+): Promise