diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index b8bb00c9..9d2f2b3f 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -11,6 +11,43 @@ concurrency: cancel-in-progress: true jobs: + rust-bridge: + name: Rust bridge (${{ matrix.os }}, ${{ matrix.python-version }}) + runs-on: ${{ matrix.os }} + strategy: + fail-fast: false + matrix: + include: + - os: ubuntu-latest + python-version: "3.10" + - os: ubuntu-latest + python-version: "3.14" + - os: macos-latest + python-version: "3.12" + - os: windows-latest + python-version: "3.12" + steps: + - uses: actions/checkout@v6 + - uses: actions/setup-python@v6 + with: + python-version: ${{ matrix.python-version }} + - name: Build and install the wheel with native extra + shell: bash + run: | + python -m pip install build + python -m build --wheel + python -c "import glob, subprocess, sys; wheel = glob.glob('dist/*.whl')[0]; subprocess.check_call([sys.executable, '-m', 'pip', 'install', wheel + '[test,cli,rust]'])" + - name: Verify installed package outside checkout imports + run: python -I scripts/check_rust_install.py + - name: Test compatibility, routing, preview and native parity + run: python -m pytest tests/test_contract_481.py tests/test_rust_backend.py tests/test_api_bridge_49.py tests/test_rust_contract.py -q + - name: Generate native parity report + run: python -m tests.rust_contract --output rust-parity.json + - uses: actions/upload-artifact@v4 + with: + name: rust-parity-${{ matrix.os }}-${{ matrix.python-version }} + path: rust-parity.json + lint: runs-on: ubuntu-latest steps: diff --git a/.gitignore b/.gitignore index f7471a1e..a73bfe4d 100644 --- a/.gitignore +++ b/.gitignore @@ -61,6 +61,7 @@ docs/* !docs/make.bat !docs/optional-surfaces.rst !docs/migration-4.8.1-contract.md +!docs/migration-4.9.md !docs/agents/ !docs/agents/** !docs/audit/ diff --git a/CHANGELOG.MD b/CHANGELOG.MD index 2dca5bba..904493f3 100644 --- a/CHANGELOG.MD +++ b/CHANGELOG.MD @@ -2,6 +2,24 @@ ## [Unreleased] +#### Added + +- Experimental `datafog[rust]` detection with keyword-only `backend="rust"` + on scan/redact entry points; Python remains the default. Core 0.3.1 is pinned, + German requests are explicitly unsupported, and ML composition is unchanged. +- `datafog.compat.v4` preserves the existing scan/redact facade and result classes; + `datafog.v5` previews actual Core types and transformation APIs. +- Native parity tests, installed-wheel CI checks, and public-call benchmarks. + +#### Deprecated + +- `detect()` and `process()` are now scheduled for removal in 5.0. This explicitly + revises the earlier promise to retain these shims throughout 5.x. They continue + working in 4.9; migrate to scan/redact and review transformation differences. +- OCR/image and Spark/distributed surfaces remain available in 4.9 with use-time + notices and are scheduled for removal in 5.0. Users needing those features can + remain on the final 4.x release. + #### Fixed - Allow installation on Python 3.14 and certify the core SDK and CLI with diff --git a/GERMAN-PII-CORE-REQUIREMENTS.md b/GERMAN-PII-CORE-REQUIREMENTS.md new file mode 100644 index 00000000..4f9e307e --- /dev/null +++ b/GERMAN-PII-CORE-REQUIREMENTS.md @@ -0,0 +1,429 @@ +# Agent brief: German structured PII detection in DataFog Core + +## Objective and scope + +Implement all seven locale-gated German entity types listed below in +`datafog-core`, with identical +behavior across Rust, Python, Node.js, and browser/WASM bindings. Make it usable +through text scanning, structured scanning, and existing transformation APIs. +All detection logic belongs in `crates/core`; bindings remain thin. + +This brief proposes concrete policy choices for implementation. In particular, +it specifies format/context-based PII detection rather than official identifier +or bank-account validation. Passport and residence-permit patterns are explicitly +legacy-compatible heuristics, not exhaustive national document-number coverage. +It does not authorize unrelated detector changes or a global locale redesign. +Do not implement this work in the legacy Python detector. + +## Existing behavior and compatibility context + +- Python 4.8.1 has `DE_IBAN` and six other locale-gated German labels. +- Its IBAN pattern is case-insensitive, accepts compact or optionally grouped + values, and does not validate a checksum. +- Core currently accepts `ScanConfig.locale`, but `scan_with_config` ignores it. +- Core scanning returns ordered findings and can retain overlapping candidates; + transformation selection resolves overlaps separately. Preserve that contract. +- Core already supplies byte/code-point offsets, binding-specific UTF-16 ranges, + structured field paths, transformations, and detector provenance. Reuse these. + +The German structure is 22 characters when normalized: `DE`, two check digits, +an eight-digit bank code, and a ten-digit account component. See the +[Deutsche Bundesbank's IBAN explanation](https://www.bundesbank.de/en/tasks/payment-systems/services/sepa/content/content-831778?index=1). + +## R1. Entity identity and findings + +Add these exact canonical labels and detector names: + +| Entity label | Detector name | +| ---------------------------- | ----------------------------------------- | +| `DE_IBAN` | `datafog-core/de-iban` | +| `DE_VAT_ID` | `datafog-core/de-vat-id` | +| `DE_TAX_ID` | `datafog-core/de-tax-id` | +| `DE_SOCIAL_SECURITY_NUMBER` | `datafog-core/de-social-security-number` | +| `DE_POSTAL_CODE` | `datafog-core/de-postal-code` | +| `DE_PASSPORT_NUMBER` | `datafog-core/de-passport-number` | +| `DE_RESIDENCE_PERMIT_NUMBER` | `datafog-core/de-residence-permit-number` | + +For every new detector: + +- Use the current crate version for `detector_version`, as existing detectors do. +- Leave confidence absent (`None`/`null`); do not invent a numeric probability. +- `matched_text` must be the exact source substring, including original casing + and internal separators. Never return a normalized substitute. +- Emit one finding per occurrence, with the existing ordering/deduplication rules. + +## R2. Activation and locale compatibility + +- Enable all seven detectors when `locale`, trimmed and compared case-insensitively, + is `de`, `de-DE`, or `de_DE`. +- Omitted locale does not enable any `DE_*` detector. Existing base detectors still run. +- Other already accepted nonempty locale values retain existing behavior and + do not activate German detection. Do not begin rejecting `en-US` or other + values as a side effect of this feature. +- Preserve existing malformed-config and empty-locale errors. +- Normalize for routing without changing the public stored locale value unless + existing API tests demonstrate that such normalization is already expected. +- Use the same routing for plain scans and structured string-value scans. +- Keep Core's current config shape: `{"locale": "de"}`. Do not introduce Python's + plural `locales`, a new engine selector, or scan-time entity selection here. +- Transformation entity selection does not activate a detector. A caller wanting + to scan and transform only German IBANs supplies both the German scan locale + and `transform.entities: ["DE_IBAN"]`. + +## R3. German IBAN lexical forms + +Use the following logical grammar, with ASCII digits only: + +```text +[Dd][Ee][0-9]{2}(SEP?[0-9]{4}){4}SEP?[0-9]{2} +SEP := one U+0020 SPACE, U+0009 TAB, U+00A0 NO-BREAK SPACE, + or U+202F NARROW NO-BREAK SPACE +``` + +- Accept compact, fully grouped, and partially grouped values. Each separator + is optional independently, but only at the group boundaries in the grammar. +- Do not require a context keyword such as `IBAN` or `Bankverbindung`. +- Do not include context labels, surrounding punctuation, or outer whitespace + in the match. +- Reject separators after `DE` but before the two check digits, multiple adjacent + separators, arbitrary grouping, hyphens, periods, and embedded newlines. +- Reject non-ASCII digits, other country prefixes, and wrong normalized length. +- Reject a candidate if its immediately preceding or following character is an + ASCII letter or digit. Start/end of input and punctuation are valid boundaries. + This preserves the legacy ASCII boundary convention; do not add Unicode-word + boundary behavior implicitly. +- Do not extract a valid-length prefix from a longer contiguous alphanumeric + identifier. Inspect the source boundary, not only the regex capture. +- Any normalization used internally must not change returned text or offsets. + +These separator and ASCII-digit rules deliberately narrow Python's broad `\s` +and Unicode `\d` acceptance. Document these as explicit migration differences: +newline-spanning and other Unicode-whitespace/digit matches are not promised. +Do not describe this feature as exact regex parity with Python 4.8.1. + +## R4. Identifier validation policy + +- Detect every candidate satisfying R3, including checksum-invalid candidates. +- Do not perform bank-directory lookup, account existence checks, network calls, + or model downloads. +- Do not silently require MOD-97 validity; this would drop PII-like values that + Python currently detects, including transcription errors. +- Explain in documentation that detection identifies sensitive-looking text and + does not establish that an account is valid or exists. +- A strict validation mode or separate validation API is outside this PR. +- Apply the same format-detection policy to the other six entities: no VAT/tax + checksum requirements, pension date validation, postal directory lookup, or + document issuance validation. Context and lexical rules below are mandatory. + +## R5. Offset and structured-data requirements + +- UTF-8 byte ranges and Unicode code-point ranges must both address the exact + original substring, with zero-based, end-exclusive offsets. +- Node and WASM UTF-16 ranges must work with JavaScript string slicing. +- Test emoji, combining marks, and CJK text before every entity type; ASCII-only fixtures + are insufficient to verify these units. +- In structured scans, preserve the existing JSON Pointer path and string-local + offset contract. Test two fields and an array element. +- Never concatenate separate fields to create a candidate. Required context must + occur in the same string value, not in a sibling field or JSON property name. + +## R6. Transformation integration + +- `scan_and_transform` must honor the scan locale and produce `[DE_IBAN]` for + the existing Core redaction strategy. Do not introduce numbered placeholders. +- `transform` must accept explicit `DE_IBAN` findings without rescanning or + requiring a locale; the locale controls detection only. +- Ensure the new label works with entity selection, per-entity overrides, exact + and full-match regex allowlists, and existing strategy/provider validation. +- Verify redaction, masking, and removal end to end; exercise provider-backed + strategies with existing test providers on supported runtimes. Do not broaden + WASM support for provider-backed strategies. +- Preserve source/output ranges and the rule that transformation records omit + original matched PII. +- Do not globally suppress generic PHONE/SSN/etc. scan findings inside IBANs. + With default transformation selection, existing overlap handling must prefer + the containing IBAN span and produce one replacement for it. Add a regression + fixture if an overlapping base detector is present. +- For IBAN-only transformation tests, set `entities: ["DE_IBAN"]` so the expected + output does not depend on unrelated detector findings. Exact allowlisting uses + the source value, not an automatically normalized IBAN. +- Apply all preceding transformation requirements to every new label, using + `[DE_VAT_ID]`, `[DE_TAX_ID]`, etc. as their redaction placeholders. +- Context text must survive replacements for tax, social-insurance, passport, + and residence-permit findings. Postal-code matches intentionally include their + prefix, so their replacement removes that prefix as well. +- Add overlap cases for VAT versus generic SSN, tax ID versus generic PHONE, and + German-prefixed postal code versus generic ZIP_CODE. Scanning may return both; + transformation must use the existing selection/overlap rules. For equal-span + built-in findings with absent confidence, the current lexical tie-break should + prefer `DE_TAX_ID` over `PHONE`. Test this; do not introduce a blanket new + priority rule that changes unrelated or caller-supplied findings. + +## R7. Required IBAN acceptance examples + +The expected count below refers to `DE_IBAN` only. Existing detectors may still +return other entity types, especially with no German locale or malformed input. + +| Input / configuration | Expected DE_IBAN behavior | +| ------------------------------------------------------- | ------------------------------------------- | +| `DE44500105175407324931`, locale `de` | One exact match | +| `DE44 5001 0517 5407 3249 31`, locale `de` | One match including internal spaces | +| `de44 5001 0517 5407 3249 31`, locale `DE-de` | One match preserving lowercase text | +| `DE44 50010517 54073249 31`, locale `de_DE` | One partially grouped match | +| Grouped value using each supported SEP | One match for each variant | +| `(DE44500105175407324931).`, locale `de` | Match excludes punctuation | +| Same IBAN twice, locale `de` | Two findings at different source ranges | +| Valid-shaped IBAN preceded by emoji/CJK/combining text | Correct byte, code-point, and UTF-16 ranges | +| `DE45500105175407324931`, locale `de` | One match despite altered checksum digits | +| A valid input without locale or with `en-US` | No DE_IBAN finding | +| `DE4450010517540732493`, locale `de` | No DE_IBAN: too short | +| `DE445001051754073249310`, locale `de` | No DE_IBAN: too long, no prefix match | +| `XDE44500105175407324931` or valid IBAN followed by `X` | No DE_IBAN: embedded identifier | +| `DE44-5001-0517-5407-3249-31` | No DE_IBAN | +| `DE44 5001 0517 5407 3249 31` | No DE_IBAN: double separator | +| `DE44\n5001 0517 5407 3249 31` | No DE_IBAN: newline is not SEP | +| `DE 44 5001 0517 5407 3249 31` | No DE_IBAN: misplaced separator | +| Same shape using Arabic-Indic/fullwidth digits | No DE_IBAN | +| Non-German prefix with the same trailing digit count | No DE_IBAN | + +Also test empty input, repeated near-matches, long digit runs, multiple adjacent +IBANs separated by punctuation, malformed config, and structured transformation. + +## R8. Conformance, performance, and delivery + +- Put positive/negative expectations in shared fixtures consumed by Rust and all + applicable bindings. Existing plain fixture runners call scan without config; + extend them to accept optional fixture scan config with unchanged defaults, + or introduce a focused locale fixture suite used by every binding. +- Run existing fixture suites unchanged. Do not rewrite old expected detections + to accommodate unrelated changes. +- Keep matching bounded and linear in input size; compile patterns once and + avoid per-candidate whole-input copying. No new dependency without a concrete + need that existing Rust facilities cannot reasonably meet. +- Compare no-locale scanning and German scanning on short, mixed, long, and + adversarial synthetic texts. Investigate a reproducible greater-than-10% + regression in existing no-locale workloads before merging. +- Run `cargo fmt --all --check`, + `cargo clippy --workspace --all-targets --all-features -- -D warnings`, and + `cargo test --workspace --all-features`. +- Run installed Python/Node binding tests and browser/WASM package tests, not + merely Rust unit tests. Verify config forwarding and offset conversions. +- Update README, entity/configuration references, binding type declarations if + needed, and migration documentation. State the checksum and separator policy. +- Deliver a focused PR with test results, performance observations, and known + migration differences. Do not publish packages without release authorization. +- Python 4.9 integration requires a subsequently published compatible Core wheel; + a source-only Core change is not sufficient to update the Python extra pin. + +## R9. Shared lexical and context rules for the other six entities + +Use these definitions in R10-R15: + +```text +D := one ASCII digit [0-9] +L := one ASCII letter [A-Za-z] +H := one of the four horizontal separators defined as SEP in R3 +GAP := H* [:#-]? H* +``` + +- All literal prefixes and context labels match ASCII case-insensitively, with + original casing preserved in findings. Do not use Unicode case folding that + expands the accepted alphabet. +- Apply R3's immediate ASCII-alphanumeric boundary checks to every value span. + Consequently a label ending in a letter needs whitespace or punctuation before + a value; `IdNr.12345678901` is allowed, `IdNr12345678901` is not. +- For context-required types, accept only `CONTEXT GAP VALUE`, with no intervening + prose or newline. A context label must not begin inside an ASCII word/number. + Include neither context nor GAP in the returned finding. +- A bare identifier or an unrelated preceding label must not activate a + context-required detector. German locale alone does not supply missing context. +- Do not search an entire document or previous line for context, nor use one + context marker to activate an arbitrary list of later identifiers. +- A structured key such as `tax_id` is not textual context for these detectors; + schema-driven detection is a separate feature. +- H may repeat in GAP and where explicitly written H+, but value-group separators + written H? allow at most one character. No newline, vertical whitespace, + Unicode digit, extra letter/digit, or punctuation substitution is accepted. +- These restrictions deliberately narrow legacy Python's Unicode `\d`, broad + `\s`, and context-substring matching. Record the differences in migration docs + and fixtures rather than changing the frozen 4.8.1 expectations. + +## R10. German VAT identifier: DE_VAT_ID + +```text +VALUE := DE (H | -)? D{9} +CONTEXT := not required +``` + +- Return the complete DE prefix, optional separator, and nine digits. +- Do not group the nine digits internally or accept multiple prefix separators. +- Examples with German locale: + +| Input | Expected matched text, or no DE_VAT_ID | +| -------------------------------------------------- | -------------------------------------- | +| `USt-IdNr DE123456789 ist gesetzt.` | `DE123456789` | +| `USt-IdNr DE 123456789 ist gesetzt.` | `DE 123456789` | +| `de-123456789` | `de-123456789` | +| `(DE123456789)` | `DE123456789` | +| `DE12345678` / `DE1234567890` / `DE123456789A` | None | +| `XDE123456789` / `DE--123456789` / `DE 123456789` | None | +| `DE123 456 789` / `AT123456789` | None | + +## R11. German tax identifier: DE_TAX_ID + +```text +VALUE := D{11} | D{2} H? D{3} H? D{3} H? D{3} +CONTEXT := Steuer(H|-)?ID | Steueridentifikationsnummer | + Identifikationsnummer | IdNr[.]? | Tax(H|-)?ID +``` + +- Return only the value, preserving optional internal separators. +- This is the legacy eleven-digit personal tax-ID detector; do not add tax-office + Steuernummer, slash-separated identifiers, or business identifiers to this label. + +| Input | Expected matched text, or no DE_TAX_ID | +| -------------------------------------------------------- | -------------------------------------- | +| `Steuer-ID 12345678901 liegt vor.` | `12345678901` | +| `Steueridentifikationsnummer: 12 345 678 901` | `12 345 678 901` | +| `IdNr.12345678901` | `12345678901` | +| `tax id # 12345678901` | `12345678901` | +| `Invoice 12345678901` / bare `12345678901` | None | +| `Steuer-ID 1234567890` / `Steuer-ID 123456789012` | None | +| `Steuer-ID A12345678901` / `Steuer-ID 12345678901Z` | None | +| `Steuer-ID 123 456 789 01` / `Steuer-ID 12 345 678 901` | None | +| `NotSteuer-ID 12345678901` / `Steuer-ID\n12345678901` | None | + +## R12. German social-insurance identifier: DE_SOCIAL_SECURITY_NUMBER + +```text +VALUE := D{2} H? D{6} H? L H? D{3} +CONTEXT := Rentenversicherungsnummer | Sozialversicherungsnummer | RVNR | SVNR +``` + +- Return only the value, preserving grouping and letter case. +- This is the legacy pension/social-insurance pattern. Do not use it to identify + health-insurance numbers, generic US SSNs, or arbitrary alphanumeric IDs. + +| Input | Expected matched text, or no DE_SOCIAL_SECURITY_NUMBER | +| --------------------------------------------------- | ------------------------------------------------------ | +| `Rentenversicherungsnummer 65150804A123 liegt vor.` | `65150804A123` | +| `SVNR: 65 150804 A123` | `65 150804 A123` | +| `rvnr # 65 150804 a 123` | `65 150804 a 123` | +| `Build 65150804A123 failed.` / bare `65150804A123` | None | +| `RVNR 65150804A12` / `RVNR 65150804A1234` | None | +| `RVNR 651508041123` / `RVNR 65150804AA123` | None | +| `RVNR 65-150804-A123` / `RVNR\n65150804A123` | None | + +## R13. German-prefixed postal code: DE_POSTAL_CODE + +```text +VALUE := (PLZ (H | : | -)? | DE (H | -) | D (H | -)) D{5} +CONTEXT := not required beyond the prefix included in VALUE +``` + +- Return the prefix, optional/required separator, and five digits together. + This deliberately preserves Python's span semantics, even though other types + exclude context. Do not silently shorten the match to the digits. +- A bare five-digit value does not become DE_POSTAL_CODE merely because locale + is German. Existing generic ZIP_CODE detection may still apply. +- Do not add city inference, a postal database, or validity checks. + +| Input | Expected matched text, or no DE_POSTAL_CODE | +| ----------------------------------------- | ------------------------------------------- | +| `PLZ10115 Berlin.` / `PLZ:10115 Berlin.` | `PLZ10115` / `PLZ:10115` | +| `PLZ 10115` / `PLZ-10115` | Complete corresponding value | +| `DE-10115 Berlin.` / `D 10115 Berlin.` | `DE-10115` / `D 10115` | +| `de 10115` | `de 10115` | +| `10115 Berlin` / `DE10115` / `D10115` | None | +| `PLZ1011` / `PLZ101150` / `PLZ10115A` | None | +| `SKU D12345` / `Release DE12345` | None | +| `PLZ: 10115` / `DE--10115` / `PLZ 10115` | None | + +`PLZ: 10115` is a useful possible future enhancement, but the legacy pattern +allows only one prefix separator. Keep its rejection explicit in this scope; +expanding it requires a separately documented detection change. + +## R14. Passport-context identifier: DE_PASSPORT_NUMBER + +```text +VALUE := L D{8} +CONTEXT := Passnummer | Reisepass(nummer)? | Passport(H+ No[.]? | H+ Number)? +``` + +The accepted labels include `Reisepass`, `Reisepassnummer`, +`Passport`, `Passport No.`, and `Passport Number`. + +- Return only the value. This is the inherited one-letter/eight-digit heuristic; + it does not claim to cover all actual German passport-number formats. +- Do not add Personalausweis detection or broader alphanumeric document formats + under this label without separately sourced requirements and fixtures. + +| Input | Expected matched text, or no DE_PASSPORT_NUMBER | +| -------------------------------------------------- | ----------------------------------------------- | +| `Passnummer C12345678 wurde geprueft.` | `C12345678` | +| `Reisepassnummer: c12345678` | `c12345678` | +| `Passport No. # A12345678` | `A12345678` | +| `Ticket A12345678 was shipped.` / bare `C12345678` | None | +| `Passnummer C1234567` / `Passnummer C123456789` | None | +| `Passnummer CC1234567` / `Passnummer 123456789` | None | +| `Passnummer C12A45678` / `Passnummer\nC12345678` | None | + +## R15. Residence-permit-context identifier: DE_RESIDENCE_PERMIT_NUMBER + +```text +VALUE := AT D{7} +CONTEXT := Aufenthaltstitel | Aufenthaltserlaubnis | Residence H+ Permit | eAT +``` + +- Return only the AT-prefixed value; no separator after AT is allowed. +- This is the inherited AT-plus-seven-digits heuristic, not a statement of + exhaustive or authoritative residence-permit numbering rules. + +| Input | Expected matched text, or no DE_RESIDENCE_PERMIT_NUMBER | +| ------------------------------------------------- | ------------------------------------------------------- | +| `Aufenthaltstitel AT1234567 gueltig.` | `AT1234567` | +| `Aufenthaltserlaubnis: at1234567` | `at1234567` | +| `Residence Permit # AT1234567` / `eAT-AT1234567` | `AT1234567` | +| `Order AT1234567 is internal.` / bare `AT1234567` | None | +| `eAT AT123456` / `eAT AT12345678` | None | +| `eAT AT 1234567` / `eAT DE1234567` | None | +| `eAT AT1234567X` / `eAT\nAT1234567` | None | + +## R16. Completion matrix and Python migration boundary + +For each of the seven labels, require shared fixtures proving: + +1. German locale aliases activate it; omitted/non-German locale does not. +2. All listed accepted forms and context aliases work; wrong lengths, unrelated + context, embedded ASCII identifiers and forbidden separators do not. +3. Matched text and byte/code-point/UTF-16 ranges preserve original input. +4. Repeated occurrences and structured values retain separate locations. +5. Selection, allowlists, overrides and supported transformations work, with + context retained or removed exactly as specified for the entity. +6. At least one synthetic input contains all seven types, with expected ordered + German findings and an end-to-end transformation result across bindings. +7. The seven-type suite passes against installed packages, with no network or + models required at inference time and no regression to existing fixtures. + +Every rejection assertion concerns the target German label, not necessarily an +empty overall scan. Generic detectors may legitimately identify the same input. + +Use the published Python 4.8.1 contract and `tests/test_de_pii_regex.py` as +comparison evidence. Keep their source fixtures unchanged. Record intentional +whitespace, digit-alphabet and context-boundary differences in a separate matrix. + +Python enables German detectors through explicit entity selection even without +a locale. Core currently selects entities only for transformations. Keep that +Core API distinction: the Python adapter may route explicit German entity +requests to locale `de` then filter, but must separately test legacy overlap and +selection semantics. Do not add a new scan selector casually within this work. + +Implementation may be split into focused PRs: locale routing plus IBAN/VAT; +contextual tax/social-insurance detectors; postal/document heuristics; then +cross-binding conformance and documentation. All seven must meet this brief +before claiming coverage of the legacy German entity set or unblocking general +German locale requests in the 4.9 adapter. Partial delivery must stay explicit. + +Out of scope: identifiers beyond these seven, exhaustive official format +validation, checksum-gated detection, language inference, schema-key inference, +and changes to existing generic detectors or Core's global overlap policy. diff --git a/PLAN-4.9.md b/PLAN-4.9.md new file mode 100644 index 00000000..c41739fa --- /dev/null +++ b/PLAN-4.9.md @@ -0,0 +1,145 @@ +# DataFog 4.9: incremental Core migration + +## Outcome + +Deliver a bridge release that lets users adopt Rust detection and the Core schema +without changing the default Python behavior. Keep the published 4.8.1 contract +as immutable evidence, with narrowly documented 4.9 API additions and revised +deprecation messages. This work does not publish a release or perform the 5.0 +cutover. + +## Release decisions + +- Keep `pip install datafog` and existing top-level imports working in 4.9. +- Add optional `datafog[rust]`, pinned to the published Core version verified by + this change. Loading the base package must not import the native extension. +- Add keyword-only `backend="python"` to scan/redact entry points. Rust is an + explicit, experimental alternative for `engine="regex"` only. +- Introduce `datafog.compat.v4` for existing scan/redact APIs and result classes; + top-level scan/redact delegate there while preserving class identity. +- Introduce `datafog.v5` as a preview of Core's actual public types and functions, + without translating them into legacy result objects or strategies. +- Revise `detect()` and `process()` warnings: removal is planned for 5.0, replacing + the earlier promise to retain them throughout 5.x. Explain the change publicly. +- Deprecate OCR and Spark in 4.9; remove them in 5.0. Preserve existing optional + installations and behavior during 4.9, and warn at meaningful use sites. +- Leave ML engines, smart/auto composition, CLI text commands, application + integrations, and service APIs on their existing backend in this increment. +- Keep the existing package version until release preparation; this branch + implements 4.9 behavior but does not publish or stamp a final release. + +## Increment 1: opt-in Rust detection + +1. Extend `datafog.engine.scan` and `scan_and_redact` with keyword-only backend + selection, preserving original parameters, validation and Python defaults. +2. Lazily call the installed `datafog_core.scan`; convert findings to existing + Entity objects using code-point offsets, not byte offsets. Preserve document + order, original text, and legacy regex provenance/confidence conventions. +3. Keep Python allowlist validation/filtering, aliases and transformation logic. + Do not reimplement detectors in the adapter or invoke Core transformations. +4. Reject Rust plus non-regex engines, unsupported German locale/entity requests, + and invalid backend values explicitly. Missing native dependencies must give + an actionable installation error. Never silently retry with Python or return + an empty result on native failure. +5. Explicit-entity redaction must not scan or import the native extension. +6. Add focused tests for routing, Unicode, validation, selection, overlaps, + allowlists, replacement strategies, missing dependencies and propagated errors. + +## Increment 2: compatibility namespace and schema preview + +1. Make `datafog.compat.v4` the facade for existing scan/redact return shapes and + result classes; retain legacy engine implementation in place where relocation + would break internal imports or monkeypatching contracts. +2. Keep top-level functions and types available and forward backend selection. + Existing agent convenience functions already accept forwarding keyword args. +3. Export the tested Core API through `datafog.v5`, preserving native type + identity. No Core import on plain `import datafog`; clear optional-dependency + errors when the preview is used without the extra. +4. Update detect/process warnings and test the explicit revised removal policy. + Do not expose these functions through the new preview. +5. Test import isolation, facade identity, unchanged legacy results, and real + Core scanning/transformation through the preview. + +## Increment 3: retirement notices for optional legacy surfaces + +1. Warn on meaningful OCR/Spark use, including direct supported entry points, + without importing heavy dependencies merely to issue a warning. +2. Keep those extras and functionality available throughout 4.9. Document removal + in 5.0 and the option to remain on the final 4.x release. Do not create new + replacement packages as part of this work. +3. Add tests proving notices are emitted and core text imports stay unaffected. + +## Increment 4: integration and evidence + +1. Keep the 4.8.1 fixture unchanged. Represent approved signature additions and + warning-message changes explicitly in the 4.9 contract checker; compare all + other observed behavior exactly. No blanket skips or recapture from dev. +2. Run applicable cases through the real published Core wheel and produce a + per-case parity report. Separate matches, deliberate unsupported requests, + known detector differences and regressions. Experimental Rust detection must + not be advertised as a drop-in equivalent before gaps are closed. +3. Add CI jobs with and without the Rust extra; smoke-test installed wheels and + native preview behavior on supported Python/platform combinations. +4. Benchmark the public Python and Rust-backed APIs, including conversion and + cold startup, with short, mixed and large synthetic inputs. Keep results + reproducible and make no unsupported speedup claim. +5. Update README, migration documentation and release notes for the actual scope. +6. Run focused suites, broader core regression tests, applicable benchmarks and + pre-commit checks; open a PR against dev and resolve CI failures. + +## Work ownership + +- Backend agent: engine backend adapter and focused backend tests. +- API agent: compatibility namespace, preview API, top-level delegation and + detect/process retirement warnings, with focused tests. +- Retirement agent: OCR/Spark notices and related tests/documentation. +- Coordinator: packaging, contract integration, parity report, CI, benchmarks, + cross-agent review, final verification and PR. + +Agents share one feature branch and have disjoint file ownership. They must not +commit, push, merge, overwrite another agent's files or edit frozen fixtures. +The coordinator reviews and integrates each change before committing. + +## Completion criteria + +- Default legacy behavior is unchanged except for the documented revised notices. +- Rust use is explicit, installation is optional, failures are visible, and + coverage limitations are documented and exercised by tests. +- Compatibility and native-preview APIs coexist without duplicated native builds. +- OCR/Spark and detect/process have actionable 5.0 retirement notices. +- Tests, installed-wheel checks, parity evidence and benchmarks are reproducible. +- A reviewed, passing PR is ready for dev; merging is subject to user direction + and existing repository approval rules. + +## Confirmed scope and delivery + +The user confirmed retaining deprecated OCR/Spark functionality in 4.9 and +removing it in 5.0. The coordinator is authorized to merge after checks and +required approvals pass; repository protections remain in effect. + +German detector expansion is specified separately in +`GERMAN-PII-CORE-REQUIREMENTS.md`. Implementing or publishing those Core changes +is outside this Python bridge increment. Rust calls requiring that unavailable +coverage must continue to fail explicitly until a tested Core release supports it. + +## Implementation evidence + +All four increments are implemented on `feature/4.9-core-migration`. Backend, +API, and retirement changes were delegated with disjoint ownership; cross-review +also corrected Windows UTF-8 fixture handling and default-filter CLI notice +visibility. The frozen 4.8.1 fixture is unchanged. + +- Base/CLI regression run: 782 passed, with expected optional-dependency skips + and pre-existing corpus xfails. +- Focused native integration run: 347 passed against published Core 0.3.1. +- Python 3.10 and 3.14 base-only runs: 181 passed each; native/CLI-specific tests + skip when their optional dependencies are absent. +- Clean installed-wheel smoke test: compatibility facade, native detection, + native preview, Unicode offsets and transformation all passed. +- Native parity: 61 exact matches, two reviewed detector differences, 17 explicit + unsupported German requests, 31 cases outside this backend's scope. +- Reproducible local timings are in `benchmarks/results-4.9.json`; they include + public-call overhead and cold startup, not just Rust scanning time. + +CI and required PR approval remain the final merge gates. No package publication +or Core-repository implementation is part of this delivery. diff --git a/README.md b/README.md index 8b9718fe..97741a60 100644 --- a/README.md +++ b/README.md @@ -188,9 +188,10 @@ scan/redact helpers, or guardrail helpers. model. - A Java runtime is required by PySpark. -OCR and Spark are not deprecated. Their broader API and packaging overhaul is -deferred; the 4.x goal is to keep them explicit, documented, and isolated from -the lightweight core path. +The upcoming 4.9 release deprecates OCR and Spark with visible use-time +warnings; their APIs and extras will be removed in 5.0. They remain functional +in 4.9. Users who need these features can stay on the final 4.x release. See +the [4.9 migration guide](docs/migration-4.9.md) for the transition plan. ## Backward-Compatible APIs @@ -256,6 +257,10 @@ Telemetry does not include input text or detected PII values. ## Development +The [4.9 migration guide](docs/migration-4.9.md) explains opt-in Rust detection, +the native `datafog.v5` preview, and the revised 5.0 retirement schedule for +`detect`/`process`, OCR, and Spark. The Python detector remains the default. + The [4.8.1 compatibility contract](docs/migration-4.8.1-contract.md) records published Python behavior for the Rust migration, with frozen fixtures and instructions for independently reproducing them from the release wheel. diff --git a/benchmarks/compare_detection_backends.py b/benchmarks/compare_detection_backends.py new file mode 100644 index 00000000..d5c89a00 --- /dev/null +++ b/benchmarks/compare_detection_backends.py @@ -0,0 +1,92 @@ +"""Measure public Python/Rust bridge calls, including adaptation and cold startup.""" + +import argparse +import importlib.metadata +import json +import os +import platform +import statistics +import subprocess +import sys +import time +from pathlib import Path + +os.environ["DATAFOG_NO_TELEMETRY"] = "1" + +PAYLOADS = { + "short": ("Contact alice@example.com", 200), + "mixed": ( + "Email alice@example.com or call (555) 123-4567. " + "SSN 123-45-6789; card 4111 1111 1111 1111; host 192.168.1.1.", + 100, + ), + "large_sparse": ("ordinary prose " * 70_000 + " alice@example.com", 3), +} + + +def measure(fn, text, backend, loops, samples): + result = fn(text, backend=backend) + durations = [] + for _ in range(samples): + start = time.perf_counter() + for _ in range(loops): + fn(text, backend=backend) + durations.append((time.perf_counter() - start) / loops * 1_000_000) + return { + "median_us": statistics.median(durations), + "samples_us": durations, + "entities": len(result.entities), + "iterations_per_sample": loops, + } + + +def cold_start(backend, samples): + source = ( + "import datafog; " + f"datafog.scan('Contact alice@example.com', backend={backend!r})" + ) + durations = [] + for _ in range(samples): + start = time.perf_counter() + subprocess.run([sys.executable, "-c", source], check=True, capture_output=True) + durations.append((time.perf_counter() - start) * 1_000) + return {"median_ms": statistics.median(durations), "samples_ms": durations} + + +def main(): + import datafog + + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--samples", type=int, default=5) + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args() + if args.samples < 1: + parser.error("samples must be positive") + report = { + "python": platform.python_version(), + "platform": platform.platform(), + "core_version": importlib.metadata.version("datafog-core"), + "method": "One warmup call, median repeated calls; no ML. Cold = process+import+first scan.", + "warm": [], + "cold": {}, + } + for name, (text, loops) in PAYLOADS.items(): + for operation in ("scan", "redact"): + row = { + "payload": name, + "operation": operation, + "utf8_bytes": len(text.encode()), + } + for backend in ("python", "rust"): + row[backend] = measure( + getattr(datafog, operation), text, backend, loops, args.samples + ) + report["warm"].append(row) + for backend in ("python", "rust"): + report["cold"][backend] = cold_start(backend, args.samples) + args.output.write_text(json.dumps(report, indent=2) + "\n") + print(f"Saved end-to-end measurements to {args.output}") + + +if __name__ == "__main__": + main() diff --git a/benchmarks/results-4.9.json b/benchmarks/results-4.9.json new file mode 100644 index 00000000..6f8c47e8 --- /dev/null +++ b/benchmarks/results-4.9.json @@ -0,0 +1,162 @@ +{ + "python": "3.12.13", + "platform": "macOS-26.6.2-arm64-arm-64bit", + "core_version": "0.3.1", + "method": "One warmup call, median repeated calls; no ML. Cold = process+import+first scan.", + "warm": [ + { + "payload": "short", + "operation": "scan", + "utf8_bytes": 25, + "python": { + "median_us": 19.735415116883814, + "samples_us": [ + 19.692290225066245, 19.87395982723683, 19.735415116883814, + 19.453955464996397, 20.307295490056276 + ], + "entities": 1, + "iterations_per_sample": 200 + }, + "rust": { + "median_us": 2.3012497695162892, + "samples_us": [ + 2.292499993927777, 2.3364595836028457, 2.3025000700727105, + 2.3012497695162892, 2.210834645666182 + ], + "entities": 1, + "iterations_per_sample": 200 + } + }, + { + "payload": "short", + "operation": "redact", + "utf8_bytes": 25, + "python": { + "median_us": 22.42000016849488, + "samples_us": [ + 22.9199999012053, 21.847704774700105, 22.270414629019797, + 22.495624725706875, 22.42000016849488 + ], + "entities": 1, + "iterations_per_sample": 200 + }, + "rust": { + "median_us": 3.6647950764745474, + "samples_us": [ + 3.821664722636342, 3.54208517819643, 3.6647950764745474, + 3.674789913929999, 3.5991653567180037 + ], + "entities": 1, + "iterations_per_sample": 200 + } + }, + { + "payload": "mixed", + "operation": "scan", + "utf8_bytes": 108, + "python": { + "median_us": 39.49917037971318, + "samples_us": [ + 39.52000057324767, 39.80332985520363, 39.49917037971318, + 38.85417012497783, 39.10957952030003 + ], + "entities": 5, + "iterations_per_sample": 100 + }, + "rust": { + "median_us": 7.581659592688084, + "samples_us": [ + 7.581659592688084, 7.61916977353394, 7.474579615518451, + 7.530420552939177, 7.725419709458948 + ], + "entities": 5, + "iterations_per_sample": 100 + } + }, + { + "payload": "mixed", + "operation": "redact", + "utf8_bytes": 108, + "python": { + "median_us": 42.573329992592335, + "samples_us": [ + 43.91749971546233, 42.02042007818818, 42.573329992592335, + 42.929170886054635, 41.73457971774042 + ], + "entities": 5, + "iterations_per_sample": 100 + }, + "rust": { + "median_us": 12.040419969707727, + "samples_us": [ + 11.556250974535942, 12.040419969707727, 11.692079715430737, + 12.406249297782779, 12.319160159677267 + ], + "entities": 5, + "iterations_per_sample": 100 + } + }, + { + "payload": "large_sparse", + "operation": "scan", + "utf8_bytes": 1050018, + "python": { + "median_us": 121914.90270197392, + "samples_us": [ + 120976.49998031557, 121914.90270197392, 124721.45829815418, + 123755.50003101428, 121670.05562999596 + ], + "entities": 1, + "iterations_per_sample": 3 + }, + "rust": { + "median_us": 5334.583343937993, + "samples_us": [ + 5415.152680749695, 5282.319655331473, 5320.333332444231, + 5334.583343937993, 5347.4583352605505 + ], + "entities": 1, + "iterations_per_sample": 3 + } + }, + { + "payload": "large_sparse", + "operation": "redact", + "utf8_bytes": 1050018, + "python": { + "median_us": 122195.88865991682, + "samples_us": [ + 122800.74999822925, 123273.65266780059, 122195.88865991682, + 120245.19433422635, 121185.16664486378 + ], + "entities": 1, + "iterations_per_sample": 3 + }, + "rust": { + "median_us": 5254.485993646085, + "samples_us": [ + 5248.666663343708, 5268.95831959943, 5254.485993646085, + 5269.472332050403, 5233.638648254176 + ], + "entities": 1, + "iterations_per_sample": 3 + } + } + ], + "cold": { + "python": { + "median_ms": 80.99587506148964, + "samples_ms": [ + 79.40420799423009, 78.51329201366752, 80.99587506148964, + 82.50495803076774, 83.314000046812 + ] + }, + "rust": { + "median_ms": 81.11529203597456, + "samples_ms": [ + 80.53745795041323, 81.226791953668, 81.9647500757128, 81.11529203597456, + 80.21229202859104 + ] + } + } +} diff --git a/datafog/__init__.py b/datafog/__init__.py index 7236ac10..591e0ef4 100644 --- a/datafog/__init__.py +++ b/datafog/__init__.py @@ -15,11 +15,9 @@ from .agent import create_guardrail, filter_output, sanitize, scan_prompt # Core API functions - always available (lightweight) +from .compat import v4 as _compat_v4 from .core import anonymize_text, detect_pii, get_supported_entities, scan_text from .engine import Entity, RedactResult, ScanResult -from .engine import redact as _redact_entities -from .engine import scan as _scan -from .engine import scan_and_redact as _scan_and_redact # Essential models - always available from .models.common import EntityTypes @@ -130,20 +128,10 @@ def _missing_dependency(*args, **kwargs): } -_REDACT_PRESETS = { - "default": "token", - "llm": "token", - "mask": "mask", - "hash": "hash", - "replace": "pseudonymize", - "pseudonymize": "pseudonymize", -} - - def _warn_v5_replacement(old_api: str, replacement: str) -> None: warnings.warn( - f"datafog.{old_api}() is deprecated for v5. Use {replacement} instead. " - "This compatibility shim will remain through the v5.x line.", + f"datafog.{old_api}() is deprecated and will be removed in 5.0. Use {replacement} instead. " + "The earlier promise to retain this shim through 5.x has been revised.", FutureWarning, stacklevel=3, ) @@ -156,25 +144,18 @@ def scan( locales: list[str] | None = None, allowlist: list[str] | None = None, allowlist_patterns: list[str] | None = None, + *, + backend: str = "python", ) -> ScanResult: - """ - v5-preview scan entrypoint. - - Defaults to the lightweight regex engine so the core install works without - optional dependency fallback warnings. - - ``allowlist`` exempts exact entity texts (your own support address, doc - placeholders); ``allowlist_patterns`` exempts entities whose full text - matches a regex (e.g. ``^\\d{10}$`` so unix timestamps stop matching as - phone numbers). - """ - return _scan( - text=text, - engine=engine, - entity_types=entity_types, - locales=locales, - allowlist=allowlist, - allowlist_patterns=allowlist_patterns, + """Scan using the 4.x compatibility API; Rust detection is opt-in.""" + return _compat_v4.scan( + text, + engine, + entity_types, + locales, + allowlist, + allowlist_patterns, + backend=backend, ) @@ -188,39 +169,21 @@ def redact( locales: list[str] | None = None, allowlist: list[str] | None = None, allowlist_patterns: list[str] | None = None, + *, + backend: str = "python", ) -> RedactResult: - """ - v5-preview redaction entrypoint. - - If entities are provided, redact those spans. Otherwise, scan text first - using the selected engine and redact the detected entities. ``allowlist`` - and ``allowlist_patterns`` exempt findings from redaction (exact text and - full-text regex match respectively); they apply to the scan path and are - rejected when explicit ``entities`` are supplied. - """ - if preset is not None: - try: - strategy = _REDACT_PRESETS[preset] - except KeyError as exc: - allowed = ", ".join(sorted(_REDACT_PRESETS)) - raise ValueError(f"preset must be one of: {allowed}") from exc - - if entities is not None: - if allowlist or allowlist_patterns: - raise ValueError( - "allowlist/allowlist_patterns cannot be combined with explicit " - "entities; filter the entities before calling redact" - ) - return _redact_entities(text=text, entities=entities, strategy=strategy) - - return _scan_and_redact( - text=text, - engine=engine, - entity_types=entity_types, - strategy=strategy, - locales=locales, - allowlist=allowlist, - allowlist_patterns=allowlist_patterns, + """Redact using the 4.x compatibility API and legacy strategies.""" + return _compat_v4.redact( + text, + entities, + engine, + entity_types, + strategy, + preset, + locales, + allowlist, + allowlist_patterns, + backend=backend, ) diff --git a/datafog/_legacy_retirement.py b/datafog/_legacy_retirement.py new file mode 100644 index 00000000..5bc96c13 --- /dev/null +++ b/datafog/_legacy_retirement.py @@ -0,0 +1,17 @@ +"""Dependency-free retirement notices for optional legacy functionality.""" + +import warnings + + +class LegacySurfaceWarning(FutureWarning): + """An optional DataFog surface scheduled for removal in 5.0.""" + + +def warn_legacy_surface(surface: str) -> None: + """Warn at the caller's public API boundary, before optional imports.""" + warnings.warn( + f"DataFog {surface} support is deprecated in 4.9 and will be removed in 5.0. " + "Remain on the final 4.x release if you need continued support.", + LegacySurfaceWarning, + stacklevel=3, + ) diff --git a/datafog/client.py b/datafog/client.py index eab203b8..fd9a303d 100644 --- a/datafog/client.py +++ b/datafog/client.py @@ -10,6 +10,8 @@ import typer +from datafog._legacy_retirement import LegacySurfaceWarning, warn_legacy_surface + from .config import OperationType, get_config from .engine import scan_and_redact from .main import DataFog @@ -63,6 +65,7 @@ def scan_image( Prints results or exits with error on failure. """ + warn_legacy_surface("OCR") if not image_urls: typer.echo("No image URLs or file paths provided. Please provide at least one.") raise typer.Exit(code=1) @@ -87,6 +90,8 @@ def scan_image( except Exception: pass except Exception as e: + if isinstance(e, LegacySurfaceWarning): + raise logging.exception("Error in run_ocr_pipeline") try: from .telemetry import track_error diff --git a/datafog/compat/__init__.py b/datafog/compat/__init__.py new file mode 100644 index 00000000..55f15891 --- /dev/null +++ b/datafog/compat/__init__.py @@ -0,0 +1 @@ +"""Compatibility APIs for migrations between DataFog major versions.""" diff --git a/datafog/compat/v4.py b/datafog/compat/v4.py new file mode 100644 index 00000000..0be6b4bb --- /dev/null +++ b/datafog/compat/v4.py @@ -0,0 +1,105 @@ +"""Legacy scan/redact facade, preserving the 4.x Python result schema. + +The result classes retain their original engine identity. This namespace does +not include the deprecated detect/process shims. +""" + +from ..engine import Entity, RedactResult, ScanResult +from ..engine import redact as _redact_entities +from ..engine import scan as _scan +from ..engine import scan_and_redact as _scan_and_redact + +__all__ = ["Entity", "ScanResult", "RedactResult", "scan", "redact"] + +_REDACT_PRESETS = { + "default": "token", + "llm": "token", + "mask": "mask", + "hash": "hash", + "replace": "pseudonymize", + "pseudonymize": "pseudonymize", +} + + +def scan( + text: str, + engine: str = "regex", + entity_types: list[str] | None = None, + locales: list[str] | None = None, + allowlist: list[str] | None = None, + allowlist_patterns: list[str] | None = None, + *, + backend: str = "python", +) -> ScanResult: + """ + Scan with the legacy result schema; Rust detection is opt-in. + + Defaults to the lightweight regex engine so the core install works without + optional dependency fallback warnings. + + ``allowlist`` exempts exact entity texts (your own support address, doc + placeholders); ``allowlist_patterns`` exempts entities whose full text + matches a regex (e.g. ``^\\d{10}$`` so unix timestamps stop matching as + phone numbers). + """ + return _scan( + text=text, + engine=engine, + entity_types=entity_types, + locales=locales, + allowlist=allowlist, + allowlist_patterns=allowlist_patterns, + backend=backend, + ) + + +def redact( + text: str, + entities: list[Entity] | None = None, + engine: str = "regex", + entity_types: list[str] | None = None, + strategy: str = "token", + preset: str | None = None, + locales: list[str] | None = None, + allowlist: list[str] | None = None, + allowlist_patterns: list[str] | None = None, + *, + backend: str = "python", +) -> RedactResult: + """ + Redact with legacy strategies and result schema. + + If entities are provided, redact those spans. Otherwise, scan text first + using the selected engine and redact the detected entities. ``allowlist`` + and ``allowlist_patterns`` exempt findings from redaction (exact text and + full-text regex match respectively); they apply to the scan path and are + rejected when explicit ``entities`` are supplied. + """ + if backend not in ("python", "rust"): + raise ValueError("backend must be one of: python, rust") + + if preset is not None: + try: + strategy = _REDACT_PRESETS[preset] + except KeyError as exc: + allowed = ", ".join(sorted(_REDACT_PRESETS)) + raise ValueError(f"preset must be one of: {allowed}") from exc + + if entities is not None: + if allowlist or allowlist_patterns: + raise ValueError( + "allowlist/allowlist_patterns cannot be combined with explicit " + "entities; filter the entities before calling redact" + ) + return _redact_entities(text=text, entities=entities, strategy=strategy) + + return _scan_and_redact( + text=text, + engine=engine, + entity_types=entity_types, + strategy=strategy, + locales=locales, + allowlist=allowlist, + allowlist_patterns=allowlist_patterns, + backend=backend, + ) diff --git a/datafog/engine.py b/datafog/engine.py index 419558e9..3873a3d2 100644 --- a/datafog/engine.py +++ b/datafog/engine.py @@ -209,6 +209,56 @@ def _regex_entities( return _suppress_overlapping_entities(entities) +def _rust_entities( + text: str, + entity_types: Optional[list[str]] = None, + locales: Optional[list[str]] = None, +) -> list[Entity]: + """Adapt native findings to the legacy regex result contract. + + Core supplies candidate findings. Python applies the legacy overlap + suppression before entity selection and allowlists, preserving that order + without duplicating the native detectors. + """ + normalized_locales = RegexAnnotator._normalize_locales(locales) + requested = {_canonical_type(value) for value in entity_types or []} + if normalized_locales or any(value.startswith("DE_") for value in requested): + raise ValueError( + "backend='rust' does not yet support German locales or DE_* entity " + "types; use backend='python' for German detection" + ) + + try: + import datafog_core + except ImportError as exc: + raise ImportError( + "The Rust backend requires datafog-core. " + 'Install with: pip install "datafog[rust]"' + ) from exc + + entities: list[Entity] = [] + for finding in datafog_core.scan(text): + canonical_type = _canonical_type(finding.entity_type) + if canonical_type not in ALL_ENTITY_TYPES: + raise RuntimeError( + f"Rust backend returned unsupported entity type: " + f"{finding.entity_type!r}; check the installed datafog-core version" + ) + if not finding.matched_text.strip(): + continue + entities.append( + Entity( + type=canonical_type, + text=finding.matched_text, + start=finding.codepoint_range.start, + end=finding.codepoint_range.end, + confidence=1.0, + engine="regex", + ) + ) + return _suppress_overlapping_entities(entities) + + def _spacy_entities(text: str) -> list[Entity]: annotator = _get_spacy_annotator() if isinstance(annotator, _UnavailableAnnotator): @@ -357,6 +407,13 @@ def _needs_ner(entity_types: Optional[list[str]]) -> bool: return bool(requested & NER_ENTITY_TYPES) +def _validate_backend(backend: str, engine: str) -> None: + if backend not in {"python", "rust"}: + raise ValueError("backend must be one of: python, rust") + if backend == "rust" and engine != "regex": + raise ValueError("backend='rust' supports only engine='regex'") + + def scan( text: str, engine: str = "smart", @@ -364,9 +421,15 @@ def scan( locales: Optional[list[str]] = None, allowlist: Optional[list[str]] = None, allowlist_patterns: Optional[list[str]] = None, + *, + backend: str = "python", ) -> ScanResult: """Scan text for PII entities. + ``backend="rust"`` opts into experimental native detection for + ``engine="regex"``. Results retain legacy regex provenance and confidence; + these are compatibility values, not native confidence estimates. + ``allowlist`` exempts exact entity texts (e.g. your own support email); ``allowlist_patterns`` exempts entities whose full text matches a regex (e.g. ``^\\d{10}$`` to stop unix timestamps matching as phone numbers). @@ -377,11 +440,14 @@ def scan( if engine not in {"regex", "spacy", "gliner", "smart"}: raise ValueError("engine must be one of: regex, spacy, gliner, smart") + _validate_backend(backend, engine) + # Validate patterns up front so config errors fail fast even when the # text contains no entities. _compile_allowlist_patterns(allowlist_patterns) - regex_entities = _regex_entities( + detector = _rust_entities if backend == "rust" else _regex_entities + regex_entities = detector( text, entity_types=entity_types, locales=locales, @@ -547,8 +613,10 @@ def scan_and_redact( locales: Optional[list[str]] = None, allowlist: Optional[list[str]] = None, allowlist_patterns: Optional[list[str]] = None, + *, + backend: str = "python", ) -> RedactResult: - """Convenience wrapper: scan then redact.""" + """Scan with the selected backend, then apply legacy Python redaction.""" scan_result = scan( text=text, engine=engine, @@ -556,5 +624,6 @@ def scan_and_redact( locales=locales, allowlist=allowlist, allowlist_patterns=allowlist_patterns, + backend=backend, ) return redact(text=text, entities=scan_result.entities, strategy=strategy) diff --git a/datafog/main.py b/datafog/main.py index 62abaaff..0dc6f5d5 100644 --- a/datafog/main.py +++ b/datafog/main.py @@ -12,6 +12,8 @@ import logging from typing import List +from datafog._legacy_retirement import warn_legacy_surface + from .config import OperationType from .engine import scan, scan_and_redact from .models.anonymizer import Anonymizer, AnonymizerType, HashType @@ -79,6 +81,7 @@ def __init__( async def run_ocr_pipeline(self, image_urls: List[str]) -> List[str]: """Run OCR + text pipeline for CLI/backward compatibility.""" + warn_legacy_surface("OCR") from .services.image_service import ImageService image_service = ImageService() diff --git a/datafog/processing/image_processing/donut_processor.py b/datafog/processing/image_processing/donut_processor.py index 50022b6a..8e3e842d 100644 --- a/datafog/processing/image_processing/donut_processor.py +++ b/datafog/processing/image_processing/donut_processor.py @@ -12,6 +12,8 @@ import re from typing import TYPE_CHECKING, Any +from datafog._legacy_retirement import LegacySurfaceWarning, warn_legacy_surface + from .image_downloader import ImageDownloader if TYPE_CHECKING: @@ -35,6 +37,7 @@ class DonutProcessor: """ def __init__(self, model_path="naver-clova-ix/donut-base-finetuned-cord-v2"): + warn_legacy_surface("OCR") # Store model path for lazy loading self.model_path = model_path self.downloader = ImageDownloader() @@ -47,6 +50,7 @@ def _missing_dependency_message(package_name: str) -> str: ) def preprocess_image(self, image: "Image.Image") -> Any: + warn_legacy_surface("OCR") import numpy as np # Convert to RGB if the image is not already in RGB mode @@ -65,6 +69,7 @@ def preprocess_image(self, image: "Image.Image") -> Any: async def extract_text_from_image(self, image: "Image.Image") -> str: """Extract text from an image using the Donut model""" + warn_legacy_surface("OCR") logging.info("DonutProcessor.extract_text_from_image called") # If we're in a test environment and PYTEST_DONUT is not enabled, return a mock response @@ -148,6 +153,8 @@ async def extract_text_from_image(self, image: "Image.Image") -> str: result = processor.token2json(sequence) return json.dumps(result) + except LegacySurfaceWarning: + raise except (ImportError, RuntimeError): raise except Exception as e: diff --git a/datafog/processing/image_processing/image_downloader.py b/datafog/processing/image_processing/image_downloader.py index b7bf338f..46aca082 100644 --- a/datafog/processing/image_processing/image_downloader.py +++ b/datafog/processing/image_processing/image_downloader.py @@ -9,6 +9,8 @@ from io import BytesIO from typing import TYPE_CHECKING, List +from datafog._legacy_retirement import warn_legacy_surface + if TYPE_CHECKING: from PIL import Image @@ -26,6 +28,7 @@ def __init__(self): async def download_image(self, image_url: str) -> "Image.Image": """Download a single image from a URL.""" + warn_legacy_surface("OCR") try: import aiohttp from PIL import Image @@ -45,4 +48,5 @@ async def download_image(self, image_url: str) -> "Image.Image": async def download_images(self, urls: List[str]) -> List["Image.Image"]: """Download multiple images from a list of URLs concurrently.""" + warn_legacy_surface("OCR") return await asyncio.gather(*[self.download_image(url) for url in urls]) diff --git a/datafog/processing/image_processing/pytesseract_processor.py b/datafog/processing/image_processing/pytesseract_processor.py index f7291470..1b39f19d 100644 --- a/datafog/processing/image_processing/pytesseract_processor.py +++ b/datafog/processing/image_processing/pytesseract_processor.py @@ -10,6 +10,8 @@ import pytesseract from PIL import Image +from datafog._legacy_retirement import warn_legacy_surface + class PytesseractProcessor: """ @@ -20,6 +22,7 @@ class PytesseractProcessor: """ async def extract_text_from_image(self, image: Image.Image) -> str: + warn_legacy_surface("OCR") try: return pytesseract.image_to_string(image) except Exception as e: diff --git a/datafog/processing/spark_processing/pyspark_udfs.py b/datafog/processing/spark_processing/pyspark_udfs.py index 2d7e2bc5..aaa53ece 100644 --- a/datafog/processing/spark_processing/pyspark_udfs.py +++ b/datafog/processing/spark_processing/pyspark_udfs.py @@ -9,6 +9,8 @@ import importlib +from datafog._legacy_retirement import warn_legacy_surface + PII_ANNOTATION_LABELS = ["DATE_TIME", "LOC", "NRP", "ORG", "PER"] MAXIMAL_STRING_SIZE = 1000000 DEFAULT_SPACY_MODEL = "en_core_web_lg" @@ -20,6 +22,7 @@ def pii_annotator(text: str, broadcasted_nlp) -> list[list[str]]: Returns: list[list[str]]: Values as arrays in order defined in the PII_ANNOTATION_LABELS. """ + warn_legacy_surface("Spark") ensure_installed("pyspark") ensure_installed("spacy") @@ -47,6 +50,7 @@ def broadcast_pii_annotator_udf( spark_session=None, spacy_model: str = DEFAULT_SPACY_MODEL ): """Broadcast PII annotator across Spark cluster and create UDF""" + warn_legacy_surface("Spark") ensure_installed("pyspark") ensure_installed("spacy") import spacy diff --git a/datafog/services/image_service.py b/datafog/services/image_service.py index 893e2f72..453e19b6 100644 --- a/datafog/services/image_service.py +++ b/datafog/services/image_service.py @@ -13,6 +13,8 @@ import ssl from typing import TYPE_CHECKING, Any, List, Union +from datafog._legacy_retirement import LegacySurfaceWarning, warn_legacy_surface + if TYPE_CHECKING: from PIL import Image @@ -24,6 +26,7 @@ class ImageDownloader: """Asynchronous image downloader with SSL support.""" async def download_image(self, url: str) -> "Image.Image": + warn_legacy_surface("OCR") try: import aiohttp import certifi @@ -57,6 +60,7 @@ class ImageService: """ def __init__(self, use_donut: bool = False, use_tesseract: bool = True): + warn_legacy_surface("OCR") self.downloader = ImageDownloader() # Check if we're in a test environment @@ -133,12 +137,14 @@ def _get_donut_processor(self): async def download_images( self, urls: List[str] ) -> List[Union["Image.Image", BaseException]]: + warn_legacy_surface("OCR") tasks = [ asyncio.create_task(self.downloader.download_image(url)) for url in urls ] return await asyncio.gather(*tasks, return_exceptions=True) async def ocr_extract(self, image_paths: List[str]) -> List[str]: + warn_legacy_surface("OCR") from PIL import Image results = [] @@ -167,6 +173,8 @@ async def ocr_extract(self, image_paths: List[str]) -> List[str]: raise ValueError("No OCR processor selected") results.append(text) + except LegacySurfaceWarning: + raise except Exception as e: error_msg = f"Error processing image {path}: {str(e)}" logging.error(error_msg) @@ -175,6 +183,7 @@ async def ocr_extract(self, image_paths: List[str]) -> List[str]: return results async def process_images(self, image_urls, operation): + warn_legacy_surface("OCR") results = [] for url in image_urls: logging.info(f"Fetching image from {url}") @@ -185,6 +194,7 @@ async def process_images(self, image_urls, operation): return results async def process_image(self, image, operation): + warn_legacy_surface("OCR") # Implement image processing logic logging.info(f"Processed image with operation: {operation}") pass diff --git a/datafog/services/spark_service.py b/datafog/services/spark_service.py index bf7d2e48..1ace78fd 100644 --- a/datafog/services/spark_service.py +++ b/datafog/services/spark_service.py @@ -9,6 +9,8 @@ import os from typing import List +from datafog._legacy_retirement import warn_legacy_surface + class SparkService: """ @@ -19,6 +21,7 @@ class SparkService: """ def __init__(self, master=None): + warn_legacy_surface("Spark") self.master = master self.ensure_installed("pyspark") diff --git a/datafog/v5.py b/datafog/v5.py new file mode 100644 index 00000000..caa270b4 --- /dev/null +++ b/datafog/v5.py @@ -0,0 +1,84 @@ +"""Preview of the Core-native API planned for DataFog 5.0. + +Names are the actual datafog_core objects: no legacy schema or strategy +translation is applied. Install ``datafog[rust]`` to use this module. Importing +``datafog`` or this namespace alone does not load the native extension. +""" + +from importlib import import_module +from typing import TYPE_CHECKING + +if TYPE_CHECKING: # pragma: no cover - static imports are never executed at runtime + from datafog_core import DataFogConfigurationError as DataFogConfigurationError + from datafog_core import DataFogFindingError as DataFogFindingError + from datafog_core import DataFogInternalError as DataFogInternalError + from datafog_core import DataFogKeyProviderError as DataFogKeyProviderError + from datafog_core import FieldMapping as FieldMapping + from datafog_core import Finding as Finding + from datafog_core import PrivacyManager as PrivacyManager + from datafog_core import Restoration as Restoration + from datafog_core import RestoreResult as RestoreResult + from datafog_core import StructuredFinding as StructuredFinding + from datafog_core import StructuredRestoration as StructuredRestoration + from datafog_core import StructuredRestoreResult as StructuredRestoreResult + from datafog_core import StructuredScanResult as StructuredScanResult + from datafog_core import StructuredTransformation as StructuredTransformation + from datafog_core import StructuredTransformResult as StructuredTransformResult + from datafog_core import TextRange as TextRange + from datafog_core import Transformation as Transformation + from datafog_core import TransformResult as TransformResult + from datafog_core import discover_fields as discover_fields + from datafog_core import scan as scan + from datafog_core import scan_and_transform as scan_and_transform + from datafog_core import ( + scan_and_transform_structured as scan_and_transform_structured, + ) + from datafog_core import scan_structured as scan_structured + from datafog_core import transform as transform + from datafog_core import transform_structured as transform_structured + +__all__ = [ + "DataFogConfigurationError", + "DataFogFindingError", + "DataFogInternalError", + "DataFogKeyProviderError", + "TextRange", + "Finding", + "Transformation", + "TransformResult", + "Restoration", + "RestoreResult", + "FieldMapping", + "StructuredFinding", + "StructuredScanResult", + "StructuredTransformation", + "StructuredTransformResult", + "StructuredRestoration", + "StructuredRestoreResult", + "PrivacyManager", + "scan", + "transform", + "scan_and_transform", + "discover_fields", + "scan_structured", + "transform_structured", + "scan_and_transform_structured", +] + + +def __getattr__(name: str): + if name not in __all__: + raise AttributeError(f"module {__name__!r} has no attribute {name!r}") + try: + core = import_module("datafog_core") + except ImportError as exc: + raise ImportError( + 'datafog.v5 requires DataFog Core. Install with: pip install "datafog[rust]"' + ) from exc + value = getattr(core, name) + globals()[name] = value + return value + + +def __dir__(): + return sorted(set(globals()) | set(__all__)) diff --git a/docs/cli.rst b/docs/cli.rst index 8dfcc9f8..e27f4d72 100644 --- a/docs/cli.rst +++ b/docs/cli.rst @@ -8,8 +8,13 @@ The main entrypoint for the CLI is through the DataFog client file, defined in : We use Typer to build the CLI, with each command defined as a separate function. Core text commands such as ``scan-text``, ``redact-text``, ``replace-text``, -and ``hash-text`` are the primary 4.5 CLI path. OCR commands remain available -for existing users, but they are optional: +and ``hash-text`` are the primary CLI path. The unreleased 4.9 bridge leaves text commands on their +existing Python detection paths; the opt-in Rust backend is a Python API option, +not a new CLI flag. Install ``datafog[cli]`` for the command-line dependencies. + +OCR commands remain functional in 4.9 but are deprecated for removal in 5.0. +``scan-image`` emits a visible ``FutureWarning`` at use time. Its optional +dependencies are: * Local image OCR requires ``datafog[ocr]`` and any needed system OCR binaries such as Tesseract. @@ -17,7 +22,11 @@ for existing users, but they are optional: * Donut OCR requires ``datafog[nlp-advanced,ocr]`` and a local model. Spark/distributed workflows are Python SDK surfaces rather than first-path CLI -commands. Install ``datafog[distributed]`` when using ``SparkService``. +commands. Spark support is also deprecated in 4.9 for removal in 5.0. Install +``datafog[distributed]`` when using ``SparkService`` during 4.x. Users needing +OCR/Spark after the cutover can remain on the final 4.x release. See +:doc:`optional-surfaces` and the +:download:`unreleased 4.9 migration guide `. German locale support --------------------- diff --git a/docs/getting-started.rst b/docs/getting-started.rst index cf6efca7..8df06248 100644 --- a/docs/getting-started.rst +++ b/docs/getting-started.rst @@ -1,8 +1,8 @@ ================================ -Getting Started With DataFog 4.5 +Getting Started With DataFog ================================ -DataFog 4.5 focuses on lightweight text PII screening. A core install should +DataFog focuses on lightweight text PII screening. A core install should let you scan and redact common structured PII without installing OCR, Spark, large NLP models, or middleware integrations. @@ -45,10 +45,43 @@ Optional extras are explicit: - ``pip install "datafog[all]"`` - You are developing or deliberately want every optional surface. +Unreleased 4.9 bridge +===================== + +The following APIs are development previews, not a claim that 4.9 is published. +From a checkout containing the 4.9 implementation, install the explicit Rust +extra to evaluate them: + +.. code-block:: bash + + python -m pip install -e ".[rust]" + +The extra pins ``datafog-core==0.3.1``. It does not change the default Python +backend, and the existing ``all`` extra does not include Rust. To opt in: + +.. code-block:: python + + import datafog + + result = datafog.scan("Contact jane@example.com", engine="regex", backend="rust") + print(result.entities) + +Only ``engine="regex"`` supports this backend. German locales and ``DE_*`` entity +selections are unsupported by the pinned Core version and explicitly rejected +by the legacy Rust adapter; use the Python backend for German detection. Other +detector differences remain, so this is an experimental comparison path. + +For native Core types use ``datafog.v5``; for the explicit legacy facade use +``datafog.compat.v4``. See :doc:`python-sdk` and the +:download:`complete migration guide ` before switching schemas. +``detect()`` and ``process()`` remain available in 4.9 with revised 5.0 removal +warnings. OCR and Spark are also deprecated in 4.9 for removal in 5.0; existing +extras remain available during 4.9. + Python Usage ============ -Use the top-level helpers for the 4.5 core path: +Use the top-level helpers for the core text path: .. code-block:: python @@ -120,7 +153,8 @@ The CLI core path is text-first: datafog hash-text "Contact jane@example.com" datafog redact-text "Steuer-ID 12345678901" --locale de -Image commands are optional. Install ``datafog[ocr]`` for local OCR and +Image commands are optional and scheduled for deprecation in 4.9 and removal +in 5.0. They remain functional in 4.9. Install ``datafog[ocr]`` for local OCR and ``datafog[web,ocr]`` when the CLI needs to download image inputs. What 4.x Is Not diff --git a/docs/important-concepts.rst b/docs/important-concepts.rst index d08c932c..a1655593 100644 --- a/docs/important-concepts.rst +++ b/docs/important-concepts.rst @@ -8,7 +8,12 @@ Overview Data Models ^^^^^^^^^^^ -Key data models to support PII annotation and OCR analysis. +Existing models support legacy PII annotation and optional OCR analysis. +The unreleased 4.9 bridge retains them and adds ``datafog.compat.v4`` for the +legacy scan/redact result classes (``Entity``, ``ScanResult``, ``RedactResult``). +``datafog.v5`` separately previews native Core ``Finding`` and ``TransformResult`` +objects; these have different fields and transformation semantics. See +:doc:`python-sdk` and the :download:`migration guide `. * AnalysisExplanation * AnnotationResult @@ -19,7 +24,8 @@ Key data models to support PII annotation and OCR analysis. Processors ^^^^^^^^^^^ -Main processors: +Text processors remain available. OCR processors below are deprecated in the +unreleased 4.9 bridge and scheduled for removal in 5.0: * SpacyAnnotator Text annotation with spaCy @@ -30,7 +36,12 @@ Main processors: Services ^^^^^^^^^^^ -Core services: +``TextService`` remains available. ``ImageService`` and ``SparkService`` are +optional legacy services, deprecated in 4.9 for removal in 5.0. They remain +functional throughout 4.9; users requiring them after the cutover can remain on +the final 4.x release. See :doc:`optional-surfaces`. + +Existing services: * ImageService Image handling and OCR diff --git a/docs/index.rst b/docs/index.rst index 57c41a3e..44f94e63 100644 --- a/docs/index.rst +++ b/docs/index.rst @@ -2,21 +2,41 @@ DataFog Documentation ===================== -DataFog 4.5 is a lightweight text PII screening package for Python. The +DataFog is a lightweight text PII screening package for Python. The primary path is a small core install, fast regex-based scanning and redaction, agent-friendly guardrail helpers, and explicit optional extras when you need NLP, OCR, Spark, or web inputs. Start with :doc:`getting-started` if you want the shortest route from install to scanning text. The roadmap and historical planning pages remain available, -but the live user docs are the first path for 4.5. +but the live user docs are the first path for current text APIs. -Use DataFog 4.5 -=============== +4.9 development preview +======================= + +.. note:: + + The 4.9 bridge described here is unreleased development work. These pages do + not announce a published 4.9 package. A normal PyPI install does not imply + availability of the new APIs below. + +4.9 preserves existing imports, result classes, Python detection defaults, and +redaction behavior. It adds explicit experimental Rust detection through +``backend="rust"``, the ``datafog.compat.v4`` facade, and the ``datafog.v5`` native +Core schema preview. See :doc:`python-sdk` and the +:download:`complete 4.9 migration guide `. + +The planned 4.9 release deprecates ``detect()``/``process()``, OCR, and Spark for +removal in 5.0. The earlier promise to retain ``detect()``/``process()`` throughout +5.x is revised. OCR/Spark remain functional in 4.9; users needing them after the +cutover can remain on the final 4.x release. See :doc:`optional-surfaces`. + +Use DataFog +=========== .. toctree:: :maxdepth: 2 - :caption: Use DataFog 4.5 + :caption: Use DataFog getting-started python-sdk @@ -49,7 +69,7 @@ Planning And History ==================== The pages below document release planning, migration history, and future -direction. They are useful context, but they are secondary to the live 4.5 +direction. They are useful context, but they are secondary to the live usage path above. .. toctree:: diff --git a/docs/migration-4.9.md b/docs/migration-4.9.md new file mode 100644 index 00000000..72233796 --- /dev/null +++ b/docs/migration-4.9.md @@ -0,0 +1,166 @@ +# Migrating incrementally with DataFog 4.9 + +> **Unreleased:** This guide describes the upcoming 4.9 bridge. Until it is +> published, evaluate these APIs from the development checkout with +> `python -m pip install -e ".[rust]"`; a normal PyPI install does not include them. + +4.9 is a bridge to the Rust-backed 5.0 API. The default Python detector, existing +imports, result objects, and redaction strategies continue to work. The optional +Rust backend and native API preview are experimental and explicitly selected. + +## Opt into Rust detection + +```bash +pip install "datafog[rust]" +``` + +The extra pins the tested `datafog-core==0.3.1` wheel. The base package needs no +Rust installation or native module. The existing `all` extra retains its legacy +dependency set; request `rust` explicitly, or use `datafog[all,rust]`. + +```python +import datafog + +result = datafog.scan("Contact alice@example.com", backend="rust") +assert result.entities[0].type == "EMAIL" + +result = datafog.redact("Contact alice@example.com", backend="rust") +assert result.redacted_text == "Contact [EMAIL_1]" +assert result.mapping == {"[EMAIL_1]": "alice@example.com"} +``` + +`backend` is keyword-only and defaults to `python`. Only `engine="regex"` supports +the Rust backend in 4.9. Low-level `datafog.engine.scan` and `scan_and_redact` +default to `smart`, so pass `engine="regex"` explicitly there. Missing native +dependencies and unsupported combinations raise errors; there is no automatic +fallback to Python. Native scanning failures propagate instead of becoming empty +results. + +The legacy adapter converts Core code-point offsets to Python indices and keeps +legacy `regex` provenance and confidence `1.0` as compatibility values, not native +probability estimates. It retains Python aliases, full-match allowlists (including +Python regex syntax), overlap selection, and transformations. Supplying explicit +entities to `redact` performs no detection and requires no native dependency. + +`sanitize`, `scan_prompt`, and `filter_output` forward the backend keyword. The +`DataFog` class, `TextService`, legacy convenience APIs, CLI, guardrail objects, +application adapters, and ML composition retain their existing detection paths. +Installing the extra alone does not change any of those paths. + +## Existing and new result schemas + +The compatibility namespace provides the scan/redact API explicitly: + +```python +from datafog.compat.v4 import Entity, ScanResult, RedactResult, scan, redact +``` + +Top-level scan/redact delegate to this facade. Result classes retain their +existing identity; their fields and positional calling conventions are preserved. +The compatibility namespace excludes `detect` and `process`. + +The native preview exports the actual Core objects, including structured scanning, +transformations, and provider-backed operations: + +```python +from datafog.v5 import Finding, scan, scan_and_transform + +findings = scan("Contact alice@example.com") +assert isinstance(findings[0], Finding) +assert findings[0].entity_type == "EMAIL" + +result = scan_and_transform( + "Contact alice@example.com", + {"transform": {"default": {"strategy": "redact"}}}, +) +assert result.text == "Contact [EMAIL]" +``` + +Importing `datafog` or the preview namespace alone does not import Core. Accessing +preview functions/types requires the Rust extra. Preview names are reexports, +not a second implementation or a legacy-schema adapter. + +| Existing Python API | Core preview | +| ------------------------------------------ | -------------------------------------------------------------- | +| `ScanResult.entities` | `list[Finding]` | +| `Entity.type`, `text`, `start`, `end` | `entity_type`, `matched_text`, explicit byte/code-point ranges | +| `RedactResult.redacted_text` | `TransformResult.text` | +| Numbered type tokens and plaintext mapping | Unnumbered redaction placeholders and transformation records | +| Legacy `hash` and numbered `pseudonymize` | Explicit keyed/provider-backed Core strategies | + +Legacy `token` is not Core `tokenize`. Do not mechanically rename strategies or +assume that the two schemas and provider requirements are interchangeable. + +## Known detection differences + +Core 0.3.1 does not implement the seven German detectors. The legacy Rust adapter +rejects German locales and `DE_*` selections, directing callers to Python. +The raw native preview retains Core's own API: its accepted locale configuration +does not imply German detector coverage. Continue using Python for those workloads. + +The frozen 111-case baseline yields the following Rust-backend comparison: + +| Outcome | Cases | Interpretation | +| ---------------------------- | ----: | ------------------------------------------------------------------------- | +| Exact match | 61 | Same observable result on these inputs | +| Reviewed detector difference | 2 | Invalid-checksum card and alphanumeric-embedded SSN are rejected by Core | +| Explicitly unsupported | 17 | German requests fail rather than silently lose coverage | +| Outside backend scope | 31 | Signatures, explicit-span transformations, legacy/service/guardrail paths | + +These counts describe this finite synthetic corpus, not universal detection +equivalence or precision/recall. See `tests/contracts/rust-0.3.1.json` for exact +reviewed outcomes and reasons. Each applicable case is asserted independently; +unknown differences fail CI. Generate the full per-case report with: + +```bash +python -m tests.rust_contract --output /tmp/rust-parity.json +``` + +The published `tests/contracts/4.8.1.json` remains unchanged. The 4.9 checker +permits only four added keyword-only backend parameters and the explicitly +revised detect/process warning messages; all other legacy observations remain +exact comparisons. The standalone oracle runner still checks exact 4.8.1 behavior +and should be used with the released 4.8.1 wheel, not the 4.9 checkout. + +## Retirement schedule + +| Surface | 4.9 | 5.0 plan | +| ---------------------------------------- | ---------------------------------------- | -------------------------------------------------------------------------- | +| `detect()` / `process()` | Still work; updated `FutureWarning` | Remove | +| OCR / Donut / Tesseract / image services | Still work; use-time deprecation notices | Remove supported OCR surfaces and extras | +| Spark / distributed processing | Still work; use-time deprecation notices | Remove supported Spark surfaces and extras | +| Legacy scan/redact facade | Available at top level and `compat.v4` | Native schema becomes primary; compatibility lifetime to be set separately | +| spaCy / GLiNER | Unchanged | No removal decision in this increment | + +**The earlier promise to retain `detect` and `process` throughout 5.x is revised.** +They are now scheduled for removal in 5.0. Use `scan` or `redact` in 4.9 and evaluate +the native preview before upgrading to 5.0. `process(anonymize=True)` used older +placeholder/hash semantics; migrating to `redact` can intentionally change output. + +Users needing OCR/Spark can remain on the final 4.x release. No successor package +or indefinite support commitment is introduced here. Deprecation notices do not +trigger downloads or import heavy dependencies during ordinary text imports. + +## Verification and performance + +Run the contract/backend suites with and without the Rust extra. CI also builds +and installs a wheel, then runs an isolated-interpreter smoke test on Linux, +macOS and Windows; it does not rely solely on imports from the checkout. + +```bash +python -m pytest tests/test_contract_481.py tests/test_rust_backend.py \ + tests/test_api_bridge_49.py tests/test_rust_contract.py -q +python benchmarks/compare_detection_backends.py --output /tmp/backend-timings.json +``` + +The benchmark measures public calls, including native/Python conversion, +redaction, and fresh-process startup on short, mixed, and roughly 1 MB sparse +synthetic text. It reports entity counts alongside timings. No general speedup +claim is made from a single machine or from cases with different outputs. + +A local CPython 3.12 macOS ARM64 reference run is recorded in +`benchmarks/results-4.9.json`. Median scan latency was 19.74 versus 2.30 microseconds +for the short payload, 39.50 versus 7.58 microseconds for mixed PII, and 121.91 +versus 5.33 milliseconds for the large sparse payload (Python versus Rust). +Fresh-process import plus first scan was approximately 81 milliseconds for both. +These are local measurements, not release performance guarantees. diff --git a/docs/optional-surfaces.rst b/docs/optional-surfaces.rst index 57ea5994..96d1494d 100644 --- a/docs/optional-surfaces.rst +++ b/docs/optional-surfaces.rst @@ -2,7 +2,7 @@ Optional OCR And Spark ========================= -DataFog 4.5 keeps the core package focused on lightweight text PII screening. +DataFog 4.9 keeps the core package focused on lightweight text PII screening. The default path is: .. code-block:: bash @@ -16,7 +16,16 @@ The default path is: result = datafog.redact("Email jane@example.com", engine="regex") print(result.redacted_text) -OCR and Spark are supported optional surfaces. They are useful for image and +OCR and Spark are deprecated optional surfaces in 4.9 and will be removed in 5.0. +Their APIs, installation extras, and existing behavior remain available throughout +4.9. Remain on the final 4.x release if you need continued OCR or Spark support. +No replacement package is introduced by this migration. + +Use sites emit ``FutureWarning`` notices, visible with Python's default warning +filters, including the ``scan-image`` CLI command. Importing DataFog or using +its text APIs does not emit OCR/Spark retirement notices. + +These optional surfaces They are useful for image and distributed workflows, but they should not be treated as required for the core install, package import, text scanning, text redaction, or guardrail helpers. @@ -51,8 +60,8 @@ Notes: and system Tesseract smoke checks. * Donut OCR requires a model that is already available locally. DataFog should not download models implicitly during normal runtime usage. -* OCR is not deprecated. A broader OCR API and packaging overhaul is deferred - beyond the 4.5 focus release. +* OCR APIs, image download/processing helpers, and the ``scan-image`` CLI + command are scheduled for removal in 5.0, along with OCR-only dependencies. Example local OCR flow: @@ -89,8 +98,8 @@ Notes: * ``SparkService`` requires PySpark and a Java runtime. * Spark PII UDF helpers also require spaCy and an installed spaCy model. -* Spark is not deprecated. A broader Spark overhaul is deferred beyond the 4.5 - focus release. +* Spark services, PII UDF helpers, and the ``distributed`` extra are scheduled + for removal in 5.0. Example local Spark flow: diff --git a/docs/python-sdk.rst b/docs/python-sdk.rst index ce70f577..7e2a2283 100644 --- a/docs/python-sdk.rst +++ b/docs/python-sdk.rst @@ -4,7 +4,7 @@ DataFog Python SDK Overview -------- -The primary 4.5 SDK path is lightweight text PII screening through the +The primary SDK path is lightweight text PII screening through the top-level ``datafog`` helpers. These helpers use the regex engine by default and do not require OCR, Spark, model downloads, or distributed dependencies. @@ -27,10 +27,73 @@ available for existing users. ``TextService(engine="regex")`` is the dependency-light service path; ``spacy``, ``gliner``, ``smart``, OCR, and Spark surfaces require their explicit extras. +4.9 compatibility and Core preview (unreleased) +----------------------------------------------- + +These additions describe development work for 4.9; they do not indicate a +published release. Existing top-level ``scan``/``redact`` functions retain their +result shapes and use the Python backend by default. They delegate through the +new facade, whose classes are the same objects as the established result types: + +.. code-block:: python + + from datafog.compat.v4 import Entity, ScanResult, RedactResult, scan, redact + +The facade does not include ``detect`` or ``process``. Those legacy helpers +continue to work at the top level in 4.9 but warn of removal in 5.0, revising the +previous promise to retain them throughout 5.x. Moving from ``process`` to +``redact`` can change old placeholder and hash output; compare results explicitly. + +Install ``.[rust]`` from the development checkout to evaluate Rust detection: + +.. code-block:: python + + import datafog + + result = datafog.redact("Contact jane@example.com", engine="regex", backend="rust") + assert result.redacted_text == "Contact [EMAIL_1]" + +``backend`` is keyword-only, defaults to ``python``, and supports Rust only with +``engine="regex"``. Detection uses Core; Python still handles legacy aliases, +full-match allowlists, result conversion, and transformation strategies. Explicit +entities passed to ``redact`` require no native scanning. Missing native modules, +unsupported combinations, and native errors are not silently retried in Python. +The ``DataFog`` class, ``TextService``, CLI, and ML composition keep their existing +backend paths. + +The native ``datafog.v5`` preview exposes actual Core types and operations rather +than adapting them to legacy classes: + +.. code-block:: python + + from datafog.v5 import Finding, scan, scan_and_transform + + findings = scan("Contact jane@example.com") + assert isinstance(findings[0], Finding) + result = scan_and_transform( + "Contact jane@example.com", + {"transform": {"default": {"strategy": "redact"}}}, + ) + assert result.text == "Contact [EMAIL]" + +Native scanning returns ``list[Finding]`` with ``entity_type``, ``matched_text``, +and explicit offset ranges. Native transformation returns ``TransformResult`` +with ``text`` and transformation records. Legacy numbered tokens, plaintext +mappings, and hash/pseudonymization strategies are not interchangeable with +native strategies. Importing the namespace is lazy; accessing its exports +requires the Rust extra. The supported compatibility lifetime after 5.0 remains +a separate decision. + +The pinned Core 0.3.1 lacks German detectors and differs on some structured +inputs. The legacy Rust adapter rejects German requests; the raw native preview +retains Core's own behavior and must not be assumed to provide German coverage. +Read the :download:`complete migration guide ` for the exact +schema comparison, finite-corpus parity results, and verification commands. + German locale coverage ---------------------- -DataFog 4.5 includes regex-only German structured PII support without adding +The Python backend includes regex-only German structured PII support without adding dependencies. German-only identifiers are opt-in because their raw shapes are country-specific or common in ordinary product, ticket, invoice, and order data. @@ -62,7 +125,7 @@ The opt-in German set currently covers ``DE_VAT_ID``, ``DE_IBAN``, Optional services ----------------- -OCR and Spark are supported optional surfaces, not the primary 4.5 path: +OCR and Spark remain available as optional surfaces throughout 4.9: * Use ``datafog[ocr]`` for local OCR helpers such as ``ImageService`` and ``PytesseractProcessor``. @@ -73,9 +136,11 @@ OCR and Spark are supported optional surfaces, not the primary 4.5 path: * Use ``datafog[distributed,nlp]`` plus an installed spaCy model for Spark PII UDF helpers. -OCR and Spark are not deprecated. Their broader overhaul is deferred so the -4.5 release can keep the core package tight while preserving existing optional -usage. See :doc:`optional-surfaces` for install notes and limitations. +The unreleased 4.9 bridge deprecates OCR and Spark for removal in 5.0. Use-time +``FutureWarning`` notices are visible under normal Python warning filters. Users +needing these features can remain on the final 4.x release; this migration does +not introduce successor packages or promise indefinite maintenance. See +:doc:`optional-surfaces` for install notes and limitations. Definitions ----------- diff --git a/docs/roadmap.rst b/docs/roadmap.rst index 080a6b0f..a4aee1f8 100644 --- a/docs/roadmap.rst +++ b/docs/roadmap.rst @@ -2,6 +2,14 @@ Release Roadmap ================ +.. note:: + + This earlier planning document is retained for context. The upcoming 4.9 + bridge revises its compatibility commitments: ``detect``/``process``, OCR, + and Spark are deprecated in 4.9 and removed in 5.0. The former promise to + retain the shims through 5.x no longer applies. See the + :download:`current migration plan ` for the agreed scope. + Where DataFog is today (4.8.x) and where it is going (v5.0.0). The 4.x line delivered the lightweight-core architecture and, from 4.6.0 on, an offline PII firewall for AI agents and gateways. The v5 cycle turns that diff --git a/docs/v5-compatibility-matrix.rst b/docs/v5-compatibility-matrix.rst index a0e2691c..ac29c378 100644 --- a/docs/v5-compatibility-matrix.rst +++ b/docs/v5-compatibility-matrix.rst @@ -2,6 +2,14 @@ v5 Compatibility Matrix ======================= +.. note:: + + This earlier planning document is retained for context. The upcoming 4.9 + bridge revises its compatibility commitments: ``detect``/``process``, OCR, + and Spark are deprecated in 4.9 and removed in 5.0. The former promise to + retain the shims through 5.x no longer applies. See the + :download:`current migration plan ` for the agreed scope. + Status ------ diff --git a/scripts/capture_481_contract.py b/scripts/capture_481_contract.py index 7aadeb46..42300351 100644 --- a/scripts/capture_481_contract.py +++ b/scripts/capture_481_contract.py @@ -25,7 +25,7 @@ def main(): fixture_path = root / "tests/contracts/4.8.1.json" if args.output.resolve() == fixture_path: parser.error("Capture to a separate candidate file for review") - fixture = json.loads(fixture_path.read_text()) + fixture = json.loads(fixture_path.read_text(encoding="utf-8")) provenance = fixture["provenance"] digest = hashlib.sha256(args.wheel.read_bytes()).hexdigest() if digest != provenance["sha256"]: @@ -69,7 +69,7 @@ def main(): runner = runpy.run_path(str(root / "tests/contract_481.py")) for case in fixture["cases"]: case["expected"] = runner["observe"](case) - with args.output.open("x") as output: + with args.output.open("x", encoding="utf-8") as output: output.write(json.dumps(fixture, indent=2, ensure_ascii=False) + "\n") print(f"Captured {len(fixture['cases'])} cases to {args.output}") diff --git a/scripts/check_rust_install.py b/scripts/check_rust_install.py new file mode 100644 index 00000000..f0e63fd9 --- /dev/null +++ b/scripts/check_rust_install.py @@ -0,0 +1,41 @@ +"""Smoke-test installed 4.9 bridge wheels with python -I, outside checkout imports.""" + +import importlib.metadata +import sys + + +def main(): + assert sys.flags.isolated, "Run with python -I" + + import datafog_core + + import datafog + from datafog import v5 + from datafog.compat import v4 + + assert importlib.metadata.version("datafog-core") == "0.3.1" + assert v4.Entity is datafog.Entity + assert v5.Finding is datafog_core.Finding + text = "πŸ‘‹ Contact alice@example.com" + legacy = datafog.scan(text, backend="rust") + assert isinstance(legacy, datafog.ScanResult) + assert legacy.entities[0].text == "alice@example.com" + assert ( + text[legacy.entities[0].start : legacy.entities[0].end] == "alice@example.com" + ) + assert datafog.redact(text, backend="rust").redacted_text == "πŸ‘‹ Contact [EMAIL_1]" + native = v5.scan(text) + assert isinstance(native[0], datafog_core.Finding) + assert ( + v5.scan_and_transform( + text, {"transform": {"default": {"strategy": "redact"}}} + ).text + == "πŸ‘‹ Contact [EMAIL]" + ) + print( + "Installed wheel: compatibility facade, Rust backend and native preview passed" + ) + + +if __name__ == "__main__": + main() diff --git a/setup.py b/setup.py index 1bf72344..4f7d7e90 100644 --- a/setup.py +++ b/setup.py @@ -1,7 +1,7 @@ from setuptools import find_packages, setup # Read README for the long description -with open("README.md", "r") as f: +with open("README.md", "r", encoding="utf-8") as f: long_description = f.read() # Use a single source of truth for the version from __about__.py @@ -84,6 +84,7 @@ ] extras_require = { + "rust": ["datafog-core==0.3.1"], "nlp": nlp_deps, "nlp-advanced": nlp_advanced_deps, "ocr": ocr_deps, diff --git a/tests/contract_481.py b/tests/contract_481.py index 075d9c71..a2150a51 100644 --- a/tests/contract_481.py +++ b/tests/contract_481.py @@ -105,7 +105,7 @@ def main(): parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--fixture", type=Path, default=FIXTURE) args = parser.parse_args() - fixture = json.loads(args.fixture.read_text()) + fixture = json.loads(args.fixture.read_text(encoding="utf-8")) failures = [] for case in fixture["cases"]: actual = observe(case) diff --git a/tests/contracts/rust-0.3.1.json b/tests/contracts/rust-0.3.1.json new file mode 100644 index 00000000..35976c35 --- /dev/null +++ b/tests/contracts/rust-0.3.1.json @@ -0,0 +1,222 @@ +{ + "core_version": "0.3.1", + "cases": { + "scan-observation-invalid-card": { + "classification": "detector-difference", + "reason": "Core validates card checksums; Python 4.8.1 accepts this card-shaped value. Experimental precision difference, not a claim of full parity.", + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [], + "text": "4111 1111 1111 1112", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + "scan-observation-boundary": { + "classification": "detector-difference", + "reason": "Core rejects the SSN embedded in an ASCII alphanumeric token; Python 4.8.1 detects the numeric substring.", + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [], + "text": "x123-45-6789y", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + "locale-de-DE_VAT_ID": { + "classification": "unsupported", + "reason": "Published Core 0.3.1 lacks German detection. The adapter rejects the request instead of silently returning incomplete findings.", + "expected": { + "error": { + "type": "ValueError", + "message": "backend='rust' does not yet support German locales or DE_* entity types; use backend='python' for German detection" + }, + "warnings": [] + } + }, + "locale-explicit-DE_VAT_ID": { + "classification": "unsupported", + "reason": "Published Core 0.3.1 lacks German detection. The adapter rejects the request instead of silently returning incomplete findings.", + "expected": { + "error": { + "type": "ValueError", + "message": "backend='rust' does not yet support German locales or DE_* entity types; use backend='python' for German detection" + }, + "warnings": [] + } + }, + "locale-de-DE_IBAN": { + "classification": "unsupported", + "reason": "Published Core 0.3.1 lacks German detection. The adapter rejects the request instead of silently returning incomplete findings.", + "expected": { + "error": { + "type": "ValueError", + "message": "backend='rust' does not yet support German locales or DE_* entity types; use backend='python' for German detection" + }, + "warnings": [] + } + }, + "locale-explicit-DE_IBAN": { + "classification": "unsupported", + "reason": "Published Core 0.3.1 lacks German detection. The adapter rejects the request instead of silently returning incomplete findings.", + "expected": { + "error": { + "type": "ValueError", + "message": "backend='rust' does not yet support German locales or DE_* entity types; use backend='python' for German detection" + }, + "warnings": [] + } + }, + "locale-de-DE_TAX_ID": { + "classification": "unsupported", + "reason": "Published Core 0.3.1 lacks German detection. The adapter rejects the request instead of silently returning incomplete findings.", + "expected": { + "error": { + "type": "ValueError", + "message": "backend='rust' does not yet support German locales or DE_* entity types; use backend='python' for German detection" + }, + "warnings": [] + } + }, + "locale-explicit-DE_TAX_ID": { + "classification": "unsupported", + "reason": "Published Core 0.3.1 lacks German detection. The adapter rejects the request instead of silently returning incomplete findings.", + "expected": { + "error": { + "type": "ValueError", + "message": "backend='rust' does not yet support German locales or DE_* entity types; use backend='python' for German detection" + }, + "warnings": [] + } + }, + "locale-de-DE_SOCIAL_SECURITY_NUMBER": { + "classification": "unsupported", + "reason": "Published Core 0.3.1 lacks German detection. The adapter rejects the request instead of silently returning incomplete findings.", + "expected": { + "error": { + "type": "ValueError", + "message": "backend='rust' does not yet support German locales or DE_* entity types; use backend='python' for German detection" + }, + "warnings": [] + } + }, + "locale-explicit-DE_SOCIAL_SECURITY_NUMBER": { + "classification": "unsupported", + "reason": "Published Core 0.3.1 lacks German detection. The adapter rejects the request instead of silently returning incomplete findings.", + "expected": { + "error": { + "type": "ValueError", + "message": "backend='rust' does not yet support German locales or DE_* entity types; use backend='python' for German detection" + }, + "warnings": [] + } + }, + "locale-de-DE_POSTAL_CODE": { + "classification": "unsupported", + "reason": "Published Core 0.3.1 lacks German detection. The adapter rejects the request instead of silently returning incomplete findings.", + "expected": { + "error": { + "type": "ValueError", + "message": "backend='rust' does not yet support German locales or DE_* entity types; use backend='python' for German detection" + }, + "warnings": [] + } + }, + "locale-explicit-DE_POSTAL_CODE": { + "classification": "unsupported", + "reason": "Published Core 0.3.1 lacks German detection. The adapter rejects the request instead of silently returning incomplete findings.", + "expected": { + "error": { + "type": "ValueError", + "message": "backend='rust' does not yet support German locales or DE_* entity types; use backend='python' for German detection" + }, + "warnings": [] + } + }, + "locale-de-DE_PASSPORT_NUMBER": { + "classification": "unsupported", + "reason": "Published Core 0.3.1 lacks German detection. The adapter rejects the request instead of silently returning incomplete findings.", + "expected": { + "error": { + "type": "ValueError", + "message": "backend='rust' does not yet support German locales or DE_* entity types; use backend='python' for German detection" + }, + "warnings": [] + } + }, + "locale-explicit-DE_PASSPORT_NUMBER": { + "classification": "unsupported", + "reason": "Published Core 0.3.1 lacks German detection. The adapter rejects the request instead of silently returning incomplete findings.", + "expected": { + "error": { + "type": "ValueError", + "message": "backend='rust' does not yet support German locales or DE_* entity types; use backend='python' for German detection" + }, + "warnings": [] + } + }, + "locale-de-DE_RESIDENCE_PERMIT_NUMBER": { + "classification": "unsupported", + "reason": "Published Core 0.3.1 lacks German detection. The adapter rejects the request instead of silently returning incomplete findings.", + "expected": { + "error": { + "type": "ValueError", + "message": "backend='rust' does not yet support German locales or DE_* entity types; use backend='python' for German detection" + }, + "warnings": [] + } + }, + "locale-explicit-DE_RESIDENCE_PERMIT_NUMBER": { + "classification": "unsupported", + "reason": "Published Core 0.3.1 lacks German detection. The adapter rejects the request instead of silently returning incomplete findings.", + "expected": { + "error": { + "type": "ValueError", + "message": "backend='rust' does not yet support German locales or DE_* entity types; use backend='python' for German detection" + }, + "warnings": [] + } + }, + "locale-negative-context": { + "classification": "unsupported", + "reason": "Published Core 0.3.1 lacks German detection. The adapter rejects the request instead of silently returning incomplete findings.", + "expected": { + "error": { + "type": "ValueError", + "message": "backend='rust' does not yet support German locales or DE_* entity types; use backend='python' for German detection" + }, + "warnings": [] + } + }, + "locale-alias-de-DE": { + "classification": "unsupported", + "reason": "Published Core 0.3.1 lacks German detection. The adapter rejects the request instead of silently returning incomplete findings.", + "expected": { + "error": { + "type": "ValueError", + "message": "backend='rust' does not yet support German locales or DE_* entity types; use backend='python' for German detection" + }, + "warnings": [] + } + }, + "locale-alias-de_de": { + "classification": "unsupported", + "reason": "Published Core 0.3.1 lacks German detection. The adapter rejects the request instead of silently returning incomplete findings.", + "expected": { + "error": { + "type": "ValueError", + "message": "backend='rust' does not yet support German locales or DE_* entity types; use backend='python' for German detection" + }, + "warnings": [] + } + } + } +} diff --git a/tests/rust_contract.py b/tests/rust_contract.py new file mode 100644 index 00000000..a63184b1 --- /dev/null +++ b/tests/rust_contract.py @@ -0,0 +1,76 @@ +"""Compare applicable 4.8.1 observations with the opt-in published Rust backend.""" + +import argparse +import copy +import importlib.metadata +import json +from collections import Counter +from pathlib import Path + +from tests.contract_481 import FIXTURE, observe + +DEVIATIONS = Path(__file__).parent / "contracts" / "rust-0.3.1.json" +TARGETS = { + "datafog:scan", + "datafog:redact", + "datafog.engine:scan_and_redact", + "datafog:sanitize", + "datafog:scan_prompt", + "datafog:filter_output", +} + + +def applicable(case): + return ( + case.get("operation", "call") == "call" + and case.get("target") in TARGETS + and "entities" not in case.get("kwargs", {}) + ) + + +def native_observation(case): + native_case = copy.deepcopy(case) + native_case.setdefault("kwargs", {})["backend"] = "rust" + return observe(native_case) + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args() + contract = json.loads(FIXTURE.read_text(encoding="utf-8")) + deviations = json.loads(DEVIATIONS.read_text(encoding="utf-8")) + results = [] + for case in contract["cases"]: + if not applicable(case): + results.append({"id": case["id"], "status": "outside-backend-scope"}) + continue + actual = native_observation(case) + deviation = deviations["cases"].get(case["id"]) + expected = deviation["expected"] if deviation else case["expected"] + status = "regression" if actual != expected else "match" + if status == "match" and deviation: + status = deviation["classification"] + results.append( + { + "id": case["id"], + "status": status, + "reason": deviation["reason"] if deviation else None, + "legacy": case["expected"], + "actual": actual, + } + ) + report = { + "core_version": importlib.metadata.version("datafog-core"), + "counts": dict(Counter(row["status"] for row in results)), + "cases": results, + } + args.output.write_text( + json.dumps(report, indent=2, ensure_ascii=False) + "\n", encoding="utf-8" + ) + print(json.dumps(report["counts"], sort_keys=True)) + return bool(report["counts"].get("regression")) + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/test_api_bridge_49.py b/tests/test_api_bridge_49.py new file mode 100644 index 00000000..483aa7f8 --- /dev/null +++ b/tests/test_api_bridge_49.py @@ -0,0 +1,141 @@ +"""Compatibility facade and native-schema preview release gates.""" + +import importlib.util +import inspect +import subprocess +import sys +from types import SimpleNamespace +from unittest.mock import Mock, patch + +import pytest + +import datafog +from datafog import engine, v5 +from datafog.compat import v4 + + +@pytest.mark.parametrize("name", ["Entity", "ScanResult", "RedactResult"]) +def test_legacy_types_retain_identity(name): + assert getattr(datafog, name) is getattr(v4, name) is getattr(engine, name) + + +def test_existing_positional_calls_match_compatibility_facade(): + text = "Email jane@example.com" + args = (text, "regex", ["EMAIL"], None, None, None) + assert datafog.scan(*args) == v4.scan(*args) + args = (text, None, "regex", ["EMAIL"], "token", "llm", None, None, None) + assert datafog.redact(*args) == v4.redact(*args) + assert datafog.redact(*args).redacted_text == "Email [EMAIL_1]" + + +@pytest.mark.parametrize("function", [datafog.scan, datafog.redact, v4.scan, v4.redact]) +def test_backend_is_additive_keyword_only(function): + parameter = inspect.signature(function).parameters["backend"] + assert parameter.kind is inspect.Parameter.KEYWORD_ONLY + assert parameter.default == "python" + + +@pytest.mark.parametrize("name", ["scan", "redact"]) +def test_top_level_delegates_backend_to_facade(name): + with patch.object(v4, name, return_value="sentinel") as delegate: + assert getattr(datafog, name)("text", backend="rust") == "sentinel" + assert delegate.call_args.kwargs["backend"] == "rust" + + +@pytest.mark.parametrize("backend", ["python", "rust"]) +def test_explicit_entities_do_not_scan(backend): + with patch.object(v4, "_scan_and_redact", side_effect=AssertionError("scanned")): + assert ( + datafog.redact("text", entities=[], backend=backend).redacted_text == "text" + ) + + +def test_explicit_entities_validate_backend(): + with pytest.raises(ValueError, match="backend"): + datafog.redact("text", entities=[], backend="unknown") + + +def test_import_and_explicit_redaction_without_native_dependency(): + code = """ +import sys +class BlockNative: + def find_spec(self, fullname, path=None, target=None): + if fullname == "datafog_core" or fullname.startswith("datafog_core."): + raise AssertionError("native import attempted") +sys.meta_path.insert(0, BlockNative()) +import datafog +import datafog.v5 +from datafog.compat import v4 +assert datafog.redact("text", entities=[], backend="rust").redacted_text == "text" +assert "datafog_core" not in sys.modules +""" + subprocess.run([sys.executable, "-c", code], check=True) + + +def test_missing_preview_dependency_has_actionable_error(): + with patch.object(v5, "import_module", side_effect=ImportError("missing")): + with pytest.raises(ImportError, match=r"datafog\[rust\]"): + v5.__getattr__("scan") + + +@pytest.mark.parametrize("namespace", [v4, v5]) +@pytest.mark.parametrize("name", ["detect", "process"]) +def test_new_namespaces_do_not_export_deprecated_shims(namespace, name): + assert name not in namespace.__all__ + assert not hasattr(namespace, name) + + +@pytest.mark.parametrize("name", ["detect", "process"]) +def test_shims_explicitly_revise_removal_promise(name): + with pytest.warns(FutureWarning, match="removed in 5.0") as captured: + getattr(datafog, name)("plain text") + assert "earlier promise" in str(captured[0].message) + assert "has been revised" in str(captured[0].message) + + +def test_real_native_exports_and_transformation(): + core = pytest.importorskip("datafog_core") + for name in v5.__all__: + assert getattr(v5, name) is getattr(core, name) + findings = v5.scan("Email jane@example.com") + assert isinstance(findings[0], v5.Finding) + result = v5.scan_and_transform( + "Email jane@example.com", {"transform": {"default": {"strategy": "redact"}}} + ) + assert isinstance(result, v5.TransformResult) + assert result.text == "Email [EMAIL]" + + +def test_preview_discovery_is_lazy_and_exports_are_cached(monkeypatch): + # A separate module object avoids cached exports from native integration tests. + spec = importlib.util.spec_from_file_location("isolated_preview", v5.__file__) + preview = importlib.util.module_from_spec(spec) + spec.loader.exec_module(preview) + core = SimpleNamespace(**{name: object() for name in preview.__all__}) + loader = Mock(return_value=core) + monkeypatch.setattr(preview, "import_module", loader) + + assert set(preview.__all__) <= set(dir(preview)) + with pytest.raises(AttributeError, match="unrecognized"): + preview.unrecognized + loader.assert_not_called() + + for name in preview.__all__: + assert getattr(preview, name) is getattr(core, name) + assert loader.call_count == len(preview.__all__) + loader.reset_mock() + for name in preview.__all__: + assert getattr(preview, name) is getattr(core, name) + loader.assert_not_called() + + +@pytest.mark.parametrize("redact", [datafog.redact, v4.redact]) +def test_compatibility_facade_rejects_unknown_preset(redact): + with pytest.raises(ValueError, match="preset must be one of"): + redact("plain text", preset="unknown") + + +@pytest.mark.parametrize("option", ["allowlist", "allowlist_patterns"]) +def test_compatibility_facade_rejects_allowlists_with_explicit_entities(option): + with pytest.raises(ValueError, match="cannot be combined with explicit entities"): + v4.redact("plain text", entities=[], **{option: ["plain text"]}) diff --git a/tests/test_contract_481.py b/tests/test_contract_481.py index 427401dd..5aa5b767 100644 --- a/tests/test_contract_481.py +++ b/tests/test_contract_481.py @@ -1,17 +1,56 @@ """Frozen observations from the published 4.8.1 wheel; never regenerate in CI.""" +import copy import json import pytest from tests.contract_481 import FIXTURE, observe -CONTRACT = json.loads(FIXTURE.read_text()) +CONTRACT = json.loads(FIXTURE.read_text(encoding="utf-8")) + + +def expected_49(case): + """Allow only the four added backend parameters and revised shim notices.""" + expected = copy.deepcopy(case["expected"]) + if case["id"] == "public-signatures": + for target in ( + "datafog:scan", + "datafog:redact", + "datafog.engine:scan", + "datafog.engine:scan_and_redact", + ): + expected["value"][target]["parameters"].append( + { + "name": "backend", + "kind": "KEYWORD_ONLY", + "required": False, + "default": "python", + } + ) + if case.get("target") in {"datafog:detect", "datafog:process"}: + name = case["target"].split(":")[1] + replacement = ( + "datafog.scan()" + if name == "detect" + else "datafog.scan() or datafog.redact()" + ) + expected["warnings"] = [ + { + "type": "FutureWarning", + "message": ( + f"datafog.{name}() is deprecated and will be removed in 5.0. " + f"Use {replacement} instead. " + "The earlier promise to retain this shim through 5.x has been revised." + ), + } + ] + return expected @pytest.mark.parametrize("case", CONTRACT["cases"], ids=lambda case: case["id"]) def test_published_481_contract(case): - assert observe(case) == case["expected"], ( + assert observe(case) == expected_49(case), ( f"4.8.1 contract changed: {case['id']} ({case['classification']}). " "Review the migration policy before changing the frozen baseline." ) diff --git a/tests/test_legacy_retirement.py b/tests/test_legacy_retirement.py new file mode 100644 index 00000000..f5dd80e3 --- /dev/null +++ b/tests/test_legacy_retirement.py @@ -0,0 +1,268 @@ +"""Retirement notices do not require installed OCR/Spark/model dependencies.""" + +import asyncio +import importlib +import os +import subprocess +import sys +import warnings +from pathlib import Path +from types import SimpleNamespace +from unittest.mock import AsyncMock, Mock + +import pytest + +from datafog._legacy_retirement import LegacySurfaceWarning +from datafog.processing.image_processing.donut_processor import DonutProcessor +from datafog.processing.spark_processing import pyspark_udfs +from datafog.services.image_service import ImageService +from datafog.services.spark_service import SparkService + + +@pytest.mark.parametrize("surface", ["OCR", "Spark"]) +def test_optional_service_notice_precedes_dependencies(monkeypatch, surface): + def unexpected_import(*args): + pytest.fail("Optional dependency loading must follow the retirement notice") + + monkeypatch.setattr(importlib, "import_module", unexpected_import) + with warnings.catch_warnings(): + warnings.simplefilter("error", LegacySurfaceWarning) + with pytest.raises( + LegacySurfaceWarning, + match=f"{surface} support is deprecated in 4.9 and will be removed in 5.0", + ): + (ImageService if surface == "OCR" else SparkService)() + + +def test_spark_missing_dependency_error_is_preserved(monkeypatch): + def unavailable(name): + raise ModuleNotFoundError(name) + + monkeypatch.setattr(importlib, "import_module", unavailable) + with ( + pytest.warns(LegacySurfaceWarning, match="final 4.x release"), + pytest.raises(ImportError, match=r"datafog\[distributed\]"), + ): + SparkService() + + +@pytest.mark.parametrize("factory", [False, True]) +def test_spark_udf_paths_warn_before_dependency_loading(monkeypatch, factory): + dependencies = Mock(side_effect=ImportError("optional dependency unavailable")) + monkeypatch.setattr(pyspark_udfs, "ensure_installed", dependencies) + with warnings.catch_warnings(): + warnings.simplefilter("error", LegacySurfaceWarning) + with pytest.raises(LegacySurfaceWarning, match="Spark.*removed in 5.0"): + if factory: + pyspark_udfs.broadcast_pii_annotator_udf() + else: + pyspark_udfs.pii_annotator("hello", None) + dependencies.assert_not_called() + + +def test_spark_udf_output_is_preserved(monkeypatch): + monkeypatch.setattr(pyspark_udfs, "ensure_installed", lambda _: None) + nlp = Mock( + return_value=SimpleNamespace(ents=[SimpleNamespace(label_="PER", text="Jane")]) + ) + with pytest.warns(LegacySurfaceWarning): + assert pyspark_udfs.pii_annotator("Jane", SimpleNamespace(value=nlp)) == [ + [], + [], + [], + [], + ["Jane"], + ] + + +def test_donut_mock_output_is_preserved(monkeypatch): + from datafog.processing.image_processing import donut_processor + + monkeypatch.setattr(donut_processor, "IN_TEST_ENV", True) + monkeypatch.setattr(donut_processor, "DONUT_TESTING_ENABLED", False) + with pytest.warns(LegacySurfaceWarning): + processor = DonutProcessor() + with pytest.warns(LegacySurfaceWarning, match="final 4.x release"): + assert asyncio.run(processor.extract_text_from_image(object())) == ( + '{"text": "Mock OCR text for testing"}' + ) + + +def test_direct_tesseract_processor_warns_and_preserves_output(monkeypatch): + # Stub optional dependencies even when they are absent from the test environment. + tesseract = SimpleNamespace(image_to_string=Mock(return_value="OCR text")) + monkeypatch.setitem(sys.modules, "pytesseract", tesseract) + monkeypatch.setitem( + sys.modules, "PIL", SimpleNamespace(Image=SimpleNamespace(Image=object)) + ) + module_name = "datafog.processing.image_processing.pytesseract_processor" + monkeypatch.delitem(sys.modules, module_name, raising=False) + processor_module = importlib.import_module(module_name) + try: + with pytest.warns(LegacySurfaceWarning, match="OCR.*removed in 5.0"): + assert ( + asyncio.run( + processor_module.PytesseractProcessor().extract_text_from_image( + object() + ) + ) + == "OCR text" + ) + finally: + sys.modules.pop(module_name, None) + + +def test_ocr_warning_as_error_is_not_converted_to_processing_result(monkeypatch): + with pytest.warns(LegacySurfaceWarning): + service = ImageService() + # Simulate a warning from a nested public processor after outer warning handling. + monkeypatch.setattr( + "datafog.services.image_service.warn_legacy_surface", lambda _: None + ) + monkeypatch.setitem(sys.modules, "PIL", SimpleNamespace(Image=object)) + service.downloader.download_image = AsyncMock( + side_effect=LegacySurfaceWarning("nested notice") + ) + with pytest.raises(LegacySurfaceWarning, match="nested notice"): + asyncio.run(service.ocr_extract(["https://example.test/image.png"])) + + +def test_ocr_pipeline_warns_before_loading_service(monkeypatch): + from datafog.main import DataFog + + datafog = DataFog() + service = Mock() + monkeypatch.setattr("datafog.services.image_service.ImageService", service) + with warnings.catch_warnings(): + warnings.simplefilter("error", LegacySurfaceWarning) + with pytest.raises(LegacySurfaceWarning): + asyncio.run(datafog.run_ocr_pipeline([])) + service.assert_not_called() + + +def test_core_and_optional_module_imports_are_quiet_and_lightweight(): + script = """ +import importlib.abc +import sys +import warnings + +blocked = {"PIL", "pytesseract", "pyspark", "torch", "transformers", "spacy", "aiohttp", "numpy"} +class BlockOptional(importlib.abc.MetaPathFinder): + def find_spec(self, fullname, *args): + if fullname.split(".")[0] in blocked: + raise AssertionError("Unexpected optional import: " + fullname) +sys.meta_path.insert(0, BlockOptional()) +with warnings.catch_warnings(record=True) as notices: + warnings.simplefilter("always") + import datafog + from datafog.services.image_service import ImageService + from datafog.services.spark_service import SparkService + from datafog.processing.image_processing.donut_processor import DonutProcessor + from datafog.processing.spark_processing import pyspark_udfs + result = datafog.scan("jane@example.com", engine="regex") + assert result.entities + from datafog._legacy_retirement import LegacySurfaceWarning + assert not [n for n in notices if issubclass(n.category, LegacySurfaceWarning)] +assert not (blocked & set(sys.modules)) +""" + env = dict( + os.environ, + PYTHONPATH=str(Path.cwd()), + DATAFOG_NO_TELEMETRY="1", + DO_NOT_TRACK="1", + ) + subprocess.run( + [sys.executable, "-c", script], + env=env, + check=True, + capture_output=True, + text=True, + ) + + +@pytest.mark.parametrize( + "module_name", + [ + "datafog.services.image_service", + "datafog.processing.image_processing.image_downloader", + ], +) +def test_direct_image_download_warns_before_optional_import(module_name): + module = importlib.import_module(module_name) + with warnings.catch_warnings(): + warnings.simplefilter("error", LegacySurfaceWarning) + with pytest.raises(LegacySurfaceWarning, match="OCR.*removed in 5.0"): + asyncio.run( + module.ImageDownloader().download_image("https://example.test/a.png") + ) + + +def test_image_cli_warns_before_running_pipeline(monkeypatch): + pytest.importorskip("typer") + from datafog import client + + datafog = Mock() + monkeypatch.setattr(client, "DataFog", datafog) + with warnings.catch_warnings(): + warnings.simplefilter("error", LegacySurfaceWarning) + with pytest.raises(LegacySurfaceWarning, match="OCR.*removed in 5.0"): + client.scan_image(["image.png"], "scan") + datafog.assert_not_called() + + +def test_image_cli_does_not_swallow_nested_retirement_warning(monkeypatch): + pytest.importorskip("typer") + from datafog import client + + monkeypatch.setattr(client, "warn_legacy_surface", lambda _: None) + pipeline = AsyncMock(side_effect=LegacySurfaceWarning("nested notice")) + monkeypatch.setattr( + client, "DataFog", Mock(return_value=SimpleNamespace(run_ocr_pipeline=pipeline)) + ) + with pytest.raises(LegacySurfaceWarning, match="nested notice"): + client.scan_image(["image.png"], "scan") + + +@pytest.mark.parametrize("image_argument", [False, True]) +def test_image_cli_notice_visible_under_default_python_filters(image_argument): + pytest.importorskip("typer") + # A fresh interpreter avoids pytest's warning filters and cached warning sites. + # Do not enable warnings here: the normal interpreter policy must show this. + script = """ +import sys +from types import SimpleNamespace +from unittest.mock import AsyncMock, Mock +from typer.testing import CliRunner +from datafog import client + +pipeline = AsyncMock(return_value=["OCR text"]) +client.DataFog = Mock(return_value=SimpleNamespace(run_ocr_pipeline=pipeline)) +args = ["scan-image"] +if sys.argv[1] == "True": + args.append("image.png") +result = CliRunner().invoke(client.app, args) +assert "deprecated in 4.9 and will be removed in 5.0" in result.stderr, result.output +assert "final 4.x release" in result.stderr, result.output +if sys.argv[1] == "True": + assert result.exit_code == 0, result.output + assert "OCR Pipeline Results: ['OCR text']" in result.stdout + pipeline.assert_awaited_once_with(image_urls=["image.png"]) +else: + assert result.exit_code == 1, result.output + assert "No image URLs or file paths provided" in result.stdout + pipeline.assert_not_awaited() +""" + env = dict( + os.environ, + PYTHONPATH=str(Path.cwd()), + DATAFOG_NO_TELEMETRY="1", + DO_NOT_TRACK="1", + ) + env.pop("PYTHONWARNINGS", None) + subprocess.run( + [sys.executable, "-c", script, str(image_argument)], + env=env, + check=True, + capture_output=True, + text=True, + ) diff --git a/tests/test_rust_backend.py b/tests/test_rust_backend.py new file mode 100644 index 00000000..97644322 --- /dev/null +++ b/tests/test_rust_backend.py @@ -0,0 +1,204 @@ +"""Behavioral coverage for the opt-in native detection adapter.""" + +import builtins +import sys +from types import SimpleNamespace + +import pytest + +from datafog import engine + + +def finding(label, text, start, end): + return SimpleNamespace( + entity_type=label, + matched_text=text, + codepoint_range=SimpleNamespace(start=start, end=end), + byte_range=SimpleNamespace(start=999, end=1000), + ) + + +@pytest.fixture +def native(monkeypatch): + calls = [] + findings = [] + + def scan(text): + calls.append(text) + return findings + + monkeypatch.setitem(sys.modules, "datafog_core", SimpleNamespace(scan=scan)) + return findings, calls + + +def test_native_offsets_duplicates_order_and_provenance(native): + findings, calls = native + text = "πŸ˜€ alice@example.com / alice@example.com" + findings.extend( + [ + finding("EMAIL", "alice@example.com", 22, 39), + finding("EMAIL", "alice@example.com", 2, 19), + ] + ) + result = engine.scan(text, engine="regex", backend="rust") + assert calls == [text] + assert result.text == text + assert result.engine_used == "regex" + assert [item.start for item in result.entities] == [2, 22] + assert all(text[item.start : item.end] == item.text for item in result.entities) + assert all( + item.engine == "regex" and item.confidence == 1.0 for item in result.entities + ) + + +@pytest.mark.parametrize("selection", [None, [], [" email_address "]]) +def test_selection_alias_and_empty_selection(native, selection): + findings, _ = native + findings.append(finding("EMAIL", "a@example.com", 0, 13)) + result = engine.scan( + "a@example.com", "regex", entity_types=selection, backend="rust" + ) + assert len(result.entities) == 1 + + +def test_unknown_selection_preserves_legacy_empty_result(native): + findings, _ = native + findings.append(finding("EMAIL", "a@example.com", 0, 13)) + assert ( + engine.scan( + "a@example.com", "regex", entity_types=["UNKNOWN"], backend="rust" + ).entities + == [] + ) + + +def test_unknown_native_label_raises_instead_of_silently_dropping_pii(native): + findings, _ = native + findings.append(finding("FUTURE_LABEL", "example", 0, 7)) + with pytest.raises(RuntimeError, match="unsupported entity type.*FUTURE_LABEL"): + engine.scan("example", "regex", backend="rust") + + +def test_python_overlap_priority_precedes_selection(native): + findings, _ = native + text = "1234567890123456" + findings.extend( + [finding("PHONE", text[:10], 0, 10), finding("CREDIT_CARD", text, 0, 16)] + ) + result = engine.scan(text, "regex", backend="rust") + assert [item.type for item in result.entities] == ["CREDIT_CARD"] + assert ( + engine.scan(text, "regex", entity_types=["PHONE"], backend="rust").entities + == [] + ) + + +@pytest.mark.parametrize( + "kwargs, count", + [ + ({"allowlist": ["a@example.com"]}, 0), + ({"allowlist": ["A@example.com"]}, 1), + ({"allowlist_patterns": [r"(?=a@).*\.com"]}, 0), + ({"allowlist_patterns": ["example"]}, 1), + ], +) +def test_python_allowlist_semantics(native, kwargs, count): + findings, _ = native + findings.append(finding("EMAIL", "a@example.com", 0, 13)) + assert ( + len(engine.scan("a@example.com", "regex", backend="rust", **kwargs).entities) + == count + ) + + +@pytest.mark.parametrize("pattern", ["[", "(a+)+", "x" * 513]) +def test_allowlist_validation_before_native(native, pattern): + _, calls = native + with pytest.raises(ValueError): + engine.scan("", "regex", backend="rust", allowlist_patterns=[pattern]) + assert not calls + + +@pytest.mark.parametrize("locale", [["de"], [" DE-DE "], "de_de"]) +def test_german_locales_rejected(native, locale): + _, calls = native + with pytest.raises(ValueError, match="German"): + engine.scan("", "regex", locales=locale, backend="rust") + assert not calls + + +@pytest.mark.parametrize("label", engine.RegexAnnotator.GERMAN_LABELS + ["DE_FUTURE"]) +def test_german_selection_rejected(native, label): + with pytest.raises(ValueError, match="German"): + engine.scan("", "regex", entity_types=[label.lower()], backend="rust") + + +def test_unknown_locale_validation_preserved(native): + with pytest.raises(ValueError, match="locale must be one of"): + engine.scan("", "regex", locales=["fr"], backend="rust") + + +@pytest.mark.parametrize("name", ["smart", "spacy", "gliner"]) +def test_nonregex_engine_rejected(native, name): + _, calls = native + with pytest.raises(ValueError, match="only engine='regex'"): + engine.scan("", name, backend="rust") + assert not calls + + +def test_bad_backend_rejected(): + with pytest.raises(ValueError, match="backend must be one of"): + engine.scan("", "regex", backend="auto") + + +def test_no_native_import_on_python_path_and_actionable_missing_error(monkeypatch): + original_import = builtins.__import__ + + def guarded_import(name, *args, **kwargs): + if name == "datafog_core": + raise ModuleNotFoundError("native absent", name=name) + return original_import(name, *args, **kwargs) + + monkeypatch.setattr(builtins, "__import__", guarded_import) + assert engine.scan("a@example.com", "regex").entities + with pytest.raises(ImportError, match=r'pip install "datafog\[rust\]"'): + engine.scan("a@example.com", "regex", backend="rust") + + +def test_native_failure_propagates_without_fallback(monkeypatch): + error = RuntimeError("native failure") + + def fail(text): + raise error + + monkeypatch.setitem(sys.modules, "datafog_core", SimpleNamespace(scan=fail)) + with pytest.raises(RuntimeError) as caught: + engine.scan("a@example.com", "regex", backend="rust") + assert caught.value is error + + +@pytest.mark.parametrize("strategy", ["token", "mask", "hash", "pseudonymize"]) +def test_all_legacy_transformations_use_native_findings(native, strategy): + findings, calls = native + text = "a@example.com a@example.com" + findings.extend( + [ + finding("EMAIL", "a@example.com", 0, 13), + finding("EMAIL", "a@example.com", 14, 27), + ] + ) + expected = engine.scan_and_redact(text, "regex", strategy=strategy) + actual = engine.scan_and_redact(text, "regex", strategy=strategy, backend="rust") + assert actual == expected + assert calls == [text] + + +@pytest.mark.parametrize("strategy", ["token", "mask", "hash", "pseudonymize"]) +def test_real_native_unicode_and_transformations(strategy): + pytest.importorskip("datafog_core") + text = "πŸ˜€ MΓΌnchen a@example.com and a@example.com" + python_result = engine.scan_and_redact(text, "regex", strategy=strategy) + native_result = engine.scan_and_redact( + text, "regex", strategy=strategy, backend="rust" + ) + assert native_result == python_result diff --git a/tests/test_rust_contract.py b/tests/test_rust_contract.py new file mode 100644 index 00000000..52ee52e1 --- /dev/null +++ b/tests/test_rust_contract.py @@ -0,0 +1,29 @@ +"""Exact native outcomes, with individually reviewed differences from Python.""" + +import importlib.metadata +import json + +import pytest + +from tests.contract_481 import FIXTURE +from tests.rust_contract import DEVIATIONS, applicable, native_observation + +pytest.importorskip("datafog_core") +CONTRACT = json.loads(FIXTURE.read_text(encoding="utf-8")) +REVIEWED = json.loads(DEVIATIONS.read_text(encoding="utf-8")) +CASES = [case for case in CONTRACT["cases"] if applicable(case)] + + +def test_native_version_and_review_inventory(): + assert importlib.metadata.version("datafog-core") == REVIEWED["core_version"] + assert set(REVIEWED["cases"]) <= {case["id"] for case in CASES} + for item in REVIEWED["cases"].values(): + assert item["classification"] in {"unsupported", "detector-difference"} + assert item["reason"] + + +@pytest.mark.parametrize("case", CASES, ids=lambda case: case["id"]) +def test_native_contract(case): + deviation = REVIEWED["cases"].get(case["id"]) + expected = deviation["expected"] if deviation else case["expected"] + assert native_observation(case) == expected, case["id"]