From c7b498ce03fcb1e4f3c9e55c04997b2b9e704846 Mon Sep 17 00:00:00 2001 From: GitHub Action Date: Tue, 28 Jul 2026 05:21:42 +0000 Subject: [PATCH 01/37] chore: bump version to 4.8.0a4 [skip ci] --- datafog/__about__.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/datafog/__about__.py b/datafog/__about__.py index 14d37982..f6e9c40b 100644 --- a/datafog/__about__.py +++ b/datafog/__about__.py @@ -1 +1 @@ -__version__ = "4.8.0b3" +__version__ = "4.8.0a4" From f13ac8bd43e126466aa867453cabdbcf9af05985 Mon Sep 17 00:00:00 2001 From: GitHub Action Date: Thu, 30 Jul 2026 05:03:39 +0000 Subject: [PATCH 02/37] chore: bump version to 4.8.0b4 [skip ci] --- datafog/__about__.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/datafog/__about__.py b/datafog/__about__.py index f6e9c40b..bea6e0a4 100644 --- a/datafog/__about__.py +++ b/datafog/__about__.py @@ -1 +1 @@ -__version__ = "4.8.0a4" +__version__ = "4.8.0b4" From de3d8eba32e969209030f16e1d308b7f36a2c14f Mon Sep 17 00:00:00 2001 From: GitHub Action Date: Mon, 3 Aug 2026 05:49:22 +0000 Subject: [PATCH 03/37] chore: bump version to 4.8.0a5 [skip ci] --- datafog/__about__.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/datafog/__about__.py b/datafog/__about__.py index bea6e0a4..cd0f3be3 100644 --- a/datafog/__about__.py +++ b/datafog/__about__.py @@ -1 +1 @@ -__version__ = "4.8.0b4" +__version__ = "4.8.0a5" From 73846cbfb447a89c50cea5bb38a7be543b803684 Mon Sep 17 00:00:00 2001 From: Sid Mohan <61345237+sidmohan0@users.noreply.github.com> Date: Wed, 5 Aug 2026 10:06:13 -0700 Subject: [PATCH 04/37] docs: explain first-time CLA signing flow --- CONTRIBUTING.md | 22 +++++++++++++++++++--- 1 file changed, 19 insertions(+), 3 deletions(-) diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 6e7e416e..059f7164 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -105,6 +105,8 @@ Before requesting review: - Keep public API changes explicit in the PR description. - Note any optional dependency profile you tested, such as `core`, `nlp`, or `nlp-advanced`. +- Complete the CLA Assistant check. First-time contributors receive a comment + on the pull request with a link to review and accept the agreement. ## Commit Messages @@ -117,9 +119,23 @@ but not required, for example: ## Legal -By submitting a pull request, you license your contribution under the project -[license](LICENSE). You also affirm that you authored the contribution or have -the right to submit it under the project license. +DataFog requires each contributor to accept the +[Individual Contributor License Agreement](CLA.md). CLA Assistant handles the +signature directly from the pull request: + +1. Open your pull request against `dev`. +2. Follow the link in the CLA Assistant comment and sign in with the same + GitHub account that authored the commits. +3. Review the agreement, complete the short signing form, and accept it. The + `cla-assistant` check updates automatically; no document upload is needed. + +You normally sign once. A new signature is requested only when DataFog updates +the agreement. If a commit has multiple authors, every human co-author must +sign. Bot accounts are handled separately by maintainers. + +If your employer or another organization may own your work, confirm that you +are allowed to contribute before signing. Questions can be sent to +`sid@datafog.ai`. ## Contributors From 191747684156c044f7abf838bdc35618c4f982b4 Mon Sep 17 00:00:00 2001 From: Sid Mohan <61345237+sidmohan0@users.noreply.github.com> Date: Wed, 5 Aug 2026 10:06:17 -0700 Subject: [PATCH 05/37] docs: add DataFog contributor license agreement --- CLA.md | 105 +++++++++++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 105 insertions(+) create mode 100644 CLA.md diff --git a/CLA.md b/CLA.md new file mode 100644 index 00000000..1c2aaed0 --- /dev/null +++ b/CLA.md @@ -0,0 +1,105 @@ +# DataFog Individual Contributor License Agreement + +Version 1.0 + +Thank you for your interest in contributing to software and documentation +managed by DataFog, Inc. ("DataFog"). This Contributor License Agreement +("Agreement") clarifies the intellectual-property rights granted with your +contributions. It protects you, DataFog, and the people who use DataFog +projects. It does not prevent you from using your own contributions for any +other purpose. + +By electronically accepting this Agreement through DataFog's CLA Assistant, +you agree to the following terms for all past, present, and future +Contributions that you submit to DataFog. + +## 1. Definitions + +"You" means the individual accepting this Agreement. If you submit a +Contribution on behalf of a legal entity, "You" also includes that entity to +the extent you are authorized to bind it. + +"Contribution" means any original work of authorship, including any change or +addition to an existing work, that you intentionally submit to DataFog for +inclusion in, or documentation of, a project owned or managed by DataFog. + +"Submit" means any electronic, verbal, or written communication sent to +DataFog or its representatives through a source-code control system, issue +tracker, code-review system, mailing list, or another communication channel +used to discuss or improve a DataFog project. A communication that you +conspicuously mark in writing as "Not a Contribution" is excluded. + +"Work" means the DataFog project to which the Contribution is submitted and +any derivative or collective work based on that project. + +## 2. Copyright license + +You grant DataFog and recipients of software distributed by DataFog a +perpetual, worldwide, non-exclusive, no-charge, royalty-free, irrevocable +copyright license to reproduce, prepare derivative works of, publicly display, +publicly perform, sublicense, and distribute your Contributions and derivative +works of those Contributions. + +## 3. Patent license + +You grant DataFog and recipients of software distributed by DataFog a +perpetual, worldwide, non-exclusive, no-charge, royalty-free, irrevocable +(except as stated in this section) patent license to make, have made, use, +offer to sell, sell, import, and otherwise transfer the Work, where the license +applies only to patent claims licensable by you that are necessarily infringed +by your Contribution alone or by combination of your Contribution with the +Work to which it was submitted. + +If you or an entity acting on your behalf files patent litigation against +DataFog or another entity alleging that a Contribution incorporated in the +Work constitutes direct or contributory patent infringement, the patent +licenses granted to you under this Agreement for that Work terminate as of the +date the litigation is filed. + +## 4. Your representations + +You represent that: + +- You are legally entitled to grant the licenses in this Agreement. +- Each Contribution is your original creation, except for material that you + identify in writing together with its source and applicable license. +- If your employer or another party may own rights in a Contribution, you have + received permission to contribute it, that party has waived those rights for + the Contribution, or that party has separately granted the necessary rights + to DataFog. +- You will notify DataFog promptly at `sid@datafog.ai` if you become aware + that any representation in this Agreement is inaccurate. + +## 5. No support obligation + +You are not expected to provide support for your Contributions unless you and +DataFog agree otherwise in writing. Unless required by applicable law or +agreed in writing, you provide each Contribution "as is," without warranties +or conditions of any kind, express or implied, including warranties of title, +non-infringement, merchantability, or fitness for a particular purpose. + +## 6. Personal information + +DataFog may retain your GitHub identity, acceptance timestamp, Agreement +version, and the information you provide in the signing form as evidence of +this Agreement. DataFog will use that information to administer contributions +and establish the provenance of project intellectual property. + +## 7. General + +This Agreement is the entire agreement between you and DataFog concerning its +subject matter and replaces prior understandings about the rights granted with +your Contributions. If any provision is unenforceable, it will be modified +only to the minimum extent necessary, and the remaining provisions will +continue in effect. A failure to enforce a provision is not a waiver of the +right to enforce it later. + +Electronic acceptance through the GitHub account authenticated by DataFog's +CLA Assistant has the same effect as your signature. + +--- + +This Agreement is based in part on the Apache Software Foundation Individual +Contributor License Agreement and the Harmony Individual Contributor License +Agreement. The template adaptations are provided under their respective +licenses. From 4f03c544d91feda191d00aecaba42b954ea5aa30 Mon Sep 17 00:00:00 2001 From: Sid Mohan <61345237+sidmohan0@users.noreply.github.com> Date: Wed, 5 Aug 2026 10:06:18 -0700 Subject: [PATCH 06/37] docs: add contributor agreement checklist --- .github/PULL_REQUEST_TEMPLATE.md | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/.github/PULL_REQUEST_TEMPLATE.md b/.github/PULL_REQUEST_TEMPLATE.md index c3a6b2d3..35da4cbb 100644 --- a/.github/PULL_REQUEST_TEMPLATE.md +++ b/.github/PULL_REQUEST_TEMPLATE.md @@ -32,6 +32,12 @@ Optional profiles tested: - [ ] ocr - [ ] distributed +## Contributor Agreement + +- [ ] I have reviewed the [DataFog CLA](../CLA.md) and will complete the + CLA Assistant check on this pull request. If my employer or another party + owns rights in this work, I have obtained permission to contribute it. + ## Notes For Reviewers Mention API changes, migrations, warnings, or release-note needs. From f48c12b9a9d373e548b3971d38b32298ad3746c6 Mon Sep 17 00:00:00 2001 From: Sid Mohan <61345237+sidmohan0@users.noreply.github.com> Date: Wed, 5 Aug 2026 10:06:19 -0700 Subject: [PATCH 07/37] docs: add CLA Assistant operations runbook --- docs/cla-assistant-operations.rst | 106 ++++++++++++++++++++++++++++++ 1 file changed, 106 insertions(+) create mode 100644 docs/cla-assistant-operations.rst diff --git a/docs/cla-assistant-operations.rst b/docs/cla-assistant-operations.rst new file mode 100644 index 00000000..823de629 --- /dev/null +++ b/docs/cla-assistant-operations.rst @@ -0,0 +1,106 @@ +======================== +CLA Assistant Operations +======================== + +This runbook records the intended production configuration for DataFog's +self-hosted CLA Assistant. Do not place credentials, private keys, database +URIs, or exported signature records in this repository. + +Production Identity +=================== + +* Public URL: ``https://cla.datafog.ai`` +* GitHub organization: ``DataFog`` +* Initial protected repository: ``DataFog/datafog-python`` +* Agreement: `CLA.md <../CLA.md>`_, version 1.0 +* Administrator allowlist: ``sidmohan0`` + +The service runs separately from the static ``datafog.ai`` landing page. The +``cla.datafog.ai`` DNS record points at the production CLA service without +changing the apex or ``www`` landing-page records. + +GitHub Configuration +==================== + +The production instance uses both a GitHub OAuth App for interactive login and +a GitHub App for repository installation and webhooks. + +OAuth App: + +* Homepage URL: ``https://cla.datafog.ai`` +* Authorization callback URL: + ``https://cla.datafog.ai/auth/github/callback`` + +GitHub App: + +* Homepage URL: ``https://cla.datafog.ai`` +* Callback URL: ``https://cla.datafog.ai/auth/github/app-callback`` +* Webhook URL: ``https://cla.datafog.ai/github/webhooks`` +* Install only on the repositories that require the CLA check. +* Grant only the repository permissions required by the upstream CLA Assistant + release. Re-check the upstream documentation before expanding permissions. + +The app secrets and private key belong in the hosting provider's secret store. +Rotate a credential immediately if it appears in logs, shell history, a pull +request, or a repository file. + +Agreement Linkage +================= + +The CLA Assistant record must point to an immutable, reviewable copy of +``CLA.md``. The public repository copy is the canonical text. If the +application requires a GitHub Gist, keep the Gist byte-for-byte identical to +``CLA.md`` and record the source commit in the Gist description. + +The signing form should request only: + +* Full legal name (required) +* Email address (required, prefilled from GitHub when available) +* Signing capacity: individual or on behalf of an organization (required) +* Organization name (only when signing for an organization) +* Confirmation that the signer has authority to submit the contribution + (required) + +Avoid collecting postal addresses, phone numbers, or other information that is +not needed to establish the agreement. + +Merge Protection +================ + +After one test pull request completes the signing flow: + +#. Add the CLA Assistant status to the ``dev`` branch's required checks. +#. Add the same check to ``main`` if pull requests can target ``main`` + directly. +#. Confirm that unsigned, signed, multi-author, and approved bot pull requests + produce the expected result. +#. Do not merge by bypassing the check except during a documented service + incident. + +Dependabot and other approved bots cannot sign. Add bot identities through the +CLA Assistant administration UI; never import a human contributor as signed +without evidence of acceptance. + +Updating The Agreement +====================== + +Treat any text change as a new agreement version: + +#. Obtain legal review of the proposed change. +#. Merge the reviewed ``CLA.md`` update. +#. Update the CLA Assistant source to the exact merged text. +#. Verify that the application requests a new signature when required. +#. Export and securely archive the prior version's signature list before the + new version becomes active. + +Backup And Incident Response +============================ + +* Back up the database on a schedule appropriate for legal records and test a + restore at least quarterly. +* Export the signature list after every agreement-version change and before a + hosting migration. +* Monitor the public health endpoint, GitHub webhook deliveries, database + availability, and TLS certificate expiration. +* If the service is unavailable, keep the required check enabled and pause + merges rather than silently accepting unsigned contributions. From 426496bed2d50c2f3a9723efbd661eba84f0a680 Mon Sep 17 00:00:00 2001 From: Sid Mohan <61345237+sidmohan0@users.noreply.github.com> Date: Wed, 5 Aug 2026 10:06:20 -0700 Subject: [PATCH 08/37] docs: index CLA Assistant operations runbook --- docs/index.rst | 1 + 1 file changed, 1 insertion(+) diff --git a/docs/index.rst b/docs/index.rst index d5cfdc66..57c41a3e 100644 --- a/docs/index.rst +++ b/docs/index.rst @@ -41,6 +41,7 @@ Contributing :caption: Contributing contributing + cla-assistant-operations v45-release-readiness live-module-map From 7587914aa8a43b1cf6b7bf157ddcd248050ece47 Mon Sep 17 00:00:00 2001 From: Sid Mohan <61345237+sidmohan0@users.noreply.github.com> Date: Wed, 5 Aug 2026 10:20:32 -0700 Subject: [PATCH 09/37] docs: record managed CLA Assistant configuration --- docs/cla-assistant-operations.rst | 144 ++++++++++++++++-------------- 1 file changed, 75 insertions(+), 69 deletions(-) diff --git a/docs/cla-assistant-operations.rst b/docs/cla-assistant-operations.rst index 823de629..32de7586 100644 --- a/docs/cla-assistant-operations.rst +++ b/docs/cla-assistant-operations.rst @@ -2,105 +2,111 @@ CLA Assistant Operations ======================== -This runbook records the intended production configuration for DataFog's -self-hosted CLA Assistant. Do not place credentials, private keys, database -URIs, or exported signature records in this repository. +This runbook records DataFog's production CLA Assistant configuration. Do not +place exported signature records or other contributor personal information in +this repository. -Production Identity -=================== +Production Configuration +======================== -* Public URL: ``https://cla.datafog.ai`` +* Branded entry point: ``https://cla.datafog.ai`` +* Hosted service: ``https://cla-assistant.io`` * GitHub organization: ``DataFog`` -* Initial protected repository: ``DataFog/datafog-python`` +* Protected repository: ``DataFog/datafog-python`` * Agreement: `CLA.md <../CLA.md>`_, version 1.0 -* Administrator allowlist: ``sidmohan0`` - -The service runs separately from the static ``datafog.ai`` landing page. The -``cla.datafog.ai`` DNS record points at the production CLA service without -changing the apex or ``www`` landing-page records. - -GitHub Configuration -==================== - -The production instance uses both a GitHub OAuth App for interactive login and -a GitHub App for repository installation and webhooks. - -OAuth App: +* Agreement Gist: + ``https://gist.github.com/sidmohan0/c7f98b0c28a9d827c0a1a570102e0b89`` +* Required GitHub status context: ``license/cla`` -* Homepage URL: ``https://cla.datafog.ai`` -* Authorization callback URL: - ``https://cla.datafog.ai/auth/github/callback`` +SAP's managed CLA Assistant deployment handles GitHub authentication, webhook +processing, and signature storage. DataFog does not operate a separate CLA +Assistant application or database. -GitHub App: +The branded entry point is a redirect-only Vercel project named +``datafog-cla-redirect``. The ``cla.datafog.ai`` DNS record is a DNS-only CNAME +to ``e2c831ea94af969f.vercel-dns-016.com``. It must redirect every path to +``https://cla-assistant.io/DataFog/datafog-python`` without changing the apex +or ``www`` landing-page records. -* Homepage URL: ``https://cla.datafog.ai`` -* Callback URL: ``https://cla.datafog.ai/auth/github/app-callback`` -* Webhook URL: ``https://cla.datafog.ai/github/webhooks`` -* Install only on the repositories that require the CLA check. -* Grant only the repository permissions required by the upstream CLA Assistant - release. Re-check the upstream documentation before expanding permissions. +Agreement And Signing Form +========================== -The app secrets and private key belong in the hosting provider's secret store. -Rotate a credential immediately if it appears in logs, shell history, a pull -request, or a repository file. +The repository ``CLA.md`` file is the canonical agreement. CLA Assistant reads +an unlisted GitHub Gist containing two files: -Agreement Linkage -================= +* ``DataFog-CLA.md`` must remain byte-for-byte identical to ``CLA.md``. +* ``metadata`` defines the contributor form fields. -The CLA Assistant record must point to an immutable, reviewable copy of -``CLA.md``. The public repository copy is the canonical text. If the -application requires a GitHub Gist, keep the Gist byte-for-byte identical to -``CLA.md`` and record the source commit in the Gist description. +The signing form requests only: -The signing form should request only: - -* Full legal name (required) -* Email address (required, prefilled from GitHub when available) +* Full legal name (required and prefilled from GitHub when available) +* Email address (required and prefilled from GitHub when available) * Signing capacity: individual or on behalf of an organization (required) -* Organization name (only when signing for an organization) +* Organization name (optional; used when signing for an organization) * Confirmation that the signer has authority to submit the contribution (required) Avoid collecting postal addresses, phone numbers, or other information that is -not needed to establish the agreement. +not needed to establish the agreement. Because the Gist is unlisted rather +than private access-controlled storage, do not put secrets or signature data in +it. + +Contributor Flow +================ + +#. A contributor opens a pull request against ``dev``. +#. CLA Assistant comments with the signing link and sets ``license/cla`` to + pending. +#. The contributor signs in with the GitHub account associated with the pull + request, reviews the agreement, completes the form, and selects ``I agree``. +#. CLA Assistant records the acceptance and changes ``license/cla`` to success. +#. If a pull request has multiple human authors, every author must sign. + +Dependabot and other approved bots cannot sign. Add bot identities through the +CLA Assistant administration UI; never import a human contributor as signed +without evidence of acceptance. Merge Protection ================ After one test pull request completes the signing flow: -#. Add the CLA Assistant status to the ``dev`` branch's required checks. +#. Add ``license/cla`` to the required checks for the ``dev`` branch. #. Add the same check to ``main`` if pull requests can target ``main`` directly. -#. Confirm that unsigned, signed, multi-author, and approved bot pull requests +#. Confirm that unsigned, signed, multi-author, and approved-bot pull requests produce the expected result. -#. Do not merge by bypassing the check except during a documented service - incident. - -Dependabot and other approved bots cannot sign. Add bot identities through the -CLA Assistant administration UI; never import a human contributor as signed -without evidence of acceptance. +#. Do not bypass the check except during a documented service incident. Updating The Agreement ====================== Treat any text change as a new agreement version: +#. Export and securely archive the current signature list from the CLA + Assistant dashboard. #. Obtain legal review of the proposed change. #. Merge the reviewed ``CLA.md`` update. -#. Update the CLA Assistant source to the exact merged text. -#. Verify that the application requests a new signature when required. -#. Export and securely archive the prior version's signature list before the - new version becomes active. - -Backup And Incident Response -============================ - -* Back up the database on a schedule appropriate for legal records and test a - restore at least quarterly. -* Export the signature list after every agreement-version change and before a - hosting migration. -* Monitor the public health endpoint, GitHub webhook deliveries, database - availability, and TLS certificate expiration. -* If the service is unavailable, keep the required check enabled and pause - merges rather than silently accepting unsigned contributions. +#. Replace only the ``DataFog-CLA.md`` Gist file with the exact merged text; + retain the ``metadata`` file unless the form is intentionally changing. +#. Verify that CLA Assistant displays the new version and requests a new + signature when required. + +Operations And Incident Response +================================ + +* Restrict CLA Assistant administration to DataFog maintainers who need it. +* Export signature records after an agreement-version change and before any + service migration; store exports in DataFog's access-controlled records + system, not GitHub. +* Monitor the hosted service, the ``license/cla`` check, the branded redirect, + and TLS certificate health. +* If the hosted service is unavailable, keep the required check enabled and + pause merges rather than silently accepting unsigned contributions. +* Review the managed service's terms and privacy policy when DataFog changes + the personal information collected in the signing form. + +Self-hosting is not the current production architecture. Re-evaluate it only +if DataFog needs controls the managed service cannot provide and after the +upstream runtime, dependencies, database, backups, security updates, and +on-call ownership have been reviewed. From 74836c1d38f49e5b3df940bd937347d0b6a8f966 Mon Sep 17 00:00:00 2001 From: GitHub Action Date: Thu, 6 Aug 2026 05:21:34 +0000 Subject: [PATCH 10/37] chore: bump version to 4.8.0b5 [skip ci] --- datafog/__about__.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/datafog/__about__.py b/datafog/__about__.py index cd0f3be3..0c0c0de0 100644 --- a/datafog/__about__.py +++ b/datafog/__about__.py @@ -1 +1 @@ -__version__ = "4.8.0a5" +__version__ = "4.8.0b5" From 61eea19260628ea54b2df1227df61e7bf099c718 Mon Sep 17 00:00:00 2001 From: GitHub Action Date: Mon, 10 Aug 2026 04:02:30 +0000 Subject: [PATCH 11/37] chore: bump version to 4.8.0a6 [skip ci] --- datafog/__about__.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/datafog/__about__.py b/datafog/__about__.py index 0c0c0de0..736d6993 100644 --- a/datafog/__about__.py +++ b/datafog/__about__.py @@ -1 +1 @@ -__version__ = "4.8.0b5" +__version__ = "4.8.0a6" From 0b61d645ff8a7f35d8a91ad0a977b070198159ec Mon Sep 17 00:00:00 2001 From: GitHub Action Date: Thu, 13 Aug 2026 04:15:29 +0000 Subject: [PATCH 12/37] chore: bump version to 4.8.0b6 [skip ci] --- datafog/__about__.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/datafog/__about__.py b/datafog/__about__.py index 736d6993..019b2304 100644 --- a/datafog/__about__.py +++ b/datafog/__about__.py @@ -1 +1 @@ -__version__ = "4.8.0a6" +__version__ = "4.8.0b6" From 1dc8cc57ca82bd45ad2e60ac9529fd922937f25a Mon Sep 17 00:00:00 2001 From: GitHub Action Date: Mon, 17 Aug 2026 03:06:35 +0000 Subject: [PATCH 13/37] chore: bump version to 4.8.0a7 [skip ci] --- datafog/__about__.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/datafog/__about__.py b/datafog/__about__.py index 019b2304..4ebdc351 100644 --- a/datafog/__about__.py +++ b/datafog/__about__.py @@ -1 +1 @@ -__version__ = "4.8.0b6" +__version__ = "4.8.0a7" From a8bcc2c984585a4a58966358dc4f4bfbe010ebdb Mon Sep 17 00:00:00 2001 From: GitHub Action Date: Thu, 20 Aug 2026 03:00:33 +0000 Subject: [PATCH 14/37] chore: bump version to 4.8.0b7 [skip ci] --- datafog/__about__.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/datafog/__about__.py b/datafog/__about__.py index 4ebdc351..23ae8c6e 100644 --- a/datafog/__about__.py +++ b/datafog/__about__.py @@ -1 +1 @@ -__version__ = "4.8.0a7" +__version__ = "4.8.0b7" From d411a386a3c6ae31816a4e4194ff95c011f8521e Mon Sep 17 00:00:00 2001 From: GitHub Action Date: Mon, 24 Aug 2026 03:10:11 +0000 Subject: [PATCH 15/37] chore: bump version to 4.8.0a8 [skip ci] --- datafog/__about__.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/datafog/__about__.py b/datafog/__about__.py index 23ae8c6e..7aad9231 100644 --- a/datafog/__about__.py +++ b/datafog/__about__.py @@ -1 +1 @@ -__version__ = "4.8.0b7" +__version__ = "4.8.0a8" From 65c4865d31554a48e959e7176709378fc717fc67 Mon Sep 17 00:00:00 2001 From: GitHub Action Date: Thu, 27 Aug 2026 12:27:04 +0000 Subject: [PATCH 16/37] chore: bump version to 4.8.0b8 [skip ci] --- datafog/__about__.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/datafog/__about__.py b/datafog/__about__.py index 7aad9231..11dfaef6 100644 --- a/datafog/__about__.py +++ b/datafog/__about__.py @@ -1 +1 @@ -__version__ = "4.8.0a8" +__version__ = "4.8.0b8" From e6e5667a5d23afdec5a804656f44b6d7279a4f71 Mon Sep 17 00:00:00 2001 From: GitHub Action Date: Mon, 31 Aug 2026 08:19:30 +0000 Subject: [PATCH 17/37] chore: bump version to 4.8.0a9 [skip ci] --- datafog/__about__.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/datafog/__about__.py b/datafog/__about__.py index 11dfaef6..db31b7b9 100644 --- a/datafog/__about__.py +++ b/datafog/__about__.py @@ -1 +1 @@ -__version__ = "4.8.0b8" +__version__ = "4.8.0a9" From 6e7911dbc7bfdb24c508051804e0f3e4ff4eeeb7 Mon Sep 17 00:00:00 2001 From: GitHub Action Date: Thu, 3 Sep 2026 06:57:27 +0000 Subject: [PATCH 18/37] chore: bump version to 4.8.0b9 [skip ci] --- datafog/__about__.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/datafog/__about__.py b/datafog/__about__.py index db31b7b9..99d24a35 100644 --- a/datafog/__about__.py +++ b/datafog/__about__.py @@ -1 +1 @@ -__version__ = "4.8.0a9" +__version__ = "4.8.0b9" From f7ea8a14c936e1994b273fa351a87061df79521b Mon Sep 17 00:00:00 2001 From: GitHub Action Date: Mon, 7 Sep 2026 07:10:34 +0000 Subject: [PATCH 19/37] chore: bump version to 4.8.0a10 [skip ci] --- datafog/__about__.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/datafog/__about__.py b/datafog/__about__.py index 99d24a35..0b863f4a 100644 --- a/datafog/__about__.py +++ b/datafog/__about__.py @@ -1 +1 @@ -__version__ = "4.8.0b9" +__version__ = "4.8.0a10" From 9e4caf36f2bf637d1fbf6cf3e7c6f6a4591076ac Mon Sep 17 00:00:00 2001 From: GitHub Action Date: Thu, 10 Sep 2026 07:06:08 +0000 Subject: [PATCH 20/37] chore: bump version to 4.8.0b10 [skip ci] --- datafog/__about__.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/datafog/__about__.py b/datafog/__about__.py index 0b863f4a..0b40db4c 100644 --- a/datafog/__about__.py +++ b/datafog/__about__.py @@ -1 +1 @@ -__version__ = "4.8.0a10" +__version__ = "4.8.0b10" From 66c62ef8446e2bc4d99fe203d5fd6c53c7c62e9f Mon Sep 17 00:00:00 2001 From: GitHub Action Date: Mon, 14 Sep 2026 07:44:16 +0000 Subject: [PATCH 21/37] chore: bump version to 4.8.0a11 [skip ci] --- datafog/__about__.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/datafog/__about__.py b/datafog/__about__.py index 0b40db4c..7c3aecfe 100644 --- a/datafog/__about__.py +++ b/datafog/__about__.py @@ -1 +1 @@ -__version__ = "4.8.0b10" +__version__ = "4.8.0a11" From a4974fcdb72408b7a1214a3e6f6cf080c10dd2f6 Mon Sep 17 00:00:00 2001 From: GitHub Action Date: Thu, 17 Sep 2026 07:15:23 +0000 Subject: [PATCH 22/37] chore: bump version to 4.8.0b11 [skip ci] --- datafog/__about__.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/datafog/__about__.py b/datafog/__about__.py index 7c3aecfe..de5bc8a6 100644 --- a/datafog/__about__.py +++ b/datafog/__about__.py @@ -1 +1 @@ -__version__ = "4.8.0a11" +__version__ = "4.8.0b11" From 4c2c997e51ee25f8a712063c8c8e37f4db055e38 Mon Sep 17 00:00:00 2001 From: GitHub Action Date: Mon, 21 Sep 2026 07:48:51 +0000 Subject: [PATCH 23/37] chore: bump version to 4.8.0a12 [skip ci] --- datafog/__about__.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/datafog/__about__.py b/datafog/__about__.py index de5bc8a6..b4ed9f6a 100644 --- a/datafog/__about__.py +++ b/datafog/__about__.py @@ -1 +1 @@ -__version__ = "4.8.0b11" +__version__ = "4.8.0a12" From 63d8858b875efea53727e76e3bede2883a44094e Mon Sep 17 00:00:00 2001 From: GitHub Action Date: Thu, 24 Sep 2026 07:16:35 +0000 Subject: [PATCH 24/37] chore: bump version to 4.8.0b12 [skip ci] --- datafog/__about__.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/datafog/__about__.py b/datafog/__about__.py index b4ed9f6a..de278028 100644 --- a/datafog/__about__.py +++ b/datafog/__about__.py @@ -1 +1 @@ -__version__ = "4.8.0a12" +__version__ = "4.8.0b12" From df69bd67b41112a3e4f5165c42eec972a0c156fb Mon Sep 17 00:00:00 2001 From: GitHub Action Date: Mon, 28 Sep 2026 08:25:58 +0000 Subject: [PATCH 25/37] chore: bump version to 4.8.0a13 [skip ci] --- datafog/__about__.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/datafog/__about__.py b/datafog/__about__.py index de278028..c9d2d193 100644 --- a/datafog/__about__.py +++ b/datafog/__about__.py @@ -1 +1 @@ -__version__ = "4.8.0b12" +__version__ = "4.8.0a13" From 308abd4aabbca430e376b3291c9a543d7ae06122 Mon Sep 17 00:00:00 2001 From: Sid Mohan <61345237+sidmohan0@users.noreply.github.com> Date: Mon, 28 Sep 2026 14:35:37 -0700 Subject: [PATCH 26/37] test: freeze published 4.8.1 text API contract --- .gitignore | 1 + README.md | 4 + docs/migration-4.8.1-contract.md | 123 + scripts/capture_481_contract.py | 78 + tests/contract_481.py | 120 + tests/contracts/4.8.1.json | 4329 ++++++++++++++++++++++++ tests/contracts/requirements-4.8.1.txt | 9 + tests/test_contract_481.py | 17 + 8 files changed, 4681 insertions(+) create mode 100644 docs/migration-4.8.1-contract.md create mode 100644 scripts/capture_481_contract.py create mode 100644 tests/contract_481.py create mode 100644 tests/contracts/4.8.1.json create mode 100644 tests/contracts/requirements-4.8.1.txt create mode 100644 tests/test_contract_481.py diff --git a/.gitignore b/.gitignore index cf11a42c..f7471a1e 100644 --- a/.gitignore +++ b/.gitignore @@ -60,6 +60,7 @@ docs/* !docs/Makefile !docs/make.bat !docs/optional-surfaces.rst +!docs/migration-4.8.1-contract.md !docs/agents/ !docs/agents/** !docs/audit/ diff --git a/README.md b/README.md index b38345b3..29179119 100644 --- a/README.md +++ b/README.md @@ -253,6 +253,10 @@ Telemetry does not include input text or detected PII values. ## Development +The [4.8.1 compatibility contract](docs/migration-4.8.1-contract.md) records +published Python behavior for the Rust migration, with frozen fixtures and +instructions for independently reproducing them from the release wheel. + ```bash git clone https://github.com/datafog/datafog-python cd datafog-python diff --git a/docs/migration-4.8.1-contract.md b/docs/migration-4.8.1-contract.md new file mode 100644 index 00000000..9d66001d --- /dev/null +++ b/docs/migration-4.8.1-contract.md @@ -0,0 +1,123 @@ +# DataFog 4.8.1 compatibility contract + +This is the release baseline for migrating the Python structured-text API to +DataFog Core. It freezes **111 black-box observations from the published 4.8.1 +wheel**, not expectations generated from the development checkout. It does not +claim that every observed detector behavior is desirable or that Core already +matches this contract. + +## Source and scope + +The source is [datafog 4.8.1 on PyPI](https://pypi.org/project/datafog/4.8.1/), +uploaded July 28, 2026. The fixture records the wheel URL, filename, SHA-256, +capture interpreter, and dependency versions. The wheel SHA-256 is: + +```text +d91cfb6d93568c9d073ca3ec93b492d29195f44828584556bd7bc443d3347040 +``` + +- `tests/contracts/4.8.1.json`: synthetic inputs and frozen results, warnings, + exceptions, selected function signatures, and dataclass field names. +- `tests/contract_481.py`: shared black-box runner for installed distributions + and the checkout. No detector or replacement algorithms are duplicated here. +- `tests/test_contract_481.py`: one independently identified pytest case per + observation, automatically included in the existing CI `pytest tests/` runs. +- `scripts/capture_481_contract.py`: verifies the wheel digest and installed + package bytes before producing a separate candidate baseline for review. + +Covered surfaces are top-level `scan`, `redact`, the engine's explicit-span +redaction and scan/redact helper, agent convenience functions, guardrail +filter/block behavior, legacy `detect`/`process` and core convenience functions, +and synchronous regex `TextService` annotation/batching. Signatures cover the +principal scan/redact/guardrail entry points and result constructors. + +This is the **structured synchronous text migration contract**, not a complete +contract for every package export. ML inference, smart/auto fallback behavior, +OCR, Spark, async/concurrency behavior, CLI output/exit codes, `DataFog` class +workflows, and Claude/LiteLLM adapters require their existing dedicated suites +and separate migration acceptance checks. Model quality and performance are +not inferred from these snapshots. No models or external services are needed. + +## Compatibility requirements + +| Concern | 4.8.1 behavior to preserve at existing Python entry points | +| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| Results | `ScanResult(entities, text, engine_used)`; `Entity(type, text, start, end, confidence, engine)`; `RedactResult(redacted_text, mapping, entities)` | +| Defaults | Top-level scan/redact and agent helpers use regex; the engine scan/helper defaults remain `smart` in the signature snapshot | +| Offsets | Zero-based, end-exclusive Python string indices. Emoji, modifiers, combining characters, CJK, and CRLF are preserved without normalization. Use Core code-point ranges, never byte ranges | +| Findings | Ordered findings with original matched text, numeric regex confidence `1.0`, and `regex` provenance; repeated occurrences retain separate spans | +| Selection | Canonical labels and aliases such as `EMAIL_ADDRESS`, `US_SSN`, `PHONE_NUMBER`, `DOB`, and `ZIP`; case/whitespace normalization. Empty selection means no restriction; an unknown-only selection yields no findings | +| Locales | Seven German entity families enabled by `de`/locale aliases or explicit entity selection; unsupported locale raises `ValueError` | +| Allowlists | Case-sensitive exact values or full-match Python regexes; validation occurs even for empty input. Lookahead is included because moving to Rust regex syntax can change accepted patterns | +| Explicit spans | Unsorted input is accepted; duplicate/overlapping spans are suppressed. Longest wins, then entity priority, confidence, and deterministic position/type ordering; adjacent spans remain distinct | +| `token` | Per-type numbering in document order, including repeated values: `[EMAIL_1]`, `[EMAIL_2]`; counters restart per call | +| `mask` | One `*` per Python character; mappings use numbered keys such as `[EMAIL_MASK_1]` so equal-length values do not collide | +| Engine `hash` | SHA-256 of the source value, truncated to 12 hex digits, inside a typed placeholder | +| Engine `pseudonymize` | Per-call numbered placeholders reused for repeated `(type, value)` pairs; no key provider and no cross-call linkage guarantee | +| Mappings | Replacement keys map to original plaintext; returned entities are the applied, surviving spans. Preserve only on compatibility APIs, without adding plaintext to Core transformation records | +| Presets | `default`/`llm`, `mask`, `hash`, `replace`, and `pseudonymize` retain their existing mappings | +| Validation | Exception classes and observed messages are captured, including rejected strategy/preset names and allowlists combined with explicit spans | +| Legacy functions | `detect` and `process` emit `FutureWarning` promising shims through 5.x. `process(anonymize=True)` has its own older placeholders and 8-digit MD5 hash output; it is not equivalent to engine redaction | +| Service API | Regex `TextService` returns a label-to-values dictionary including empty buckets and legacy `DOB`/`ZIP` labels; the modern scan facade returns canonical entities | + +The JSON is the exact executable evidence for the selected inputs. This table +explains the intended compatibility policy; finite examples are not proof of +complete detector coverage. In particular, legacy `token` must not silently +become Core's provider-backed `tokenize`, and legacy `pseudonymize` must not +silently acquire key-provider requirements. + +## Observations requiring migration review + +Cases marked `review-required` remain checked so a change cannot pass unnoticed. +They are **not instructions to reimplement bugs in Rust**. + +| Case IDs | Observed behavior | Proposed treatment | +| --------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------- | +| `scan-observation-invalid-card` | Card-shaped input with an invalid checksum is detected | Review as a detector-quality change; document any intentional difference | +| `scan-observation-boundary`, `scan-observation-ambiguous-number` | Embedded SSN-like values and arbitrary ten-digit numbers are detected | Evaluate precision/recall before deciding parity | +| `locale-default-DE_VAT_ID`, `locale-default-DE_TAX_ID`, `locale-negative-context` | German-looking numbers can still match generic SSN/PHONE detectors without a German finding | Preserve opt-in locale semantics; review generic detector differences separately | +| `explicit-invalid-bounds` | Invalid bounds are silently discarded | Consider explicit validation in the new API; retain or explicitly migrate legacy behavior | +| `explicit-mismatched-text` | Replacement uses the source slice even when `Entity.text` disagrees; returned entity retains the supplied text | Prefer Core validation for the new API; explicitly decide compatibility behavior | +| `process-invalid` | Unknown legacy anonymization method produces an unnumbered type placeholder | Preserve in the promised shim or announce an intentional change | + +For any future mismatch, record the case ID, old/new result, classification +(regression, intentional improvement, or deliberate breaking change), rationale, +and targeted test in the migration PR. Do not regenerate the 4.8.1 fixture from +the implementation being migrated. If approved deviations become necessary, +record them separately by case ID with rationale; never blanket-skip this suite. + +## Run and reproduce + +Run the checkout contract with the normal development environment: + +```bash +python -m pytest tests/test_contract_481.py -q +``` + +To independently check the published wheel, run from the repository root using +an isolated Python 3.12 environment. Substitute an unused temporary directory: + +```bash +python3.12 -m venv /tmp/datafog-481-oracle +/tmp/datafog-481-oracle/bin/python -m pip download --no-deps --only-binary=:all: \ + --index-url https://pypi.org/simple datafog==4.8.1 -d /tmp/datafog-481-oracle +/tmp/datafog-481-oracle/bin/python -m pip install \ + -r tests/contracts/requirements-4.8.1.txt \ + /tmp/datafog-481-oracle/datafog-4.8.1-py3-none-any.whl +/tmp/datafog-481-oracle/bin/python -I tests/contract_481.py +/tmp/datafog-481-oracle/bin/python -I scripts/capture_481_contract.py \ + --wheel /tmp/datafog-481-oracle/datafog-4.8.1-py3-none-any.whl \ + --output /tmp/datafog-481-oracle/candidate.json +``` + +The isolated flag excludes the checkout and `PYTHONPATH` from imports. Capture +checks version, wheel SHA-256, installed file bytes, and import location. It +refuses to overwrite the committed baseline or an existing candidate file. +Compare candidate cases to the committed cases; interpreter/dependency metadata +may differ on another environment. Routine CI is offline with respect to the +oracle: it reads committed expectations and never downloads or regenerates them. +The runner disables DataFog telemetry for its synthetic inputs. + +When introducing the Rust adapter, run these same cases through the existing +Python facade with that backend selected. Direct Core result objects have a +different API and should retain their own cross-runtime conformance suite. diff --git a/scripts/capture_481_contract.py b/scripts/capture_481_contract.py new file mode 100644 index 00000000..7aadeb46 --- /dev/null +++ b/scripts/capture_481_contract.py @@ -0,0 +1,78 @@ +"""Reproduce the oracle from the exact published wheel in an isolated environment. + +Writes a separate candidate file; never overwrites the committed contract. +""" + +import argparse +import hashlib +import importlib.metadata +import json +import platform +import runpy +import sys +import zipfile +from pathlib import Path + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--wheel", type=Path, required=True) + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args() + if not sys.flags.isolated: + parser.error("Use python -I to exclude the checkout and PYTHONPATH") + root = Path(__file__).resolve().parents[1] + fixture_path = root / "tests/contracts/4.8.1.json" + if args.output.resolve() == fixture_path: + parser.error("Capture to a separate candidate file for review") + fixture = json.loads(fixture_path.read_text()) + provenance = fixture["provenance"] + digest = hashlib.sha256(args.wheel.read_bytes()).hexdigest() + if digest != provenance["sha256"]: + parser.error("Wheel does not match the frozen PyPI SHA-256") + dist = importlib.metadata.distribution("datafog") + if dist.version != "4.8.1": + parser.error("Install the published 4.8.1 wheel first") + # A version string alone cannot establish release provenance. Check installed + # package bytes, then ensure imports resolve to that verified installation. + with zipfile.ZipFile(args.wheel) as archive: + for name in archive.namelist(): + if ( + name.startswith("datafog/") + and not name.endswith("/") + and Path(dist.locate_file(name)).read_bytes() != archive.read(name) + ): + parser.error(f"Installed file differs from published wheel: {name}") + import datafog + + if ( + Path(datafog.__file__).resolve() + != Path(dist.locate_file("datafog/__init__.py")).resolve() + ): + parser.error("datafog import is shadowed by another source tree") + fixture["capture_environment"] = { + "python": platform.python_version(), + "implementation": platform.python_implementation(), + "dependencies": { + name: importlib.metadata.version(name) + for name in ( + "pydantic", + "pydantic-core", + "pydantic-settings", + "typing-extensions", + "annotated-types", + "python-dotenv", + "typing-inspection", + ) + }, + } + runner = runpy.run_path(str(root / "tests/contract_481.py")) + for case in fixture["cases"]: + case["expected"] = runner["observe"](case) + with args.output.open("x") as output: + output.write(json.dumps(fixture, indent=2, ensure_ascii=False) + "\n") + print(f"Captured {len(fixture['cases'])} cases to {args.output}") + + +if __name__ == "__main__": + main() diff --git a/tests/contract_481.py b/tests/contract_481.py new file mode 100644 index 00000000..075d9c71 --- /dev/null +++ b/tests/contract_481.py @@ -0,0 +1,120 @@ +"""Black-box runner shared by the released-wheel oracle and checkout tests. + +Run with ``python -I tests/contract_481.py`` to check an installed package. +This runner deliberately contains no detection or transformation implementation. +""" + +from __future__ import annotations + +import argparse +import dataclasses +import importlib +import inspect +import json +import os +import warnings +from pathlib import Path +from unittest.mock import patch + +FIXTURE = Path(__file__).parent / "contracts" / "4.8.1.json" + + +def normalize(value): + if dataclasses.is_dataclass(value): + return { + "result_type": type(value).__name__, + "fields": { + field.name: normalize(getattr(value, field.name)) + for field in dataclasses.fields(value) + }, + } + if isinstance(value, dict): + return {key: normalize(item) for key, item in value.items()} + if isinstance(value, (list, tuple)): + return [normalize(item) for item in value] + if value is None or isinstance(value, (str, int, float, bool)): + return value + raise TypeError(f"Unsupported contract result: {type(value).__name__}") + + +def resolve(target): + module, name = target.rsplit(":", 1) + return getattr(importlib.import_module(module), name) + + +def invoke(case): + kwargs = dict(case.get("kwargs", {})) + operation = case.get("operation", "call") + if "entities" in kwargs: + entity = resolve("datafog.engine:Entity") + kwargs["entities"] = [entity(**item) for item in kwargs["entities"]] + if operation == "schema": + result = {} + for target in case["targets"]: + value = resolve(target) + result[target] = { + "parameters": [ + { + "name": param.name, + "kind": param.kind.name, + "required": param.default is inspect.Parameter.empty, + "default": ( + None + if param.default is inspect.Parameter.empty + else normalize(param.default) + ), + } + for param in inspect.signature(value).parameters.values() + ], + } + if dataclasses.is_dataclass(value): + result[target]["fields"] = [ + field.name for field in dataclasses.fields(value) + ] + return result + if operation == "service": + service = resolve("datafog.services.text_service:TextService")( + **case.get("constructor", {}) + ) + return getattr(service, case["method"])(**kwargs) + if operation == "guardrail": + guardrail = resolve("datafog:create_guardrail")(**case.get("constructor", {})) + return getattr(guardrail, case["method"])(**kwargs) + return resolve(case["target"])(**kwargs) + + +def observe(case): + # Never send synthetic contract inputs to telemetry, even on opted-in hosts. + with ( + patch.dict(os.environ, {"DATAFOG_NO_TELEMETRY": "1"}), + warnings.catch_warnings(record=True) as caught, + ): + warnings.simplefilter("always") + try: + result = {"value": normalize(invoke(case))} + except Exception as exc: # noqa: BLE001 - exception behavior is oracle output + result = {"error": {"type": type(exc).__name__, "message": str(exc)}} + result["warnings"] = [ + {"type": item.category.__name__, "message": str(item.message)} + for item in caught + ] + return result + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--fixture", type=Path, default=FIXTURE) + args = parser.parse_args() + fixture = json.loads(args.fixture.read_text()) + failures = [] + for case in fixture["cases"]: + actual = observe(case) + if actual != case["expected"]: + failures.append(case["id"]) + print(json.dumps({"id": case["id"], "actual": actual}, indent=2)) + print(f"{len(fixture['cases']) - len(failures)}/{len(fixture['cases'])} matched") + return bool(failures) + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/contracts/4.8.1.json b/tests/contracts/4.8.1.json new file mode 100644 index 00000000..4b898838 --- /dev/null +++ b/tests/contracts/4.8.1.json @@ -0,0 +1,4329 @@ +{ + "schema_version": 1, + "provenance": { + "distribution": "datafog", + "version": "4.8.1", + "filename": "datafog-4.8.1-py3-none-any.whl", + "url": "https://files.pythonhosted.org/packages/80/b5/93063b6df06da145aeef015aaf4a74764c96e16201ec08c412a0a813b800/datafog-4.8.1-py3-none-any.whl", + "sha256": "d91cfb6d93568c9d073ca3ec93b492d29195f44828584556bd7bc443d3347040", + "uploaded_at": "2026-07-28T00:10:58.733942Z" + }, + "cases": [ + { + "id": "scan-empty", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "" + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [], + "text": "", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "scan-negative", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "The quick brown fox jumps over the lazy dog." + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [], + "text": "The quick brown fox jumps over the lazy dog.", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "scan-email", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "Email alice@example.com." + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "alice@example.com", + "start": 6, + "end": 23, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "Email alice@example.com.", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "scan-phone", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "Call (555) 123-4567." + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "PHONE", + "text": "(555) 123-4567", + "start": 5, + "end": 19, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "Call (555) 123-4567.", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "scan-ssn", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "SSN 123-45-6789." + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "SSN", + "text": "123-45-6789", + "start": 4, + "end": 15, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "SSN 123-45-6789.", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "scan-card", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "Card 4111 1111 1111 1111." + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "CREDIT_CARD", + "text": "4111 1111 1111 1111", + "start": 5, + "end": 24, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "Card 4111 1111 1111 1111.", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "scan-ip", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "Host 192.168.1.1." + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "IP_ADDRESS", + "text": "192.168.1.1", + "start": 5, + "end": 16, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "Host 192.168.1.1.", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "scan-date", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "Born 01/15/1990." + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "DATE", + "text": "01/15/1990", + "start": 5, + "end": 15, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "Born 01/15/1990.", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "scan-zip", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "ZIP 94105." + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "ZIP_CODE", + "text": "94105", + "start": 4, + "end": 9, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "ZIP 94105.", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "scan-unicode", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "👩🏽‍💻 é 中文 alice@example.com\r\nBob bob@example.org" + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "alice@example.com", + "start": 11, + "end": 28, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "bob@example.org", + "start": 34, + "end": 49, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "👩🏽‍💻 é 中文 alice@example.com\r\nBob bob@example.org", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "scan-repeated", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "alice@example.com then alice@example.com" + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "alice@example.com", + "start": 0, + "end": 17, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "alice@example.com", + "start": 23, + "end": 40, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "alice@example.com then alice@example.com", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "scan-mixed", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "alice@example.com; 123-45-6789; (555) 123-4567; 4111 1111 1111 1111; 192.168.1.1; 01/15/1990; 94105" + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "alice@example.com", + "start": 0, + "end": 17, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "SSN", + "text": "123-45-6789", + "start": 19, + "end": 30, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "PHONE", + "text": "(555) 123-4567", + "start": 32, + "end": 46, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "CREDIT_CARD", + "text": "4111 1111 1111 1111", + "start": 48, + "end": 67, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "IP_ADDRESS", + "text": "192.168.1.1", + "start": 69, + "end": 80, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "DATE", + "text": "01/15/1990", + "start": 82, + "end": 92, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "ZIP_CODE", + "text": "94105", + "start": 94, + "end": 99, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "alice@example.com; 123-45-6789; (555) 123-4567; 4111 1111 1111 1111; 192.168.1.1; 01/15/1990; 94105", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "scan-observation-invalid-email", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "alice@ and @example.com" + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [], + "text": "alice@ and @example.com", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "scan-observation-invalid-ip", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "999.999.999.999" + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [], + "text": "999.999.999.999", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "scan-observation-invalid-card", + "classification": "review-required", + "target": "datafog:scan", + "kwargs": { + "text": "4111 1111 1111 1112" + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "CREDIT_CARD", + "text": "4111 1111 1111 1112", + "start": 0, + "end": 19, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "4111 1111 1111 1112", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "scan-observation-boundary", + "classification": "review-required", + "target": "datafog:scan", + "kwargs": { + "text": "x123-45-6789y" + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "SSN", + "text": "123-45-6789", + "start": 1, + "end": 12, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "x123-45-6789y", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "scan-observation-ambiguous-number", + "classification": "review-required", + "target": "datafog:scan", + "kwargs": { + "text": "Invoice 1234567890" + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "PHONE", + "text": "1234567890", + "start": 8, + "end": 18, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "Invoice 1234567890", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "selection-email", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "alice@example.com 123-45-6789 (555) 123-4567 01/15/1990 94105", + "entity_types": ["EMAIL"] + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "alice@example.com", + "start": 0, + "end": 17, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "alice@example.com 123-45-6789 (555) 123-4567 01/15/1990 94105", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "selection-aliases", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "alice@example.com 123-45-6789 (555) 123-4567 01/15/1990 94105", + "entity_types": [ + "EMAIL_ADDRESS", + "US_SSN", + "PHONE_NUMBER", + "DOB", + "ZIP" + ] + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "alice@example.com", + "start": 0, + "end": 17, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "SSN", + "text": "123-45-6789", + "start": 18, + "end": 29, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "PHONE", + "text": "(555) 123-4567", + "start": 30, + "end": 44, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "DATE", + "text": "01/15/1990", + "start": 45, + "end": 55, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "ZIP_CODE", + "text": "94105", + "start": 56, + "end": 61, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "alice@example.com 123-45-6789 (555) 123-4567 01/15/1990 94105", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "selection-normalized", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "alice@example.com 123-45-6789 (555) 123-4567 01/15/1990 94105", + "entity_types": [" email_address "] + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "alice@example.com", + "start": 0, + "end": 17, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "alice@example.com 123-45-6789 (555) 123-4567 01/15/1990 94105", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "selection-unknown", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "alice@example.com 123-45-6789 (555) 123-4567 01/15/1990 94105", + "entity_types": ["NOT_A_TYPE"] + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [], + "text": "alice@example.com 123-45-6789 (555) 123-4567 01/15/1990 94105", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "selection-empty", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "alice@example.com 123-45-6789 (555) 123-4567 01/15/1990 94105", + "entity_types": [] + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "alice@example.com", + "start": 0, + "end": 17, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "SSN", + "text": "123-45-6789", + "start": 18, + "end": 29, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "PHONE", + "text": "(555) 123-4567", + "start": 30, + "end": 44, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "DATE", + "text": "01/15/1990", + "start": 45, + "end": 55, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "ZIP_CODE", + "text": "94105", + "start": 56, + "end": 61, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "alice@example.com 123-45-6789 (555) 123-4567 01/15/1990 94105", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "allowlist-exact", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "alice@example.com bob@example.org alice@example.com", + "allowlist": ["alice@example.com"] + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "bob@example.org", + "start": 18, + "end": 33, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "alice@example.com bob@example.org alice@example.com", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "allowlist-case-sensitive", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "alice@example.com bob@example.org alice@example.com", + "allowlist": ["ALICE@example.com"] + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "alice@example.com", + "start": 0, + "end": 17, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "bob@example.org", + "start": 18, + "end": 33, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "alice@example.com", + "start": 34, + "end": 51, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "alice@example.com bob@example.org alice@example.com", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "allowlist-regex", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "alice@example.com bob@example.org alice@example.com", + "allowlist_patterns": [".+@example\\.com"] + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "bob@example.org", + "start": 18, + "end": 33, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "alice@example.com bob@example.org alice@example.com", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "allowlist-full-match", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "alice@example.com bob@example.org alice@example.com", + "allowlist_patterns": ["example"] + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "alice@example.com", + "start": 0, + "end": 17, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "bob@example.org", + "start": 18, + "end": 33, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "alice@example.com", + "start": 34, + "end": 51, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "alice@example.com bob@example.org alice@example.com", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "allowlist-combined", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "alice@example.com bob@example.org alice@example.com", + "allowlist": ["alice@example.com"], + "allowlist_patterns": [".+@example\\.org"] + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [], + "text": "alice@example.com bob@example.org alice@example.com", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "locale-default-DE_VAT_ID", + "classification": "review-required", + "target": "datafog:scan", + "kwargs": { + "text": "USt-IdNr DE 123456789 ist gesetzt." + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "SSN", + "text": "123456789", + "start": 12, + "end": 21, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "USt-IdNr DE 123456789 ist gesetzt.", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "locale-de-DE_VAT_ID", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "USt-IdNr DE 123456789 ist gesetzt.", + "locales": ["de"] + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "DE_VAT_ID", + "text": "DE 123456789", + "start": 9, + "end": 21, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "USt-IdNr DE 123456789 ist gesetzt.", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "locale-explicit-DE_VAT_ID", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "USt-IdNr DE 123456789 ist gesetzt.", + "entity_types": ["DE_VAT_ID"] + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "DE_VAT_ID", + "text": "DE 123456789", + "start": 9, + "end": 21, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "USt-IdNr DE 123456789 ist gesetzt.", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "locale-default-DE_IBAN", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "IBAN DE44 5001 0517 5407 3249 31 ist gueltig." + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [], + "text": "IBAN DE44 5001 0517 5407 3249 31 ist gueltig.", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "locale-de-DE_IBAN", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "IBAN DE44 5001 0517 5407 3249 31 ist gueltig.", + "locales": ["de"] + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "DE_IBAN", + "text": "DE44 5001 0517 5407 3249 31", + "start": 5, + "end": 32, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "IBAN DE44 5001 0517 5407 3249 31 ist gueltig.", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "locale-explicit-DE_IBAN", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "IBAN DE44 5001 0517 5407 3249 31 ist gueltig.", + "entity_types": ["DE_IBAN"] + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "DE_IBAN", + "text": "DE44 5001 0517 5407 3249 31", + "start": 5, + "end": 32, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "IBAN DE44 5001 0517 5407 3249 31 ist gueltig.", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "locale-default-DE_TAX_ID", + "classification": "review-required", + "target": "datafog:scan", + "kwargs": { + "text": "Steuer-ID 12345678901 liegt vor." + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "PHONE", + "text": "12345678901", + "start": 10, + "end": 21, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "Steuer-ID 12345678901 liegt vor.", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "locale-de-DE_TAX_ID", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "Steuer-ID 12345678901 liegt vor.", + "locales": ["de"] + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "DE_TAX_ID", + "text": "12345678901", + "start": 10, + "end": 21, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "Steuer-ID 12345678901 liegt vor.", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "locale-explicit-DE_TAX_ID", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "Steuer-ID 12345678901 liegt vor.", + "entity_types": ["DE_TAX_ID"] + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "DE_TAX_ID", + "text": "12345678901", + "start": 10, + "end": 21, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "Steuer-ID 12345678901 liegt vor.", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "locale-default-DE_SOCIAL_SECURITY_NUMBER", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "Rentenversicherungsnummer 65150804A123 liegt vor." + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [], + "text": "Rentenversicherungsnummer 65150804A123 liegt vor.", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "locale-de-DE_SOCIAL_SECURITY_NUMBER", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "Rentenversicherungsnummer 65150804A123 liegt vor.", + "locales": ["de"] + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "DE_SOCIAL_SECURITY_NUMBER", + "text": "65150804A123", + "start": 26, + "end": 38, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "Rentenversicherungsnummer 65150804A123 liegt vor.", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "locale-explicit-DE_SOCIAL_SECURITY_NUMBER", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "Rentenversicherungsnummer 65150804A123 liegt vor.", + "entity_types": ["DE_SOCIAL_SECURITY_NUMBER"] + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "DE_SOCIAL_SECURITY_NUMBER", + "text": "65150804A123", + "start": 26, + "end": 38, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "Rentenversicherungsnummer 65150804A123 liegt vor.", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "locale-default-DE_POSTAL_CODE", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "PLZ10115 Berlin." + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [], + "text": "PLZ10115 Berlin.", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "locale-de-DE_POSTAL_CODE", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "PLZ10115 Berlin.", + "locales": ["de"] + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "DE_POSTAL_CODE", + "text": "PLZ10115", + "start": 0, + "end": 8, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "PLZ10115 Berlin.", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "locale-explicit-DE_POSTAL_CODE", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "PLZ10115 Berlin.", + "entity_types": ["DE_POSTAL_CODE"] + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "DE_POSTAL_CODE", + "text": "PLZ10115", + "start": 0, + "end": 8, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "PLZ10115 Berlin.", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "locale-default-DE_PASSPORT_NUMBER", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "Passnummer C12345678 wurde geprueft." + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [], + "text": "Passnummer C12345678 wurde geprueft.", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "locale-de-DE_PASSPORT_NUMBER", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "Passnummer C12345678 wurde geprueft.", + "locales": ["de"] + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "DE_PASSPORT_NUMBER", + "text": "C12345678", + "start": 11, + "end": 20, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "Passnummer C12345678 wurde geprueft.", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "locale-explicit-DE_PASSPORT_NUMBER", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "Passnummer C12345678 wurde geprueft.", + "entity_types": ["DE_PASSPORT_NUMBER"] + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "DE_PASSPORT_NUMBER", + "text": "C12345678", + "start": 11, + "end": 20, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "Passnummer C12345678 wurde geprueft.", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "locale-default-DE_RESIDENCE_PERMIT_NUMBER", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "Aufenthaltstitel AT1234567 gueltig." + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [], + "text": "Aufenthaltstitel AT1234567 gueltig.", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "locale-de-DE_RESIDENCE_PERMIT_NUMBER", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "Aufenthaltstitel AT1234567 gueltig.", + "locales": ["de"] + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "DE_RESIDENCE_PERMIT_NUMBER", + "text": "AT1234567", + "start": 17, + "end": 26, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "Aufenthaltstitel AT1234567 gueltig.", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "locale-explicit-DE_RESIDENCE_PERMIT_NUMBER", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "Aufenthaltstitel AT1234567 gueltig.", + "entity_types": ["DE_RESIDENCE_PERMIT_NUMBER"] + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "DE_RESIDENCE_PERMIT_NUMBER", + "text": "AT1234567", + "start": 17, + "end": 26, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "Aufenthaltstitel AT1234567 gueltig.", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "locale-negative-context", + "classification": "review-required", + "target": "datafog:scan", + "kwargs": { + "text": "Invoice 12345678901; Ticket A12345678; Order AT1234567.", + "locales": ["de"] + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "PHONE", + "text": "12345678901", + "start": 8, + "end": 19, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "Invoice 12345678901; Ticket A12345678; Order AT1234567.", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "locale-unknown", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "alice@example.com", + "locales": ["xx"] + }, + "expected": { + "error": { + "type": "ValueError", + "message": "locale must be one of: de, de-de, de_de" + }, + "warnings": [] + } + }, + { + "id": "redact-token", + "classification": "compatibility", + "target": "datafog:redact", + "kwargs": { + "text": "a@example.com b@example.com a@example.com 123-45-6789", + "strategy": "token" + }, + "expected": { + "value": { + "result_type": "RedactResult", + "fields": { + "redacted_text": "[EMAIL_1] [EMAIL_2] [EMAIL_3] [SSN_1]", + "mapping": { + "[EMAIL_1]": "a@example.com", + "[EMAIL_2]": "b@example.com", + "[EMAIL_3]": "a@example.com", + "[SSN_1]": "123-45-6789" + }, + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "a@example.com", + "start": 0, + "end": 13, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "b@example.com", + "start": 14, + "end": 27, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "a@example.com", + "start": 28, + "end": 41, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "SSN", + "text": "123-45-6789", + "start": 42, + "end": 53, + "confidence": 1.0, + "engine": "regex" + } + } + ] + } + }, + "warnings": [] + } + }, + { + "id": "engine-redact-token", + "classification": "compatibility", + "target": "datafog.engine:scan_and_redact", + "kwargs": { + "text": "a@example.com b@example.com a@example.com", + "strategy": "token", + "engine": "regex" + }, + "expected": { + "value": { + "result_type": "RedactResult", + "fields": { + "redacted_text": "[EMAIL_1] [EMAIL_2] [EMAIL_3]", + "mapping": { + "[EMAIL_1]": "a@example.com", + "[EMAIL_2]": "b@example.com", + "[EMAIL_3]": "a@example.com" + }, + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "a@example.com", + "start": 0, + "end": 13, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "b@example.com", + "start": 14, + "end": 27, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "a@example.com", + "start": 28, + "end": 41, + "confidence": 1.0, + "engine": "regex" + } + } + ] + } + }, + "warnings": [] + } + }, + { + "id": "redact-mask", + "classification": "compatibility", + "target": "datafog:redact", + "kwargs": { + "text": "a@example.com b@example.com a@example.com 123-45-6789", + "strategy": "mask" + }, + "expected": { + "value": { + "result_type": "RedactResult", + "fields": { + "redacted_text": "************* ************* ************* ***********", + "mapping": { + "[EMAIL_MASK_1]": "a@example.com", + "[EMAIL_MASK_2]": "b@example.com", + "[EMAIL_MASK_3]": "a@example.com", + "[SSN_MASK_1]": "123-45-6789" + }, + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "a@example.com", + "start": 0, + "end": 13, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "b@example.com", + "start": 14, + "end": 27, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "a@example.com", + "start": 28, + "end": 41, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "SSN", + "text": "123-45-6789", + "start": 42, + "end": 53, + "confidence": 1.0, + "engine": "regex" + } + } + ] + } + }, + "warnings": [] + } + }, + { + "id": "engine-redact-mask", + "classification": "compatibility", + "target": "datafog.engine:scan_and_redact", + "kwargs": { + "text": "a@example.com b@example.com a@example.com", + "strategy": "mask", + "engine": "regex" + }, + "expected": { + "value": { + "result_type": "RedactResult", + "fields": { + "redacted_text": "************* ************* *************", + "mapping": { + "[EMAIL_MASK_1]": "a@example.com", + "[EMAIL_MASK_2]": "b@example.com", + "[EMAIL_MASK_3]": "a@example.com" + }, + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "a@example.com", + "start": 0, + "end": 13, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "b@example.com", + "start": 14, + "end": 27, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "a@example.com", + "start": 28, + "end": 41, + "confidence": 1.0, + "engine": "regex" + } + } + ] + } + }, + "warnings": [] + } + }, + { + "id": "redact-hash", + "classification": "compatibility", + "target": "datafog:redact", + "kwargs": { + "text": "a@example.com b@example.com a@example.com 123-45-6789", + "strategy": "hash" + }, + "expected": { + "value": { + "result_type": "RedactResult", + "fields": { + "redacted_text": "[EMAIL_08168cd80dfd] [EMAIL_e8f39b3e1382] [EMAIL_08168cd80dfd] [SSN_01a54629efb9]", + "mapping": { + "[EMAIL_08168cd80dfd]": "a@example.com", + "[EMAIL_e8f39b3e1382]": "b@example.com", + "[SSN_01a54629efb9]": "123-45-6789" + }, + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "a@example.com", + "start": 0, + "end": 13, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "b@example.com", + "start": 14, + "end": 27, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "a@example.com", + "start": 28, + "end": 41, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "SSN", + "text": "123-45-6789", + "start": 42, + "end": 53, + "confidence": 1.0, + "engine": "regex" + } + } + ] + } + }, + "warnings": [] + } + }, + { + "id": "engine-redact-hash", + "classification": "compatibility", + "target": "datafog.engine:scan_and_redact", + "kwargs": { + "text": "a@example.com b@example.com a@example.com", + "strategy": "hash", + "engine": "regex" + }, + "expected": { + "value": { + "result_type": "RedactResult", + "fields": { + "redacted_text": "[EMAIL_08168cd80dfd] [EMAIL_e8f39b3e1382] [EMAIL_08168cd80dfd]", + "mapping": { + "[EMAIL_08168cd80dfd]": "a@example.com", + "[EMAIL_e8f39b3e1382]": "b@example.com" + }, + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "a@example.com", + "start": 0, + "end": 13, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "b@example.com", + "start": 14, + "end": 27, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "a@example.com", + "start": 28, + "end": 41, + "confidence": 1.0, + "engine": "regex" + } + } + ] + } + }, + "warnings": [] + } + }, + { + "id": "redact-pseudonymize", + "classification": "compatibility", + "target": "datafog:redact", + "kwargs": { + "text": "a@example.com b@example.com a@example.com 123-45-6789", + "strategy": "pseudonymize" + }, + "expected": { + "value": { + "result_type": "RedactResult", + "fields": { + "redacted_text": "[EMAIL_PSEUDO_1] [EMAIL_PSEUDO_2] [EMAIL_PSEUDO_1] [SSN_PSEUDO_1]", + "mapping": { + "[EMAIL_PSEUDO_1]": "a@example.com", + "[EMAIL_PSEUDO_2]": "b@example.com", + "[SSN_PSEUDO_1]": "123-45-6789" + }, + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "a@example.com", + "start": 0, + "end": 13, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "b@example.com", + "start": 14, + "end": 27, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "a@example.com", + "start": 28, + "end": 41, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "SSN", + "text": "123-45-6789", + "start": 42, + "end": 53, + "confidence": 1.0, + "engine": "regex" + } + } + ] + } + }, + "warnings": [] + } + }, + { + "id": "engine-redact-pseudonymize", + "classification": "compatibility", + "target": "datafog.engine:scan_and_redact", + "kwargs": { + "text": "a@example.com b@example.com a@example.com", + "strategy": "pseudonymize", + "engine": "regex" + }, + "expected": { + "value": { + "result_type": "RedactResult", + "fields": { + "redacted_text": "[EMAIL_PSEUDO_1] [EMAIL_PSEUDO_2] [EMAIL_PSEUDO_1]", + "mapping": { + "[EMAIL_PSEUDO_1]": "a@example.com", + "[EMAIL_PSEUDO_2]": "b@example.com" + }, + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "a@example.com", + "start": 0, + "end": 13, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "b@example.com", + "start": 14, + "end": 27, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "a@example.com", + "start": 28, + "end": 41, + "confidence": 1.0, + "engine": "regex" + } + } + ] + } + }, + "warnings": [] + } + }, + { + "id": "redact-unicode", + "classification": "compatibility", + "target": "datafog:redact", + "kwargs": { + "text": "👩🏽‍💻 é 中文 alice@example.com" + }, + "expected": { + "value": { + "result_type": "RedactResult", + "fields": { + "redacted_text": "👩🏽‍💻 é 中文 [EMAIL_1]", + "mapping": { + "[EMAIL_1]": "alice@example.com" + }, + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "alice@example.com", + "start": 11, + "end": 28, + "confidence": 1.0, + "engine": "regex" + } + } + ] + } + }, + "warnings": [] + } + }, + { + "id": "redact-empty", + "classification": "compatibility", + "target": "datafog:redact", + "kwargs": { + "text": "" + }, + "expected": { + "value": { + "result_type": "RedactResult", + "fields": { + "redacted_text": "", + "mapping": {}, + "entities": [] + } + }, + "warnings": [] + } + }, + { + "id": "redact-allowlist", + "classification": "compatibility", + "target": "datafog:redact", + "kwargs": { + "text": "alice@example.com bob@example.org", + "allowlist": ["alice@example.com"] + }, + "expected": { + "value": { + "result_type": "RedactResult", + "fields": { + "redacted_text": "alice@example.com [EMAIL_1]", + "mapping": { + "[EMAIL_1]": "bob@example.org" + }, + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "bob@example.org", + "start": 18, + "end": 33, + "confidence": 1.0, + "engine": "regex" + } + } + ] + } + }, + "warnings": [] + } + }, + { + "id": "preset-default", + "classification": "compatibility", + "target": "datafog:redact", + "kwargs": { + "text": "alice@example.com", + "preset": "default" + }, + "expected": { + "value": { + "result_type": "RedactResult", + "fields": { + "redacted_text": "[EMAIL_1]", + "mapping": { + "[EMAIL_1]": "alice@example.com" + }, + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "alice@example.com", + "start": 0, + "end": 17, + "confidence": 1.0, + "engine": "regex" + } + } + ] + } + }, + "warnings": [] + } + }, + { + "id": "preset-llm", + "classification": "compatibility", + "target": "datafog:redact", + "kwargs": { + "text": "alice@example.com", + "preset": "llm" + }, + "expected": { + "value": { + "result_type": "RedactResult", + "fields": { + "redacted_text": "[EMAIL_1]", + "mapping": { + "[EMAIL_1]": "alice@example.com" + }, + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "alice@example.com", + "start": 0, + "end": 17, + "confidence": 1.0, + "engine": "regex" + } + } + ] + } + }, + "warnings": [] + } + }, + { + "id": "preset-mask", + "classification": "compatibility", + "target": "datafog:redact", + "kwargs": { + "text": "alice@example.com", + "preset": "mask" + }, + "expected": { + "value": { + "result_type": "RedactResult", + "fields": { + "redacted_text": "*****************", + "mapping": { + "[EMAIL_MASK_1]": "alice@example.com" + }, + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "alice@example.com", + "start": 0, + "end": 17, + "confidence": 1.0, + "engine": "regex" + } + } + ] + } + }, + "warnings": [] + } + }, + { + "id": "preset-hash", + "classification": "compatibility", + "target": "datafog:redact", + "kwargs": { + "text": "alice@example.com", + "preset": "hash" + }, + "expected": { + "value": { + "result_type": "RedactResult", + "fields": { + "redacted_text": "[EMAIL_ff8d9819fc0e]", + "mapping": { + "[EMAIL_ff8d9819fc0e]": "alice@example.com" + }, + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "alice@example.com", + "start": 0, + "end": 17, + "confidence": 1.0, + "engine": "regex" + } + } + ] + } + }, + "warnings": [] + } + }, + { + "id": "preset-replace", + "classification": "compatibility", + "target": "datafog:redact", + "kwargs": { + "text": "alice@example.com", + "preset": "replace" + }, + "expected": { + "value": { + "result_type": "RedactResult", + "fields": { + "redacted_text": "[EMAIL_PSEUDO_1]", + "mapping": { + "[EMAIL_PSEUDO_1]": "alice@example.com" + }, + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "alice@example.com", + "start": 0, + "end": 17, + "confidence": 1.0, + "engine": "regex" + } + } + ] + } + }, + "warnings": [] + } + }, + { + "id": "preset-pseudonymize", + "classification": "compatibility", + "target": "datafog:redact", + "kwargs": { + "text": "alice@example.com", + "preset": "pseudonymize" + }, + "expected": { + "value": { + "result_type": "RedactResult", + "fields": { + "redacted_text": "[EMAIL_PSEUDO_1]", + "mapping": { + "[EMAIL_PSEUDO_1]": "alice@example.com" + }, + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "alice@example.com", + "start": 0, + "end": 17, + "confidence": 1.0, + "engine": "regex" + } + } + ] + } + }, + "warnings": [] + } + }, + { + "id": "explicit-unsorted", + "classification": "compatibility", + "target": "datafog:redact", + "kwargs": { + "text": "a@example.com b@example.com", + "entities": [ + { + "type": "EMAIL", + "text": "b@example.com", + "start": 14, + "end": 27, + "confidence": 1.0, + "engine": "regex" + }, + { + "type": "EMAIL", + "text": "a@example.com", + "start": 0, + "end": 13, + "confidence": 1.0, + "engine": "regex" + } + ] + }, + "expected": { + "value": { + "result_type": "RedactResult", + "fields": { + "redacted_text": "[EMAIL_1] [EMAIL_2]", + "mapping": { + "[EMAIL_1]": "a@example.com", + "[EMAIL_2]": "b@example.com" + }, + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "a@example.com", + "start": 0, + "end": 13, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "b@example.com", + "start": 14, + "end": 27, + "confidence": 1.0, + "engine": "regex" + } + } + ] + } + }, + "warnings": [] + } + }, + { + "id": "explicit-duplicates", + "classification": "compatibility", + "target": "datafog:redact", + "kwargs": { + "text": "a@example.com b@example.com", + "entities": [ + { + "type": "EMAIL", + "text": "a@example.com", + "start": 0, + "end": 13, + "confidence": 1.0, + "engine": "regex" + }, + { + "type": "EMAIL", + "text": "a@example.com", + "start": 0, + "end": 13, + "confidence": 1.0, + "engine": "regex" + }, + { + "type": "EMAIL", + "text": "b@example.com", + "start": 14, + "end": 27, + "confidence": 1.0, + "engine": "regex" + } + ] + }, + "expected": { + "value": { + "result_type": "RedactResult", + "fields": { + "redacted_text": "[EMAIL_1] [EMAIL_2]", + "mapping": { + "[EMAIL_1]": "a@example.com", + "[EMAIL_2]": "b@example.com" + }, + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "a@example.com", + "start": 0, + "end": 13, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "b@example.com", + "start": 14, + "end": 27, + "confidence": 1.0, + "engine": "regex" + } + } + ] + } + }, + "warnings": [] + } + }, + { + "id": "explicit-empty", + "classification": "compatibility", + "target": "datafog:redact", + "kwargs": { + "text": "a@example.com b@example.com", + "entities": [] + }, + "expected": { + "value": { + "result_type": "RedactResult", + "fields": { + "redacted_text": "a@example.com b@example.com", + "mapping": {}, + "entities": [] + } + }, + "warnings": [] + } + }, + { + "id": "explicit-overlap", + "classification": "compatibility", + "target": "datafog:redact", + "kwargs": { + "text": "a@example.com b@example.com", + "entities": [ + { + "type": "PHONE", + "text": "example.", + "start": 2, + "end": 10, + "confidence": 1.0, + "engine": "regex" + }, + { + "type": "EMAIL", + "text": "a@example.com", + "start": 0, + "end": 13, + "confidence": 1.0, + "engine": "regex" + }, + { + "type": "EMAIL", + "text": "b@example.com", + "start": 14, + "end": 27, + "confidence": 1.0, + "engine": "regex" + } + ] + }, + "expected": { + "value": { + "result_type": "RedactResult", + "fields": { + "redacted_text": "[EMAIL_1] [EMAIL_2]", + "mapping": { + "[EMAIL_1]": "a@example.com", + "[EMAIL_2]": "b@example.com" + }, + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "a@example.com", + "start": 0, + "end": 13, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "b@example.com", + "start": 14, + "end": 27, + "confidence": 1.0, + "engine": "regex" + } + } + ] + } + }, + "warnings": [] + } + }, + { + "id": "explicit-type-priority", + "classification": "compatibility", + "target": "datafog:redact", + "kwargs": { + "text": "a@example.com b@example.com", + "entities": [ + { + "type": "EMAIL", + "text": "a@example.com", + "start": 0, + "end": 13, + "confidence": 1.0, + "engine": "regex" + }, + { + "type": "PHONE", + "text": "a@example.com", + "start": 0, + "end": 13, + "confidence": 1.0, + "engine": "regex" + } + ] + }, + "expected": { + "value": { + "result_type": "RedactResult", + "fields": { + "redacted_text": "[PHONE_1] b@example.com", + "mapping": { + "[PHONE_1]": "a@example.com" + }, + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "PHONE", + "text": "a@example.com", + "start": 0, + "end": 13, + "confidence": 1.0, + "engine": "regex" + } + } + ] + } + }, + "warnings": [] + } + }, + { + "id": "explicit-confidence-tie", + "classification": "compatibility", + "target": "datafog:redact", + "kwargs": { + "text": "a@example.com b@example.com", + "entities": [ + { + "type": "EMAIL", + "text": "a@example.com", + "start": 0, + "end": 13, + "confidence": 0.5, + "engine": "regex" + }, + { + "type": "EMAIL", + "text": "a@example.com", + "start": 0, + "end": 13, + "confidence": 1.0, + "engine": "regex" + } + ] + }, + "expected": { + "value": { + "result_type": "RedactResult", + "fields": { + "redacted_text": "[EMAIL_1] b@example.com", + "mapping": { + "[EMAIL_1]": "a@example.com" + }, + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "a@example.com", + "start": 0, + "end": 13, + "confidence": 1.0, + "engine": "regex" + } + } + ] + } + }, + "warnings": [] + } + }, + { + "id": "explicit-adjacent", + "classification": "compatibility", + "target": "datafog:redact", + "kwargs": { + "text": "abcd", + "entities": [ + { + "type": "EMAIL", + "text": "ab", + "start": 0, + "end": 2, + "confidence": 1.0, + "engine": "regex" + }, + { + "type": "EMAIL", + "text": "cd", + "start": 2, + "end": 4, + "confidence": 1.0, + "engine": "regex" + } + ] + }, + "expected": { + "value": { + "result_type": "RedactResult", + "fields": { + "redacted_text": "[EMAIL_1][EMAIL_2]", + "mapping": { + "[EMAIL_1]": "ab", + "[EMAIL_2]": "cd" + }, + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "ab", + "start": 0, + "end": 2, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "cd", + "start": 2, + "end": 4, + "confidence": 1.0, + "engine": "regex" + } + } + ] + } + }, + "warnings": [] + } + }, + { + "id": "explicit-invalid-bounds", + "classification": "review-required", + "target": "datafog:redact", + "kwargs": { + "text": "a@example.com b@example.com", + "entities": [ + { + "type": "EMAIL", + "text": "", + "start": -1, + "end": 5, + "confidence": 1.0, + "engine": "regex" + }, + { + "type": "EMAIL", + "text": "a@example.com b@example.com", + "start": 0, + "end": 100, + "confidence": 1.0, + "engine": "regex" + }, + { + "type": "EMAIL", + "text": "", + "start": 3, + "end": 3, + "confidence": 1.0, + "engine": "regex" + }, + { + "type": "EMAIL", + "text": "a@example.com", + "start": 0, + "end": 13, + "confidence": 1.0, + "engine": "regex" + } + ] + }, + "expected": { + "value": { + "result_type": "RedactResult", + "fields": { + "redacted_text": "[EMAIL_1] b@example.com", + "mapping": { + "[EMAIL_1]": "a@example.com" + }, + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "a@example.com", + "start": 0, + "end": 13, + "confidence": 1.0, + "engine": "regex" + } + } + ] + } + }, + "warnings": [] + } + }, + { + "id": "explicit-mismatched-text", + "classification": "review-required", + "target": "datafog:redact", + "kwargs": { + "text": "a@example.com b@example.com", + "entities": [ + { + "type": "EMAIL", + "text": "wrong@example.org", + "start": 0, + "end": 13, + "confidence": 1.0, + "engine": "regex" + } + ] + }, + "expected": { + "value": { + "result_type": "RedactResult", + "fields": { + "redacted_text": "[EMAIL_1] b@example.com", + "mapping": { + "[EMAIL_1]": "a@example.com" + }, + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "wrong@example.org", + "start": 0, + "end": 13, + "confidence": 1.0, + "engine": "regex" + } + } + ] + } + }, + "warnings": [] + } + }, + { + "id": "error-non-string", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": null + }, + "expected": { + "error": { + "type": "TypeError", + "message": "text must be a string" + }, + "warnings": [] + } + }, + { + "id": "error-invalid-engine", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "test", + "engine": "unknown" + }, + "expected": { + "error": { + "type": "ValueError", + "message": "engine must be one of: regex, spacy, gliner, smart" + }, + "warnings": [] + } + }, + { + "id": "error-invalid-pattern", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "", + "allowlist_patterns": ["["] + }, + "expected": { + "error": { + "type": "ValueError", + "message": "allowlist_patterns contains an invalid regex: '[' (unterminated character set at position 0)" + }, + "warnings": [] + } + }, + { + "id": "error-unsafe-pattern", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "", + "allowlist_patterns": ["(a+)+"] + }, + "expected": { + "error": { + "type": "ValueError", + "message": "allowlist_patterns contains a quantified group with a nested quantifier ('(a+)+'), which risks catastrophic backtracking; rewrite the pattern without nesting quantifiers" + }, + "warnings": [] + } + }, + { + "id": "error-oversized-pattern", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "", + "allowlist_patterns": [ + "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" + ] + }, + "expected": { + "error": { + "type": "ValueError", + "message": "allowlist_patterns entries must be at most 512 characters" + }, + "warnings": [] + } + }, + { + "id": "error-strategy", + "classification": "compatibility", + "target": "datafog:redact", + "kwargs": { + "text": "alice@example.com", + "strategy": "tokenize" + }, + "expected": { + "error": { + "type": "ValueError", + "message": "strategy must be one of: token, mask, hash, pseudonymize" + }, + "warnings": [] + } + }, + { + "id": "error-preset", + "classification": "compatibility", + "target": "datafog:redact", + "kwargs": { + "text": "alice@example.com", + "preset": "unknown" + }, + "expected": { + "error": { + "type": "ValueError", + "message": "preset must be one of: default, hash, llm, mask, pseudonymize, replace" + }, + "warnings": [] + } + }, + { + "id": "error-explicit-allowlist", + "classification": "compatibility", + "target": "datafog:redact", + "kwargs": { + "text": "a@example.com b@example.com", + "entities": [ + { + "type": "EMAIL", + "text": "a@example.com", + "start": 0, + "end": 13, + "confidence": 1.0, + "engine": "regex" + } + ], + "allowlist": ["a@example.com"] + }, + "expected": { + "error": { + "type": "ValueError", + "message": "allowlist/allowlist_patterns cannot be combined with explicit entities; filter the entities before calling redact" + }, + "warnings": [] + } + }, + { + "id": "error-explicit-pattern", + "classification": "compatibility", + "target": "datafog:redact", + "kwargs": { + "text": "a@example.com b@example.com", + "entities": [ + { + "type": "EMAIL", + "text": "a@example.com", + "start": 0, + "end": 13, + "confidence": 1.0, + "engine": "regex" + } + ], + "allowlist_patterns": [".*"] + }, + "expected": { + "error": { + "type": "ValueError", + "message": "allowlist/allowlist_patterns cannot be combined with explicit entities; filter the entities before calling redact" + }, + "warnings": [] + } + }, + { + "id": "facade-sanitize", + "classification": "compatibility", + "target": "datafog:sanitize", + "kwargs": { + "text": "alice@example.com alice@example.com" + }, + "expected": { + "value": "[EMAIL_1] [EMAIL_2]", + "warnings": [] + } + }, + { + "id": "facade-filter_output", + "classification": "compatibility", + "target": "datafog:filter_output", + "kwargs": { + "output": "alice@example.com alice@example.com" + }, + "expected": { + "value": { + "result_type": "RedactResult", + "fields": { + "redacted_text": "[EMAIL_1] [EMAIL_2]", + "mapping": { + "[EMAIL_1]": "alice@example.com", + "[EMAIL_2]": "alice@example.com" + }, + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "alice@example.com", + "start": 0, + "end": 17, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "alice@example.com", + "start": 18, + "end": 35, + "confidence": 1.0, + "engine": "regex" + } + } + ] + } + }, + "warnings": [] + } + }, + { + "id": "facade-scan_prompt", + "classification": "compatibility", + "target": "datafog:scan_prompt", + "kwargs": { + "prompt": "alice@example.com alice@example.com" + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "alice@example.com", + "start": 0, + "end": 17, + "confidence": 1.0, + "engine": "regex" + } + }, + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "alice@example.com", + "start": 18, + "end": 35, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "alice@example.com alice@example.com", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "facade-detect", + "classification": "compatibility", + "target": "datafog:detect", + "kwargs": { + "text": "alice@example.com alice@example.com" + }, + "expected": { + "value": [ + { + "type": "EMAIL", + "value": "alice@example.com", + "start": 0, + "end": 17 + }, + { + "type": "EMAIL", + "value": "alice@example.com", + "start": 18, + "end": 35 + } + ], + "warnings": [ + { + "type": "FutureWarning", + "message": "datafog.detect() is deprecated for v5. Use datafog.scan() instead. This compatibility shim will remain through the v5.x line." + } + ] + } + }, + { + "id": "facade-detect_pii", + "classification": "compatibility", + "target": "datafog:detect_pii", + "kwargs": { + "text": "alice@example.com alice@example.com" + }, + "expected": { + "value": { + "EMAIL": ["alice@example.com", "alice@example.com"] + }, + "warnings": [] + } + }, + { + "id": "facade-scan_text", + "classification": "compatibility", + "target": "datafog:scan_text", + "kwargs": { + "text": "alice@example.com alice@example.com" + }, + "expected": { + "value": true, + "warnings": [] + } + }, + { + "id": "facade-anonymize_text", + "classification": "compatibility", + "target": "datafog:anonymize_text", + "kwargs": { + "text": "alice@example.com alice@example.com" + }, + "expected": { + "value": "[EMAIL_1] [EMAIL_2]", + "warnings": [] + } + }, + { + "id": "facade-process", + "classification": "compatibility", + "target": "datafog:process", + "kwargs": { + "text": "alice@example.com alice@example.com" + }, + "expected": { + "value": { + "original": "alice@example.com alice@example.com", + "findings": [ + { + "type": "EMAIL", + "value": "alice@example.com", + "start": 0, + "end": 17 + }, + { + "type": "EMAIL", + "value": "alice@example.com", + "start": 18, + "end": 35 + } + ] + }, + "warnings": [ + { + "type": "FutureWarning", + "message": "datafog.process() is deprecated for v5. Use datafog.scan() or datafog.redact() instead. This compatibility shim will remain through the v5.x line." + } + ] + } + }, + { + "id": "process-redact", + "classification": "compatibility", + "target": "datafog:process", + "kwargs": { + "text": "alice@example.com", + "method": "redact", + "anonymize": true + }, + "expected": { + "value": { + "original": "alice@example.com", + "findings": [ + { + "type": "EMAIL", + "value": "alice@example.com", + "start": 0, + "end": 17 + } + ], + "anonymized": "[EMAIL_REDACTED]" + }, + "warnings": [ + { + "type": "FutureWarning", + "message": "datafog.process() is deprecated for v5. Use datafog.scan() or datafog.redact() instead. This compatibility shim will remain through the v5.x line." + } + ] + } + }, + { + "id": "process-replace", + "classification": "compatibility", + "target": "datafog:process", + "kwargs": { + "text": "alice@example.com", + "method": "replace", + "anonymize": true + }, + "expected": { + "value": { + "original": "alice@example.com", + "findings": [ + { + "type": "EMAIL", + "value": "alice@example.com", + "start": 0, + "end": 17 + } + ], + "anonymized": "[EMAIL_XXXXX]" + }, + "warnings": [ + { + "type": "FutureWarning", + "message": "datafog.process() is deprecated for v5. Use datafog.scan() or datafog.redact() instead. This compatibility shim will remain through the v5.x line." + } + ] + } + }, + { + "id": "process-hash", + "classification": "compatibility", + "target": "datafog:process", + "kwargs": { + "text": "alice@example.com", + "method": "hash", + "anonymize": true + }, + "expected": { + "value": { + "original": "alice@example.com", + "findings": [ + { + "type": "EMAIL", + "value": "alice@example.com", + "start": 0, + "end": 17 + } + ], + "anonymized": "[EMAIL_c160f8cc]" + }, + "warnings": [ + { + "type": "FutureWarning", + "message": "datafog.process() is deprecated for v5. Use datafog.scan() or datafog.redact() instead. This compatibility shim will remain through the v5.x line." + } + ] + } + }, + { + "id": "process-invalid", + "classification": "review-required", + "target": "datafog:process", + "kwargs": { + "text": "alice@example.com", + "method": "invalid", + "anonymize": true + }, + "expected": { + "value": { + "original": "alice@example.com", + "findings": [ + { + "type": "EMAIL", + "value": "alice@example.com", + "start": 0, + "end": 17 + } + ], + "anonymized": "[EMAIL]" + }, + "warnings": [ + { + "type": "FutureWarning", + "message": "datafog.process() is deprecated for v5. Use datafog.scan() or datafog.redact() instead. This compatibility shim will remain through the v5.x line." + } + ] + } + }, + { + "id": "guardrail-redact", + "classification": "compatibility", + "target": "datafog:scan", + "operation": "guardrail", + "method": "filter", + "constructor": { + "on_detect": "redact" + }, + "kwargs": { + "text": "alice@example.com" + }, + "expected": { + "value": { + "result_type": "RedactResult", + "fields": { + "redacted_text": "[EMAIL_1]", + "mapping": { + "[EMAIL_1]": "alice@example.com" + }, + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "alice@example.com", + "start": 0, + "end": 17, + "confidence": 1.0, + "engine": "regex" + } + } + ] + } + }, + "warnings": [] + } + }, + { + "id": "guardrail-block", + "classification": "compatibility", + "target": "datafog:scan", + "operation": "guardrail", + "method": "filter", + "constructor": { + "on_detect": "block" + }, + "kwargs": { + "text": "alice@example.com" + }, + "expected": { + "error": { + "type": "GuardrailBlockedError", + "message": "Guardrail blocked text containing 1 PII entities." + }, + "warnings": [] + } + }, + { + "id": "guardrail-clean", + "classification": "compatibility", + "target": "datafog:scan", + "operation": "guardrail", + "method": "filter", + "constructor": { + "on_detect": "block" + }, + "kwargs": { + "text": "nothing sensitive here" + }, + "expected": { + "value": { + "result_type": "RedactResult", + "fields": { + "redacted_text": "nothing sensitive here", + "mapping": {}, + "entities": [] + } + }, + "warnings": [] + } + }, + { + "id": "service-default", + "classification": "compatibility", + "target": "datafog:scan", + "operation": "service", + "constructor": {}, + "method": "annotate_text_sync", + "kwargs": { + "text": "alice@example.com" + }, + "expected": { + "value": { + "EMAIL": ["alice@example.com"], + "PHONE": [], + "SSN": [], + "CREDIT_CARD": [], + "IP_ADDRESS": [], + "DOB": [], + "ZIP": [], + "DE_VAT_ID": [], + "DE_IBAN": [], + "DE_TAX_ID": [], + "DE_SOCIAL_SECURITY_NUMBER": [], + "DE_POSTAL_CODE": [], + "DE_PASSPORT_NUMBER": [], + "DE_RESIDENCE_PERMIT_NUMBER": [] + }, + "warnings": [] + } + }, + { + "id": "service-german", + "classification": "compatibility", + "target": "datafog:scan", + "operation": "service", + "constructor": { + "locales": ["de"] + }, + "method": "annotate_text_sync", + "kwargs": { + "text": "Steuer-ID 12345678901 liegt vor." + }, + "expected": { + "value": { + "EMAIL": [], + "PHONE": ["12345678901"], + "SSN": [], + "CREDIT_CARD": [], + "IP_ADDRESS": [], + "DOB": [], + "ZIP": [], + "DE_VAT_ID": [], + "DE_IBAN": [], + "DE_TAX_ID": ["12345678901"], + "DE_SOCIAL_SECURITY_NUMBER": [], + "DE_POSTAL_CODE": [], + "DE_PASSPORT_NUMBER": [], + "DE_RESIDENCE_PERMIT_NUMBER": [] + }, + "warnings": [] + } + }, + { + "id": "service-batch", + "classification": "compatibility", + "target": "datafog:scan", + "operation": "service", + "constructor": {}, + "method": "batch_annotate_text_sync", + "kwargs": { + "texts": ["alice@example.com", "no pii"] + }, + "expected": { + "value": [ + { + "EMAIL": ["alice@example.com"], + "PHONE": [], + "SSN": [], + "CREDIT_CARD": [], + "IP_ADDRESS": [], + "DOB": [], + "ZIP": [], + "DE_VAT_ID": [], + "DE_IBAN": [], + "DE_TAX_ID": [], + "DE_SOCIAL_SECURITY_NUMBER": [], + "DE_POSTAL_CODE": [], + "DE_PASSPORT_NUMBER": [], + "DE_RESIDENCE_PERMIT_NUMBER": [] + }, + { + "EMAIL": [], + "PHONE": [], + "SSN": [], + "CREDIT_CARD": [], + "IP_ADDRESS": [], + "DOB": [], + "ZIP": [], + "DE_VAT_ID": [], + "DE_IBAN": [], + "DE_TAX_ID": [], + "DE_SOCIAL_SECURITY_NUMBER": [], + "DE_POSTAL_CODE": [], + "DE_PASSPORT_NUMBER": [], + "DE_RESIDENCE_PERMIT_NUMBER": [] + } + ], + "warnings": [] + } + }, + { + "id": "public-signatures", + "classification": "compatibility", + "target": "datafog:scan", + "operation": "schema", + "targets": [ + "datafog:scan", + "datafog:redact", + "datafog:protect", + "datafog:detect", + "datafog:process", + "datafog:sanitize", + "datafog:scan_prompt", + "datafog:filter_output", + "datafog:create_guardrail", + "datafog.engine:Entity", + "datafog.engine:ScanResult", + "datafog.engine:RedactResult", + "datafog.engine:scan", + "datafog.engine:redact", + "datafog.engine:scan_and_redact" + ], + "expected": { + "value": { + "datafog:scan": { + "parameters": [ + { + "name": "text", + "kind": "POSITIONAL_OR_KEYWORD", + "required": true, + "default": null + }, + { + "name": "engine", + "kind": "POSITIONAL_OR_KEYWORD", + "required": false, + "default": "regex" + }, + { + "name": "entity_types", + "kind": "POSITIONAL_OR_KEYWORD", + "required": false, + "default": null + }, + { + "name": "locales", + "kind": "POSITIONAL_OR_KEYWORD", + "required": false, + "default": null + }, + { + "name": "allowlist", + "kind": "POSITIONAL_OR_KEYWORD", + "required": false, + "default": null + }, + { + "name": "allowlist_patterns", + "kind": "POSITIONAL_OR_KEYWORD", + "required": false, + "default": null + } + ] + }, + "datafog:redact": { + "parameters": [ + { + "name": "text", + "kind": "POSITIONAL_OR_KEYWORD", + "required": true, + "default": null + }, + { + "name": "entities", + "kind": "POSITIONAL_OR_KEYWORD", + "required": false, + "default": null + }, + { + "name": "engine", + "kind": "POSITIONAL_OR_KEYWORD", + "required": false, + "default": "regex" + }, + { + "name": "entity_types", + "kind": "POSITIONAL_OR_KEYWORD", + "required": false, + "default": null + }, + { + "name": "strategy", + "kind": "POSITIONAL_OR_KEYWORD", + "required": false, + "default": "token" + }, + { + "name": "preset", + "kind": "POSITIONAL_OR_KEYWORD", + "required": false, + "default": null + }, + { + "name": "locales", + "kind": "POSITIONAL_OR_KEYWORD", + "required": false, + "default": null + }, + { + "name": "allowlist", + "kind": "POSITIONAL_OR_KEYWORD", + "required": false, + "default": null + }, + { + "name": "allowlist_patterns", + "kind": "POSITIONAL_OR_KEYWORD", + "required": false, + "default": null + } + ] + }, + "datafog:protect": { + "parameters": [ + { + "name": "entity_types", + "kind": "POSITIONAL_OR_KEYWORD", + "required": false, + "default": null + }, + { + "name": "engine", + "kind": "POSITIONAL_OR_KEYWORD", + "required": false, + "default": "regex" + }, + { + "name": "strategy", + "kind": "POSITIONAL_OR_KEYWORD", + "required": false, + "default": "token" + }, + { + "name": "on_detect", + "kind": "POSITIONAL_OR_KEYWORD", + "required": false, + "default": "redact" + }, + { + "name": "locales", + "kind": "POSITIONAL_OR_KEYWORD", + "required": false, + "default": null + } + ] + }, + "datafog:detect": { + "parameters": [ + { + "name": "text", + "kind": "POSITIONAL_OR_KEYWORD", + "required": true, + "default": null + } + ] + }, + "datafog:process": { + "parameters": [ + { + "name": "text", + "kind": "POSITIONAL_OR_KEYWORD", + "required": true, + "default": null + }, + { + "name": "anonymize", + "kind": "POSITIONAL_OR_KEYWORD", + "required": false, + "default": false + }, + { + "name": "method", + "kind": "POSITIONAL_OR_KEYWORD", + "required": false, + "default": "redact" + } + ] + }, + "datafog:sanitize": { + "parameters": [ + { + "name": "text", + "kind": "POSITIONAL_OR_KEYWORD", + "required": true, + "default": null + }, + { + "name": "engine", + "kind": "POSITIONAL_OR_KEYWORD", + "required": false, + "default": "regex" + }, + { + "name": "kwargs", + "kind": "VAR_KEYWORD", + "required": true, + "default": null + } + ] + }, + "datafog:scan_prompt": { + "parameters": [ + { + "name": "prompt", + "kind": "POSITIONAL_OR_KEYWORD", + "required": true, + "default": null + }, + { + "name": "engine", + "kind": "POSITIONAL_OR_KEYWORD", + "required": false, + "default": "regex" + }, + { + "name": "kwargs", + "kind": "VAR_KEYWORD", + "required": true, + "default": null + } + ] + }, + "datafog:filter_output": { + "parameters": [ + { + "name": "output", + "kind": "POSITIONAL_OR_KEYWORD", + "required": true, + "default": null + }, + { + "name": "engine", + "kind": "POSITIONAL_OR_KEYWORD", + "required": false, + "default": "regex" + }, + { + "name": "kwargs", + "kind": "VAR_KEYWORD", + "required": true, + "default": null + } + ] + }, + "datafog:create_guardrail": { + "parameters": [ + { + "name": "entity_types", + "kind": "POSITIONAL_OR_KEYWORD", + "required": false, + "default": null + }, + { + "name": "locales", + "kind": "POSITIONAL_OR_KEYWORD", + "required": false, + "default": null + }, + { + "name": "engine", + "kind": "POSITIONAL_OR_KEYWORD", + "required": false, + "default": "regex" + }, + { + "name": "strategy", + "kind": "POSITIONAL_OR_KEYWORD", + "required": false, + "default": "token" + }, + { + "name": "on_detect", + "kind": "POSITIONAL_OR_KEYWORD", + "required": false, + "default": "redact" + } + ] + }, + "datafog.engine:Entity": { + "parameters": [ + { + "name": "type", + "kind": "POSITIONAL_OR_KEYWORD", + "required": true, + "default": null + }, + { + "name": "text", + "kind": "POSITIONAL_OR_KEYWORD", + "required": true, + "default": null + }, + { + "name": "start", + "kind": "POSITIONAL_OR_KEYWORD", + "required": true, + "default": null + }, + { + "name": "end", + "kind": "POSITIONAL_OR_KEYWORD", + "required": true, + "default": null + }, + { + "name": "confidence", + "kind": "POSITIONAL_OR_KEYWORD", + "required": true, + "default": null + }, + { + "name": "engine", + "kind": "POSITIONAL_OR_KEYWORD", + "required": true, + "default": null + } + ], + "fields": ["type", "text", "start", "end", "confidence", "engine"] + }, + "datafog.engine:ScanResult": { + "parameters": [ + { + "name": "entities", + "kind": "POSITIONAL_OR_KEYWORD", + "required": true, + "default": null + }, + { + "name": "text", + "kind": "POSITIONAL_OR_KEYWORD", + "required": true, + "default": null + }, + { + "name": "engine_used", + "kind": "POSITIONAL_OR_KEYWORD", + "required": true, + "default": null + } + ], + "fields": ["entities", "text", "engine_used"] + }, + "datafog.engine:RedactResult": { + "parameters": [ + { + "name": "redacted_text", + "kind": "POSITIONAL_OR_KEYWORD", + "required": true, + "default": null + }, + { + "name": "mapping", + "kind": "POSITIONAL_OR_KEYWORD", + "required": true, + "default": null + }, + { + "name": "entities", + "kind": "POSITIONAL_OR_KEYWORD", + "required": true, + "default": null + } + ], + "fields": ["redacted_text", "mapping", "entities"] + }, + "datafog.engine:scan": { + "parameters": [ + { + "name": "text", + "kind": "POSITIONAL_OR_KEYWORD", + "required": true, + "default": null + }, + { + "name": "engine", + "kind": "POSITIONAL_OR_KEYWORD", + "required": false, + "default": "smart" + }, + { + "name": "entity_types", + "kind": "POSITIONAL_OR_KEYWORD", + "required": false, + "default": null + }, + { + "name": "locales", + "kind": "POSITIONAL_OR_KEYWORD", + "required": false, + "default": null + }, + { + "name": "allowlist", + "kind": "POSITIONAL_OR_KEYWORD", + "required": false, + "default": null + }, + { + "name": "allowlist_patterns", + "kind": "POSITIONAL_OR_KEYWORD", + "required": false, + "default": null + } + ] + }, + "datafog.engine:redact": { + "parameters": [ + { + "name": "text", + "kind": "POSITIONAL_OR_KEYWORD", + "required": true, + "default": null + }, + { + "name": "entities", + "kind": "POSITIONAL_OR_KEYWORD", + "required": true, + "default": null + }, + { + "name": "strategy", + "kind": "POSITIONAL_OR_KEYWORD", + "required": false, + "default": "token" + } + ] + }, + "datafog.engine:scan_and_redact": { + "parameters": [ + { + "name": "text", + "kind": "POSITIONAL_OR_KEYWORD", + "required": true, + "default": null + }, + { + "name": "engine", + "kind": "POSITIONAL_OR_KEYWORD", + "required": false, + "default": "smart" + }, + { + "name": "entity_types", + "kind": "POSITIONAL_OR_KEYWORD", + "required": false, + "default": null + }, + { + "name": "strategy", + "kind": "POSITIONAL_OR_KEYWORD", + "required": false, + "default": "token" + }, + { + "name": "locales", + "kind": "POSITIONAL_OR_KEYWORD", + "required": false, + "default": null + }, + { + "name": "allowlist", + "kind": "POSITIONAL_OR_KEYWORD", + "required": false, + "default": null + }, + { + "name": "allowlist_patterns", + "kind": "POSITIONAL_OR_KEYWORD", + "required": false, + "default": null + } + ] + } + }, + "warnings": [] + } + }, + { + "id": "anonymize-text-replace", + "classification": "compatibility", + "target": "datafog:anonymize_text", + "kwargs": { + "text": "alice@example.com", + "method": "replace" + }, + "expected": { + "value": "[EMAIL_PSEUDO_1]", + "warnings": [] + } + }, + { + "id": "anonymize-text-hash", + "classification": "compatibility", + "target": "datafog:anonymize_text", + "kwargs": { + "text": "alice@example.com", + "method": "hash" + }, + "expected": { + "value": "[EMAIL_ff8d9819fc0e]", + "warnings": [] + } + }, + { + "id": "anonymize-text-invalid", + "classification": "compatibility", + "target": "datafog:anonymize_text", + "kwargs": { + "text": "alice@example.com", + "method": "invalid" + }, + "expected": { + "error": { + "type": "ValueError", + "message": "Invalid method: invalid. Use 'redact', 'replace', or 'hash'" + }, + "warnings": [] + } + }, + { + "id": "locale-alias-de-DE", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "Steuer-ID 12345678901 liegt vor.", + "locales": ["de-DE"] + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "DE_TAX_ID", + "text": "12345678901", + "start": 10, + "end": 21, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "Steuer-ID 12345678901 liegt vor.", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "locale-alias-de_de", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "Steuer-ID 12345678901 liegt vor.", + "locales": ["de_de"] + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "DE_TAX_ID", + "text": "12345678901", + "start": 10, + "end": 21, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "Steuer-ID 12345678901 liegt vor.", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "allowlist-lookahead", + "classification": "compatibility", + "target": "datafog:scan", + "kwargs": { + "text": "alice@example.com bob@example.org", + "allowlist_patterns": ["(?=alice).*"] + }, + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "EMAIL", + "text": "bob@example.org", + "start": 18, + "end": 33, + "confidence": 1.0, + "engine": "regex" + } + } + ], + "text": "alice@example.com bob@example.org", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + { + "id": "redact-unicode-explicit-mask", + "classification": "compatibility", + "target": "datafog:redact", + "kwargs": { + "text": "👩🏽‍💻é", + "strategy": "mask", + "entities": [ + { + "type": "PERSON", + "text": "👩🏽‍💻é", + "start": 0, + "end": 6, + "confidence": 1.0, + "engine": "regex" + } + ] + }, + "expected": { + "value": { + "result_type": "RedactResult", + "fields": { + "redacted_text": "******", + "mapping": { + "[PERSON_MASK_1]": "👩🏽‍💻é" + }, + "entities": [ + { + "result_type": "Entity", + "fields": { + "type": "PERSON", + "text": "👩🏽‍💻é", + "start": 0, + "end": 6, + "confidence": 1.0, + "engine": "regex" + } + } + ] + } + }, + "warnings": [] + } + } + ], + "capture_environment": { + "python": "3.12.13", + "implementation": "CPython", + "dependencies": { + "pydantic": "2.13.5", + "pydantic-core": "2.46.5", + "pydantic-settings": "2.15.0", + "typing-extensions": "4.16.0", + "annotated-types": "0.8.0", + "python-dotenv": "1.2.3", + "typing-inspection": "0.4.4" + } + } +} diff --git a/tests/contracts/requirements-4.8.1.txt b/tests/contracts/requirements-4.8.1.txt new file mode 100644 index 00000000..f1dc083c --- /dev/null +++ b/tests/contracts/requirements-4.8.1.txt @@ -0,0 +1,9 @@ +# Dependencies used when capturing the published wheel on CPython 3.12.13. +# These pins belong to the oracle only, not to the product's dependency policy. +annotated-types==0.8.0 +pydantic==2.13.5 +pydantic-core==2.46.5 +pydantic-settings==2.15.0 +python-dotenv==1.2.3 +typing-extensions==4.16.0 +typing-inspection==0.4.4 diff --git a/tests/test_contract_481.py b/tests/test_contract_481.py new file mode 100644 index 00000000..427401dd --- /dev/null +++ b/tests/test_contract_481.py @@ -0,0 +1,17 @@ +"""Frozen observations from the published 4.8.1 wheel; never regenerate in CI.""" + +import json + +import pytest + +from tests.contract_481 import FIXTURE, observe + +CONTRACT = json.loads(FIXTURE.read_text()) + + +@pytest.mark.parametrize("case", CONTRACT["cases"], ids=lambda case: case["id"]) +def test_published_481_contract(case): + assert observe(case) == case["expected"], ( + f"4.8.1 contract changed: {case['id']} ({case['classification']}). " + "Review the migration policy before changing the frozen baseline." + ) From 4129613d97f27cdc365f866a95b9c2d20e500e62 Mon Sep 17 00:00:00 2001 From: Sid Mohan <61345237+sidmohan0@users.noreply.github.com> Date: Mon, 28 Sep 2026 14:49:06 -0700 Subject: [PATCH 27/37] fix: include GLiNER tokenizer dependencies in NLP extra --- README.md | 3 +++ setup.py | 2 ++ tests/test_install_profiles.py | 5 +++++ 3 files changed, 10 insertions(+) diff --git a/README.md b/README.md index 29179119..8b9718fe 100644 --- a/README.md +++ b/README.md @@ -70,6 +70,9 @@ pip install datafog[distributed] pip install datafog[all] ``` +The `nlp-advanced` extra includes SentencePiece and protobuf for GLiNER's +multilingual tokenizer; installing the OCR extra is not required for GLiNER. + Python 3.13 support is certified for the core SDK, CLI, `nlp`, `nlp-advanced`, and `ocr` install profiles. Donut OCR still requires a model that is available locally before runtime use. `distributed` and `all` remain diff --git a/setup.py b/setup.py index 1b4723ef..1bf72344 100644 --- a/setup.py +++ b/setup.py @@ -34,6 +34,8 @@ "torch>=2.1.0,<2.7", "transformers>=4.20.0", "huggingface-hub>=0.16.0", + "sentencepiece>=0.2.0", + "protobuf>=4.0.0", ] ocr_deps = [ diff --git a/tests/test_install_profiles.py b/tests/test_install_profiles.py index 2680543b..a34259d9 100644 --- a/tests/test_install_profiles.py +++ b/tests/test_install_profiles.py @@ -35,12 +35,17 @@ def test_install_profile_import_surface() -> None: assert SpacyPIIAnnotator is not None elif profile == "nlp-advanced": import gliner # noqa: F401 + import sentencepiece # noqa: F401 import torch # noqa: F401 import transformers # noqa: F401 + from transformers.convert_slow_tokenizer import import_protobuf from datafog.processing.text_processing.gliner_annotator import GLiNERAnnotator assert GLiNERAnnotator is not None + # GLiNER's multilingual DeBERTa tokenizer needs the protobuf schema, + # even when no OCR extra is installed. + assert import_protobuf().ModelProto is not None elif profile == "ocr": import numpy # noqa: F401 import pytesseract # noqa: F401 From 46fd8b0115f06ee497576e4e39b2902aeba2e995 Mon Sep 17 00:00:00 2001 From: Sid Mohan <61345237+sidmohan0@users.noreply.github.com> Date: Mon, 28 Sep 2026 15:15:55 -0700 Subject: [PATCH 28/37] docs: specify the 4.9 bridge and German Core requirements --- GERMAN-PII-CORE-REQUIREMENTS.md | 429 ++++++++++++++++++++++++++++++++ PLAN-4.9.md | 145 +++++++++++ 2 files changed, 574 insertions(+) create mode 100644 GERMAN-PII-CORE-REQUIREMENTS.md create mode 100644 PLAN-4.9.md diff --git a/GERMAN-PII-CORE-REQUIREMENTS.md b/GERMAN-PII-CORE-REQUIREMENTS.md new file mode 100644 index 00000000..4f9e307e --- /dev/null +++ b/GERMAN-PII-CORE-REQUIREMENTS.md @@ -0,0 +1,429 @@ +# Agent brief: German structured PII detection in DataFog Core + +## Objective and scope + +Implement all seven locale-gated German entity types listed below in +`datafog-core`, with identical +behavior across Rust, Python, Node.js, and browser/WASM bindings. Make it usable +through text scanning, structured scanning, and existing transformation APIs. +All detection logic belongs in `crates/core`; bindings remain thin. + +This brief proposes concrete policy choices for implementation. In particular, +it specifies format/context-based PII detection rather than official identifier +or bank-account validation. Passport and residence-permit patterns are explicitly +legacy-compatible heuristics, not exhaustive national document-number coverage. +It does not authorize unrelated detector changes or a global locale redesign. +Do not implement this work in the legacy Python detector. + +## Existing behavior and compatibility context + +- Python 4.8.1 has `DE_IBAN` and six other locale-gated German labels. +- Its IBAN pattern is case-insensitive, accepts compact or optionally grouped + values, and does not validate a checksum. +- Core currently accepts `ScanConfig.locale`, but `scan_with_config` ignores it. +- Core scanning returns ordered findings and can retain overlapping candidates; + transformation selection resolves overlaps separately. Preserve that contract. +- Core already supplies byte/code-point offsets, binding-specific UTF-16 ranges, + structured field paths, transformations, and detector provenance. Reuse these. + +The German structure is 22 characters when normalized: `DE`, two check digits, +an eight-digit bank code, and a ten-digit account component. See the +[Deutsche Bundesbank's IBAN explanation](https://www.bundesbank.de/en/tasks/payment-systems/services/sepa/content/content-831778?index=1). + +## R1. Entity identity and findings + +Add these exact canonical labels and detector names: + +| Entity label | Detector name | +| ---------------------------- | ----------------------------------------- | +| `DE_IBAN` | `datafog-core/de-iban` | +| `DE_VAT_ID` | `datafog-core/de-vat-id` | +| `DE_TAX_ID` | `datafog-core/de-tax-id` | +| `DE_SOCIAL_SECURITY_NUMBER` | `datafog-core/de-social-security-number` | +| `DE_POSTAL_CODE` | `datafog-core/de-postal-code` | +| `DE_PASSPORT_NUMBER` | `datafog-core/de-passport-number` | +| `DE_RESIDENCE_PERMIT_NUMBER` | `datafog-core/de-residence-permit-number` | + +For every new detector: + +- Use the current crate version for `detector_version`, as existing detectors do. +- Leave confidence absent (`None`/`null`); do not invent a numeric probability. +- `matched_text` must be the exact source substring, including original casing + and internal separators. Never return a normalized substitute. +- Emit one finding per occurrence, with the existing ordering/deduplication rules. + +## R2. Activation and locale compatibility + +- Enable all seven detectors when `locale`, trimmed and compared case-insensitively, + is `de`, `de-DE`, or `de_DE`. +- Omitted locale does not enable any `DE_*` detector. Existing base detectors still run. +- Other already accepted nonempty locale values retain existing behavior and + do not activate German detection. Do not begin rejecting `en-US` or other + values as a side effect of this feature. +- Preserve existing malformed-config and empty-locale errors. +- Normalize for routing without changing the public stored locale value unless + existing API tests demonstrate that such normalization is already expected. +- Use the same routing for plain scans and structured string-value scans. +- Keep Core's current config shape: `{"locale": "de"}`. Do not introduce Python's + plural `locales`, a new engine selector, or scan-time entity selection here. +- Transformation entity selection does not activate a detector. A caller wanting + to scan and transform only German IBANs supplies both the German scan locale + and `transform.entities: ["DE_IBAN"]`. + +## R3. German IBAN lexical forms + +Use the following logical grammar, with ASCII digits only: + +```text +[Dd][Ee][0-9]{2}(SEP?[0-9]{4}){4}SEP?[0-9]{2} +SEP := one U+0020 SPACE, U+0009 TAB, U+00A0 NO-BREAK SPACE, + or U+202F NARROW NO-BREAK SPACE +``` + +- Accept compact, fully grouped, and partially grouped values. Each separator + is optional independently, but only at the group boundaries in the grammar. +- Do not require a context keyword such as `IBAN` or `Bankverbindung`. +- Do not include context labels, surrounding punctuation, or outer whitespace + in the match. +- Reject separators after `DE` but before the two check digits, multiple adjacent + separators, arbitrary grouping, hyphens, periods, and embedded newlines. +- Reject non-ASCII digits, other country prefixes, and wrong normalized length. +- Reject a candidate if its immediately preceding or following character is an + ASCII letter or digit. Start/end of input and punctuation are valid boundaries. + This preserves the legacy ASCII boundary convention; do not add Unicode-word + boundary behavior implicitly. +- Do not extract a valid-length prefix from a longer contiguous alphanumeric + identifier. Inspect the source boundary, not only the regex capture. +- Any normalization used internally must not change returned text or offsets. + +These separator and ASCII-digit rules deliberately narrow Python's broad `\s` +and Unicode `\d` acceptance. Document these as explicit migration differences: +newline-spanning and other Unicode-whitespace/digit matches are not promised. +Do not describe this feature as exact regex parity with Python 4.8.1. + +## R4. Identifier validation policy + +- Detect every candidate satisfying R3, including checksum-invalid candidates. +- Do not perform bank-directory lookup, account existence checks, network calls, + or model downloads. +- Do not silently require MOD-97 validity; this would drop PII-like values that + Python currently detects, including transcription errors. +- Explain in documentation that detection identifies sensitive-looking text and + does not establish that an account is valid or exists. +- A strict validation mode or separate validation API is outside this PR. +- Apply the same format-detection policy to the other six entities: no VAT/tax + checksum requirements, pension date validation, postal directory lookup, or + document issuance validation. Context and lexical rules below are mandatory. + +## R5. Offset and structured-data requirements + +- UTF-8 byte ranges and Unicode code-point ranges must both address the exact + original substring, with zero-based, end-exclusive offsets. +- Node and WASM UTF-16 ranges must work with JavaScript string slicing. +- Test emoji, combining marks, and CJK text before every entity type; ASCII-only fixtures + are insufficient to verify these units. +- In structured scans, preserve the existing JSON Pointer path and string-local + offset contract. Test two fields and an array element. +- Never concatenate separate fields to create a candidate. Required context must + occur in the same string value, not in a sibling field or JSON property name. + +## R6. Transformation integration + +- `scan_and_transform` must honor the scan locale and produce `[DE_IBAN]` for + the existing Core redaction strategy. Do not introduce numbered placeholders. +- `transform` must accept explicit `DE_IBAN` findings without rescanning or + requiring a locale; the locale controls detection only. +- Ensure the new label works with entity selection, per-entity overrides, exact + and full-match regex allowlists, and existing strategy/provider validation. +- Verify redaction, masking, and removal end to end; exercise provider-backed + strategies with existing test providers on supported runtimes. Do not broaden + WASM support for provider-backed strategies. +- Preserve source/output ranges and the rule that transformation records omit + original matched PII. +- Do not globally suppress generic PHONE/SSN/etc. scan findings inside IBANs. + With default transformation selection, existing overlap handling must prefer + the containing IBAN span and produce one replacement for it. Add a regression + fixture if an overlapping base detector is present. +- For IBAN-only transformation tests, set `entities: ["DE_IBAN"]` so the expected + output does not depend on unrelated detector findings. Exact allowlisting uses + the source value, not an automatically normalized IBAN. +- Apply all preceding transformation requirements to every new label, using + `[DE_VAT_ID]`, `[DE_TAX_ID]`, etc. as their redaction placeholders. +- Context text must survive replacements for tax, social-insurance, passport, + and residence-permit findings. Postal-code matches intentionally include their + prefix, so their replacement removes that prefix as well. +- Add overlap cases for VAT versus generic SSN, tax ID versus generic PHONE, and + German-prefixed postal code versus generic ZIP_CODE. Scanning may return both; + transformation must use the existing selection/overlap rules. For equal-span + built-in findings with absent confidence, the current lexical tie-break should + prefer `DE_TAX_ID` over `PHONE`. Test this; do not introduce a blanket new + priority rule that changes unrelated or caller-supplied findings. + +## R7. Required IBAN acceptance examples + +The expected count below refers to `DE_IBAN` only. Existing detectors may still +return other entity types, especially with no German locale or malformed input. + +| Input / configuration | Expected DE_IBAN behavior | +| ------------------------------------------------------- | ------------------------------------------- | +| `DE44500105175407324931`, locale `de` | One exact match | +| `DE44 5001 0517 5407 3249 31`, locale `de` | One match including internal spaces | +| `de44 5001 0517 5407 3249 31`, locale `DE-de` | One match preserving lowercase text | +| `DE44 50010517 54073249 31`, locale `de_DE` | One partially grouped match | +| Grouped value using each supported SEP | One match for each variant | +| `(DE44500105175407324931).`, locale `de` | Match excludes punctuation | +| Same IBAN twice, locale `de` | Two findings at different source ranges | +| Valid-shaped IBAN preceded by emoji/CJK/combining text | Correct byte, code-point, and UTF-16 ranges | +| `DE45500105175407324931`, locale `de` | One match despite altered checksum digits | +| A valid input without locale or with `en-US` | No DE_IBAN finding | +| `DE4450010517540732493`, locale `de` | No DE_IBAN: too short | +| `DE445001051754073249310`, locale `de` | No DE_IBAN: too long, no prefix match | +| `XDE44500105175407324931` or valid IBAN followed by `X` | No DE_IBAN: embedded identifier | +| `DE44-5001-0517-5407-3249-31` | No DE_IBAN | +| `DE44 5001 0517 5407 3249 31` | No DE_IBAN: double separator | +| `DE44\n5001 0517 5407 3249 31` | No DE_IBAN: newline is not SEP | +| `DE 44 5001 0517 5407 3249 31` | No DE_IBAN: misplaced separator | +| Same shape using Arabic-Indic/fullwidth digits | No DE_IBAN | +| Non-German prefix with the same trailing digit count | No DE_IBAN | + +Also test empty input, repeated near-matches, long digit runs, multiple adjacent +IBANs separated by punctuation, malformed config, and structured transformation. + +## R8. Conformance, performance, and delivery + +- Put positive/negative expectations in shared fixtures consumed by Rust and all + applicable bindings. Existing plain fixture runners call scan without config; + extend them to accept optional fixture scan config with unchanged defaults, + or introduce a focused locale fixture suite used by every binding. +- Run existing fixture suites unchanged. Do not rewrite old expected detections + to accommodate unrelated changes. +- Keep matching bounded and linear in input size; compile patterns once and + avoid per-candidate whole-input copying. No new dependency without a concrete + need that existing Rust facilities cannot reasonably meet. +- Compare no-locale scanning and German scanning on short, mixed, long, and + adversarial synthetic texts. Investigate a reproducible greater-than-10% + regression in existing no-locale workloads before merging. +- Run `cargo fmt --all --check`, + `cargo clippy --workspace --all-targets --all-features -- -D warnings`, and + `cargo test --workspace --all-features`. +- Run installed Python/Node binding tests and browser/WASM package tests, not + merely Rust unit tests. Verify config forwarding and offset conversions. +- Update README, entity/configuration references, binding type declarations if + needed, and migration documentation. State the checksum and separator policy. +- Deliver a focused PR with test results, performance observations, and known + migration differences. Do not publish packages without release authorization. +- Python 4.9 integration requires a subsequently published compatible Core wheel; + a source-only Core change is not sufficient to update the Python extra pin. + +## R9. Shared lexical and context rules for the other six entities + +Use these definitions in R10-R15: + +```text +D := one ASCII digit [0-9] +L := one ASCII letter [A-Za-z] +H := one of the four horizontal separators defined as SEP in R3 +GAP := H* [:#-]? H* +``` + +- All literal prefixes and context labels match ASCII case-insensitively, with + original casing preserved in findings. Do not use Unicode case folding that + expands the accepted alphabet. +- Apply R3's immediate ASCII-alphanumeric boundary checks to every value span. + Consequently a label ending in a letter needs whitespace or punctuation before + a value; `IdNr.12345678901` is allowed, `IdNr12345678901` is not. +- For context-required types, accept only `CONTEXT GAP VALUE`, with no intervening + prose or newline. A context label must not begin inside an ASCII word/number. + Include neither context nor GAP in the returned finding. +- A bare identifier or an unrelated preceding label must not activate a + context-required detector. German locale alone does not supply missing context. +- Do not search an entire document or previous line for context, nor use one + context marker to activate an arbitrary list of later identifiers. +- A structured key such as `tax_id` is not textual context for these detectors; + schema-driven detection is a separate feature. +- H may repeat in GAP and where explicitly written H+, but value-group separators + written H? allow at most one character. No newline, vertical whitespace, + Unicode digit, extra letter/digit, or punctuation substitution is accepted. +- These restrictions deliberately narrow legacy Python's Unicode `\d`, broad + `\s`, and context-substring matching. Record the differences in migration docs + and fixtures rather than changing the frozen 4.8.1 expectations. + +## R10. German VAT identifier: DE_VAT_ID + +```text +VALUE := DE (H | -)? D{9} +CONTEXT := not required +``` + +- Return the complete DE prefix, optional separator, and nine digits. +- Do not group the nine digits internally or accept multiple prefix separators. +- Examples with German locale: + +| Input | Expected matched text, or no DE_VAT_ID | +| -------------------------------------------------- | -------------------------------------- | +| `USt-IdNr DE123456789 ist gesetzt.` | `DE123456789` | +| `USt-IdNr DE 123456789 ist gesetzt.` | `DE 123456789` | +| `de-123456789` | `de-123456789` | +| `(DE123456789)` | `DE123456789` | +| `DE12345678` / `DE1234567890` / `DE123456789A` | None | +| `XDE123456789` / `DE--123456789` / `DE 123456789` | None | +| `DE123 456 789` / `AT123456789` | None | + +## R11. German tax identifier: DE_TAX_ID + +```text +VALUE := D{11} | D{2} H? D{3} H? D{3} H? D{3} +CONTEXT := Steuer(H|-)?ID | Steueridentifikationsnummer | + Identifikationsnummer | IdNr[.]? | Tax(H|-)?ID +``` + +- Return only the value, preserving optional internal separators. +- This is the legacy eleven-digit personal tax-ID detector; do not add tax-office + Steuernummer, slash-separated identifiers, or business identifiers to this label. + +| Input | Expected matched text, or no DE_TAX_ID | +| -------------------------------------------------------- | -------------------------------------- | +| `Steuer-ID 12345678901 liegt vor.` | `12345678901` | +| `Steueridentifikationsnummer: 12 345 678 901` | `12 345 678 901` | +| `IdNr.12345678901` | `12345678901` | +| `tax id # 12345678901` | `12345678901` | +| `Invoice 12345678901` / bare `12345678901` | None | +| `Steuer-ID 1234567890` / `Steuer-ID 123456789012` | None | +| `Steuer-ID A12345678901` / `Steuer-ID 12345678901Z` | None | +| `Steuer-ID 123 456 789 01` / `Steuer-ID 12 345 678 901` | None | +| `NotSteuer-ID 12345678901` / `Steuer-ID\n12345678901` | None | + +## R12. German social-insurance identifier: DE_SOCIAL_SECURITY_NUMBER + +```text +VALUE := D{2} H? D{6} H? L H? D{3} +CONTEXT := Rentenversicherungsnummer | Sozialversicherungsnummer | RVNR | SVNR +``` + +- Return only the value, preserving grouping and letter case. +- This is the legacy pension/social-insurance pattern. Do not use it to identify + health-insurance numbers, generic US SSNs, or arbitrary alphanumeric IDs. + +| Input | Expected matched text, or no DE_SOCIAL_SECURITY_NUMBER | +| --------------------------------------------------- | ------------------------------------------------------ | +| `Rentenversicherungsnummer 65150804A123 liegt vor.` | `65150804A123` | +| `SVNR: 65 150804 A123` | `65 150804 A123` | +| `rvnr # 65 150804 a 123` | `65 150804 a 123` | +| `Build 65150804A123 failed.` / bare `65150804A123` | None | +| `RVNR 65150804A12` / `RVNR 65150804A1234` | None | +| `RVNR 651508041123` / `RVNR 65150804AA123` | None | +| `RVNR 65-150804-A123` / `RVNR\n65150804A123` | None | + +## R13. German-prefixed postal code: DE_POSTAL_CODE + +```text +VALUE := (PLZ (H | : | -)? | DE (H | -) | D (H | -)) D{5} +CONTEXT := not required beyond the prefix included in VALUE +``` + +- Return the prefix, optional/required separator, and five digits together. + This deliberately preserves Python's span semantics, even though other types + exclude context. Do not silently shorten the match to the digits. +- A bare five-digit value does not become DE_POSTAL_CODE merely because locale + is German. Existing generic ZIP_CODE detection may still apply. +- Do not add city inference, a postal database, or validity checks. + +| Input | Expected matched text, or no DE_POSTAL_CODE | +| ----------------------------------------- | ------------------------------------------- | +| `PLZ10115 Berlin.` / `PLZ:10115 Berlin.` | `PLZ10115` / `PLZ:10115` | +| `PLZ 10115` / `PLZ-10115` | Complete corresponding value | +| `DE-10115 Berlin.` / `D 10115 Berlin.` | `DE-10115` / `D 10115` | +| `de 10115` | `de 10115` | +| `10115 Berlin` / `DE10115` / `D10115` | None | +| `PLZ1011` / `PLZ101150` / `PLZ10115A` | None | +| `SKU D12345` / `Release DE12345` | None | +| `PLZ: 10115` / `DE--10115` / `PLZ 10115` | None | + +`PLZ: 10115` is a useful possible future enhancement, but the legacy pattern +allows only one prefix separator. Keep its rejection explicit in this scope; +expanding it requires a separately documented detection change. + +## R14. Passport-context identifier: DE_PASSPORT_NUMBER + +```text +VALUE := L D{8} +CONTEXT := Passnummer | Reisepass(nummer)? | Passport(H+ No[.]? | H+ Number)? +``` + +The accepted labels include `Reisepass`, `Reisepassnummer`, +`Passport`, `Passport No.`, and `Passport Number`. + +- Return only the value. This is the inherited one-letter/eight-digit heuristic; + it does not claim to cover all actual German passport-number formats. +- Do not add Personalausweis detection or broader alphanumeric document formats + under this label without separately sourced requirements and fixtures. + +| Input | Expected matched text, or no DE_PASSPORT_NUMBER | +| -------------------------------------------------- | ----------------------------------------------- | +| `Passnummer C12345678 wurde geprueft.` | `C12345678` | +| `Reisepassnummer: c12345678` | `c12345678` | +| `Passport No. # A12345678` | `A12345678` | +| `Ticket A12345678 was shipped.` / bare `C12345678` | None | +| `Passnummer C1234567` / `Passnummer C123456789` | None | +| `Passnummer CC1234567` / `Passnummer 123456789` | None | +| `Passnummer C12A45678` / `Passnummer\nC12345678` | None | + +## R15. Residence-permit-context identifier: DE_RESIDENCE_PERMIT_NUMBER + +```text +VALUE := AT D{7} +CONTEXT := Aufenthaltstitel | Aufenthaltserlaubnis | Residence H+ Permit | eAT +``` + +- Return only the AT-prefixed value; no separator after AT is allowed. +- This is the inherited AT-plus-seven-digits heuristic, not a statement of + exhaustive or authoritative residence-permit numbering rules. + +| Input | Expected matched text, or no DE_RESIDENCE_PERMIT_NUMBER | +| ------------------------------------------------- | ------------------------------------------------------- | +| `Aufenthaltstitel AT1234567 gueltig.` | `AT1234567` | +| `Aufenthaltserlaubnis: at1234567` | `at1234567` | +| `Residence Permit # AT1234567` / `eAT-AT1234567` | `AT1234567` | +| `Order AT1234567 is internal.` / bare `AT1234567` | None | +| `eAT AT123456` / `eAT AT12345678` | None | +| `eAT AT 1234567` / `eAT DE1234567` | None | +| `eAT AT1234567X` / `eAT\nAT1234567` | None | + +## R16. Completion matrix and Python migration boundary + +For each of the seven labels, require shared fixtures proving: + +1. German locale aliases activate it; omitted/non-German locale does not. +2. All listed accepted forms and context aliases work; wrong lengths, unrelated + context, embedded ASCII identifiers and forbidden separators do not. +3. Matched text and byte/code-point/UTF-16 ranges preserve original input. +4. Repeated occurrences and structured values retain separate locations. +5. Selection, allowlists, overrides and supported transformations work, with + context retained or removed exactly as specified for the entity. +6. At least one synthetic input contains all seven types, with expected ordered + German findings and an end-to-end transformation result across bindings. +7. The seven-type suite passes against installed packages, with no network or + models required at inference time and no regression to existing fixtures. + +Every rejection assertion concerns the target German label, not necessarily an +empty overall scan. Generic detectors may legitimately identify the same input. + +Use the published Python 4.8.1 contract and `tests/test_de_pii_regex.py` as +comparison evidence. Keep their source fixtures unchanged. Record intentional +whitespace, digit-alphabet and context-boundary differences in a separate matrix. + +Python enables German detectors through explicit entity selection even without +a locale. Core currently selects entities only for transformations. Keep that +Core API distinction: the Python adapter may route explicit German entity +requests to locale `de` then filter, but must separately test legacy overlap and +selection semantics. Do not add a new scan selector casually within this work. + +Implementation may be split into focused PRs: locale routing plus IBAN/VAT; +contextual tax/social-insurance detectors; postal/document heuristics; then +cross-binding conformance and documentation. All seven must meet this brief +before claiming coverage of the legacy German entity set or unblocking general +German locale requests in the 4.9 adapter. Partial delivery must stay explicit. + +Out of scope: identifiers beyond these seven, exhaustive official format +validation, checksum-gated detection, language inference, schema-key inference, +and changes to existing generic detectors or Core's global overlap policy. diff --git a/PLAN-4.9.md b/PLAN-4.9.md new file mode 100644 index 00000000..c41739fa --- /dev/null +++ b/PLAN-4.9.md @@ -0,0 +1,145 @@ +# DataFog 4.9: incremental Core migration + +## Outcome + +Deliver a bridge release that lets users adopt Rust detection and the Core schema +without changing the default Python behavior. Keep the published 4.8.1 contract +as immutable evidence, with narrowly documented 4.9 API additions and revised +deprecation messages. This work does not publish a release or perform the 5.0 +cutover. + +## Release decisions + +- Keep `pip install datafog` and existing top-level imports working in 4.9. +- Add optional `datafog[rust]`, pinned to the published Core version verified by + this change. Loading the base package must not import the native extension. +- Add keyword-only `backend="python"` to scan/redact entry points. Rust is an + explicit, experimental alternative for `engine="regex"` only. +- Introduce `datafog.compat.v4` for existing scan/redact APIs and result classes; + top-level scan/redact delegate there while preserving class identity. +- Introduce `datafog.v5` as a preview of Core's actual public types and functions, + without translating them into legacy result objects or strategies. +- Revise `detect()` and `process()` warnings: removal is planned for 5.0, replacing + the earlier promise to retain them throughout 5.x. Explain the change publicly. +- Deprecate OCR and Spark in 4.9; remove them in 5.0. Preserve existing optional + installations and behavior during 4.9, and warn at meaningful use sites. +- Leave ML engines, smart/auto composition, CLI text commands, application + integrations, and service APIs on their existing backend in this increment. +- Keep the existing package version until release preparation; this branch + implements 4.9 behavior but does not publish or stamp a final release. + +## Increment 1: opt-in Rust detection + +1. Extend `datafog.engine.scan` and `scan_and_redact` with keyword-only backend + selection, preserving original parameters, validation and Python defaults. +2. Lazily call the installed `datafog_core.scan`; convert findings to existing + Entity objects using code-point offsets, not byte offsets. Preserve document + order, original text, and legacy regex provenance/confidence conventions. +3. Keep Python allowlist validation/filtering, aliases and transformation logic. + Do not reimplement detectors in the adapter or invoke Core transformations. +4. Reject Rust plus non-regex engines, unsupported German locale/entity requests, + and invalid backend values explicitly. Missing native dependencies must give + an actionable installation error. Never silently retry with Python or return + an empty result on native failure. +5. Explicit-entity redaction must not scan or import the native extension. +6. Add focused tests for routing, Unicode, validation, selection, overlaps, + allowlists, replacement strategies, missing dependencies and propagated errors. + +## Increment 2: compatibility namespace and schema preview + +1. Make `datafog.compat.v4` the facade for existing scan/redact return shapes and + result classes; retain legacy engine implementation in place where relocation + would break internal imports or monkeypatching contracts. +2. Keep top-level functions and types available and forward backend selection. + Existing agent convenience functions already accept forwarding keyword args. +3. Export the tested Core API through `datafog.v5`, preserving native type + identity. No Core import on plain `import datafog`; clear optional-dependency + errors when the preview is used without the extra. +4. Update detect/process warnings and test the explicit revised removal policy. + Do not expose these functions through the new preview. +5. Test import isolation, facade identity, unchanged legacy results, and real + Core scanning/transformation through the preview. + +## Increment 3: retirement notices for optional legacy surfaces + +1. Warn on meaningful OCR/Spark use, including direct supported entry points, + without importing heavy dependencies merely to issue a warning. +2. Keep those extras and functionality available throughout 4.9. Document removal + in 5.0 and the option to remain on the final 4.x release. Do not create new + replacement packages as part of this work. +3. Add tests proving notices are emitted and core text imports stay unaffected. + +## Increment 4: integration and evidence + +1. Keep the 4.8.1 fixture unchanged. Represent approved signature additions and + warning-message changes explicitly in the 4.9 contract checker; compare all + other observed behavior exactly. No blanket skips or recapture from dev. +2. Run applicable cases through the real published Core wheel and produce a + per-case parity report. Separate matches, deliberate unsupported requests, + known detector differences and regressions. Experimental Rust detection must + not be advertised as a drop-in equivalent before gaps are closed. +3. Add CI jobs with and without the Rust extra; smoke-test installed wheels and + native preview behavior on supported Python/platform combinations. +4. Benchmark the public Python and Rust-backed APIs, including conversion and + cold startup, with short, mixed and large synthetic inputs. Keep results + reproducible and make no unsupported speedup claim. +5. Update README, migration documentation and release notes for the actual scope. +6. Run focused suites, broader core regression tests, applicable benchmarks and + pre-commit checks; open a PR against dev and resolve CI failures. + +## Work ownership + +- Backend agent: engine backend adapter and focused backend tests. +- API agent: compatibility namespace, preview API, top-level delegation and + detect/process retirement warnings, with focused tests. +- Retirement agent: OCR/Spark notices and related tests/documentation. +- Coordinator: packaging, contract integration, parity report, CI, benchmarks, + cross-agent review, final verification and PR. + +Agents share one feature branch and have disjoint file ownership. They must not +commit, push, merge, overwrite another agent's files or edit frozen fixtures. +The coordinator reviews and integrates each change before committing. + +## Completion criteria + +- Default legacy behavior is unchanged except for the documented revised notices. +- Rust use is explicit, installation is optional, failures are visible, and + coverage limitations are documented and exercised by tests. +- Compatibility and native-preview APIs coexist without duplicated native builds. +- OCR/Spark and detect/process have actionable 5.0 retirement notices. +- Tests, installed-wheel checks, parity evidence and benchmarks are reproducible. +- A reviewed, passing PR is ready for dev; merging is subject to user direction + and existing repository approval rules. + +## Confirmed scope and delivery + +The user confirmed retaining deprecated OCR/Spark functionality in 4.9 and +removing it in 5.0. The coordinator is authorized to merge after checks and +required approvals pass; repository protections remain in effect. + +German detector expansion is specified separately in +`GERMAN-PII-CORE-REQUIREMENTS.md`. Implementing or publishing those Core changes +is outside this Python bridge increment. Rust calls requiring that unavailable +coverage must continue to fail explicitly until a tested Core release supports it. + +## Implementation evidence + +All four increments are implemented on `feature/4.9-core-migration`. Backend, +API, and retirement changes were delegated with disjoint ownership; cross-review +also corrected Windows UTF-8 fixture handling and default-filter CLI notice +visibility. The frozen 4.8.1 fixture is unchanged. + +- Base/CLI regression run: 782 passed, with expected optional-dependency skips + and pre-existing corpus xfails. +- Focused native integration run: 347 passed against published Core 0.3.1. +- Python 3.10 and 3.14 base-only runs: 181 passed each; native/CLI-specific tests + skip when their optional dependencies are absent. +- Clean installed-wheel smoke test: compatibility facade, native detection, + native preview, Unicode offsets and transformation all passed. +- Native parity: 61 exact matches, two reviewed detector differences, 17 explicit + unsupported German requests, 31 cases outside this backend's scope. +- Reproducible local timings are in `benchmarks/results-4.9.json`; they include + public-call overhead and cold startup, not just Rust scanning time. + +CI and required PR approval remain the final merge gates. No package publication +or Core-repository implementation is part of this delivery. From 79a5b09573dcac2787d9defb49eb0b9908c19350 Mon Sep 17 00:00:00 2001 From: Sid Mohan <61345237+sidmohan0@users.noreply.github.com> Date: Mon, 28 Sep 2026 15:15:55 -0700 Subject: [PATCH 29/37] feat: add opt-in Rust detection and 4.9 migration bridge --- .github/workflows/ci.yml | 37 +++ .gitignore | 1 + CHANGELOG.MD | 18 ++ README.md | 4 + benchmarks/compare_detection_backends.py | 92 ++++++ benchmarks/results-4.9.json | 162 +++++++++++ datafog/__init__.py | 93 ++---- datafog/_legacy_retirement.py | 17 ++ datafog/client.py | 5 + datafog/compat/__init__.py | 1 + datafog/compat/v4.py | 105 +++++++ datafog/engine.py | 73 ++++- datafog/main.py | 3 + .../image_processing/donut_processor.py | 7 + .../image_processing/image_downloader.py | 4 + .../image_processing/pytesseract_processor.py | 3 + .../spark_processing/pyspark_udfs.py | 4 + datafog/services/image_service.py | 10 + datafog/services/spark_service.py | 3 + datafog/v5.py | 84 ++++++ docs/migration-4.9.md | 162 +++++++++++ docs/optional-surfaces.rst | 21 +- scripts/capture_481_contract.py | 4 +- scripts/check_rust_install.py | 41 +++ setup.py | 3 +- tests/contract_481.py | 2 +- tests/contracts/rust-0.3.1.json | 222 +++++++++++++++ tests/rust_contract.py | 76 +++++ tests/test_api_bridge_49.py | 104 +++++++ tests/test_contract_481.py | 43 ++- tests/test_legacy_retirement.py | 268 ++++++++++++++++++ tests/test_rust_backend.py | 204 +++++++++++++ tests/test_rust_contract.py | 29 ++ 33 files changed, 1826 insertions(+), 79 deletions(-) create mode 100644 benchmarks/compare_detection_backends.py create mode 100644 benchmarks/results-4.9.json create mode 100644 datafog/_legacy_retirement.py create mode 100644 datafog/compat/__init__.py create mode 100644 datafog/compat/v4.py create mode 100644 datafog/v5.py create mode 100644 docs/migration-4.9.md create mode 100644 scripts/check_rust_install.py create mode 100644 tests/contracts/rust-0.3.1.json create mode 100644 tests/rust_contract.py create mode 100644 tests/test_api_bridge_49.py create mode 100644 tests/test_legacy_retirement.py create mode 100644 tests/test_rust_backend.py create mode 100644 tests/test_rust_contract.py diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index b8bb00c9..9d2f2b3f 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -11,6 +11,43 @@ concurrency: cancel-in-progress: true jobs: + rust-bridge: + name: Rust bridge (${{ matrix.os }}, ${{ matrix.python-version }}) + runs-on: ${{ matrix.os }} + strategy: + fail-fast: false + matrix: + include: + - os: ubuntu-latest + python-version: "3.10" + - os: ubuntu-latest + python-version: "3.14" + - os: macos-latest + python-version: "3.12" + - os: windows-latest + python-version: "3.12" + steps: + - uses: actions/checkout@v6 + - uses: actions/setup-python@v6 + with: + python-version: ${{ matrix.python-version }} + - name: Build and install the wheel with native extra + shell: bash + run: | + python -m pip install build + python -m build --wheel + python -c "import glob, subprocess, sys; wheel = glob.glob('dist/*.whl')[0]; subprocess.check_call([sys.executable, '-m', 'pip', 'install', wheel + '[test,cli,rust]'])" + - name: Verify installed package outside checkout imports + run: python -I scripts/check_rust_install.py + - name: Test compatibility, routing, preview and native parity + run: python -m pytest tests/test_contract_481.py tests/test_rust_backend.py tests/test_api_bridge_49.py tests/test_rust_contract.py -q + - name: Generate native parity report + run: python -m tests.rust_contract --output rust-parity.json + - uses: actions/upload-artifact@v4 + with: + name: rust-parity-${{ matrix.os }}-${{ matrix.python-version }} + path: rust-parity.json + lint: runs-on: ubuntu-latest steps: diff --git a/.gitignore b/.gitignore index f7471a1e..a73bfe4d 100644 --- a/.gitignore +++ b/.gitignore @@ -61,6 +61,7 @@ docs/* !docs/make.bat !docs/optional-surfaces.rst !docs/migration-4.8.1-contract.md +!docs/migration-4.9.md !docs/agents/ !docs/agents/** !docs/audit/ diff --git a/CHANGELOG.MD b/CHANGELOG.MD index 2dca5bba..904493f3 100644 --- a/CHANGELOG.MD +++ b/CHANGELOG.MD @@ -2,6 +2,24 @@ ## [Unreleased] +#### Added + +- Experimental `datafog[rust]` detection with keyword-only `backend="rust"` + on scan/redact entry points; Python remains the default. Core 0.3.1 is pinned, + German requests are explicitly unsupported, and ML composition is unchanged. +- `datafog.compat.v4` preserves the existing scan/redact facade and result classes; + `datafog.v5` previews actual Core types and transformation APIs. +- Native parity tests, installed-wheel CI checks, and public-call benchmarks. + +#### Deprecated + +- `detect()` and `process()` are now scheduled for removal in 5.0. This explicitly + revises the earlier promise to retain these shims throughout 5.x. They continue + working in 4.9; migrate to scan/redact and review transformation differences. +- OCR/image and Spark/distributed surfaces remain available in 4.9 with use-time + notices and are scheduled for removal in 5.0. Users needing those features can + remain on the final 4.x release. + #### Fixed - Allow installation on Python 3.14 and certify the core SDK and CLI with diff --git a/README.md b/README.md index 8b9718fe..a97744d7 100644 --- a/README.md +++ b/README.md @@ -256,6 +256,10 @@ Telemetry does not include input text or detected PII values. ## Development +The [4.9 migration guide](docs/migration-4.9.md) explains opt-in Rust detection, +the native `datafog.v5` preview, and the revised 5.0 retirement schedule for +`detect`/`process`, OCR, and Spark. The Python detector remains the default. + The [4.8.1 compatibility contract](docs/migration-4.8.1-contract.md) records published Python behavior for the Rust migration, with frozen fixtures and instructions for independently reproducing them from the release wheel. diff --git a/benchmarks/compare_detection_backends.py b/benchmarks/compare_detection_backends.py new file mode 100644 index 00000000..d5c89a00 --- /dev/null +++ b/benchmarks/compare_detection_backends.py @@ -0,0 +1,92 @@ +"""Measure public Python/Rust bridge calls, including adaptation and cold startup.""" + +import argparse +import importlib.metadata +import json +import os +import platform +import statistics +import subprocess +import sys +import time +from pathlib import Path + +os.environ["DATAFOG_NO_TELEMETRY"] = "1" + +PAYLOADS = { + "short": ("Contact alice@example.com", 200), + "mixed": ( + "Email alice@example.com or call (555) 123-4567. " + "SSN 123-45-6789; card 4111 1111 1111 1111; host 192.168.1.1.", + 100, + ), + "large_sparse": ("ordinary prose " * 70_000 + " alice@example.com", 3), +} + + +def measure(fn, text, backend, loops, samples): + result = fn(text, backend=backend) + durations = [] + for _ in range(samples): + start = time.perf_counter() + for _ in range(loops): + fn(text, backend=backend) + durations.append((time.perf_counter() - start) / loops * 1_000_000) + return { + "median_us": statistics.median(durations), + "samples_us": durations, + "entities": len(result.entities), + "iterations_per_sample": loops, + } + + +def cold_start(backend, samples): + source = ( + "import datafog; " + f"datafog.scan('Contact alice@example.com', backend={backend!r})" + ) + durations = [] + for _ in range(samples): + start = time.perf_counter() + subprocess.run([sys.executable, "-c", source], check=True, capture_output=True) + durations.append((time.perf_counter() - start) * 1_000) + return {"median_ms": statistics.median(durations), "samples_ms": durations} + + +def main(): + import datafog + + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--samples", type=int, default=5) + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args() + if args.samples < 1: + parser.error("samples must be positive") + report = { + "python": platform.python_version(), + "platform": platform.platform(), + "core_version": importlib.metadata.version("datafog-core"), + "method": "One warmup call, median repeated calls; no ML. Cold = process+import+first scan.", + "warm": [], + "cold": {}, + } + for name, (text, loops) in PAYLOADS.items(): + for operation in ("scan", "redact"): + row = { + "payload": name, + "operation": operation, + "utf8_bytes": len(text.encode()), + } + for backend in ("python", "rust"): + row[backend] = measure( + getattr(datafog, operation), text, backend, loops, args.samples + ) + report["warm"].append(row) + for backend in ("python", "rust"): + report["cold"][backend] = cold_start(backend, args.samples) + args.output.write_text(json.dumps(report, indent=2) + "\n") + print(f"Saved end-to-end measurements to {args.output}") + + +if __name__ == "__main__": + main() diff --git a/benchmarks/results-4.9.json b/benchmarks/results-4.9.json new file mode 100644 index 00000000..6f8c47e8 --- /dev/null +++ b/benchmarks/results-4.9.json @@ -0,0 +1,162 @@ +{ + "python": "3.12.13", + "platform": "macOS-26.6.2-arm64-arm-64bit", + "core_version": "0.3.1", + "method": "One warmup call, median repeated calls; no ML. Cold = process+import+first scan.", + "warm": [ + { + "payload": "short", + "operation": "scan", + "utf8_bytes": 25, + "python": { + "median_us": 19.735415116883814, + "samples_us": [ + 19.692290225066245, 19.87395982723683, 19.735415116883814, + 19.453955464996397, 20.307295490056276 + ], + "entities": 1, + "iterations_per_sample": 200 + }, + "rust": { + "median_us": 2.3012497695162892, + "samples_us": [ + 2.292499993927777, 2.3364595836028457, 2.3025000700727105, + 2.3012497695162892, 2.210834645666182 + ], + "entities": 1, + "iterations_per_sample": 200 + } + }, + { + "payload": "short", + "operation": "redact", + "utf8_bytes": 25, + "python": { + "median_us": 22.42000016849488, + "samples_us": [ + 22.9199999012053, 21.847704774700105, 22.270414629019797, + 22.495624725706875, 22.42000016849488 + ], + "entities": 1, + "iterations_per_sample": 200 + }, + "rust": { + "median_us": 3.6647950764745474, + "samples_us": [ + 3.821664722636342, 3.54208517819643, 3.6647950764745474, + 3.674789913929999, 3.5991653567180037 + ], + "entities": 1, + "iterations_per_sample": 200 + } + }, + { + "payload": "mixed", + "operation": "scan", + "utf8_bytes": 108, + "python": { + "median_us": 39.49917037971318, + "samples_us": [ + 39.52000057324767, 39.80332985520363, 39.49917037971318, + 38.85417012497783, 39.10957952030003 + ], + "entities": 5, + "iterations_per_sample": 100 + }, + "rust": { + "median_us": 7.581659592688084, + "samples_us": [ + 7.581659592688084, 7.61916977353394, 7.474579615518451, + 7.530420552939177, 7.725419709458948 + ], + "entities": 5, + "iterations_per_sample": 100 + } + }, + { + "payload": "mixed", + "operation": "redact", + "utf8_bytes": 108, + "python": { + "median_us": 42.573329992592335, + "samples_us": [ + 43.91749971546233, 42.02042007818818, 42.573329992592335, + 42.929170886054635, 41.73457971774042 + ], + "entities": 5, + "iterations_per_sample": 100 + }, + "rust": { + "median_us": 12.040419969707727, + "samples_us": [ + 11.556250974535942, 12.040419969707727, 11.692079715430737, + 12.406249297782779, 12.319160159677267 + ], + "entities": 5, + "iterations_per_sample": 100 + } + }, + { + "payload": "large_sparse", + "operation": "scan", + "utf8_bytes": 1050018, + "python": { + "median_us": 121914.90270197392, + "samples_us": [ + 120976.49998031557, 121914.90270197392, 124721.45829815418, + 123755.50003101428, 121670.05562999596 + ], + "entities": 1, + "iterations_per_sample": 3 + }, + "rust": { + "median_us": 5334.583343937993, + "samples_us": [ + 5415.152680749695, 5282.319655331473, 5320.333332444231, + 5334.583343937993, 5347.4583352605505 + ], + "entities": 1, + "iterations_per_sample": 3 + } + }, + { + "payload": "large_sparse", + "operation": "redact", + "utf8_bytes": 1050018, + "python": { + "median_us": 122195.88865991682, + "samples_us": [ + 122800.74999822925, 123273.65266780059, 122195.88865991682, + 120245.19433422635, 121185.16664486378 + ], + "entities": 1, + "iterations_per_sample": 3 + }, + "rust": { + "median_us": 5254.485993646085, + "samples_us": [ + 5248.666663343708, 5268.95831959943, 5254.485993646085, + 5269.472332050403, 5233.638648254176 + ], + "entities": 1, + "iterations_per_sample": 3 + } + } + ], + "cold": { + "python": { + "median_ms": 80.99587506148964, + "samples_ms": [ + 79.40420799423009, 78.51329201366752, 80.99587506148964, + 82.50495803076774, 83.314000046812 + ] + }, + "rust": { + "median_ms": 81.11529203597456, + "samples_ms": [ + 80.53745795041323, 81.226791953668, 81.9647500757128, 81.11529203597456, + 80.21229202859104 + ] + } + } +} diff --git a/datafog/__init__.py b/datafog/__init__.py index 7236ac10..591e0ef4 100644 --- a/datafog/__init__.py +++ b/datafog/__init__.py @@ -15,11 +15,9 @@ from .agent import create_guardrail, filter_output, sanitize, scan_prompt # Core API functions - always available (lightweight) +from .compat import v4 as _compat_v4 from .core import anonymize_text, detect_pii, get_supported_entities, scan_text from .engine import Entity, RedactResult, ScanResult -from .engine import redact as _redact_entities -from .engine import scan as _scan -from .engine import scan_and_redact as _scan_and_redact # Essential models - always available from .models.common import EntityTypes @@ -130,20 +128,10 @@ def _missing_dependency(*args, **kwargs): } -_REDACT_PRESETS = { - "default": "token", - "llm": "token", - "mask": "mask", - "hash": "hash", - "replace": "pseudonymize", - "pseudonymize": "pseudonymize", -} - - def _warn_v5_replacement(old_api: str, replacement: str) -> None: warnings.warn( - f"datafog.{old_api}() is deprecated for v5. Use {replacement} instead. " - "This compatibility shim will remain through the v5.x line.", + f"datafog.{old_api}() is deprecated and will be removed in 5.0. Use {replacement} instead. " + "The earlier promise to retain this shim through 5.x has been revised.", FutureWarning, stacklevel=3, ) @@ -156,25 +144,18 @@ def scan( locales: list[str] | None = None, allowlist: list[str] | None = None, allowlist_patterns: list[str] | None = None, + *, + backend: str = "python", ) -> ScanResult: - """ - v5-preview scan entrypoint. - - Defaults to the lightweight regex engine so the core install works without - optional dependency fallback warnings. - - ``allowlist`` exempts exact entity texts (your own support address, doc - placeholders); ``allowlist_patterns`` exempts entities whose full text - matches a regex (e.g. ``^\\d{10}$`` so unix timestamps stop matching as - phone numbers). - """ - return _scan( - text=text, - engine=engine, - entity_types=entity_types, - locales=locales, - allowlist=allowlist, - allowlist_patterns=allowlist_patterns, + """Scan using the 4.x compatibility API; Rust detection is opt-in.""" + return _compat_v4.scan( + text, + engine, + entity_types, + locales, + allowlist, + allowlist_patterns, + backend=backend, ) @@ -188,39 +169,21 @@ def redact( locales: list[str] | None = None, allowlist: list[str] | None = None, allowlist_patterns: list[str] | None = None, + *, + backend: str = "python", ) -> RedactResult: - """ - v5-preview redaction entrypoint. - - If entities are provided, redact those spans. Otherwise, scan text first - using the selected engine and redact the detected entities. ``allowlist`` - and ``allowlist_patterns`` exempt findings from redaction (exact text and - full-text regex match respectively); they apply to the scan path and are - rejected when explicit ``entities`` are supplied. - """ - if preset is not None: - try: - strategy = _REDACT_PRESETS[preset] - except KeyError as exc: - allowed = ", ".join(sorted(_REDACT_PRESETS)) - raise ValueError(f"preset must be one of: {allowed}") from exc - - if entities is not None: - if allowlist or allowlist_patterns: - raise ValueError( - "allowlist/allowlist_patterns cannot be combined with explicit " - "entities; filter the entities before calling redact" - ) - return _redact_entities(text=text, entities=entities, strategy=strategy) - - return _scan_and_redact( - text=text, - engine=engine, - entity_types=entity_types, - strategy=strategy, - locales=locales, - allowlist=allowlist, - allowlist_patterns=allowlist_patterns, + """Redact using the 4.x compatibility API and legacy strategies.""" + return _compat_v4.redact( + text, + entities, + engine, + entity_types, + strategy, + preset, + locales, + allowlist, + allowlist_patterns, + backend=backend, ) diff --git a/datafog/_legacy_retirement.py b/datafog/_legacy_retirement.py new file mode 100644 index 00000000..5bc96c13 --- /dev/null +++ b/datafog/_legacy_retirement.py @@ -0,0 +1,17 @@ +"""Dependency-free retirement notices for optional legacy functionality.""" + +import warnings + + +class LegacySurfaceWarning(FutureWarning): + """An optional DataFog surface scheduled for removal in 5.0.""" + + +def warn_legacy_surface(surface: str) -> None: + """Warn at the caller's public API boundary, before optional imports.""" + warnings.warn( + f"DataFog {surface} support is deprecated in 4.9 and will be removed in 5.0. " + "Remain on the final 4.x release if you need continued support.", + LegacySurfaceWarning, + stacklevel=3, + ) diff --git a/datafog/client.py b/datafog/client.py index eab203b8..fd9a303d 100644 --- a/datafog/client.py +++ b/datafog/client.py @@ -10,6 +10,8 @@ import typer +from datafog._legacy_retirement import LegacySurfaceWarning, warn_legacy_surface + from .config import OperationType, get_config from .engine import scan_and_redact from .main import DataFog @@ -63,6 +65,7 @@ def scan_image( Prints results or exits with error on failure. """ + warn_legacy_surface("OCR") if not image_urls: typer.echo("No image URLs or file paths provided. Please provide at least one.") raise typer.Exit(code=1) @@ -87,6 +90,8 @@ def scan_image( except Exception: pass except Exception as e: + if isinstance(e, LegacySurfaceWarning): + raise logging.exception("Error in run_ocr_pipeline") try: from .telemetry import track_error diff --git a/datafog/compat/__init__.py b/datafog/compat/__init__.py new file mode 100644 index 00000000..55f15891 --- /dev/null +++ b/datafog/compat/__init__.py @@ -0,0 +1 @@ +"""Compatibility APIs for migrations between DataFog major versions.""" diff --git a/datafog/compat/v4.py b/datafog/compat/v4.py new file mode 100644 index 00000000..0be6b4bb --- /dev/null +++ b/datafog/compat/v4.py @@ -0,0 +1,105 @@ +"""Legacy scan/redact facade, preserving the 4.x Python result schema. + +The result classes retain their original engine identity. This namespace does +not include the deprecated detect/process shims. +""" + +from ..engine import Entity, RedactResult, ScanResult +from ..engine import redact as _redact_entities +from ..engine import scan as _scan +from ..engine import scan_and_redact as _scan_and_redact + +__all__ = ["Entity", "ScanResult", "RedactResult", "scan", "redact"] + +_REDACT_PRESETS = { + "default": "token", + "llm": "token", + "mask": "mask", + "hash": "hash", + "replace": "pseudonymize", + "pseudonymize": "pseudonymize", +} + + +def scan( + text: str, + engine: str = "regex", + entity_types: list[str] | None = None, + locales: list[str] | None = None, + allowlist: list[str] | None = None, + allowlist_patterns: list[str] | None = None, + *, + backend: str = "python", +) -> ScanResult: + """ + Scan with the legacy result schema; Rust detection is opt-in. + + Defaults to the lightweight regex engine so the core install works without + optional dependency fallback warnings. + + ``allowlist`` exempts exact entity texts (your own support address, doc + placeholders); ``allowlist_patterns`` exempts entities whose full text + matches a regex (e.g. ``^\\d{10}$`` so unix timestamps stop matching as + phone numbers). + """ + return _scan( + text=text, + engine=engine, + entity_types=entity_types, + locales=locales, + allowlist=allowlist, + allowlist_patterns=allowlist_patterns, + backend=backend, + ) + + +def redact( + text: str, + entities: list[Entity] | None = None, + engine: str = "regex", + entity_types: list[str] | None = None, + strategy: str = "token", + preset: str | None = None, + locales: list[str] | None = None, + allowlist: list[str] | None = None, + allowlist_patterns: list[str] | None = None, + *, + backend: str = "python", +) -> RedactResult: + """ + Redact with legacy strategies and result schema. + + If entities are provided, redact those spans. Otherwise, scan text first + using the selected engine and redact the detected entities. ``allowlist`` + and ``allowlist_patterns`` exempt findings from redaction (exact text and + full-text regex match respectively); they apply to the scan path and are + rejected when explicit ``entities`` are supplied. + """ + if backend not in ("python", "rust"): + raise ValueError("backend must be one of: python, rust") + + if preset is not None: + try: + strategy = _REDACT_PRESETS[preset] + except KeyError as exc: + allowed = ", ".join(sorted(_REDACT_PRESETS)) + raise ValueError(f"preset must be one of: {allowed}") from exc + + if entities is not None: + if allowlist or allowlist_patterns: + raise ValueError( + "allowlist/allowlist_patterns cannot be combined with explicit " + "entities; filter the entities before calling redact" + ) + return _redact_entities(text=text, entities=entities, strategy=strategy) + + return _scan_and_redact( + text=text, + engine=engine, + entity_types=entity_types, + strategy=strategy, + locales=locales, + allowlist=allowlist, + allowlist_patterns=allowlist_patterns, + backend=backend, + ) diff --git a/datafog/engine.py b/datafog/engine.py index 419558e9..3873a3d2 100644 --- a/datafog/engine.py +++ b/datafog/engine.py @@ -209,6 +209,56 @@ def _regex_entities( return _suppress_overlapping_entities(entities) +def _rust_entities( + text: str, + entity_types: Optional[list[str]] = None, + locales: Optional[list[str]] = None, +) -> list[Entity]: + """Adapt native findings to the legacy regex result contract. + + Core supplies candidate findings. Python applies the legacy overlap + suppression before entity selection and allowlists, preserving that order + without duplicating the native detectors. + """ + normalized_locales = RegexAnnotator._normalize_locales(locales) + requested = {_canonical_type(value) for value in entity_types or []} + if normalized_locales or any(value.startswith("DE_") for value in requested): + raise ValueError( + "backend='rust' does not yet support German locales or DE_* entity " + "types; use backend='python' for German detection" + ) + + try: + import datafog_core + except ImportError as exc: + raise ImportError( + "The Rust backend requires datafog-core. " + 'Install with: pip install "datafog[rust]"' + ) from exc + + entities: list[Entity] = [] + for finding in datafog_core.scan(text): + canonical_type = _canonical_type(finding.entity_type) + if canonical_type not in ALL_ENTITY_TYPES: + raise RuntimeError( + f"Rust backend returned unsupported entity type: " + f"{finding.entity_type!r}; check the installed datafog-core version" + ) + if not finding.matched_text.strip(): + continue + entities.append( + Entity( + type=canonical_type, + text=finding.matched_text, + start=finding.codepoint_range.start, + end=finding.codepoint_range.end, + confidence=1.0, + engine="regex", + ) + ) + return _suppress_overlapping_entities(entities) + + def _spacy_entities(text: str) -> list[Entity]: annotator = _get_spacy_annotator() if isinstance(annotator, _UnavailableAnnotator): @@ -357,6 +407,13 @@ def _needs_ner(entity_types: Optional[list[str]]) -> bool: return bool(requested & NER_ENTITY_TYPES) +def _validate_backend(backend: str, engine: str) -> None: + if backend not in {"python", "rust"}: + raise ValueError("backend must be one of: python, rust") + if backend == "rust" and engine != "regex": + raise ValueError("backend='rust' supports only engine='regex'") + + def scan( text: str, engine: str = "smart", @@ -364,9 +421,15 @@ def scan( locales: Optional[list[str]] = None, allowlist: Optional[list[str]] = None, allowlist_patterns: Optional[list[str]] = None, + *, + backend: str = "python", ) -> ScanResult: """Scan text for PII entities. + ``backend="rust"`` opts into experimental native detection for + ``engine="regex"``. Results retain legacy regex provenance and confidence; + these are compatibility values, not native confidence estimates. + ``allowlist`` exempts exact entity texts (e.g. your own support email); ``allowlist_patterns`` exempts entities whose full text matches a regex (e.g. ``^\\d{10}$`` to stop unix timestamps matching as phone numbers). @@ -377,11 +440,14 @@ def scan( if engine not in {"regex", "spacy", "gliner", "smart"}: raise ValueError("engine must be one of: regex, spacy, gliner, smart") + _validate_backend(backend, engine) + # Validate patterns up front so config errors fail fast even when the # text contains no entities. _compile_allowlist_patterns(allowlist_patterns) - regex_entities = _regex_entities( + detector = _rust_entities if backend == "rust" else _regex_entities + regex_entities = detector( text, entity_types=entity_types, locales=locales, @@ -547,8 +613,10 @@ def scan_and_redact( locales: Optional[list[str]] = None, allowlist: Optional[list[str]] = None, allowlist_patterns: Optional[list[str]] = None, + *, + backend: str = "python", ) -> RedactResult: - """Convenience wrapper: scan then redact.""" + """Scan with the selected backend, then apply legacy Python redaction.""" scan_result = scan( text=text, engine=engine, @@ -556,5 +624,6 @@ def scan_and_redact( locales=locales, allowlist=allowlist, allowlist_patterns=allowlist_patterns, + backend=backend, ) return redact(text=text, entities=scan_result.entities, strategy=strategy) diff --git a/datafog/main.py b/datafog/main.py index 62abaaff..0dc6f5d5 100644 --- a/datafog/main.py +++ b/datafog/main.py @@ -12,6 +12,8 @@ import logging from typing import List +from datafog._legacy_retirement import warn_legacy_surface + from .config import OperationType from .engine import scan, scan_and_redact from .models.anonymizer import Anonymizer, AnonymizerType, HashType @@ -79,6 +81,7 @@ def __init__( async def run_ocr_pipeline(self, image_urls: List[str]) -> List[str]: """Run OCR + text pipeline for CLI/backward compatibility.""" + warn_legacy_surface("OCR") from .services.image_service import ImageService image_service = ImageService() diff --git a/datafog/processing/image_processing/donut_processor.py b/datafog/processing/image_processing/donut_processor.py index 50022b6a..8e3e842d 100644 --- a/datafog/processing/image_processing/donut_processor.py +++ b/datafog/processing/image_processing/donut_processor.py @@ -12,6 +12,8 @@ import re from typing import TYPE_CHECKING, Any +from datafog._legacy_retirement import LegacySurfaceWarning, warn_legacy_surface + from .image_downloader import ImageDownloader if TYPE_CHECKING: @@ -35,6 +37,7 @@ class DonutProcessor: """ def __init__(self, model_path="naver-clova-ix/donut-base-finetuned-cord-v2"): + warn_legacy_surface("OCR") # Store model path for lazy loading self.model_path = model_path self.downloader = ImageDownloader() @@ -47,6 +50,7 @@ def _missing_dependency_message(package_name: str) -> str: ) def preprocess_image(self, image: "Image.Image") -> Any: + warn_legacy_surface("OCR") import numpy as np # Convert to RGB if the image is not already in RGB mode @@ -65,6 +69,7 @@ def preprocess_image(self, image: "Image.Image") -> Any: async def extract_text_from_image(self, image: "Image.Image") -> str: """Extract text from an image using the Donut model""" + warn_legacy_surface("OCR") logging.info("DonutProcessor.extract_text_from_image called") # If we're in a test environment and PYTEST_DONUT is not enabled, return a mock response @@ -148,6 +153,8 @@ async def extract_text_from_image(self, image: "Image.Image") -> str: result = processor.token2json(sequence) return json.dumps(result) + except LegacySurfaceWarning: + raise except (ImportError, RuntimeError): raise except Exception as e: diff --git a/datafog/processing/image_processing/image_downloader.py b/datafog/processing/image_processing/image_downloader.py index b7bf338f..46aca082 100644 --- a/datafog/processing/image_processing/image_downloader.py +++ b/datafog/processing/image_processing/image_downloader.py @@ -9,6 +9,8 @@ from io import BytesIO from typing import TYPE_CHECKING, List +from datafog._legacy_retirement import warn_legacy_surface + if TYPE_CHECKING: from PIL import Image @@ -26,6 +28,7 @@ def __init__(self): async def download_image(self, image_url: str) -> "Image.Image": """Download a single image from a URL.""" + warn_legacy_surface("OCR") try: import aiohttp from PIL import Image @@ -45,4 +48,5 @@ async def download_image(self, image_url: str) -> "Image.Image": async def download_images(self, urls: List[str]) -> List["Image.Image"]: """Download multiple images from a list of URLs concurrently.""" + warn_legacy_surface("OCR") return await asyncio.gather(*[self.download_image(url) for url in urls]) diff --git a/datafog/processing/image_processing/pytesseract_processor.py b/datafog/processing/image_processing/pytesseract_processor.py index f7291470..1b39f19d 100644 --- a/datafog/processing/image_processing/pytesseract_processor.py +++ b/datafog/processing/image_processing/pytesseract_processor.py @@ -10,6 +10,8 @@ import pytesseract from PIL import Image +from datafog._legacy_retirement import warn_legacy_surface + class PytesseractProcessor: """ @@ -20,6 +22,7 @@ class PytesseractProcessor: """ async def extract_text_from_image(self, image: Image.Image) -> str: + warn_legacy_surface("OCR") try: return pytesseract.image_to_string(image) except Exception as e: diff --git a/datafog/processing/spark_processing/pyspark_udfs.py b/datafog/processing/spark_processing/pyspark_udfs.py index 2d7e2bc5..aaa53ece 100644 --- a/datafog/processing/spark_processing/pyspark_udfs.py +++ b/datafog/processing/spark_processing/pyspark_udfs.py @@ -9,6 +9,8 @@ import importlib +from datafog._legacy_retirement import warn_legacy_surface + PII_ANNOTATION_LABELS = ["DATE_TIME", "LOC", "NRP", "ORG", "PER"] MAXIMAL_STRING_SIZE = 1000000 DEFAULT_SPACY_MODEL = "en_core_web_lg" @@ -20,6 +22,7 @@ def pii_annotator(text: str, broadcasted_nlp) -> list[list[str]]: Returns: list[list[str]]: Values as arrays in order defined in the PII_ANNOTATION_LABELS. """ + warn_legacy_surface("Spark") ensure_installed("pyspark") ensure_installed("spacy") @@ -47,6 +50,7 @@ def broadcast_pii_annotator_udf( spark_session=None, spacy_model: str = DEFAULT_SPACY_MODEL ): """Broadcast PII annotator across Spark cluster and create UDF""" + warn_legacy_surface("Spark") ensure_installed("pyspark") ensure_installed("spacy") import spacy diff --git a/datafog/services/image_service.py b/datafog/services/image_service.py index 893e2f72..453e19b6 100644 --- a/datafog/services/image_service.py +++ b/datafog/services/image_service.py @@ -13,6 +13,8 @@ import ssl from typing import TYPE_CHECKING, Any, List, Union +from datafog._legacy_retirement import LegacySurfaceWarning, warn_legacy_surface + if TYPE_CHECKING: from PIL import Image @@ -24,6 +26,7 @@ class ImageDownloader: """Asynchronous image downloader with SSL support.""" async def download_image(self, url: str) -> "Image.Image": + warn_legacy_surface("OCR") try: import aiohttp import certifi @@ -57,6 +60,7 @@ class ImageService: """ def __init__(self, use_donut: bool = False, use_tesseract: bool = True): + warn_legacy_surface("OCR") self.downloader = ImageDownloader() # Check if we're in a test environment @@ -133,12 +137,14 @@ def _get_donut_processor(self): async def download_images( self, urls: List[str] ) -> List[Union["Image.Image", BaseException]]: + warn_legacy_surface("OCR") tasks = [ asyncio.create_task(self.downloader.download_image(url)) for url in urls ] return await asyncio.gather(*tasks, return_exceptions=True) async def ocr_extract(self, image_paths: List[str]) -> List[str]: + warn_legacy_surface("OCR") from PIL import Image results = [] @@ -167,6 +173,8 @@ async def ocr_extract(self, image_paths: List[str]) -> List[str]: raise ValueError("No OCR processor selected") results.append(text) + except LegacySurfaceWarning: + raise except Exception as e: error_msg = f"Error processing image {path}: {str(e)}" logging.error(error_msg) @@ -175,6 +183,7 @@ async def ocr_extract(self, image_paths: List[str]) -> List[str]: return results async def process_images(self, image_urls, operation): + warn_legacy_surface("OCR") results = [] for url in image_urls: logging.info(f"Fetching image from {url}") @@ -185,6 +194,7 @@ async def process_images(self, image_urls, operation): return results async def process_image(self, image, operation): + warn_legacy_surface("OCR") # Implement image processing logic logging.info(f"Processed image with operation: {operation}") pass diff --git a/datafog/services/spark_service.py b/datafog/services/spark_service.py index bf7d2e48..1ace78fd 100644 --- a/datafog/services/spark_service.py +++ b/datafog/services/spark_service.py @@ -9,6 +9,8 @@ import os from typing import List +from datafog._legacy_retirement import warn_legacy_surface + class SparkService: """ @@ -19,6 +21,7 @@ class SparkService: """ def __init__(self, master=None): + warn_legacy_surface("Spark") self.master = master self.ensure_installed("pyspark") diff --git a/datafog/v5.py b/datafog/v5.py new file mode 100644 index 00000000..8baad471 --- /dev/null +++ b/datafog/v5.py @@ -0,0 +1,84 @@ +"""Preview of the Core-native API planned for DataFog 5.0. + +Names are the actual datafog_core objects: no legacy schema or strategy +translation is applied. Install ``datafog[rust]`` to use this module. Importing +``datafog`` or this namespace alone does not load the native extension. +""" + +from importlib import import_module +from typing import TYPE_CHECKING + +if TYPE_CHECKING: + from datafog_core import DataFogConfigurationError as DataFogConfigurationError + from datafog_core import DataFogFindingError as DataFogFindingError + from datafog_core import DataFogInternalError as DataFogInternalError + from datafog_core import DataFogKeyProviderError as DataFogKeyProviderError + from datafog_core import FieldMapping as FieldMapping + from datafog_core import Finding as Finding + from datafog_core import PrivacyManager as PrivacyManager + from datafog_core import Restoration as Restoration + from datafog_core import RestoreResult as RestoreResult + from datafog_core import StructuredFinding as StructuredFinding + from datafog_core import StructuredRestoration as StructuredRestoration + from datafog_core import StructuredRestoreResult as StructuredRestoreResult + from datafog_core import StructuredScanResult as StructuredScanResult + from datafog_core import StructuredTransformation as StructuredTransformation + from datafog_core import StructuredTransformResult as StructuredTransformResult + from datafog_core import TextRange as TextRange + from datafog_core import Transformation as Transformation + from datafog_core import TransformResult as TransformResult + from datafog_core import discover_fields as discover_fields + from datafog_core import scan as scan + from datafog_core import scan_and_transform as scan_and_transform + from datafog_core import ( + scan_and_transform_structured as scan_and_transform_structured, + ) + from datafog_core import scan_structured as scan_structured + from datafog_core import transform as transform + from datafog_core import transform_structured as transform_structured + +__all__ = [ + "DataFogConfigurationError", + "DataFogFindingError", + "DataFogInternalError", + "DataFogKeyProviderError", + "TextRange", + "Finding", + "Transformation", + "TransformResult", + "Restoration", + "RestoreResult", + "FieldMapping", + "StructuredFinding", + "StructuredScanResult", + "StructuredTransformation", + "StructuredTransformResult", + "StructuredRestoration", + "StructuredRestoreResult", + "PrivacyManager", + "scan", + "transform", + "scan_and_transform", + "discover_fields", + "scan_structured", + "transform_structured", + "scan_and_transform_structured", +] + + +def __getattr__(name: str): + if name not in __all__: + raise AttributeError(f"module {__name__!r} has no attribute {name!r}") + try: + core = import_module("datafog_core") + except ImportError as exc: + raise ImportError( + 'datafog.v5 requires DataFog Core. Install with: pip install "datafog[rust]"' + ) from exc + value = getattr(core, name) + globals()[name] = value + return value + + +def __dir__(): + return sorted(set(globals()) | set(__all__)) diff --git a/docs/migration-4.9.md b/docs/migration-4.9.md new file mode 100644 index 00000000..53d4e597 --- /dev/null +++ b/docs/migration-4.9.md @@ -0,0 +1,162 @@ +# Migrating incrementally with DataFog 4.9 + +4.9 is a bridge to the Rust-backed 5.0 API. The default Python detector, existing +imports, result objects, and redaction strategies continue to work. The optional +Rust backend and native API preview are experimental and explicitly selected. + +## Opt into Rust detection + +```bash +pip install "datafog[rust]" +``` + +The extra pins the tested `datafog-core==0.3.1` wheel. The base package needs no +Rust installation or native module. The existing `all` extra retains its legacy +dependency set; request `rust` explicitly, or use `datafog[all,rust]`. + +```python +import datafog + +result = datafog.scan("Contact alice@example.com", backend="rust") +assert result.entities[0].type == "EMAIL" + +result = datafog.redact("Contact alice@example.com", backend="rust") +assert result.redacted_text == "Contact [EMAIL_1]" +assert result.mapping == {"[EMAIL_1]": "alice@example.com"} +``` + +`backend` is keyword-only and defaults to `python`. Only `engine="regex"` supports +the Rust backend in 4.9. Low-level `datafog.engine.scan` and `scan_and_redact` +default to `smart`, so pass `engine="regex"` explicitly there. Missing native +dependencies and unsupported combinations raise errors; there is no automatic +fallback to Python. Native scanning failures propagate instead of becoming empty +results. + +The legacy adapter converts Core code-point offsets to Python indices and keeps +legacy `regex` provenance and confidence `1.0` as compatibility values, not native +probability estimates. It retains Python aliases, full-match allowlists (including +Python regex syntax), overlap selection, and transformations. Supplying explicit +entities to `redact` performs no detection and requires no native dependency. + +`sanitize`, `scan_prompt`, and `filter_output` forward the backend keyword. The +`DataFog` class, `TextService`, legacy convenience APIs, CLI, guardrail objects, +application adapters, and ML composition retain their existing detection paths. +Installing the extra alone does not change any of those paths. + +## Existing and new result schemas + +The compatibility namespace provides the scan/redact API explicitly: + +```python +from datafog.compat.v4 import Entity, ScanResult, RedactResult, scan, redact +``` + +Top-level scan/redact delegate to this facade. Result classes retain their +existing identity; their fields and positional calling conventions are preserved. +The compatibility namespace excludes `detect` and `process`. + +The native preview exports the actual Core objects, including structured scanning, +transformations, and provider-backed operations: + +```python +from datafog.v5 import Finding, scan, scan_and_transform + +findings = scan("Contact alice@example.com") +assert isinstance(findings[0], Finding) +assert findings[0].entity_type == "EMAIL" + +result = scan_and_transform( + "Contact alice@example.com", + {"transform": {"default": {"strategy": "redact"}}}, +) +assert result.text == "Contact [EMAIL]" +``` + +Importing `datafog` or the preview namespace alone does not import Core. Accessing +preview functions/types requires the Rust extra. Preview names are reexports, +not a second implementation or a legacy-schema adapter. + +| Existing Python API | Core preview | +| ------------------------------------------ | -------------------------------------------------------------- | +| `ScanResult.entities` | `list[Finding]` | +| `Entity.type`, `text`, `start`, `end` | `entity_type`, `matched_text`, explicit byte/code-point ranges | +| `RedactResult.redacted_text` | `TransformResult.text` | +| Numbered type tokens and plaintext mapping | Unnumbered redaction placeholders and transformation records | +| Legacy `hash` and numbered `pseudonymize` | Explicit keyed/provider-backed Core strategies | + +Legacy `token` is not Core `tokenize`. Do not mechanically rename strategies or +assume that the two schemas and provider requirements are interchangeable. + +## Known detection differences + +Core 0.3.1 does not implement the seven German detectors. The legacy Rust adapter +rejects German locales and `DE_*` selections, directing callers to Python. +The raw native preview retains Core's own API: its accepted locale configuration +does not imply German detector coverage. Continue using Python for those workloads. + +The frozen 111-case baseline yields the following Rust-backend comparison: + +| Outcome | Cases | Interpretation | +| ---------------------------- | ----: | ------------------------------------------------------------------------- | +| Exact match | 61 | Same observable result on these inputs | +| Reviewed detector difference | 2 | Invalid-checksum card and alphanumeric-embedded SSN are rejected by Core | +| Explicitly unsupported | 17 | German requests fail rather than silently lose coverage | +| Outside backend scope | 31 | Signatures, explicit-span transformations, legacy/service/guardrail paths | + +These counts describe this finite synthetic corpus, not universal detection +equivalence or precision/recall. See `tests/contracts/rust-0.3.1.json` for exact +reviewed outcomes and reasons. Each applicable case is asserted independently; +unknown differences fail CI. Generate the full per-case report with: + +```bash +python -m tests.rust_contract --output /tmp/rust-parity.json +``` + +The published `tests/contracts/4.8.1.json` remains unchanged. The 4.9 checker +permits only four added keyword-only backend parameters and the explicitly +revised detect/process warning messages; all other legacy observations remain +exact comparisons. The standalone oracle runner still checks exact 4.8.1 behavior +and should be used with the released 4.8.1 wheel, not the 4.9 checkout. + +## Retirement schedule + +| Surface | 4.9 | 5.0 plan | +| ---------------------------------------- | ---------------------------------------- | -------------------------------------------------------------------------- | +| `detect()` / `process()` | Still work; updated `FutureWarning` | Remove | +| OCR / Donut / Tesseract / image services | Still work; use-time deprecation notices | Remove supported OCR surfaces and extras | +| Spark / distributed processing | Still work; use-time deprecation notices | Remove supported Spark surfaces and extras | +| Legacy scan/redact facade | Available at top level and `compat.v4` | Native schema becomes primary; compatibility lifetime to be set separately | +| spaCy / GLiNER | Unchanged | No removal decision in this increment | + +**The earlier promise to retain `detect` and `process` throughout 5.x is revised.** +They are now scheduled for removal in 5.0. Use `scan` or `redact` in 4.9 and evaluate +the native preview before upgrading to 5.0. `process(anonymize=True)` used older +placeholder/hash semantics; migrating to `redact` can intentionally change output. + +Users needing OCR/Spark can remain on the final 4.x release. No successor package +or indefinite support commitment is introduced here. Deprecation notices do not +trigger downloads or import heavy dependencies during ordinary text imports. + +## Verification and performance + +Run the contract/backend suites with and without the Rust extra. CI also builds +and installs a wheel, then runs an isolated-interpreter smoke test on Linux, +macOS and Windows; it does not rely solely on imports from the checkout. + +```bash +python -m pytest tests/test_contract_481.py tests/test_rust_backend.py \ + tests/test_api_bridge_49.py tests/test_rust_contract.py -q +python benchmarks/compare_detection_backends.py --output /tmp/backend-timings.json +``` + +The benchmark measures public calls, including native/Python conversion, +redaction, and fresh-process startup on short, mixed, and roughly 1 MB sparse +synthetic text. It reports entity counts alongside timings. No general speedup +claim is made from a single machine or from cases with different outputs. + +A local CPython 3.12 macOS ARM64 reference run is recorded in +`benchmarks/results-4.9.json`. Median scan latency was 19.74 versus 2.30 microseconds +for the short payload, 39.50 versus 7.58 microseconds for mixed PII, and 121.91 +versus 5.33 milliseconds for the large sparse payload (Python versus Rust). +Fresh-process import plus first scan was approximately 81 milliseconds for both. +These are local measurements, not release performance guarantees. diff --git a/docs/optional-surfaces.rst b/docs/optional-surfaces.rst index 57ea5994..96d1494d 100644 --- a/docs/optional-surfaces.rst +++ b/docs/optional-surfaces.rst @@ -2,7 +2,7 @@ Optional OCR And Spark ========================= -DataFog 4.5 keeps the core package focused on lightweight text PII screening. +DataFog 4.9 keeps the core package focused on lightweight text PII screening. The default path is: .. code-block:: bash @@ -16,7 +16,16 @@ The default path is: result = datafog.redact("Email jane@example.com", engine="regex") print(result.redacted_text) -OCR and Spark are supported optional surfaces. They are useful for image and +OCR and Spark are deprecated optional surfaces in 4.9 and will be removed in 5.0. +Their APIs, installation extras, and existing behavior remain available throughout +4.9. Remain on the final 4.x release if you need continued OCR or Spark support. +No replacement package is introduced by this migration. + +Use sites emit ``FutureWarning`` notices, visible with Python's default warning +filters, including the ``scan-image`` CLI command. Importing DataFog or using +its text APIs does not emit OCR/Spark retirement notices. + +These optional surfaces They are useful for image and distributed workflows, but they should not be treated as required for the core install, package import, text scanning, text redaction, or guardrail helpers. @@ -51,8 +60,8 @@ Notes: and system Tesseract smoke checks. * Donut OCR requires a model that is already available locally. DataFog should not download models implicitly during normal runtime usage. -* OCR is not deprecated. A broader OCR API and packaging overhaul is deferred - beyond the 4.5 focus release. +* OCR APIs, image download/processing helpers, and the ``scan-image`` CLI + command are scheduled for removal in 5.0, along with OCR-only dependencies. Example local OCR flow: @@ -89,8 +98,8 @@ Notes: * ``SparkService`` requires PySpark and a Java runtime. * Spark PII UDF helpers also require spaCy and an installed spaCy model. -* Spark is not deprecated. A broader Spark overhaul is deferred beyond the 4.5 - focus release. +* Spark services, PII UDF helpers, and the ``distributed`` extra are scheduled + for removal in 5.0. Example local Spark flow: diff --git a/scripts/capture_481_contract.py b/scripts/capture_481_contract.py index 7aadeb46..42300351 100644 --- a/scripts/capture_481_contract.py +++ b/scripts/capture_481_contract.py @@ -25,7 +25,7 @@ def main(): fixture_path = root / "tests/contracts/4.8.1.json" if args.output.resolve() == fixture_path: parser.error("Capture to a separate candidate file for review") - fixture = json.loads(fixture_path.read_text()) + fixture = json.loads(fixture_path.read_text(encoding="utf-8")) provenance = fixture["provenance"] digest = hashlib.sha256(args.wheel.read_bytes()).hexdigest() if digest != provenance["sha256"]: @@ -69,7 +69,7 @@ def main(): runner = runpy.run_path(str(root / "tests/contract_481.py")) for case in fixture["cases"]: case["expected"] = runner["observe"](case) - with args.output.open("x") as output: + with args.output.open("x", encoding="utf-8") as output: output.write(json.dumps(fixture, indent=2, ensure_ascii=False) + "\n") print(f"Captured {len(fixture['cases'])} cases to {args.output}") diff --git a/scripts/check_rust_install.py b/scripts/check_rust_install.py new file mode 100644 index 00000000..f0e63fd9 --- /dev/null +++ b/scripts/check_rust_install.py @@ -0,0 +1,41 @@ +"""Smoke-test installed 4.9 bridge wheels with python -I, outside checkout imports.""" + +import importlib.metadata +import sys + + +def main(): + assert sys.flags.isolated, "Run with python -I" + + import datafog_core + + import datafog + from datafog import v5 + from datafog.compat import v4 + + assert importlib.metadata.version("datafog-core") == "0.3.1" + assert v4.Entity is datafog.Entity + assert v5.Finding is datafog_core.Finding + text = "👋 Contact alice@example.com" + legacy = datafog.scan(text, backend="rust") + assert isinstance(legacy, datafog.ScanResult) + assert legacy.entities[0].text == "alice@example.com" + assert ( + text[legacy.entities[0].start : legacy.entities[0].end] == "alice@example.com" + ) + assert datafog.redact(text, backend="rust").redacted_text == "👋 Contact [EMAIL_1]" + native = v5.scan(text) + assert isinstance(native[0], datafog_core.Finding) + assert ( + v5.scan_and_transform( + text, {"transform": {"default": {"strategy": "redact"}}} + ).text + == "👋 Contact [EMAIL]" + ) + print( + "Installed wheel: compatibility facade, Rust backend and native preview passed" + ) + + +if __name__ == "__main__": + main() diff --git a/setup.py b/setup.py index 1bf72344..4f7d7e90 100644 --- a/setup.py +++ b/setup.py @@ -1,7 +1,7 @@ from setuptools import find_packages, setup # Read README for the long description -with open("README.md", "r") as f: +with open("README.md", "r", encoding="utf-8") as f: long_description = f.read() # Use a single source of truth for the version from __about__.py @@ -84,6 +84,7 @@ ] extras_require = { + "rust": ["datafog-core==0.3.1"], "nlp": nlp_deps, "nlp-advanced": nlp_advanced_deps, "ocr": ocr_deps, diff --git a/tests/contract_481.py b/tests/contract_481.py index 075d9c71..a2150a51 100644 --- a/tests/contract_481.py +++ b/tests/contract_481.py @@ -105,7 +105,7 @@ def main(): parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--fixture", type=Path, default=FIXTURE) args = parser.parse_args() - fixture = json.loads(args.fixture.read_text()) + fixture = json.loads(args.fixture.read_text(encoding="utf-8")) failures = [] for case in fixture["cases"]: actual = observe(case) diff --git a/tests/contracts/rust-0.3.1.json b/tests/contracts/rust-0.3.1.json new file mode 100644 index 00000000..35976c35 --- /dev/null +++ b/tests/contracts/rust-0.3.1.json @@ -0,0 +1,222 @@ +{ + "core_version": "0.3.1", + "cases": { + "scan-observation-invalid-card": { + "classification": "detector-difference", + "reason": "Core validates card checksums; Python 4.8.1 accepts this card-shaped value. Experimental precision difference, not a claim of full parity.", + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [], + "text": "4111 1111 1111 1112", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + "scan-observation-boundary": { + "classification": "detector-difference", + "reason": "Core rejects the SSN embedded in an ASCII alphanumeric token; Python 4.8.1 detects the numeric substring.", + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [], + "text": "x123-45-6789y", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + "locale-de-DE_VAT_ID": { + "classification": "unsupported", + "reason": "Published Core 0.3.1 lacks German detection. The adapter rejects the request instead of silently returning incomplete findings.", + "expected": { + "error": { + "type": "ValueError", + "message": "backend='rust' does not yet support German locales or DE_* entity types; use backend='python' for German detection" + }, + "warnings": [] + } + }, + "locale-explicit-DE_VAT_ID": { + "classification": "unsupported", + "reason": "Published Core 0.3.1 lacks German detection. The adapter rejects the request instead of silently returning incomplete findings.", + "expected": { + "error": { + "type": "ValueError", + "message": "backend='rust' does not yet support German locales or DE_* entity types; use backend='python' for German detection" + }, + "warnings": [] + } + }, + "locale-de-DE_IBAN": { + "classification": "unsupported", + "reason": "Published Core 0.3.1 lacks German detection. The adapter rejects the request instead of silently returning incomplete findings.", + "expected": { + "error": { + "type": "ValueError", + "message": "backend='rust' does not yet support German locales or DE_* entity types; use backend='python' for German detection" + }, + "warnings": [] + } + }, + "locale-explicit-DE_IBAN": { + "classification": "unsupported", + "reason": "Published Core 0.3.1 lacks German detection. The adapter rejects the request instead of silently returning incomplete findings.", + "expected": { + "error": { + "type": "ValueError", + "message": "backend='rust' does not yet support German locales or DE_* entity types; use backend='python' for German detection" + }, + "warnings": [] + } + }, + "locale-de-DE_TAX_ID": { + "classification": "unsupported", + "reason": "Published Core 0.3.1 lacks German detection. The adapter rejects the request instead of silently returning incomplete findings.", + "expected": { + "error": { + "type": "ValueError", + "message": "backend='rust' does not yet support German locales or DE_* entity types; use backend='python' for German detection" + }, + "warnings": [] + } + }, + "locale-explicit-DE_TAX_ID": { + "classification": "unsupported", + "reason": "Published Core 0.3.1 lacks German detection. The adapter rejects the request instead of silently returning incomplete findings.", + "expected": { + "error": { + "type": "ValueError", + "message": "backend='rust' does not yet support German locales or DE_* entity types; use backend='python' for German detection" + }, + "warnings": [] + } + }, + "locale-de-DE_SOCIAL_SECURITY_NUMBER": { + "classification": "unsupported", + "reason": "Published Core 0.3.1 lacks German detection. The adapter rejects the request instead of silently returning incomplete findings.", + "expected": { + "error": { + "type": "ValueError", + "message": "backend='rust' does not yet support German locales or DE_* entity types; use backend='python' for German detection" + }, + "warnings": [] + } + }, + "locale-explicit-DE_SOCIAL_SECURITY_NUMBER": { + "classification": "unsupported", + "reason": "Published Core 0.3.1 lacks German detection. The adapter rejects the request instead of silently returning incomplete findings.", + "expected": { + "error": { + "type": "ValueError", + "message": "backend='rust' does not yet support German locales or DE_* entity types; use backend='python' for German detection" + }, + "warnings": [] + } + }, + "locale-de-DE_POSTAL_CODE": { + "classification": "unsupported", + "reason": "Published Core 0.3.1 lacks German detection. The adapter rejects the request instead of silently returning incomplete findings.", + "expected": { + "error": { + "type": "ValueError", + "message": "backend='rust' does not yet support German locales or DE_* entity types; use backend='python' for German detection" + }, + "warnings": [] + } + }, + "locale-explicit-DE_POSTAL_CODE": { + "classification": "unsupported", + "reason": "Published Core 0.3.1 lacks German detection. The adapter rejects the request instead of silently returning incomplete findings.", + "expected": { + "error": { + "type": "ValueError", + "message": "backend='rust' does not yet support German locales or DE_* entity types; use backend='python' for German detection" + }, + "warnings": [] + } + }, + "locale-de-DE_PASSPORT_NUMBER": { + "classification": "unsupported", + "reason": "Published Core 0.3.1 lacks German detection. The adapter rejects the request instead of silently returning incomplete findings.", + "expected": { + "error": { + "type": "ValueError", + "message": "backend='rust' does not yet support German locales or DE_* entity types; use backend='python' for German detection" + }, + "warnings": [] + } + }, + "locale-explicit-DE_PASSPORT_NUMBER": { + "classification": "unsupported", + "reason": "Published Core 0.3.1 lacks German detection. The adapter rejects the request instead of silently returning incomplete findings.", + "expected": { + "error": { + "type": "ValueError", + "message": "backend='rust' does not yet support German locales or DE_* entity types; use backend='python' for German detection" + }, + "warnings": [] + } + }, + "locale-de-DE_RESIDENCE_PERMIT_NUMBER": { + "classification": "unsupported", + "reason": "Published Core 0.3.1 lacks German detection. The adapter rejects the request instead of silently returning incomplete findings.", + "expected": { + "error": { + "type": "ValueError", + "message": "backend='rust' does not yet support German locales or DE_* entity types; use backend='python' for German detection" + }, + "warnings": [] + } + }, + "locale-explicit-DE_RESIDENCE_PERMIT_NUMBER": { + "classification": "unsupported", + "reason": "Published Core 0.3.1 lacks German detection. The adapter rejects the request instead of silently returning incomplete findings.", + "expected": { + "error": { + "type": "ValueError", + "message": "backend='rust' does not yet support German locales or DE_* entity types; use backend='python' for German detection" + }, + "warnings": [] + } + }, + "locale-negative-context": { + "classification": "unsupported", + "reason": "Published Core 0.3.1 lacks German detection. The adapter rejects the request instead of silently returning incomplete findings.", + "expected": { + "error": { + "type": "ValueError", + "message": "backend='rust' does not yet support German locales or DE_* entity types; use backend='python' for German detection" + }, + "warnings": [] + } + }, + "locale-alias-de-DE": { + "classification": "unsupported", + "reason": "Published Core 0.3.1 lacks German detection. The adapter rejects the request instead of silently returning incomplete findings.", + "expected": { + "error": { + "type": "ValueError", + "message": "backend='rust' does not yet support German locales or DE_* entity types; use backend='python' for German detection" + }, + "warnings": [] + } + }, + "locale-alias-de_de": { + "classification": "unsupported", + "reason": "Published Core 0.3.1 lacks German detection. The adapter rejects the request instead of silently returning incomplete findings.", + "expected": { + "error": { + "type": "ValueError", + "message": "backend='rust' does not yet support German locales or DE_* entity types; use backend='python' for German detection" + }, + "warnings": [] + } + } + } +} diff --git a/tests/rust_contract.py b/tests/rust_contract.py new file mode 100644 index 00000000..a63184b1 --- /dev/null +++ b/tests/rust_contract.py @@ -0,0 +1,76 @@ +"""Compare applicable 4.8.1 observations with the opt-in published Rust backend.""" + +import argparse +import copy +import importlib.metadata +import json +from collections import Counter +from pathlib import Path + +from tests.contract_481 import FIXTURE, observe + +DEVIATIONS = Path(__file__).parent / "contracts" / "rust-0.3.1.json" +TARGETS = { + "datafog:scan", + "datafog:redact", + "datafog.engine:scan_and_redact", + "datafog:sanitize", + "datafog:scan_prompt", + "datafog:filter_output", +} + + +def applicable(case): + return ( + case.get("operation", "call") == "call" + and case.get("target") in TARGETS + and "entities" not in case.get("kwargs", {}) + ) + + +def native_observation(case): + native_case = copy.deepcopy(case) + native_case.setdefault("kwargs", {})["backend"] = "rust" + return observe(native_case) + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args() + contract = json.loads(FIXTURE.read_text(encoding="utf-8")) + deviations = json.loads(DEVIATIONS.read_text(encoding="utf-8")) + results = [] + for case in contract["cases"]: + if not applicable(case): + results.append({"id": case["id"], "status": "outside-backend-scope"}) + continue + actual = native_observation(case) + deviation = deviations["cases"].get(case["id"]) + expected = deviation["expected"] if deviation else case["expected"] + status = "regression" if actual != expected else "match" + if status == "match" and deviation: + status = deviation["classification"] + results.append( + { + "id": case["id"], + "status": status, + "reason": deviation["reason"] if deviation else None, + "legacy": case["expected"], + "actual": actual, + } + ) + report = { + "core_version": importlib.metadata.version("datafog-core"), + "counts": dict(Counter(row["status"] for row in results)), + "cases": results, + } + args.output.write_text( + json.dumps(report, indent=2, ensure_ascii=False) + "\n", encoding="utf-8" + ) + print(json.dumps(report["counts"], sort_keys=True)) + return bool(report["counts"].get("regression")) + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/test_api_bridge_49.py b/tests/test_api_bridge_49.py new file mode 100644 index 00000000..9fa814ed --- /dev/null +++ b/tests/test_api_bridge_49.py @@ -0,0 +1,104 @@ +"""Compatibility facade and native-schema preview release gates.""" + +import inspect +import subprocess +import sys +from unittest.mock import patch + +import pytest + +import datafog +from datafog import engine, v5 +from datafog.compat import v4 + + +@pytest.mark.parametrize("name", ["Entity", "ScanResult", "RedactResult"]) +def test_legacy_types_retain_identity(name): + assert getattr(datafog, name) is getattr(v4, name) is getattr(engine, name) + + +def test_existing_positional_calls_match_compatibility_facade(): + text = "Email jane@example.com" + args = (text, "regex", ["EMAIL"], None, None, None) + assert datafog.scan(*args) == v4.scan(*args) + args = (text, None, "regex", ["EMAIL"], "token", "llm", None, None, None) + assert datafog.redact(*args) == v4.redact(*args) + assert datafog.redact(*args).redacted_text == "Email [EMAIL_1]" + + +@pytest.mark.parametrize("function", [datafog.scan, datafog.redact, v4.scan, v4.redact]) +def test_backend_is_additive_keyword_only(function): + parameter = inspect.signature(function).parameters["backend"] + assert parameter.kind is inspect.Parameter.KEYWORD_ONLY + assert parameter.default == "python" + + +@pytest.mark.parametrize("name", ["scan", "redact"]) +def test_top_level_delegates_backend_to_facade(name): + with patch.object(v4, name, return_value="sentinel") as delegate: + assert getattr(datafog, name)("text", backend="rust") == "sentinel" + assert delegate.call_args.kwargs["backend"] == "rust" + + +@pytest.mark.parametrize("backend", ["python", "rust"]) +def test_explicit_entities_do_not_scan(backend): + with patch.object(v4, "_scan_and_redact", side_effect=AssertionError("scanned")): + assert ( + datafog.redact("text", entities=[], backend=backend).redacted_text == "text" + ) + + +def test_explicit_entities_validate_backend(): + with pytest.raises(ValueError, match="backend"): + datafog.redact("text", entities=[], backend="unknown") + + +def test_import_and_explicit_redaction_without_native_dependency(): + code = """ +import sys +class BlockNative: + def find_spec(self, fullname, path=None, target=None): + if fullname == "datafog_core" or fullname.startswith("datafog_core."): + raise AssertionError("native import attempted") +sys.meta_path.insert(0, BlockNative()) +import datafog +import datafog.v5 +from datafog.compat import v4 +assert datafog.redact("text", entities=[], backend="rust").redacted_text == "text" +assert "datafog_core" not in sys.modules +""" + subprocess.run([sys.executable, "-c", code], check=True) + + +def test_missing_preview_dependency_has_actionable_error(): + with patch.object(v5, "import_module", side_effect=ImportError("missing")): + with pytest.raises(ImportError, match=r"datafog\[rust\]"): + v5.__getattr__("scan") + + +@pytest.mark.parametrize("namespace", [v4, v5]) +@pytest.mark.parametrize("name", ["detect", "process"]) +def test_new_namespaces_do_not_export_deprecated_shims(namespace, name): + assert name not in namespace.__all__ + assert not hasattr(namespace, name) + + +@pytest.mark.parametrize("name", ["detect", "process"]) +def test_shims_explicitly_revise_removal_promise(name): + with pytest.warns(FutureWarning, match="removed in 5.0") as captured: + getattr(datafog, name)("plain text") + assert "earlier promise" in str(captured[0].message) + assert "has been revised" in str(captured[0].message) + + +def test_real_native_exports_and_transformation(): + core = pytest.importorskip("datafog_core") + for name in v5.__all__: + assert getattr(v5, name) is getattr(core, name) + findings = v5.scan("Email jane@example.com") + assert isinstance(findings[0], v5.Finding) + result = v5.scan_and_transform( + "Email jane@example.com", {"transform": {"default": {"strategy": "redact"}}} + ) + assert isinstance(result, v5.TransformResult) + assert result.text == "Email [EMAIL]" diff --git a/tests/test_contract_481.py b/tests/test_contract_481.py index 427401dd..5aa5b767 100644 --- a/tests/test_contract_481.py +++ b/tests/test_contract_481.py @@ -1,17 +1,56 @@ """Frozen observations from the published 4.8.1 wheel; never regenerate in CI.""" +import copy import json import pytest from tests.contract_481 import FIXTURE, observe -CONTRACT = json.loads(FIXTURE.read_text()) +CONTRACT = json.loads(FIXTURE.read_text(encoding="utf-8")) + + +def expected_49(case): + """Allow only the four added backend parameters and revised shim notices.""" + expected = copy.deepcopy(case["expected"]) + if case["id"] == "public-signatures": + for target in ( + "datafog:scan", + "datafog:redact", + "datafog.engine:scan", + "datafog.engine:scan_and_redact", + ): + expected["value"][target]["parameters"].append( + { + "name": "backend", + "kind": "KEYWORD_ONLY", + "required": False, + "default": "python", + } + ) + if case.get("target") in {"datafog:detect", "datafog:process"}: + name = case["target"].split(":")[1] + replacement = ( + "datafog.scan()" + if name == "detect" + else "datafog.scan() or datafog.redact()" + ) + expected["warnings"] = [ + { + "type": "FutureWarning", + "message": ( + f"datafog.{name}() is deprecated and will be removed in 5.0. " + f"Use {replacement} instead. " + "The earlier promise to retain this shim through 5.x has been revised." + ), + } + ] + return expected @pytest.mark.parametrize("case", CONTRACT["cases"], ids=lambda case: case["id"]) def test_published_481_contract(case): - assert observe(case) == case["expected"], ( + assert observe(case) == expected_49(case), ( f"4.8.1 contract changed: {case['id']} ({case['classification']}). " "Review the migration policy before changing the frozen baseline." ) diff --git a/tests/test_legacy_retirement.py b/tests/test_legacy_retirement.py new file mode 100644 index 00000000..f5dd80e3 --- /dev/null +++ b/tests/test_legacy_retirement.py @@ -0,0 +1,268 @@ +"""Retirement notices do not require installed OCR/Spark/model dependencies.""" + +import asyncio +import importlib +import os +import subprocess +import sys +import warnings +from pathlib import Path +from types import SimpleNamespace +from unittest.mock import AsyncMock, Mock + +import pytest + +from datafog._legacy_retirement import LegacySurfaceWarning +from datafog.processing.image_processing.donut_processor import DonutProcessor +from datafog.processing.spark_processing import pyspark_udfs +from datafog.services.image_service import ImageService +from datafog.services.spark_service import SparkService + + +@pytest.mark.parametrize("surface", ["OCR", "Spark"]) +def test_optional_service_notice_precedes_dependencies(monkeypatch, surface): + def unexpected_import(*args): + pytest.fail("Optional dependency loading must follow the retirement notice") + + monkeypatch.setattr(importlib, "import_module", unexpected_import) + with warnings.catch_warnings(): + warnings.simplefilter("error", LegacySurfaceWarning) + with pytest.raises( + LegacySurfaceWarning, + match=f"{surface} support is deprecated in 4.9 and will be removed in 5.0", + ): + (ImageService if surface == "OCR" else SparkService)() + + +def test_spark_missing_dependency_error_is_preserved(monkeypatch): + def unavailable(name): + raise ModuleNotFoundError(name) + + monkeypatch.setattr(importlib, "import_module", unavailable) + with ( + pytest.warns(LegacySurfaceWarning, match="final 4.x release"), + pytest.raises(ImportError, match=r"datafog\[distributed\]"), + ): + SparkService() + + +@pytest.mark.parametrize("factory", [False, True]) +def test_spark_udf_paths_warn_before_dependency_loading(monkeypatch, factory): + dependencies = Mock(side_effect=ImportError("optional dependency unavailable")) + monkeypatch.setattr(pyspark_udfs, "ensure_installed", dependencies) + with warnings.catch_warnings(): + warnings.simplefilter("error", LegacySurfaceWarning) + with pytest.raises(LegacySurfaceWarning, match="Spark.*removed in 5.0"): + if factory: + pyspark_udfs.broadcast_pii_annotator_udf() + else: + pyspark_udfs.pii_annotator("hello", None) + dependencies.assert_not_called() + + +def test_spark_udf_output_is_preserved(monkeypatch): + monkeypatch.setattr(pyspark_udfs, "ensure_installed", lambda _: None) + nlp = Mock( + return_value=SimpleNamespace(ents=[SimpleNamespace(label_="PER", text="Jane")]) + ) + with pytest.warns(LegacySurfaceWarning): + assert pyspark_udfs.pii_annotator("Jane", SimpleNamespace(value=nlp)) == [ + [], + [], + [], + [], + ["Jane"], + ] + + +def test_donut_mock_output_is_preserved(monkeypatch): + from datafog.processing.image_processing import donut_processor + + monkeypatch.setattr(donut_processor, "IN_TEST_ENV", True) + monkeypatch.setattr(donut_processor, "DONUT_TESTING_ENABLED", False) + with pytest.warns(LegacySurfaceWarning): + processor = DonutProcessor() + with pytest.warns(LegacySurfaceWarning, match="final 4.x release"): + assert asyncio.run(processor.extract_text_from_image(object())) == ( + '{"text": "Mock OCR text for testing"}' + ) + + +def test_direct_tesseract_processor_warns_and_preserves_output(monkeypatch): + # Stub optional dependencies even when they are absent from the test environment. + tesseract = SimpleNamespace(image_to_string=Mock(return_value="OCR text")) + monkeypatch.setitem(sys.modules, "pytesseract", tesseract) + monkeypatch.setitem( + sys.modules, "PIL", SimpleNamespace(Image=SimpleNamespace(Image=object)) + ) + module_name = "datafog.processing.image_processing.pytesseract_processor" + monkeypatch.delitem(sys.modules, module_name, raising=False) + processor_module = importlib.import_module(module_name) + try: + with pytest.warns(LegacySurfaceWarning, match="OCR.*removed in 5.0"): + assert ( + asyncio.run( + processor_module.PytesseractProcessor().extract_text_from_image( + object() + ) + ) + == "OCR text" + ) + finally: + sys.modules.pop(module_name, None) + + +def test_ocr_warning_as_error_is_not_converted_to_processing_result(monkeypatch): + with pytest.warns(LegacySurfaceWarning): + service = ImageService() + # Simulate a warning from a nested public processor after outer warning handling. + monkeypatch.setattr( + "datafog.services.image_service.warn_legacy_surface", lambda _: None + ) + monkeypatch.setitem(sys.modules, "PIL", SimpleNamespace(Image=object)) + service.downloader.download_image = AsyncMock( + side_effect=LegacySurfaceWarning("nested notice") + ) + with pytest.raises(LegacySurfaceWarning, match="nested notice"): + asyncio.run(service.ocr_extract(["https://example.test/image.png"])) + + +def test_ocr_pipeline_warns_before_loading_service(monkeypatch): + from datafog.main import DataFog + + datafog = DataFog() + service = Mock() + monkeypatch.setattr("datafog.services.image_service.ImageService", service) + with warnings.catch_warnings(): + warnings.simplefilter("error", LegacySurfaceWarning) + with pytest.raises(LegacySurfaceWarning): + asyncio.run(datafog.run_ocr_pipeline([])) + service.assert_not_called() + + +def test_core_and_optional_module_imports_are_quiet_and_lightweight(): + script = """ +import importlib.abc +import sys +import warnings + +blocked = {"PIL", "pytesseract", "pyspark", "torch", "transformers", "spacy", "aiohttp", "numpy"} +class BlockOptional(importlib.abc.MetaPathFinder): + def find_spec(self, fullname, *args): + if fullname.split(".")[0] in blocked: + raise AssertionError("Unexpected optional import: " + fullname) +sys.meta_path.insert(0, BlockOptional()) +with warnings.catch_warnings(record=True) as notices: + warnings.simplefilter("always") + import datafog + from datafog.services.image_service import ImageService + from datafog.services.spark_service import SparkService + from datafog.processing.image_processing.donut_processor import DonutProcessor + from datafog.processing.spark_processing import pyspark_udfs + result = datafog.scan("jane@example.com", engine="regex") + assert result.entities + from datafog._legacy_retirement import LegacySurfaceWarning + assert not [n for n in notices if issubclass(n.category, LegacySurfaceWarning)] +assert not (blocked & set(sys.modules)) +""" + env = dict( + os.environ, + PYTHONPATH=str(Path.cwd()), + DATAFOG_NO_TELEMETRY="1", + DO_NOT_TRACK="1", + ) + subprocess.run( + [sys.executable, "-c", script], + env=env, + check=True, + capture_output=True, + text=True, + ) + + +@pytest.mark.parametrize( + "module_name", + [ + "datafog.services.image_service", + "datafog.processing.image_processing.image_downloader", + ], +) +def test_direct_image_download_warns_before_optional_import(module_name): + module = importlib.import_module(module_name) + with warnings.catch_warnings(): + warnings.simplefilter("error", LegacySurfaceWarning) + with pytest.raises(LegacySurfaceWarning, match="OCR.*removed in 5.0"): + asyncio.run( + module.ImageDownloader().download_image("https://example.test/a.png") + ) + + +def test_image_cli_warns_before_running_pipeline(monkeypatch): + pytest.importorskip("typer") + from datafog import client + + datafog = Mock() + monkeypatch.setattr(client, "DataFog", datafog) + with warnings.catch_warnings(): + warnings.simplefilter("error", LegacySurfaceWarning) + with pytest.raises(LegacySurfaceWarning, match="OCR.*removed in 5.0"): + client.scan_image(["image.png"], "scan") + datafog.assert_not_called() + + +def test_image_cli_does_not_swallow_nested_retirement_warning(monkeypatch): + pytest.importorskip("typer") + from datafog import client + + monkeypatch.setattr(client, "warn_legacy_surface", lambda _: None) + pipeline = AsyncMock(side_effect=LegacySurfaceWarning("nested notice")) + monkeypatch.setattr( + client, "DataFog", Mock(return_value=SimpleNamespace(run_ocr_pipeline=pipeline)) + ) + with pytest.raises(LegacySurfaceWarning, match="nested notice"): + client.scan_image(["image.png"], "scan") + + +@pytest.mark.parametrize("image_argument", [False, True]) +def test_image_cli_notice_visible_under_default_python_filters(image_argument): + pytest.importorskip("typer") + # A fresh interpreter avoids pytest's warning filters and cached warning sites. + # Do not enable warnings here: the normal interpreter policy must show this. + script = """ +import sys +from types import SimpleNamespace +from unittest.mock import AsyncMock, Mock +from typer.testing import CliRunner +from datafog import client + +pipeline = AsyncMock(return_value=["OCR text"]) +client.DataFog = Mock(return_value=SimpleNamespace(run_ocr_pipeline=pipeline)) +args = ["scan-image"] +if sys.argv[1] == "True": + args.append("image.png") +result = CliRunner().invoke(client.app, args) +assert "deprecated in 4.9 and will be removed in 5.0" in result.stderr, result.output +assert "final 4.x release" in result.stderr, result.output +if sys.argv[1] == "True": + assert result.exit_code == 0, result.output + assert "OCR Pipeline Results: ['OCR text']" in result.stdout + pipeline.assert_awaited_once_with(image_urls=["image.png"]) +else: + assert result.exit_code == 1, result.output + assert "No image URLs or file paths provided" in result.stdout + pipeline.assert_not_awaited() +""" + env = dict( + os.environ, + PYTHONPATH=str(Path.cwd()), + DATAFOG_NO_TELEMETRY="1", + DO_NOT_TRACK="1", + ) + env.pop("PYTHONWARNINGS", None) + subprocess.run( + [sys.executable, "-c", script, str(image_argument)], + env=env, + check=True, + capture_output=True, + text=True, + ) diff --git a/tests/test_rust_backend.py b/tests/test_rust_backend.py new file mode 100644 index 00000000..97644322 --- /dev/null +++ b/tests/test_rust_backend.py @@ -0,0 +1,204 @@ +"""Behavioral coverage for the opt-in native detection adapter.""" + +import builtins +import sys +from types import SimpleNamespace + +import pytest + +from datafog import engine + + +def finding(label, text, start, end): + return SimpleNamespace( + entity_type=label, + matched_text=text, + codepoint_range=SimpleNamespace(start=start, end=end), + byte_range=SimpleNamespace(start=999, end=1000), + ) + + +@pytest.fixture +def native(monkeypatch): + calls = [] + findings = [] + + def scan(text): + calls.append(text) + return findings + + monkeypatch.setitem(sys.modules, "datafog_core", SimpleNamespace(scan=scan)) + return findings, calls + + +def test_native_offsets_duplicates_order_and_provenance(native): + findings, calls = native + text = "😀 alice@example.com / alice@example.com" + findings.extend( + [ + finding("EMAIL", "alice@example.com", 22, 39), + finding("EMAIL", "alice@example.com", 2, 19), + ] + ) + result = engine.scan(text, engine="regex", backend="rust") + assert calls == [text] + assert result.text == text + assert result.engine_used == "regex" + assert [item.start for item in result.entities] == [2, 22] + assert all(text[item.start : item.end] == item.text for item in result.entities) + assert all( + item.engine == "regex" and item.confidence == 1.0 for item in result.entities + ) + + +@pytest.mark.parametrize("selection", [None, [], [" email_address "]]) +def test_selection_alias_and_empty_selection(native, selection): + findings, _ = native + findings.append(finding("EMAIL", "a@example.com", 0, 13)) + result = engine.scan( + "a@example.com", "regex", entity_types=selection, backend="rust" + ) + assert len(result.entities) == 1 + + +def test_unknown_selection_preserves_legacy_empty_result(native): + findings, _ = native + findings.append(finding("EMAIL", "a@example.com", 0, 13)) + assert ( + engine.scan( + "a@example.com", "regex", entity_types=["UNKNOWN"], backend="rust" + ).entities + == [] + ) + + +def test_unknown_native_label_raises_instead_of_silently_dropping_pii(native): + findings, _ = native + findings.append(finding("FUTURE_LABEL", "example", 0, 7)) + with pytest.raises(RuntimeError, match="unsupported entity type.*FUTURE_LABEL"): + engine.scan("example", "regex", backend="rust") + + +def test_python_overlap_priority_precedes_selection(native): + findings, _ = native + text = "1234567890123456" + findings.extend( + [finding("PHONE", text[:10], 0, 10), finding("CREDIT_CARD", text, 0, 16)] + ) + result = engine.scan(text, "regex", backend="rust") + assert [item.type for item in result.entities] == ["CREDIT_CARD"] + assert ( + engine.scan(text, "regex", entity_types=["PHONE"], backend="rust").entities + == [] + ) + + +@pytest.mark.parametrize( + "kwargs, count", + [ + ({"allowlist": ["a@example.com"]}, 0), + ({"allowlist": ["A@example.com"]}, 1), + ({"allowlist_patterns": [r"(?=a@).*\.com"]}, 0), + ({"allowlist_patterns": ["example"]}, 1), + ], +) +def test_python_allowlist_semantics(native, kwargs, count): + findings, _ = native + findings.append(finding("EMAIL", "a@example.com", 0, 13)) + assert ( + len(engine.scan("a@example.com", "regex", backend="rust", **kwargs).entities) + == count + ) + + +@pytest.mark.parametrize("pattern", ["[", "(a+)+", "x" * 513]) +def test_allowlist_validation_before_native(native, pattern): + _, calls = native + with pytest.raises(ValueError): + engine.scan("", "regex", backend="rust", allowlist_patterns=[pattern]) + assert not calls + + +@pytest.mark.parametrize("locale", [["de"], [" DE-DE "], "de_de"]) +def test_german_locales_rejected(native, locale): + _, calls = native + with pytest.raises(ValueError, match="German"): + engine.scan("", "regex", locales=locale, backend="rust") + assert not calls + + +@pytest.mark.parametrize("label", engine.RegexAnnotator.GERMAN_LABELS + ["DE_FUTURE"]) +def test_german_selection_rejected(native, label): + with pytest.raises(ValueError, match="German"): + engine.scan("", "regex", entity_types=[label.lower()], backend="rust") + + +def test_unknown_locale_validation_preserved(native): + with pytest.raises(ValueError, match="locale must be one of"): + engine.scan("", "regex", locales=["fr"], backend="rust") + + +@pytest.mark.parametrize("name", ["smart", "spacy", "gliner"]) +def test_nonregex_engine_rejected(native, name): + _, calls = native + with pytest.raises(ValueError, match="only engine='regex'"): + engine.scan("", name, backend="rust") + assert not calls + + +def test_bad_backend_rejected(): + with pytest.raises(ValueError, match="backend must be one of"): + engine.scan("", "regex", backend="auto") + + +def test_no_native_import_on_python_path_and_actionable_missing_error(monkeypatch): + original_import = builtins.__import__ + + def guarded_import(name, *args, **kwargs): + if name == "datafog_core": + raise ModuleNotFoundError("native absent", name=name) + return original_import(name, *args, **kwargs) + + monkeypatch.setattr(builtins, "__import__", guarded_import) + assert engine.scan("a@example.com", "regex").entities + with pytest.raises(ImportError, match=r'pip install "datafog\[rust\]"'): + engine.scan("a@example.com", "regex", backend="rust") + + +def test_native_failure_propagates_without_fallback(monkeypatch): + error = RuntimeError("native failure") + + def fail(text): + raise error + + monkeypatch.setitem(sys.modules, "datafog_core", SimpleNamespace(scan=fail)) + with pytest.raises(RuntimeError) as caught: + engine.scan("a@example.com", "regex", backend="rust") + assert caught.value is error + + +@pytest.mark.parametrize("strategy", ["token", "mask", "hash", "pseudonymize"]) +def test_all_legacy_transformations_use_native_findings(native, strategy): + findings, calls = native + text = "a@example.com a@example.com" + findings.extend( + [ + finding("EMAIL", "a@example.com", 0, 13), + finding("EMAIL", "a@example.com", 14, 27), + ] + ) + expected = engine.scan_and_redact(text, "regex", strategy=strategy) + actual = engine.scan_and_redact(text, "regex", strategy=strategy, backend="rust") + assert actual == expected + assert calls == [text] + + +@pytest.mark.parametrize("strategy", ["token", "mask", "hash", "pseudonymize"]) +def test_real_native_unicode_and_transformations(strategy): + pytest.importorskip("datafog_core") + text = "😀 München a@example.com and a@example.com" + python_result = engine.scan_and_redact(text, "regex", strategy=strategy) + native_result = engine.scan_and_redact( + text, "regex", strategy=strategy, backend="rust" + ) + assert native_result == python_result diff --git a/tests/test_rust_contract.py b/tests/test_rust_contract.py new file mode 100644 index 00000000..52ee52e1 --- /dev/null +++ b/tests/test_rust_contract.py @@ -0,0 +1,29 @@ +"""Exact native outcomes, with individually reviewed differences from Python.""" + +import importlib.metadata +import json + +import pytest + +from tests.contract_481 import FIXTURE +from tests.rust_contract import DEVIATIONS, applicable, native_observation + +pytest.importorskip("datafog_core") +CONTRACT = json.loads(FIXTURE.read_text(encoding="utf-8")) +REVIEWED = json.loads(DEVIATIONS.read_text(encoding="utf-8")) +CASES = [case for case in CONTRACT["cases"] if applicable(case)] + + +def test_native_version_and_review_inventory(): + assert importlib.metadata.version("datafog-core") == REVIEWED["core_version"] + assert set(REVIEWED["cases"]) <= {case["id"] for case in CASES} + for item in REVIEWED["cases"].values(): + assert item["classification"] in {"unsupported", "detector-difference"} + assert item["reason"] + + +@pytest.mark.parametrize("case", CASES, ids=lambda case: case["id"]) +def test_native_contract(case): + deviation = REVIEWED["cases"].get(case["id"]) + expected = deviation["expected"] if deviation else case["expected"] + assert native_observation(case) == expected, case["id"] From c07ba009e2c1ab36c44efa8ea9d2f1b15eacf6be Mon Sep 17 00:00:00 2001 From: Sid Mohan <61345237+sidmohan0@users.noreply.github.com> Date: Mon, 28 Sep 2026 15:21:05 -0700 Subject: [PATCH 30/37] test(api): cover lazy preview and facade validation --- datafog/v5.py | 2 +- tests/test_api_bridge_49.py | 39 ++++++++++++++++++++++++++++++++++++- 2 files changed, 39 insertions(+), 2 deletions(-) diff --git a/datafog/v5.py b/datafog/v5.py index 8baad471..caa270b4 100644 --- a/datafog/v5.py +++ b/datafog/v5.py @@ -8,7 +8,7 @@ from importlib import import_module from typing import TYPE_CHECKING -if TYPE_CHECKING: +if TYPE_CHECKING: # pragma: no cover - static imports are never executed at runtime from datafog_core import DataFogConfigurationError as DataFogConfigurationError from datafog_core import DataFogFindingError as DataFogFindingError from datafog_core import DataFogInternalError as DataFogInternalError diff --git a/tests/test_api_bridge_49.py b/tests/test_api_bridge_49.py index 9fa814ed..483aa7f8 100644 --- a/tests/test_api_bridge_49.py +++ b/tests/test_api_bridge_49.py @@ -1,9 +1,11 @@ """Compatibility facade and native-schema preview release gates.""" +import importlib.util import inspect import subprocess import sys -from unittest.mock import patch +from types import SimpleNamespace +from unittest.mock import Mock, patch import pytest @@ -102,3 +104,38 @@ def test_real_native_exports_and_transformation(): ) assert isinstance(result, v5.TransformResult) assert result.text == "Email [EMAIL]" + + +def test_preview_discovery_is_lazy_and_exports_are_cached(monkeypatch): + # A separate module object avoids cached exports from native integration tests. + spec = importlib.util.spec_from_file_location("isolated_preview", v5.__file__) + preview = importlib.util.module_from_spec(spec) + spec.loader.exec_module(preview) + core = SimpleNamespace(**{name: object() for name in preview.__all__}) + loader = Mock(return_value=core) + monkeypatch.setattr(preview, "import_module", loader) + + assert set(preview.__all__) <= set(dir(preview)) + with pytest.raises(AttributeError, match="unrecognized"): + preview.unrecognized + loader.assert_not_called() + + for name in preview.__all__: + assert getattr(preview, name) is getattr(core, name) + assert loader.call_count == len(preview.__all__) + loader.reset_mock() + for name in preview.__all__: + assert getattr(preview, name) is getattr(core, name) + loader.assert_not_called() + + +@pytest.mark.parametrize("redact", [datafog.redact, v4.redact]) +def test_compatibility_facade_rejects_unknown_preset(redact): + with pytest.raises(ValueError, match="preset must be one of"): + redact("plain text", preset="unknown") + + +@pytest.mark.parametrize("option", ["allowlist", "allowlist_patterns"]) +def test_compatibility_facade_rejects_allowlists_with_explicit_entities(option): + with pytest.raises(ValueError, match="cannot be combined with explicit entities"): + v4.redact("plain text", entities=[], **{option: ["plain text"]}) From 6ebd4c86655e0561941475762a7d71e45204f60f Mon Sep 17 00:00:00 2001 From: Sid Mohan <61345237+sidmohan0@users.noreply.github.com> Date: Mon, 28 Sep 2026 15:28:18 -0700 Subject: [PATCH 31/37] docs: align live Python guides with upcoming 4.9 bridge --- docs/cli.rst | 15 ++++++-- docs/getting-started.rst | 42 ++++++++++++++++++-- docs/important-concepts.rst | 17 ++++++-- docs/index.rst | 32 ++++++++++++--- docs/migration-4.9.md | 4 ++ docs/python-sdk.rst | 77 ++++++++++++++++++++++++++++++++++--- 6 files changed, 165 insertions(+), 22 deletions(-) diff --git a/docs/cli.rst b/docs/cli.rst index 8dfcc9f8..e27f4d72 100644 --- a/docs/cli.rst +++ b/docs/cli.rst @@ -8,8 +8,13 @@ The main entrypoint for the CLI is through the DataFog client file, defined in : We use Typer to build the CLI, with each command defined as a separate function. Core text commands such as ``scan-text``, ``redact-text``, ``replace-text``, -and ``hash-text`` are the primary 4.5 CLI path. OCR commands remain available -for existing users, but they are optional: +and ``hash-text`` are the primary CLI path. The unreleased 4.9 bridge leaves text commands on their +existing Python detection paths; the opt-in Rust backend is a Python API option, +not a new CLI flag. Install ``datafog[cli]`` for the command-line dependencies. + +OCR commands remain functional in 4.9 but are deprecated for removal in 5.0. +``scan-image`` emits a visible ``FutureWarning`` at use time. Its optional +dependencies are: * Local image OCR requires ``datafog[ocr]`` and any needed system OCR binaries such as Tesseract. @@ -17,7 +22,11 @@ for existing users, but they are optional: * Donut OCR requires ``datafog[nlp-advanced,ocr]`` and a local model. Spark/distributed workflows are Python SDK surfaces rather than first-path CLI -commands. Install ``datafog[distributed]`` when using ``SparkService``. +commands. Spark support is also deprecated in 4.9 for removal in 5.0. Install +``datafog[distributed]`` when using ``SparkService`` during 4.x. Users needing +OCR/Spark after the cutover can remain on the final 4.x release. See +:doc:`optional-surfaces` and the +:download:`unreleased 4.9 migration guide `. German locale support --------------------- diff --git a/docs/getting-started.rst b/docs/getting-started.rst index cf6efca7..8df06248 100644 --- a/docs/getting-started.rst +++ b/docs/getting-started.rst @@ -1,8 +1,8 @@ ================================ -Getting Started With DataFog 4.5 +Getting Started With DataFog ================================ -DataFog 4.5 focuses on lightweight text PII screening. A core install should +DataFog focuses on lightweight text PII screening. A core install should let you scan and redact common structured PII without installing OCR, Spark, large NLP models, or middleware integrations. @@ -45,10 +45,43 @@ Optional extras are explicit: - ``pip install "datafog[all]"`` - You are developing or deliberately want every optional surface. +Unreleased 4.9 bridge +===================== + +The following APIs are development previews, not a claim that 4.9 is published. +From a checkout containing the 4.9 implementation, install the explicit Rust +extra to evaluate them: + +.. code-block:: bash + + python -m pip install -e ".[rust]" + +The extra pins ``datafog-core==0.3.1``. It does not change the default Python +backend, and the existing ``all`` extra does not include Rust. To opt in: + +.. code-block:: python + + import datafog + + result = datafog.scan("Contact jane@example.com", engine="regex", backend="rust") + print(result.entities) + +Only ``engine="regex"`` supports this backend. German locales and ``DE_*`` entity +selections are unsupported by the pinned Core version and explicitly rejected +by the legacy Rust adapter; use the Python backend for German detection. Other +detector differences remain, so this is an experimental comparison path. + +For native Core types use ``datafog.v5``; for the explicit legacy facade use +``datafog.compat.v4``. See :doc:`python-sdk` and the +:download:`complete migration guide ` before switching schemas. +``detect()`` and ``process()`` remain available in 4.9 with revised 5.0 removal +warnings. OCR and Spark are also deprecated in 4.9 for removal in 5.0; existing +extras remain available during 4.9. + Python Usage ============ -Use the top-level helpers for the 4.5 core path: +Use the top-level helpers for the core text path: .. code-block:: python @@ -120,7 +153,8 @@ The CLI core path is text-first: datafog hash-text "Contact jane@example.com" datafog redact-text "Steuer-ID 12345678901" --locale de -Image commands are optional. Install ``datafog[ocr]`` for local OCR and +Image commands are optional and scheduled for deprecation in 4.9 and removal +in 5.0. They remain functional in 4.9. Install ``datafog[ocr]`` for local OCR and ``datafog[web,ocr]`` when the CLI needs to download image inputs. What 4.x Is Not diff --git a/docs/important-concepts.rst b/docs/important-concepts.rst index d08c932c..a1655593 100644 --- a/docs/important-concepts.rst +++ b/docs/important-concepts.rst @@ -8,7 +8,12 @@ Overview Data Models ^^^^^^^^^^^ -Key data models to support PII annotation and OCR analysis. +Existing models support legacy PII annotation and optional OCR analysis. +The unreleased 4.9 bridge retains them and adds ``datafog.compat.v4`` for the +legacy scan/redact result classes (``Entity``, ``ScanResult``, ``RedactResult``). +``datafog.v5`` separately previews native Core ``Finding`` and ``TransformResult`` +objects; these have different fields and transformation semantics. See +:doc:`python-sdk` and the :download:`migration guide `. * AnalysisExplanation * AnnotationResult @@ -19,7 +24,8 @@ Key data models to support PII annotation and OCR analysis. Processors ^^^^^^^^^^^ -Main processors: +Text processors remain available. OCR processors below are deprecated in the +unreleased 4.9 bridge and scheduled for removal in 5.0: * SpacyAnnotator Text annotation with spaCy @@ -30,7 +36,12 @@ Main processors: Services ^^^^^^^^^^^ -Core services: +``TextService`` remains available. ``ImageService`` and ``SparkService`` are +optional legacy services, deprecated in 4.9 for removal in 5.0. They remain +functional throughout 4.9; users requiring them after the cutover can remain on +the final 4.x release. See :doc:`optional-surfaces`. + +Existing services: * ImageService Image handling and OCR diff --git a/docs/index.rst b/docs/index.rst index 57c41a3e..44f94e63 100644 --- a/docs/index.rst +++ b/docs/index.rst @@ -2,21 +2,41 @@ DataFog Documentation ===================== -DataFog 4.5 is a lightweight text PII screening package for Python. The +DataFog is a lightweight text PII screening package for Python. The primary path is a small core install, fast regex-based scanning and redaction, agent-friendly guardrail helpers, and explicit optional extras when you need NLP, OCR, Spark, or web inputs. Start with :doc:`getting-started` if you want the shortest route from install to scanning text. The roadmap and historical planning pages remain available, -but the live user docs are the first path for 4.5. +but the live user docs are the first path for current text APIs. -Use DataFog 4.5 -=============== +4.9 development preview +======================= + +.. note:: + + The 4.9 bridge described here is unreleased development work. These pages do + not announce a published 4.9 package. A normal PyPI install does not imply + availability of the new APIs below. + +4.9 preserves existing imports, result classes, Python detection defaults, and +redaction behavior. It adds explicit experimental Rust detection through +``backend="rust"``, the ``datafog.compat.v4`` facade, and the ``datafog.v5`` native +Core schema preview. See :doc:`python-sdk` and the +:download:`complete 4.9 migration guide `. + +The planned 4.9 release deprecates ``detect()``/``process()``, OCR, and Spark for +removal in 5.0. The earlier promise to retain ``detect()``/``process()`` throughout +5.x is revised. OCR/Spark remain functional in 4.9; users needing them after the +cutover can remain on the final 4.x release. See :doc:`optional-surfaces`. + +Use DataFog +=========== .. toctree:: :maxdepth: 2 - :caption: Use DataFog 4.5 + :caption: Use DataFog getting-started python-sdk @@ -49,7 +69,7 @@ Planning And History ==================== The pages below document release planning, migration history, and future -direction. They are useful context, but they are secondary to the live 4.5 +direction. They are useful context, but they are secondary to the live usage path above. .. toctree:: diff --git a/docs/migration-4.9.md b/docs/migration-4.9.md index 53d4e597..72233796 100644 --- a/docs/migration-4.9.md +++ b/docs/migration-4.9.md @@ -1,5 +1,9 @@ # Migrating incrementally with DataFog 4.9 +> **Unreleased:** This guide describes the upcoming 4.9 bridge. Until it is +> published, evaluate these APIs from the development checkout with +> `python -m pip install -e ".[rust]"`; a normal PyPI install does not include them. + 4.9 is a bridge to the Rust-backed 5.0 API. The default Python detector, existing imports, result objects, and redaction strategies continue to work. The optional Rust backend and native API preview are experimental and explicitly selected. diff --git a/docs/python-sdk.rst b/docs/python-sdk.rst index ce70f577..7e2a2283 100644 --- a/docs/python-sdk.rst +++ b/docs/python-sdk.rst @@ -4,7 +4,7 @@ DataFog Python SDK Overview -------- -The primary 4.5 SDK path is lightweight text PII screening through the +The primary SDK path is lightweight text PII screening through the top-level ``datafog`` helpers. These helpers use the regex engine by default and do not require OCR, Spark, model downloads, or distributed dependencies. @@ -27,10 +27,73 @@ available for existing users. ``TextService(engine="regex")`` is the dependency-light service path; ``spacy``, ``gliner``, ``smart``, OCR, and Spark surfaces require their explicit extras. +4.9 compatibility and Core preview (unreleased) +----------------------------------------------- + +These additions describe development work for 4.9; they do not indicate a +published release. Existing top-level ``scan``/``redact`` functions retain their +result shapes and use the Python backend by default. They delegate through the +new facade, whose classes are the same objects as the established result types: + +.. code-block:: python + + from datafog.compat.v4 import Entity, ScanResult, RedactResult, scan, redact + +The facade does not include ``detect`` or ``process``. Those legacy helpers +continue to work at the top level in 4.9 but warn of removal in 5.0, revising the +previous promise to retain them throughout 5.x. Moving from ``process`` to +``redact`` can change old placeholder and hash output; compare results explicitly. + +Install ``.[rust]`` from the development checkout to evaluate Rust detection: + +.. code-block:: python + + import datafog + + result = datafog.redact("Contact jane@example.com", engine="regex", backend="rust") + assert result.redacted_text == "Contact [EMAIL_1]" + +``backend`` is keyword-only, defaults to ``python``, and supports Rust only with +``engine="regex"``. Detection uses Core; Python still handles legacy aliases, +full-match allowlists, result conversion, and transformation strategies. Explicit +entities passed to ``redact`` require no native scanning. Missing native modules, +unsupported combinations, and native errors are not silently retried in Python. +The ``DataFog`` class, ``TextService``, CLI, and ML composition keep their existing +backend paths. + +The native ``datafog.v5`` preview exposes actual Core types and operations rather +than adapting them to legacy classes: + +.. code-block:: python + + from datafog.v5 import Finding, scan, scan_and_transform + + findings = scan("Contact jane@example.com") + assert isinstance(findings[0], Finding) + result = scan_and_transform( + "Contact jane@example.com", + {"transform": {"default": {"strategy": "redact"}}}, + ) + assert result.text == "Contact [EMAIL]" + +Native scanning returns ``list[Finding]`` with ``entity_type``, ``matched_text``, +and explicit offset ranges. Native transformation returns ``TransformResult`` +with ``text`` and transformation records. Legacy numbered tokens, plaintext +mappings, and hash/pseudonymization strategies are not interchangeable with +native strategies. Importing the namespace is lazy; accessing its exports +requires the Rust extra. The supported compatibility lifetime after 5.0 remains +a separate decision. + +The pinned Core 0.3.1 lacks German detectors and differs on some structured +inputs. The legacy Rust adapter rejects German requests; the raw native preview +retains Core's own behavior and must not be assumed to provide German coverage. +Read the :download:`complete migration guide ` for the exact +schema comparison, finite-corpus parity results, and verification commands. + German locale coverage ---------------------- -DataFog 4.5 includes regex-only German structured PII support without adding +The Python backend includes regex-only German structured PII support without adding dependencies. German-only identifiers are opt-in because their raw shapes are country-specific or common in ordinary product, ticket, invoice, and order data. @@ -62,7 +125,7 @@ The opt-in German set currently covers ``DE_VAT_ID``, ``DE_IBAN``, Optional services ----------------- -OCR and Spark are supported optional surfaces, not the primary 4.5 path: +OCR and Spark remain available as optional surfaces throughout 4.9: * Use ``datafog[ocr]`` for local OCR helpers such as ``ImageService`` and ``PytesseractProcessor``. @@ -73,9 +136,11 @@ OCR and Spark are supported optional surfaces, not the primary 4.5 path: * Use ``datafog[distributed,nlp]`` plus an installed spaCy model for Spark PII UDF helpers. -OCR and Spark are not deprecated. Their broader overhaul is deferred so the -4.5 release can keep the core package tight while preserving existing optional -usage. See :doc:`optional-surfaces` for install notes and limitations. +The unreleased 4.9 bridge deprecates OCR and Spark for removal in 5.0. Use-time +``FutureWarning`` notices are visible under normal Python warning filters. Users +needing these features can remain on the final 4.x release; this migration does +not introduce successor packages or promise indefinite maintenance. See +:doc:`optional-surfaces` for install notes and limitations. Definitions ----------- From 5f5bdbb38783c0b646ddd6da9762c91f656d52cd Mon Sep 17 00:00:00 2001 From: Sid Mohan <61345237+sidmohan0@users.noreply.github.com> Date: Mon, 28 Sep 2026 15:32:18 -0700 Subject: [PATCH 32/37] docs: supersede outdated retirement commitments --- README.md | 7 ++++--- docs/roadmap.rst | 8 ++++++++ docs/v5-compatibility-matrix.rst | 8 ++++++++ 3 files changed, 20 insertions(+), 3 deletions(-) diff --git a/README.md b/README.md index a97744d7..97741a60 100644 --- a/README.md +++ b/README.md @@ -188,9 +188,10 @@ scan/redact helpers, or guardrail helpers. model. - A Java runtime is required by PySpark. -OCR and Spark are not deprecated. Their broader API and packaging overhaul is -deferred; the 4.x goal is to keep them explicit, documented, and isolated from -the lightweight core path. +The upcoming 4.9 release deprecates OCR and Spark with visible use-time +warnings; their APIs and extras will be removed in 5.0. They remain functional +in 4.9. Users who need these features can stay on the final 4.x release. See +the [4.9 migration guide](docs/migration-4.9.md) for the transition plan. ## Backward-Compatible APIs diff --git a/docs/roadmap.rst b/docs/roadmap.rst index 080a6b0f..a4aee1f8 100644 --- a/docs/roadmap.rst +++ b/docs/roadmap.rst @@ -2,6 +2,14 @@ Release Roadmap ================ +.. note:: + + This earlier planning document is retained for context. The upcoming 4.9 + bridge revises its compatibility commitments: ``detect``/``process``, OCR, + and Spark are deprecated in 4.9 and removed in 5.0. The former promise to + retain the shims through 5.x no longer applies. See the + :download:`current migration plan ` for the agreed scope. + Where DataFog is today (4.8.x) and where it is going (v5.0.0). The 4.x line delivered the lightweight-core architecture and, from 4.6.0 on, an offline PII firewall for AI agents and gateways. The v5 cycle turns that diff --git a/docs/v5-compatibility-matrix.rst b/docs/v5-compatibility-matrix.rst index a0e2691c..ac29c378 100644 --- a/docs/v5-compatibility-matrix.rst +++ b/docs/v5-compatibility-matrix.rst @@ -2,6 +2,14 @@ v5 Compatibility Matrix ======================= +.. note:: + + This earlier planning document is retained for context. The upcoming 4.9 + bridge revises its compatibility commitments: ``detect``/``process``, OCR, + and Spark are deprecated in 4.9 and removed in 5.0. The former promise to + retain the shims through 5.x no longer applies. See the + :download:`current migration plan ` for the agreed scope. + Status ------ From 272bd157f5031d8d95d03fae13db5f6a8a425bcc Mon Sep 17 00:00:00 2001 From: Sid Mohan <61345237+sidmohan0@users.noreply.github.com> Date: Mon, 28 Sep 2026 15:56:20 -0700 Subject: [PATCH 33/37] docs: define Core 0.4 adapter integration gate --- .gitignore | 1 + docs/core-0.4-integration.md | 40 ++++++++++++++++++++++++++++++++++++ 2 files changed, 41 insertions(+) create mode 100644 docs/core-0.4-integration.md diff --git a/.gitignore b/.gitignore index a73bfe4d..46214511 100644 --- a/.gitignore +++ b/.gitignore @@ -62,6 +62,7 @@ docs/* !docs/optional-surfaces.rst !docs/migration-4.8.1-contract.md !docs/migration-4.9.md +!docs/core-0.4-integration.md !docs/agents/ !docs/agents/** !docs/audit/ diff --git a/docs/core-0.4-integration.md b/docs/core-0.4-integration.md new file mode 100644 index 00000000..0fe87e94 --- /dev/null +++ b/docs/core-0.4-integration.md @@ -0,0 +1,40 @@ +# Core 0.4 capability adapter integration gate + +Status: draft implementation against the proposed capability contract. No Core +candidate wheel has been validated yet. The dependency range remains unchanged +until that gate passes; this branch must not merge or publish in this state. + +## Scope + +- Keep Python detection as the default and retain legacy result classes, + transformation strategies, aliases, full-match allowlists, and overlap order. +- Require Core capability contract version 1 for the Rust backend. +- Accept future finding labels without consulting Python's detector inventory. +- Read text applicability and activation settings from `entities` metadata. + `PERSON` is structured-only; explicit text selection must fail clearly. +- Derive configuration activations, including UUID opt-in, from metadata. +- Validate explicit locales against the advertised identifiers using trimming + and ASCII case-insensitive comparison. Preserve the requested locale when + forwarding to Core. German aliases activate German detection; en-US and fr + are supported base-only locales. Do not use Python regex locale validation. +- Support Python's plural locales through singular Core scans, then deduplicate + identical entity labels and source ranges before legacy overlap handling. +- Ignore unrelated additive capability fields. Fail explicitly for incompatible + contracts or unusable metadata rather than silently falling back to Python. + +## Pending release gates + +1. Obtain the finalized candidate wheel and platform support matrix from Core. +2. Validate capability inventory and activation metadata against the candidate. +3. Exercise every new entity through scan, selection, allowlists, and redaction; + verify Unicode code-point slicing and multiple-locale deduplication. +4. Record reviewed 0.3-to-0.4 differences without changing the frozen 4.8.1 + observations: German coverage, stricter locales, and default JWT, PRIVATE_KEY, + contextual US_ROUTING_NUMBER, and contextual NPI detection. UUID stays opt-in. +5. Run the base suite without native dependencies and native candidate tests, + installed-wheel checks, and public-call performance comparisons. +6. Only after candidate validation, change the Rust extra to + `datafog-core>=0.4.0,<0.5`, update the native parity fixture and CI, and update + migration documentation to distinguish the new contract from the 0.3.1 bridge. +7. Verify the published Core wheel before publishing Python. No release is + authorized by this integration work. From 27231a1175d54e23807379ed807bd9406d36ab58 Mon Sep 17 00:00:00 2001 From: Sid Mohan <61345237+sidmohan0@users.noreply.github.com> Date: Mon, 28 Sep 2026 16:03:13 -0700 Subject: [PATCH 34/37] feat: discover Rust entities and activation from Core capabilities --- .github/workflows/ci.yml | 2 +- CHANGELOG.MD | 12 +- README.md | 10 + benchmarks/results-core-0.4.json | 162 +++++++++++ datafog/_core_capabilities.py | 141 ++++++++++ datafog/engine.py | 49 ++-- docs/core-0.4-integration.md | 53 +++- docs/getting-started.rst | 19 +- docs/migration-4.9.md | 70 ++++- docs/python-sdk.rst | 28 +- scripts/check_rust_install.py | 3 +- setup.py | 2 +- tests/contracts/rust-0.4.0.json | 46 +++ tests/rust_contract.py | 4 +- tests/test_core04_integration.py | 215 ++++++++++++++ tests/test_core_capability_adapter.py | 386 ++++++++++++++++++++++++++ tests/test_rust_backend.py | 45 +-- tests/test_rust_contract.py | 8 +- 18 files changed, 1176 insertions(+), 79 deletions(-) create mode 100644 benchmarks/results-core-0.4.json create mode 100644 datafog/_core_capabilities.py create mode 100644 tests/contracts/rust-0.4.0.json create mode 100644 tests/test_core04_integration.py create mode 100644 tests/test_core_capability_adapter.py diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 9d2f2b3f..be420465 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -40,7 +40,7 @@ jobs: - name: Verify installed package outside checkout imports run: python -I scripts/check_rust_install.py - name: Test compatibility, routing, preview and native parity - run: python -m pytest tests/test_contract_481.py tests/test_rust_backend.py tests/test_api_bridge_49.py tests/test_rust_contract.py -q + run: python -m pytest tests/test_contract_481.py tests/test_rust_backend.py tests/test_api_bridge_49.py tests/test_rust_contract.py tests/test_core_capability_adapter.py tests/test_core04_integration.py -q - name: Generate native parity report run: python -m tests.rust_contract --output rust-parity.json - uses: actions/upload-artifact@v4 diff --git a/CHANGELOG.MD b/CHANGELOG.MD index 904493f3..25e6edfd 100644 --- a/CHANGELOG.MD +++ b/CHANGELOG.MD @@ -5,8 +5,16 @@ #### Added - Experimental `datafog[rust]` detection with keyword-only `backend="rust"` - on scan/redact entry points; Python remains the default. Core 0.3.1 is pinned, - German requests are explicitly unsupported, and ML composition is unchanged. + on scan/redact entry points; Python remains the default. The follow-up requires + `datafog-core>=0.4.0,<0.5` and capability contract 1; Core 0.4.0 publication is + pending, so development validation uses candidate wheels. ML composition is unchanged. +- Capability-driven entity and locale discovery, German locale activation, + explicit UUID activation, and forward-compatible finding labels. Core's + structured-only PERSON is rejected for explicit Rust text selection. +- Core 0.4.0 adds JWT, private-key, contextual routing-number, and contextual NPI + detection, plus stricter explicit locale validation. The legacy overlap policy + still prefers PHONE over a same-span NPI; NPI filtering can return no entities. + Use native scanning/transformation when NPI must be retained. - `datafog.compat.v4` preserves the existing scan/redact facade and result classes; `datafog.v5` previews actual Core types and transformation APIs. - Native parity tests, installed-wheel CI checks, and public-call benchmarks. diff --git a/README.md b/README.md index 97741a60..a1e0c231 100644 --- a/README.md +++ b/README.md @@ -261,6 +261,16 @@ The [4.9 migration guide](docs/migration-4.9.md) explains opt-in Rust detection, the native `datafog.v5` preview, and the revised 5.0 retirement schedule for `detect`/`process`, OCR, and Spark. The Python detector remains the default. +The unreleased Rust adapter requires Core `>=0.4.0,<0.5` and capability contract 1. +Until Core 0.4.0 is published, development evaluation requires a validated candidate +wheel. Entity labels, locales, and activation settings come from the installed +Core, allowing compatible releases to add detectors without a Python update. +German detection is opt-in through locale or entity selection; UUID is opt-in +through `entity_types=["UUID"]`. Core's structured-only `PERSON` is unavailable +for explicit Rust text selection. The legacy overlap policy can suppress NPI in +favor of PHONE; the migration guide explains native alternatives. Lock the Core +version if detection output must remain reproducible. + The [4.8.1 compatibility contract](docs/migration-4.8.1-contract.md) records published Python behavior for the Rust migration, with frozen fixtures and instructions for independently reproducing them from the release wheel. diff --git a/benchmarks/results-core-0.4.json b/benchmarks/results-core-0.4.json new file mode 100644 index 00000000..4b24acc4 --- /dev/null +++ b/benchmarks/results-core-0.4.json @@ -0,0 +1,162 @@ +{ + "python": "3.12.13", + "platform": "macOS-26.6.2-arm64-arm-64bit", + "core_version": "0.4.0", + "method": "One warmup call, median repeated calls; no ML. Cold = process+import+first scan.", + "warm": [ + { + "payload": "short", + "operation": "scan", + "utf8_bytes": 25, + "python": { + "median_us": 20.800204947590828, + "samples_us": [ + 20.712295081466436, 20.800204947590828, 20.514580537565053, + 23.7722898600623, 20.953959901817143 + ], + "entities": 1, + "iterations_per_sample": 200 + }, + "rust": { + "median_us": 3.2310403184965253, + "samples_us": [ + 3.405414754524827, 3.2291648676618934, 3.268750151619315, + 3.2310403184965253, 3.2225000904873013 + ], + "entities": 1, + "iterations_per_sample": 200 + } + }, + { + "payload": "short", + "operation": "redact", + "utf8_bytes": 25, + "python": { + "median_us": 21.760420058853924, + "samples_us": [ + 21.941669983789325, 21.712500019930303, 21.760420058853924, + 22.200000239536166, 21.4781251270324 + ], + "entities": 1, + "iterations_per_sample": 200 + }, + "rust": { + "median_us": 4.631250048987567, + "samples_us": [ + 4.536875057965517, 5.366874975152314, 4.631250048987567, + 4.643334541469812, 4.628329770639539 + ], + "entities": 1, + "iterations_per_sample": 200 + } + }, + { + "payload": "mixed", + "operation": "scan", + "utf8_bytes": 108, + "python": { + "median_us": 42.857080698013306, + "samples_us": [ + 42.857080698013306, 43.618750059977174, 41.11749934963882, + 43.416660046204925, 40.88166053406894 + ], + "entities": 5, + "iterations_per_sample": 100 + }, + "rust": { + "median_us": 9.399170521646738, + "samples_us": [ + 9.652090957388282, 9.399170521646738, 9.429160272702575, + 9.3812495470047, 9.250829461961985 + ], + "entities": 5, + "iterations_per_sample": 100 + } + }, + { + "payload": "mixed", + "operation": "redact", + "utf8_bytes": 108, + "python": { + "median_us": 45.040000695735216, + "samples_us": [ + 44.95292087085545, 45.765830436721444, 44.35917013324797, + 45.040000695735216, 45.48249999061227 + ], + "entities": 5, + "iterations_per_sample": 100 + }, + "rust": { + "median_us": 13.997500063851476, + "samples_us": [ + 14.225839404389262, 13.675830559805036, 13.945830287411809, + 13.997500063851476, 14.635410625487566 + ], + "entities": 5, + "iterations_per_sample": 100 + } + }, + { + "payload": "large_sparse", + "operation": "scan", + "utf8_bytes": 1050018, + "python": { + "median_us": 127264.26400089017, + "samples_us": [ + 126637.02767652771, 126413.45833738646, 133560.56968526295, + 127264.26400089017, 131677.19434325895 + ], + "entities": 1, + "iterations_per_sample": 3 + }, + "rust": { + "median_us": 7252.249983139336, + "samples_us": [ + 7574.055343866348, 7437.528033430378, 7189.485981749992, + 7252.249983139336, 7154.236353623371 + ], + "entities": 1, + "iterations_per_sample": 3 + } + }, + { + "payload": "large_sparse", + "operation": "redact", + "utf8_bytes": 1050018, + "python": { + "median_us": 125197.09730986506, + "samples_us": [ + 124213.70833180845, 125561.3473476842, 124531.3056667025, + 127014.16663670292, 125197.09730986506 + ], + "entities": 1, + "iterations_per_sample": 3 + }, + "rust": { + "median_us": 7177.152670919895, + "samples_us": [ + 7177.152670919895, 7076.22233312577, 7222.194341011345, + 7156.6526700432105, 7303.208306742211 + ], + "entities": 1, + "iterations_per_sample": 3 + } + } + ], + "cold": { + "python": { + "median_ms": 81.64816698990762, + "samples_ms": [ + 84.22841690480709, 81.64816698990762, 81.9787080399692, + 81.22462499886751, 80.63562505412847 + ] + }, + "rust": { + "median_ms": 83.1466669915244, + "samples_ms": [ + 84.22383293509483, 82.70379202440381, 83.1466669915244, + 84.2408339958638, 82.79358292929828 + ] + } + } +} diff --git a/datafog/_core_capabilities.py b/datafog/_core_capabilities.py new file mode 100644 index 00000000..a5a0df4b --- /dev/null +++ b/datafog/_core_capabilities.py @@ -0,0 +1,141 @@ +"""Translate installed Core capabilities into native text scan configuration.""" + +from __future__ import annotations + +from copy import deepcopy +from functools import lru_cache +from typing import Any + +_ASCII_LOWER = str.maketrans("ABCDEFGHIJKLMNOPQRSTUVWXYZ", "abcdefghijklmnopqrstuvwxyz") + + +def _locale_key(value: str) -> str: + return value.strip(" \t\n\r\v\f").translate(_ASCII_LOWER) + + +def _invalid(detail: str) -> RuntimeError: + return RuntimeError(f"Incompatible datafog-core capabilities: {detail}") + + +def _labels(value: Any, field: str) -> set[str]: + if not isinstance(value, list) or any( + not isinstance(label, str) or not label for label in value + ): + raise _invalid(f"{field} must be a list of entity labels") + return set(value) + + +@lru_cache(maxsize=4) +def _read_capabilities(capabilities: Any) -> tuple[set[str], dict, dict]: + """Snapshot one installed build; replacing its reader invalidates the cache. + + The private snapshot never escapes into native scan configuration. Failed + capability reads or validation are not cached by lru_cache. + """ + payload = deepcopy(capabilities()) + if not isinstance(payload, dict): + raise _invalid("capabilities() must return a dictionary") + version = payload.get("contract_version") + if not isinstance(version, int) or isinstance(version, bool) or version != 1: + raise _invalid(f"unsupported contract_version {version!r}; expected 1") + supported = _labels(payload.get("supported_entities"), "supported_entities") + defaults = _labels(payload.get("default_entities"), "default_entities") + if not defaults <= supported: + raise _invalid("default_entities contains unsupported labels") + advertised_locales = payload.get("locales") + metadata = payload.get("entities") + if not isinstance(advertised_locales, dict) or not isinstance(metadata, dict): + raise _invalid("locales and entities must be dictionaries") + + return supported, _locale_lookup(advertised_locales, supported), metadata + + +def _locale_lookup(advertised_locales: dict, supported: set[str]) -> dict[str, str]: + locale_lookup: dict[str, str] = {} + for locale, details in advertised_locales.items(): + if not isinstance(locale, str) or not _locale_key(locale): + raise _invalid("locale identifiers must be nonempty strings") + if ( + not isinstance(details, dict) + or not _labels( + details.get("enabled_entities"), f"locales[{locale!r}].enabled_entities" + ) + <= supported + ): + raise _invalid(f"invalid locale metadata for {locale!r}") + locale_lookup[_locale_key(locale)] = locale + + return locale_lookup + + +def _resolve_locale(value: Any, locale_lookup: dict[str, str]) -> str: + if not isinstance(value, str) or _locale_key(value) not in locale_lookup: + raise ValueError(f"Unsupported locale for the Rust backend: {value!r}") + return value + + +def _requested_locales(locales: Any, lookup: dict[str, str]) -> set[str]: + if isinstance(locales, str): + locales = [locales] + if locales is not None and ( + not isinstance(locales, (list, tuple)) + or any(not isinstance(locale, str) for locale in locales) + ): + raise ValueError("locales must be a list of locale identifiers") + return {_resolve_locale(locale, lookup) for locale in locales or []} + + +def _activation_config(label: str, metadata: dict) -> dict: + details = metadata.get(label) + if not isinstance(details, dict): + raise _invalid(f"missing entity metadata for {label!r}") + scopes = details.get("scopes") + if not isinstance(scopes, list) or any( + not isinstance(scope, str) for scope in scopes + ): + raise _invalid(f"invalid scopes for {label!r}") + if "text" not in scopes: + raise ValueError( + f"Entity {label!r} is not available for Rust text scanning " + "(structured-only entity)" + ) + activation = details.get("activation") + if not isinstance(activation, dict): + raise _invalid(f"missing activation metadata for {label!r}") + kind = activation.get("kind") + if kind == "default": + return {} + if kind not in {"locale", "config"}: + raise _invalid(f"unsupported activation kind {kind!r} for {label!r}") + config = activation.get("scan_config") + if not isinstance(config, dict) or not config: + raise _invalid(f"missing scan_config for {label!r}") + if kind == "locale" and "locale" not in config: + raise _invalid(f"missing activation locale for {label!r}") + return deepcopy(config) + + +def scan_configs(core: Any, requested: set[str], locales: Any) -> list[dict]: + """Build a union of singular-locale scans without duplicating inventories.""" + capabilities = getattr(core, "capabilities", None) + if not callable(capabilities): + raise _invalid( + "contract version 1 is required; install a compatible Core 0.4.x" + ) + supported, lookup, metadata = _read_capabilities(capabilities) + selected_locales = _requested_locales(locales, lookup) + common: dict[str, Any] = {} + for label in sorted(requested & supported): + config = _activation_config(label, metadata) + for key, value in config.items(): + if key == "locale": + selected_locales.add(_resolve_locale(value, lookup)) + elif key in common and common[key] != value: + raise _invalid(f"conflicting activation values for {key!r}") + else: + common[key] = value + if not selected_locales: + return [common] + return [ + deepcopy(dict(common, locale=locale)) for locale in sorted(selected_locales) + ] diff --git a/datafog/engine.py b/datafog/engine.py index 3873a3d2..0e2a0fb4 100644 --- a/datafog/engine.py +++ b/datafog/engine.py @@ -220,14 +220,6 @@ def _rust_entities( suppression before entity selection and allowlists, preserving that order without duplicating the native detectors. """ - normalized_locales = RegexAnnotator._normalize_locales(locales) - requested = {_canonical_type(value) for value in entity_types or []} - if normalized_locales or any(value.startswith("DE_") for value in requested): - raise ValueError( - "backend='rust' does not yet support German locales or DE_* entity " - "types; use backend='python' for German detection" - ) - try: import datafog_core except ImportError as exc: @@ -236,26 +228,31 @@ def _rust_entities( 'Install with: pip install "datafog[rust]"' ) from exc + from ._core_capabilities import scan_configs + + requested = {_canonical_type(value) for value in entity_types or []} + configs = scan_configs(datafog_core, requested, locales) entities: list[Entity] = [] - for finding in datafog_core.scan(text): - canonical_type = _canonical_type(finding.entity_type) - if canonical_type not in ALL_ENTITY_TYPES: - raise RuntimeError( - f"Rust backend returned unsupported entity type: " - f"{finding.entity_type!r}; check the installed datafog-core version" - ) - if not finding.matched_text.strip(): - continue - entities.append( - Entity( - type=canonical_type, - text=finding.matched_text, - start=finding.codepoint_range.start, - end=finding.codepoint_range.end, - confidence=1.0, - engine="regex", + seen: set[tuple[str, int, int]] = set() + for config in configs: + for finding in datafog_core.scan(text, config=config): + canonical_type = _canonical_type(finding.entity_type) + start = finding.codepoint_range.start + end = finding.codepoint_range.end + key = (canonical_type, start, end) + if key in seen or not finding.matched_text.strip(): + continue + seen.add(key) + entities.append( + Entity( + type=canonical_type, + text=finding.matched_text, + start=start, + end=end, + confidence=1.0, + engine="regex", + ) ) - ) return _suppress_overlapping_entities(entities) diff --git a/docs/core-0.4-integration.md b/docs/core-0.4-integration.md index 0fe87e94..5ecec1ac 100644 --- a/docs/core-0.4-integration.md +++ b/docs/core-0.4-integration.md @@ -1,8 +1,9 @@ # Core 0.4 capability adapter integration gate -Status: draft implementation against the proposed capability contract. No Core -candidate wheel has been validated yet. The dependency range remains unchanged -until that gate passes; this branch must not merge or publish in this state. +Status: draft implementation validated against the final local Core 0.4.0 +candidate wheel. Publication and hosted native CI remain pending; this branch +must not merge or publish until those installation gates pass. See candidate +evidence below. ## Scope @@ -38,3 +39,49 @@ until that gate passes; this branch must not merge or publish in this state. migration documentation to distinguish the new contract from the 0.3.1 bridge. 7. Verify the published Core wheel before publishing Python. No release is authorized by this integration work. + +## Candidate evidence + +The final Core wheel was force-reinstalled and tested on macOS ARM64, CPython +3.12. It reports Core 0.4.0 and capability contract 1. + +- Source revision supplied by Core: `60c4636`. +- Artifact: `datafog_core-0.4.0-cp310-abi3-macosx_11_0_arm64.whl`. +- Verified SHA256: + `351ab81ae575b31a6739a43e29135884ccc0d7152f01df568cd1f7608d690b1c`. +- Focused candidate/adapter/legacy contract suites: **352 passed**. +- Base/CLI regression without native dependency: **833 passed, 19 skipped, + 295 deselected, 19 existing xfails**. +- Frozen corpus: **77 exact matches, two detector differences, one validation + difference, 31 cases outside backend scope**. All formerly unsupported German + cases now match. The invalid-card and embedded-SSN differences persist. The + unknown-locale exception remains `ValueError` with a new capability-based + diagnostic. Original `4.8.1.json` and historical `rust-0.3.1.json` are unchanged. +- Clean Python wheel installed with the exact Core candidate: isolated smoke test + and `pip check` passed. +- Sphinx documentation build passed with existing static/autodoc warnings. +- Public scan/redact and cold-start measurements are recorded in + `../benchmarks/results-core-0.4.json`; compare only like-for-like local runs. + Median warm Rust scan: 3.23 µs short, 9.40 µs mixed, 7.25 ms for 1 MB sparse. + Python equivalents: 20.80 µs, 42.86 µs, 127.26 ms. These are candidate-local + measurements, not universal speed guarantees or a claim of unchanged Core + 0.3.1 detector cost. Validated capabilities are cached per native reader + identity; reloading/replacing that reader gets a new snapshot. + +### Reviewed NPI overlap limitation + +Core returns both `PHONE` and `NPI` for `NPI 1234567893`. The legacy adapter +preserves overlap-before-selection and its existing priority: `PHONE` wins. +Consequently, explicit `entity_types=["NPI"]` can return no entities; default +legacy redaction still protects the number as a phone. Native `datafog.v5.scan` +retains both findings, and native transformation can explicitly select `NPI`. + +This is an intentional compatibility limitation, not complete NPI parity. +A regression test asserts this exact behavior. No new hardcoded entity +priorities or selection-order changes were introduced. Any future change needs +an explicit overlap-policy decision rather than silently changing legacy output. + +The tested wheel satisfies the local candidate gate. The supported dependency +range is now `>=0.4.0,<0.5`. Hosted native CI and published-wheel validation remain +blocked until Core 0.4.0 is available to those installers; do not publish Python +or treat local macOS validation as a cross-platform CI result. diff --git a/docs/getting-started.rst b/docs/getting-started.rst index 8df06248..eba0bcb6 100644 --- a/docs/getting-started.rst +++ b/docs/getting-started.rst @@ -56,7 +56,9 @@ extra to evaluate them: python -m pip install -e ".[rust]" -The extra pins ``datafog-core==0.3.1``. It does not change the default Python +The extra requires ``datafog-core>=0.4.0,<0.5`` and capability contract 1. +Core 0.4.0 is not published yet: development evaluation requires a validated +candidate wheel until the dependency is available on PyPI. It does not change the default Python backend, and the existing ``all`` extra does not include Rust. To opt in: .. code-block:: python @@ -66,10 +68,17 @@ backend, and the existing ``all`` extra does not include Rust. To opt in: result = datafog.scan("Contact jane@example.com", engine="regex", backend="rust") print(result.entities) -Only ``engine="regex"`` supports this backend. German locales and ``DE_*`` entity -selections are unsupported by the pinned Core version and explicitly rejected -by the legacy Rust adapter; use the Python backend for German detection. Other -detector differences remain, so this is an experimental comparison path. +Only ``engine="regex"`` supports this backend. The adapter discovers entities and +activation settings from Core capabilities. German detection is enabled by +``locales=["de"]`` (also ``de-DE`` or ``de_DE``), or explicit German entity +selection. ``en-US`` and ``fr`` are accepted base-only locales; other explicit +locales fail validation. Select ``entity_types=["UUID"]`` to enable UUID. +Core's structured-only ``PERSON`` cannot be explicitly selected for text scanning. + +Future Core labels pass through, but detection output may change across compatible +updates. The preserved legacy overlap policy may retain ``PHONE`` instead of a +same-span ``NPI``, including when filtering for NPI. See the migration guide for +this limitation and native alternatives. For native Core types use ``datafog.v5``; for the explicit legacy facade use ``datafog.compat.v4``. See :doc:`python-sdk` and the diff --git a/docs/migration-4.9.md b/docs/migration-4.9.md index 72233796..c1b763cf 100644 --- a/docs/migration-4.9.md +++ b/docs/migration-4.9.md @@ -1,8 +1,9 @@ # Migrating incrementally with DataFog 4.9 > **Unreleased:** This guide describes the upcoming 4.9 bridge. Until it is -> published, evaluate these APIs from the development checkout with -> `python -m pip install -e ".[rust]"`; a normal PyPI install does not include them. +> published, use the development checkout and a validated Core 0.4.0 candidate +> wheel. The declared Core dependency cannot resolve from PyPI until Core 0.4.0 +> is published. A normal released Python install does not include this follow-up. 4.9 is a bridge to the Rust-backed 5.0 API. The default Python detector, existing imports, result objects, and redaction strategies continue to work. The optional @@ -14,7 +15,10 @@ Rust backend and native API preview are experimental and explicitly selected. pip install "datafog[rust]" ``` -The extra pins the tested `datafog-core==0.3.1` wheel. The base package needs no +The extra requires `datafog-core>=0.4.0,<0.5` and capability contract version 1. +Compatible Core updates can add entity labels and locales without a Python +release. Detection results may evolve; lock Core for reproducible output. +Missing capabilities or incompatible contract versions fail clearly. The base package needs no Rust installation or native module. The existing `all` extra retains its legacy dependency set; request `rust` explicitly, or use `datafog[all,rust]`. @@ -93,22 +97,57 @@ assume that the two schemas and provider requirements are interchangeable. ## Known detection differences -Core 0.3.1 does not implement the seven German detectors. The legacy Rust adapter -rejects German locales and `DE_*` selections, directing callers to Python. -The raw native preview retains Core's own API: its accepted locale configuration -does not imply German detector coverage. Continue using Python for those workloads. +The adapter reads `datafog_core.capabilities()` from the installed build instead +of maintaining a second supported-entity inventory. Unknown future finding labels +are preserved. Entity selection derives activation settings from Core metadata. +`PERSON` is structured-only in Core 0.4.0; explicitly selecting it for Rust text +scanning raises an error. Use native structured scanning for that entity. + +Core 0.4.0 supports the seven German detectors through `de`, `de-DE`, or `de_DE`. +Locale matching trims ASCII whitespace and ignores ASCII case. `en-US` and `fr` +are accepted base-only locales; other explicit locales raise an error. Python +accepts multiple locales, scans each through Core's singular-locale interface, +and deduplicates findings before applying the existing overlap policy. + +```python +result = datafog.scan( + "DE89370400440532013000", backend="rust", locales=["de"] +) +assert result.entities[0].type == "DE_IBAN" + +result = datafog.scan( + "550e8400-e29b-41d4-a716-446655440000", + backend="rust", + entity_types=["UUID"], +) +assert result.entities[0].type == "UUID" +``` + +German entity selection also enables its advertised locale automatically. UUID +selection enables Core's `detect_uuid` setting; UUID is not enabled by default. +Core 0.4.0 adds default JWT, private-key, contextual US routing number, and +contextual NPI detection. Routing numbers and NPIs still require textual context. +These detector additions and stricter locale validation intentionally change +native behavior compared with Core 0.3.1. + +**NPI compatibility limitation:** Core can emit `PHONE` and `NPI` for the same +span. The legacy adapter preserves its existing overlap-before-filter policy, +which keeps `PHONE` in that tie. Consequently `entity_types=["NPI"]` can return +no entities. Native `datafog.v5.scan()` retains the NPI finding; native +transformation can select NPI. This limitation does not remove NPI support from +Core, and changing the legacy overlap policy is outside this increment. The frozen 111-case baseline yields the following Rust-backend comparison: -| Outcome | Cases | Interpretation | -| ---------------------------- | ----: | ------------------------------------------------------------------------- | -| Exact match | 61 | Same observable result on these inputs | -| Reviewed detector difference | 2 | Invalid-checksum card and alphanumeric-embedded SSN are rejected by Core | -| Explicitly unsupported | 17 | German requests fail rather than silently lose coverage | -| Outside backend scope | 31 | Signatures, explicit-span transformations, legacy/service/guardrail paths | +| Outcome | Cases | Interpretation | +| ------------------------------ | ----: | ------------------------------------------------------------------------- | +| Exact match | 77 | Same observable result on these inputs | +| Reviewed detector difference | 2 | Invalid-checksum card and alphanumeric-embedded SSN are rejected by Core | +| Reviewed validation difference | 1 | Core rejects an unsupported explicit locale with its stricter validation | +| Outside backend scope | 31 | Signatures, explicit-span transformations, legacy/service/guardrail paths | These counts describe this finite synthetic corpus, not universal detection -equivalence or precision/recall. See `tests/contracts/rust-0.3.1.json` for exact +equivalence or precision/recall. See `tests/contracts/rust-0.4.0.json` for exact reviewed outcomes and reasons. Each applicable case is asserted independently; unknown differences fail CI. Generate the full per-case report with: @@ -159,7 +198,8 @@ synthetic text. It reports entity counts alongside timings. No general speedup claim is made from a single machine or from cases with different outputs. A local CPython 3.12 macOS ARM64 reference run is recorded in -`benchmarks/results-4.9.json`. Median scan latency was 19.74 versus 2.30 microseconds +`benchmarks/results-4.9.json`. This historical run used Core 0.3.1 and does not +measure Core 0.4.0. Median scan latency was 19.74 versus 2.30 microseconds for the short payload, 39.50 versus 7.58 microseconds for mixed PII, and 121.91 versus 5.33 milliseconds for the large sparse payload (Python versus Rust). Fresh-process import plus first scan was approximately 81 milliseconds for both. diff --git a/docs/python-sdk.rst b/docs/python-sdk.rst index 7e2a2283..3ebe48f7 100644 --- a/docs/python-sdk.rst +++ b/docs/python-sdk.rst @@ -44,7 +44,10 @@ continue to work at the top level in 4.9 but warn of removal in 5.0, revising th previous promise to retain them throughout 5.x. Moving from ``process`` to ``redact`` can change old placeholder and hash output; compare results explicitly. -Install ``.[rust]`` from the development checkout to evaluate Rust detection: +The unreleased adapter requires ``datafog-core>=0.4.0,<0.5``. Until Core 0.4.0 +is published, use a validated candidate wheel with the development checkout; +installing ``.[rust]`` from PyPI alone cannot resolve the new dependency yet. +Once available, install ``.[rust]`` to evaluate Rust detection: .. code-block:: python @@ -84,9 +87,26 @@ native strategies. Importing the namespace is lazy; accessing its exports requires the Rust extra. The supported compatibility lifetime after 5.0 remains a separate decision. -The pinned Core 0.3.1 lacks German detectors and differs on some structured -inputs. The legacy Rust adapter rejects German requests; the raw native preview -retains Core's own behavior and must not be assumed to provide German coverage. +The adapter requires capability contract version 1 and discovers supported labels, +locales, and activation settings from the installed Core. Future finding labels +are preserved. Core 0.4.x updates may change detection output; lock the version +when reproducibility is required. German aliases ``de``, ``de-DE``, and ``de_DE`` +activate German detectors; ``en-US`` and ``fr`` activate only base detectors. +Locale validation trims ASCII whitespace and ignores ASCII case; unsupported +explicit locales raise errors. Multiple Python locales produce a deduplicated +union of Core scans. Explicit German entity selection enables the required locale. + +Select ``entity_types=["UUID"]`` with ``backend="rust"`` to enable UUID detection +through Core metadata. Core's ``PERSON`` is structured-only and explicitly +selecting it for Rust text scanning raises an error. The Python default backend +keeps its existing behavior. + +Core 0.4.0 adds JWT, private-key, contextual routing-number, and contextual NPI +findings. The legacy overlap policy can prefer ``PHONE`` over a same-span ``NPI``; +filtering for NPI then returns no entities. Native ``datafog.v5.scan`` retains NPI, +and native transformation can select it. Legacy result and transformation +semantics remain unchanged. + Read the :download:`complete migration guide ` for the exact schema comparison, finite-corpus parity results, and verification commands. diff --git a/scripts/check_rust_install.py b/scripts/check_rust_install.py index f0e63fd9..26120232 100644 --- a/scripts/check_rust_install.py +++ b/scripts/check_rust_install.py @@ -13,7 +13,8 @@ def main(): from datafog import v5 from datafog.compat import v4 - assert importlib.metadata.version("datafog-core") == "0.3.1" + assert importlib.metadata.version("datafog-core").startswith("0.4.") + assert datafog_core.capabilities()["contract_version"] == 1 assert v4.Entity is datafog.Entity assert v5.Finding is datafog_core.Finding text = "👋 Contact alice@example.com" diff --git a/setup.py b/setup.py index 4f7d7e90..72cb2da3 100644 --- a/setup.py +++ b/setup.py @@ -84,7 +84,7 @@ ] extras_require = { - "rust": ["datafog-core==0.3.1"], + "rust": ["datafog-core>=0.4.0,<0.5"], "nlp": nlp_deps, "nlp-advanced": nlp_advanced_deps, "ocr": ocr_deps, diff --git a/tests/contracts/rust-0.4.0.json b/tests/contracts/rust-0.4.0.json new file mode 100644 index 00000000..541cc3a3 --- /dev/null +++ b/tests/contracts/rust-0.4.0.json @@ -0,0 +1,46 @@ +{ + "core_version": "0.4.0", + "cases": { + "scan-observation-invalid-card": { + "classification": "detector-difference", + "reason": "Core validates card checksums; Python 4.8.1 accepts this card-shaped value. Experimental precision difference, not a claim of full parity.", + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [], + "text": "4111 1111 1111 1112", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + "scan-observation-boundary": { + "classification": "detector-difference", + "reason": "Core rejects the SSN embedded in an ASCII alphanumeric token; Python 4.8.1 detects the numeric substring.", + "expected": { + "value": { + "result_type": "ScanResult", + "fields": { + "entities": [], + "text": "x123-45-6789y", + "engine_used": "regex" + } + }, + "warnings": [] + } + }, + "locale-unknown": { + "classification": "validation-difference", + "reason": "The capability adapter rejects the unadvertised locale with a backend-specific diagnostic instead of the Python regex locale inventory. The exception remains ValueError.", + "expected": { + "error": { + "type": "ValueError", + "message": "Unsupported locale for the Rust backend: 'xx'" + }, + "warnings": [] + } + } + } +} diff --git a/tests/rust_contract.py b/tests/rust_contract.py index a63184b1..3b477231 100644 --- a/tests/rust_contract.py +++ b/tests/rust_contract.py @@ -1,4 +1,4 @@ -"""Compare applicable 4.8.1 observations with the opt-in published Rust backend.""" +"""Compare applicable 4.8.1 observations with the opt-in Rust backend.""" import argparse import copy @@ -9,7 +9,7 @@ from tests.contract_481 import FIXTURE, observe -DEVIATIONS = Path(__file__).parent / "contracts" / "rust-0.3.1.json" +DEVIATIONS = Path(__file__).parent / "contracts" / "rust-0.4.0.json" TARGETS = { "datafog:scan", "datafog:redact", diff --git a/tests/test_core04_integration.py b/tests/test_core04_integration.py new file mode 100644 index 00000000..272060af --- /dev/null +++ b/tests/test_core04_integration.py @@ -0,0 +1,215 @@ +"""Exercise actual Core 0.4 wheels through the Python compatibility adapter. + +Examples originate in Core's fixtures/{german,jwt,private-key,npi, +us-routing-number,uuid}.jsonl. Payloads are synthetic detector fixtures, +including a nonfunctional PEM block; no credentials are used. +""" + +import importlib.metadata +import re + +import pytest + +import datafog +from datafog import engine, v5 +from datafog.compat import v4 + +core = pytest.importorskip("datafog_core") + +GERMAN = [ + ("DE_IBAN", "DE44 5001 0517 5407 3249 31", "DE44 5001 0517 5407 3249 31"), + ("DE_VAT_ID", "USt-IdNr DE 123456789 ist gesetzt.", "DE 123456789"), + ("DE_TAX_ID", "Steuer-ID 12345678901 liegt vor.", "12345678901"), + ( + "DE_SOCIAL_SECURITY_NUMBER", + "Rentenversicherungsnummer 65150804A123 liegt vor.", + "65150804A123", + ), + ("DE_POSTAL_CODE", "PLZ10115 Berlin.", "PLZ10115"), + ("DE_PASSPORT_NUMBER", "Passnummer C12345678 wurde geprueft.", "C12345678"), + ( + "DE_RESIDENCE_PERMIT_NUMBER", + "Aufenthaltstitel AT1234567 gueltig.", + "AT1234567", + ), +] +JWT = "eyJhbGciOiJIUzI1NiJ9.eyJzdWIiOiJ0ZXN0IiwiZXhwIjowfQ.c2ln" +# Deliberately invalid key material: the payload decodes to ASCII "abcd". +PRIVATE_KEY = ( + "-----BEGIN PRIVATE KEY-----\nYWJjZA==\n-----END PRIVATE KEY-----" # gitleaks:allow +) +UUID = "550e8400-e29b-11d4-a716-446655440000" +NEW_DEFAULTS = [ + ("JWT", JWT, JWT), + ("PRIVATE_KEY", PRIVATE_KEY, PRIVATE_KEY), + ("NPI", "NPI 1234567893", "1234567893"), + ("US_ROUTING_NUMBER", "routing 021000021", "021000021"), +] +COMPAT_NEW_DEFAULTS = [sample for sample in NEW_DEFAULTS if sample[0] != "NPI"] + + +def test_supported_wheel_version_and_capability_contract(): + version = importlib.metadata.version("datafog-core") + assert version.split(".")[:2] == ["0", "4"], version + capabilities = core.capabilities() + assert capabilities["contract_version"] == 1 + supported = capabilities["supported_entities"] + defaults = capabilities["default_entities"] + assert supported == sorted(set(supported)) + assert defaults == sorted(set(defaults)) + assert set(defaults) <= set(supported) + assert {label for label, _, _ in NEW_DEFAULTS} <= set(defaults) + assert "PERSON" in supported and "PERSON" not in defaults + assert "UUID" in supported and "UUID" not in defaults + assert capabilities["entities"]["PERSON"]["scopes"] == ["structured"] + assert capabilities["entities"]["UUID"]["activation"]["scan_config"] == { + "detect_uuid": True + } + german_labels = {label for label, _, _ in GERMAN} + assert not german_labels & set(defaults) + for locale in ("de", "de-DE", "de_DE"): + assert german_labels <= set(capabilities["locales"][locale]["enabled_entities"]) + + +@pytest.mark.parametrize("label, sample, value", COMPAT_NEW_DEFAULTS) +@pytest.mark.parametrize("explicit_selection", [False, True]) +def test_new_default_entities_survive_adapter_and_redaction( + label, sample, value, explicit_selection +): + text = "😀 café\n" + sample + options = {"backend": "rust"} + if explicit_selection: + options["entity_types"] = [label] + result = datafog.scan(text, **options) + assert type(result) is v4.ScanResult + matches = [item for item in result.entities if item.type == label] + assert len(matches) == 1 + item = matches[0] + assert type(item) is v4.Entity + assert item.text == text[item.start : item.end] == value + redacted = datafog.redact(text, **options) + assert type(redacted) is v4.RedactResult + assert redacted.redacted_text == text.replace(value, f"[{label}_1]") + + +@pytest.mark.parametrize("label, sample, value", GERMAN) +@pytest.mark.parametrize("activation", ["locale", "selection"]) +def test_seven_german_entities_activate_and_redact(label, sample, value, activation): + text = "😀 München " + sample + assert not any( + item.type == label for item in datafog.scan(text, backend="rust").entities + ) + options = {"backend": "rust"} + if activation == "locale": + options["locales"] = ["de"] + else: + options["entity_types"] = [label] + result = datafog.scan(text, **options) + matches = [item for item in result.entities if item.type == label] + assert len(matches) == 1 + item = matches[0] + assert item.text == text[item.start : item.end] == value + assert datafog.redact(text, **options).redacted_text == text.replace( + value, f"[{label}_1]" + ) + + +def test_uuid_optin_selection_drives_native_configuration(): + text = "😀 café " + UUID + assert "UUID" not in { + item.type for item in datafog.scan(text, backend="rust").entities + } + options = {"backend": "rust", "entity_types": ["UUID"]} + matches = datafog.scan(text, **options).entities + assert [(item.type, item.text) for item in matches] == [("UUID", UUID)] + assert text[matches[0].start : matches[0].end] == UUID + assert datafog.redact(text, **options).redacted_text == "😀 café [UUID_1]" + + +@pytest.mark.parametrize("locale", ["de", "de-DE", "de_DE", " DE-dE\t"]) +def test_real_german_aliases(locale): + sample = GERMAN[0][1] + matches = datafog.scan(sample, locales=[locale], backend="rust").entities + assert [(item.type, item.text) for item in matches] == [("DE_IBAN", sample)] + + +def test_plural_locales_union_without_duplicate_redactions(): + text = "😀 a@example.com " + GERMAN[0][1] + options = {"locales": ["fr", "de_DE", "en-US", "de", "de"], "backend": "rust"} + matches = datafog.scan(text, **options).entities + assert [(item.type, item.text) for item in matches] == [ + ("EMAIL", "a@example.com"), + ("DE_IBAN", GERMAN[0][1]), + ] + assert datafog.redact(text, **options).redacted_text == "😀 [EMAIL_1] [DE_IBAN_1]" + + +@pytest.mark.parametrize("label, sample, value", COMPAT_NEW_DEFAULTS + GERMAN) +def test_new_labels_respect_existing_selection_and_allowlists(label, sample, value): + text = "😀\n" + sample + options = {"entity_types": [label], "backend": "rust"} + assert [item.text for item in datafog.scan(text, **options).entities] == [value] + assert not datafog.scan(text, allowlist=[value], **options).entities + assert not datafog.scan( + text, allowlist_patterns=[re.escape(value)], **options + ).entities + assert datafog.redact(text, allowlist=[value], **options).redacted_text == text + + +@pytest.mark.parametrize("strategy", ["token", "mask", "hash", "pseudonymize"]) +def test_native_findings_use_legacy_transformations(strategy): + text = "😀 " + GERMAN[0][1] + " and " + UUID + options = {"entity_types": ["DE_IBAN", "UUID"], "backend": "rust"} + scanned = datafog.scan(text, **options) + assert [item.type for item in scanned.entities] == ["DE_IBAN", "UUID"] + expected = engine.redact(text, scanned.entities, strategy=strategy) + assert datafog.redact(text, strategy=strategy, **options) == expected + + +@pytest.mark.parametrize( + "label, raw", [("NPI", "1234567893"), ("US_ROUTING_NUMBER", "021000021")] +) +def test_contextual_identifiers_do_not_activate_without_context(label, raw): + result = datafog.scan(raw, entity_types=[label], backend="rust") + assert not result.entities + + +def test_npi_retains_native_identity_but_legacy_overlap_prefers_phone(): + """4.x suppresses overlaps before selection; this is a reviewed difference.""" + text = "😀 café NPI 1234567893" + native = v5.scan(text) + assert [(item.entity_type, item.matched_text) for item in native] == [ + ("PHONE", "1234567893"), + ("NPI", "1234567893"), + ] + assert { + (item.codepoint_range.start, item.codepoint_range.end) for item in native + } == {(11, 21)} + legacy = datafog.scan(text, backend="rust") + assert [(item.type, item.text) for item in legacy.entities] == [ + ("PHONE", "1234567893") + ] + assert not datafog.scan(text, entity_types=["NPI"], backend="rust").entities + assert datafog.redact(text, backend="rust").redacted_text == "😀 café NPI [PHONE_1]" + # Native selection occurs before transformation overlap resolution, exposing + # the NPI-specific route without changing the 4.x compatibility contract. + result = v5.scan_and_transform( + text, + {"transform": {"default": {"strategy": "redact"}, "entities": ["NPI"]}}, + ) + assert result.text == "😀 café NPI [NPI]" + + +def test_structured_only_person_rejected_by_text_adapter(): + with pytest.raises(ValueError, match="PERSON"): + datafog.scan("Jane Doe", entity_types=["PERSON"], backend="rust") + + +def test_unsupported_locale_rejected_explicitly(): + with pytest.raises(ValueError, match="locale"): + datafog.scan("a@example.com", locales=["unsupported"], backend="rust") + + +def test_python_default_backend_remains_unchanged(): + assert not datafog.scan(JWT).entities + assert [item.type for item in datafog.scan(JWT, backend="rust").entities] == ["JWT"] diff --git a/tests/test_core_capability_adapter.py b/tests/test_core_capability_adapter.py new file mode 100644 index 00000000..dce01060 --- /dev/null +++ b/tests/test_core_capability_adapter.py @@ -0,0 +1,386 @@ +"""The Rust compatibility adapter follows installed Core capabilities. + +These tests use a small synthetic registry, not a copy of Core's entity list. +Real candidate-wheel coverage lives separately in the native contract suite. +""" + +import copy +import sys +from types import SimpleNamespace + +import pytest + +import datafog +from datafog import engine +from datafog.compat import v4 + + +def capability_fixture(): + def entity(kind="default", config=None, scopes=None): + activation = {"kind": kind} + if config is not None: + activation["scan_config"] = config + return { + "scopes": scopes or ["structured", "text"], + "activation": activation, + } + + entities = { + "EMAIL": entity(), + "FUTURE_ID": entity(), + "UUID": entity("config", {"detect_uuid": True}), + "FUTURE_OPTIN": entity("config", {"detect_future": True}), + "DE_IBAN": entity("locale", {"locale": "de"}), + "FUTURE_LOCAL": entity("locale", {"locale": "zz"}), + "PERSON": entity("structured", scopes=["structured"]), + } + return { + "contract_version": 1, + "supported_entities": sorted(entities), + "default_entities": ["EMAIL", "FUTURE_ID"], + "locales": { + "de": {"enabled_entities": ["DE_IBAN"]}, + "de-DE": {"enabled_entities": ["DE_IBAN"]}, + "de_DE": {"enabled_entities": ["DE_IBAN"]}, + "en-US": {"enabled_entities": []}, + "fr": {"enabled_entities": []}, + "zz": {"enabled_entities": ["FUTURE_LOCAL"]}, + }, + "entities": entities, + } + + +def finding(text, value, label, start=None): + start = text.index(value) if start is None else start + return SimpleNamespace( + entity_type=label, + matched_text=value, + codepoint_range=SimpleNamespace(start=start, end=start + len(value)), + byte_range=SimpleNamespace( + start=len(text[:start].encode()), + end=len(text[: start + len(value)].encode()), + ), + ) + + +@pytest.fixture +def core(monkeypatch): + state = SimpleNamespace( + metadata=capability_fixture(), calls=[], findings=[], response=None + ) + + def scan(text, config=None): + state.calls.append((text, copy.deepcopy(config))) + if state.response: + return state.response(text, config) + return state.findings + + state.module = SimpleNamespace( + capabilities=lambda: copy.deepcopy(state.metadata), scan=scan + ) + monkeypatch.setitem(sys.modules, "datafog_core", state.module) + return state + + +def test_default_scan_leaves_optin_configuration_disabled(core): + datafog.scan("safe text", backend="rust") + assert len(core.calls) == 1 + assert not core.calls[0][1].get("detect_uuid", False) + assert not core.calls[0][1].get("detect_future", False) + assert not core.calls[0][1].get("locale") + + +@pytest.mark.parametrize("label", ["FUTURE_ID", "UNADVERTISED_FUTURE"]) +def test_future_findings_preserve_legacy_classes_and_unicode_offsets(core, label): + text = "😀 München opaque-secret" + native = finding(text, "opaque-secret", label) + assert native.byte_range.start != native.codepoint_range.start + core.findings = [native] + result = datafog.scan(text, backend="rust") + assert type(result) is v4.ScanResult is engine.ScanResult + assert type(result.entities[0]) is v4.Entity is engine.Entity + item = result.entities[0] + assert item.type == label + assert text[item.start : item.end] == item.text == "opaque-secret" + assert item.engine == "regex" and item.confidence == 1.0 + redacted = datafog.redact(text, backend="rust") + assert type(redacted) is v4.RedactResult is engine.RedactResult + assert redacted.redacted_text == f"😀 München [{label}_1]" + + +@pytest.mark.parametrize( + "label, expected", + [("UUID", {"detect_uuid": True}), ("FUTURE_OPTIN", {"detect_future": True})], +) +def test_explicit_config_activation_is_metadata_driven(core, label, expected): + datafog.scan("value", entity_types=[label.lower()], backend="rust") + assert len(core.calls) == 1 + assert all(core.calls[0][1][key] == value for key, value in expected.items()) + + +@pytest.mark.parametrize("label, locale", [("DE_IBAN", "de"), ("FUTURE_LOCAL", "zz")]) +def test_explicit_entity_selection_activates_its_advertised_locale(core, label, locale): + datafog.scan("value", entity_types=[label], backend="rust") + assert any(config.get("locale") == locale for _, config in core.calls) + + +@pytest.mark.parametrize("supplied", [" DE-dE \t", "DE_de", "DE", " FR "]) +def test_locale_aliases_match_capabilities_and_preserve_source(core, supplied): + datafog.scan("value", locales=[supplied], backend="rust") + assert core.calls == [("value", {"locale": supplied})] + + +def test_single_string_locale_remains_supported(core): + datafog.scan("value", locales="de", backend="rust") + assert core.calls == [("value", {"locale": "de"})] + + +def test_plural_locales_union_and_deduplicate_findings_in_document_order(core): + text = "mail german other" + + def response(text, config): + matches = [finding(text, "mail", "EMAIL")] + if config["locale"] == "de": + matches.append(finding(text, "german", "DE_IBAN")) + if config["locale"] == "zz": + matches.insert(0, finding(text, "other", "FUTURE_LOCAL")) + return matches + + core.response = response + result = datafog.scan(text, locales=["zz", "de", "de"], backend="rust") + assert {config["locale"] for _, config in core.calls} == {"de", "zz"} + assert [(item.type, item.start) for item in result.entities] == [ + ("EMAIL", 0), + ("DE_IBAN", 5), + ("FUTURE_LOCAL", 12), + ] + + +def test_config_activation_applies_to_each_locale_scan(core): + datafog.scan("value", entity_types=["UUID"], locales=["de", "zz"], backend="rust") + assert {config["locale"] for _, config in core.calls} == {"de", "zz"} + assert all(config["detect_uuid"] is True for _, config in core.calls) + + +def test_requested_locale_and_entity_activation_are_combined(core): + datafog.scan("value", entity_types=["DE_IBAN"], locales=["zz"], backend="rust") + assert {config["locale"] for _, config in core.calls} == {"de", "zz"} + + +def test_structured_only_selection_rejected_before_scan(core): + with pytest.raises(ValueError, match="PERSON"): + datafog.scan("value", entity_types=["PERSON"], backend="rust") + assert not core.calls + + +def test_unsupported_locale_rejected_before_scan(core): + with pytest.raises(ValueError, match="locale"): + datafog.scan("value", locales=["unsupported"], backend="rust") + assert not core.calls + + +@pytest.mark.parametrize("capabilities", [None, {}, "not callable"]) +def test_missing_capability_api_fails_actionably(core, capabilities): + core.module.capabilities = capabilities + with pytest.raises(RuntimeError, match="capabilit|contract|Core|core"): + datafog.scan("value", backend="rust") + assert not core.calls + + +@pytest.mark.parametrize("version", [2, "1", None, True]) +def test_invalid_contract_version_rejected(core, version): + core.metadata["contract_version"] = version + with pytest.raises(RuntimeError, match="contract|capabilit"): + datafog.scan("value", backend="rust") + assert not core.calls + + +@pytest.mark.parametrize( + "field", + [ + "contract_version", + "supported_entities", + "default_entities", + "locales", + "entities", + ], +) +def test_missing_required_adapter_metadata_rejected(core, field): + del core.metadata[field] + with pytest.raises(RuntimeError, match="capabilit|contract"): + datafog.scan("value", backend="rust") + assert not core.calls + + +@pytest.mark.parametrize( + "field, value", + [ + ("supported_entities", "EMAIL"), + ("default_entities", ["UNSUPPORTED"]), + ("locales", {"de": {"enabled_entities": ["UNSUPPORTED"]}}), + ("entities", []), + ], +) +def test_malformed_capability_inventory_rejected(core, field, value): + core.metadata[field] = value + with pytest.raises(RuntimeError, match="capabilit"): + datafog.scan("value", backend="rust") + assert not core.calls + + +@pytest.mark.parametrize( + "metadata", + [ + None, + {"scopes": "text", "activation": {"kind": "default"}}, + {"scopes": ["text"]}, + {"scopes": ["text"], "activation": {"kind": "future-kind"}}, + {"scopes": ["text"], "activation": {"kind": "config"}}, + { + "scopes": ["text"], + "activation": {"kind": "locale", "scan_config": {"other": True}}, + }, + ], +) +def test_incomplete_selected_entity_activation_rejected(core, metadata): + core.metadata["entities"]["FUTURE_ID"] = metadata + with pytest.raises(RuntimeError, match="capabilit"): + datafog.scan("value", entity_types=["FUTURE_ID"], backend="rust") + assert not core.calls + + +def test_conflicting_activation_configuration_rejected_without_partial_scan(core): + core.metadata["entities"]["FUTURE_OPTIN"]["activation"]["scan_config"] = { + "detect_uuid": False + } + with pytest.raises(RuntimeError, match="conflicting"): + datafog.scan("value", entity_types=["UUID", "FUTURE_OPTIN"], backend="rust") + assert not core.calls + + +def test_locale_normalization_does_not_strip_non_ascii_whitespace(core): + with pytest.raises(ValueError, match="locale"): + datafog.scan("value", locales=["\u00a0de\u00a0"], backend="rust") + assert not core.calls + + +def test_additive_metadata_does_not_break_compatible_contract(core): + core.metadata["future_metadata"] = {"anything": True} + core.metadata["entities"]["EMAIL"]["future_field"] = 42 + datafog.scan("value", backend="rust") + assert core.calls + + +def test_capability_errors_propagate_without_fallback(core): + failure = OSError("registry unavailable") + + def fail(): + raise failure + + core.module.capabilities = fail + with pytest.raises(OSError) as caught: + datafog.scan("value", backend="rust") + assert caught.value is failure + assert not core.calls + + +def test_native_scan_errors_propagate_without_fallback(core): + failure = RuntimeError("native scan failed") + + def fail(text, config): + raise failure + + core.response = fail + with pytest.raises(RuntimeError) as caught: + datafog.scan("value", backend="rust") + assert caught.value is failure + + +@pytest.mark.parametrize("strategy", ["token", "mask", "hash", "pseudonymize"]) +def test_future_labels_preserve_filter_allowlist_and_redaction_behavior(core, strategy): + text = "keep secret a@example.com secret" + core.findings = [ + finding(text, "keep", "FUTURE_ID"), + finding(text, "secret", "FUTURE_ID", 5), + finding(text, "a@example.com", "EMAIL"), + finding(text, "secret", "FUTURE_ID", 26), + ] + options = { + "entity_types": ["future_id"], + "allowlist": ["keep"], + "backend": "rust", + } + result = datafog.scan(text, **options) + assert [item.text for item in result.entities] == ["secret", "secret"] + expected = engine.redact(text, result.entities, strategy=strategy) + assert datafog.redact(text, strategy=strategy, **options) == expected + assert not datafog.scan(text, allowlist_patterns=[r"secret"], **options).entities + + +def test_capability_reader_cached_and_replacement_invalidates(core): + reads = [] + + def read(): + reads.append(1) + return core.metadata + + core.module.capabilities = read + datafog.scan("safe", backend="rust") + datafog.scan("safe", backend="rust") + assert reads == [1] + + def replacement(): + reads.append(2) + return core.metadata + + core.module.capabilities = replacement + datafog.scan("safe", backend="rust") + assert reads == [1, 2] + + +@pytest.mark.parametrize("failure", ["exception", "version", "locale"]) +def test_failed_capability_reads_are_not_cached(core, failure): + reads = [] + + def read(): + reads.append(1) + if len(reads) == 1: + if failure == "exception": + raise RuntimeError("temporary read failure") + broken = copy.deepcopy(core.metadata) + if failure == "version": + broken["contract_version"] = 999 + else: + broken["locales"]["de"]["enabled_entities"] = ["NOT_SUPPORTED"] + return broken + return core.metadata + + core.module.capabilities = read + with pytest.raises(RuntimeError): + datafog.scan("safe", backend="rust") + datafog.scan("safe", backend="rust") + assert reads == [1, 1] + + +def test_cached_snapshot_and_nested_configs_are_private(core): + core.metadata["entities"]["FUTURE_OPTIN"]["activation"]["scan_config"] = { + "future_options": {"enabled": ["original"]} + } + core.module.capabilities = lambda: core.metadata + + def mutate_config(text, config): + config["future_options"]["enabled"].append("native mutation") + return [] + + core.response = mutate_config + datafog.scan( + "safe", backend="rust", entity_types=["FUTURE_OPTIN"], locales=["de", "fr"] + ) + core.metadata["entities"]["FUTURE_OPTIN"]["activation"]["scan_config"][ + "future_options" + ]["enabled"].append("caller mutation") + datafog.scan("safe", backend="rust", entity_types=["FUTURE_OPTIN"]) + assert len(core.calls) == 3 + assert all( + config["future_options"]["enabled"] == ["original"] for _, config in core.calls + ) diff --git a/tests/test_rust_backend.py b/tests/test_rust_backend.py index 97644322..74922c8f 100644 --- a/tests/test_rust_backend.py +++ b/tests/test_rust_backend.py @@ -7,6 +7,7 @@ import pytest from datafog import engine +from tests.test_core_capability_adapter import capability_fixture def finding(label, text, start, end): @@ -23,11 +24,15 @@ def native(monkeypatch): calls = [] findings = [] - def scan(text): + def scan(text, config=None): calls.append(text) return findings - monkeypatch.setitem(sys.modules, "datafog_core", SimpleNamespace(scan=scan)) + monkeypatch.setitem( + sys.modules, + "datafog_core", + SimpleNamespace(scan=scan, capabilities=capability_fixture), + ) return findings, calls @@ -72,11 +77,11 @@ def test_unknown_selection_preserves_legacy_empty_result(native): ) -def test_unknown_native_label_raises_instead_of_silently_dropping_pii(native): +def test_unknown_native_label_preserved_without_silently_dropping_pii(native): findings, _ = native findings.append(finding("FUTURE_LABEL", "example", 0, 7)) - with pytest.raises(RuntimeError, match="unsupported entity type.*FUTURE_LABEL"): - engine.scan("example", "regex", backend="rust") + result = engine.scan("example", "regex", backend="rust") + assert [item.type for item in result.entities] == ["FUTURE_LABEL"] def test_python_overlap_priority_precedes_selection(native): @@ -120,22 +125,24 @@ def test_allowlist_validation_before_native(native, pattern): @pytest.mark.parametrize("locale", [["de"], [" DE-DE "], "de_de"]) -def test_german_locales_rejected(native, locale): +def test_advertised_german_locales_accepted(native, locale): _, calls = native - with pytest.raises(ValueError, match="German"): - engine.scan("", "regex", locales=locale, backend="rust") - assert not calls + assert not engine.scan("", "regex", locales=locale, backend="rust").entities + assert calls == [""] -@pytest.mark.parametrize("label", engine.RegexAnnotator.GERMAN_LABELS + ["DE_FUTURE"]) -def test_german_selection_rejected(native, label): - with pytest.raises(ValueError, match="German"): - engine.scan("", "regex", entity_types=[label.lower()], backend="rust") +def test_german_selection_is_filtered_after_native_detection(native): + findings, _ = native + findings.append(finding("DE_IBAN", "DE-example", 0, 10)) + result = engine.scan( + "DE-example", "regex", entity_types=["de_iban"], backend="rust" + ) + assert [item.type for item in result.entities] == ["DE_IBAN"] def test_unknown_locale_validation_preserved(native): - with pytest.raises(ValueError, match="locale must be one of"): - engine.scan("", "regex", locales=["fr"], backend="rust") + with pytest.raises(ValueError, match="locale"): + engine.scan("", "regex", locales=["unsupported"], backend="rust") @pytest.mark.parametrize("name", ["smart", "spacy", "gliner"]) @@ -168,10 +175,14 @@ def guarded_import(name, *args, **kwargs): def test_native_failure_propagates_without_fallback(monkeypatch): error = RuntimeError("native failure") - def fail(text): + def fail(text, config=None): raise error - monkeypatch.setitem(sys.modules, "datafog_core", SimpleNamespace(scan=fail)) + monkeypatch.setitem( + sys.modules, + "datafog_core", + SimpleNamespace(scan=fail, capabilities=capability_fixture), + ) with pytest.raises(RuntimeError) as caught: engine.scan("a@example.com", "regex", backend="rust") assert caught.value is error diff --git a/tests/test_rust_contract.py b/tests/test_rust_contract.py index 52ee52e1..9bea3641 100644 --- a/tests/test_rust_contract.py +++ b/tests/test_rust_contract.py @@ -15,10 +15,14 @@ def test_native_version_and_review_inventory(): - assert importlib.metadata.version("datafog-core") == REVIEWED["core_version"] + assert importlib.metadata.version("datafog-core").startswith("0.4.") assert set(REVIEWED["cases"]) <= {case["id"] for case in CASES} for item in REVIEWED["cases"].values(): - assert item["classification"] in {"unsupported", "detector-difference"} + assert item["classification"] in { + "unsupported", + "detector-difference", + "validation-difference", + } assert item["reason"] From 16c2300470216b5b5ff3c2773421d08b6f21ef41 Mon Sep 17 00:00:00 2001 From: Sid Mohan <61345237+sidmohan0@users.noreply.github.com> Date: Mon, 28 Sep 2026 16:22:08 -0700 Subject: [PATCH 35/37] docs: verify published Core 0.4.0 integration --- CHANGELOG.MD | 5 +++-- README.md | 5 +++-- docs/core-0.4-integration.md | 40 ++++++++++++++++++++++++++++-------- docs/getting-started.rst | 6 +++--- docs/migration-4.9.md | 6 +++--- docs/python-sdk.rst | 7 +++---- 6 files changed, 46 insertions(+), 23 deletions(-) diff --git a/CHANGELOG.MD b/CHANGELOG.MD index 25e6edfd..b5327ede 100644 --- a/CHANGELOG.MD +++ b/CHANGELOG.MD @@ -6,8 +6,9 @@ - Experimental `datafog[rust]` detection with keyword-only `backend="rust"` on scan/redact entry points; Python remains the default. The follow-up requires - `datafog-core>=0.4.0,<0.5` and capability contract 1; Core 0.4.0 publication is - pending, so development validation uses candidate wheels. ML composition is unchanged. + `datafog-core>=0.4.0,<0.5` and capability contract 1. Core 0.4.0 is published + and registry-wheel validation passed; DataFog Python remains unreleased. + ML composition is unchanged. - Capability-driven entity and locale discovery, German locale activation, explicit UUID activation, and forward-compatible finding labels. Core's structured-only PERSON is rejected for explicit Rust text selection. diff --git a/README.md b/README.md index a1e0c231..799fd689 100644 --- a/README.md +++ b/README.md @@ -262,8 +262,9 @@ the native `datafog.v5` preview, and the revised 5.0 retirement schedule for `detect`/`process`, OCR, and Spark. The Python detector remains the default. The unreleased Rust adapter requires Core `>=0.4.0,<0.5` and capability contract 1. -Until Core 0.4.0 is published, development evaluation requires a validated candidate -wheel. Entity labels, locales, and activation settings come from the installed +Core 0.4.0 is available on PyPI; install the development checkout with +`python -m pip install -e ".[rust]"` to evaluate this unreleased Python adapter. +Entity labels, locales, and activation settings come from the installed Core, allowing compatible releases to add detectors without a Python update. German detection is opt-in through locale or entity selection; UUID is opt-in through `entity_types=["UUID"]`. Core's structured-only `PERSON` is unavailable diff --git a/docs/core-0.4-integration.md b/docs/core-0.4-integration.md index 5ecec1ac..36d3dcde 100644 --- a/docs/core-0.4-integration.md +++ b/docs/core-0.4-integration.md @@ -1,9 +1,8 @@ # Core 0.4 capability adapter integration gate -Status: draft implementation validated against the final local Core 0.4.0 -candidate wheel. Publication and hosted native CI remain pending; this branch -must not merge or publish until those installation gates pass. See candidate -evidence below. +Status: validated against both the final local candidate and published Core +0.4.0 wheel. Hosted CI results are tracked on Python PR #179. DataFog Python +remains unpublished; merging and release require separate authorization. ## Scope @@ -23,7 +22,7 @@ evidence below. - Ignore unrelated additive capability fields. Fail explicitly for incompatible contracts or unusable metadata rather than silently falling back to Python. -## Pending release gates +## Release validation checklist 1. Obtain the finalized candidate wheel and platform support matrix from Core. 2. Validate capability inventory and activation metadata against the candidate. @@ -81,7 +80,30 @@ A regression test asserts this exact behavior. No new hardcoded entity priorities or selection-order changes were introduced. Any future change needs an explicit overlap-policy decision rather than silently changing legacy output. -The tested wheel satisfies the local candidate gate. The supported dependency -range is now `>=0.4.0,<0.5`. Hosted native CI and published-wheel validation remain -blocked until Core 0.4.0 is available to those installers; do not publish Python -or treat local macOS validation as a cross-platform CI result. +The supported dependency range is `>=0.4.0,<0.5`. Local candidate validation +was followed by published-artifact validation below. Do not treat local macOS +validation as a cross-platform CI result. + +## Published artifact validation + +A fresh virtual environment installed the built Python wheel with `[test,cli,rust]` +using `--no-cache-dir --index-url https://pypi.org/simple`, without a local Core +wheel or editable Core checkout. The normal resolver selected published Core +0.4.0 after an initial index propagation delay. + +- Publication source: `133bceff0d2a53a7f6d1a75693330c1797285654`, tag + `python-v0.4.0` (Core publish run `36496967055`). +- Registry artifact: `datafog_core-0.4.0-cp310-abi3-macosx_11_0_arm64.whl`. +- Download origin: `files.pythonhosted.org`, recorded by pip's installation report. +- Published SHA256, matching PyPI release metadata: + `b4217c2a9834cb774d89a13b9543166dd7f38512a233b1c80984ea9c07c5162c`. +- Installed-wheel smoke with `python -I`: passed. +- `pip check`: passed. +- Published Core / adapter / unchanged legacy contract tests: **352 passed**. +- Frozen native comparison: **77 matches, two detector differences, one + validation difference, 31 outside-scope cases**, unchanged from the candidate. + +The earlier hosted Rust jobs failed only because PyPI did not yet offer Core +0.4.0. Those failed jobs were retried after publication; the current PR head's +CI checks remain the authority for cross-platform readiness. Nothing in this +validation publishes or merges DataFog Python. diff --git a/docs/getting-started.rst b/docs/getting-started.rst index eba0bcb6..74234da6 100644 --- a/docs/getting-started.rst +++ b/docs/getting-started.rst @@ -57,9 +57,9 @@ extra to evaluate them: python -m pip install -e ".[rust]" The extra requires ``datafog-core>=0.4.0,<0.5`` and capability contract 1. -Core 0.4.0 is not published yet: development evaluation requires a validated -candidate wheel until the dependency is available on PyPI. It does not change the default Python -backend, and the existing ``all`` extra does not include Rust. To opt in: +Core 0.4.0 is available on PyPI; DataFog Python 4.9 remains unreleased. The extra +does not change the default Python backend, and the existing ``all`` extra does +not include Rust. To opt in: .. code-block:: python diff --git a/docs/migration-4.9.md b/docs/migration-4.9.md index c1b763cf..dff739ae 100644 --- a/docs/migration-4.9.md +++ b/docs/migration-4.9.md @@ -1,9 +1,9 @@ # Migrating incrementally with DataFog 4.9 > **Unreleased:** This guide describes the upcoming 4.9 bridge. Until it is -> published, use the development checkout and a validated Core 0.4.0 candidate -> wheel. The declared Core dependency cannot resolve from PyPI until Core 0.4.0 -> is published. A normal released Python install does not include this follow-up. +> published, install the development checkout with `python -m pip install -e +".[rust]"`. Core 0.4.0 is available on PyPI. A normal released DataFog Python +> install does not include this follow-up. 4.9 is a bridge to the Rust-backed 5.0 API. The default Python detector, existing imports, result objects, and redaction strategies continue to work. The optional diff --git a/docs/python-sdk.rst b/docs/python-sdk.rst index 3ebe48f7..e126c94d 100644 --- a/docs/python-sdk.rst +++ b/docs/python-sdk.rst @@ -44,10 +44,9 @@ continue to work at the top level in 4.9 but warn of removal in 5.0, revising th previous promise to retain them throughout 5.x. Moving from ``process`` to ``redact`` can change old placeholder and hash output; compare results explicitly. -The unreleased adapter requires ``datafog-core>=0.4.0,<0.5``. Until Core 0.4.0 -is published, use a validated candidate wheel with the development checkout; -installing ``.[rust]`` from PyPI alone cannot resolve the new dependency yet. -Once available, install ``.[rust]`` to evaluate Rust detection: +The unreleased adapter requires ``datafog-core>=0.4.0,<0.5``. Core 0.4.0 is +available on PyPI. Install ``.[rust]`` from the development checkout to evaluate +Rust detection; DataFog Python 4.9 itself is not yet released: .. code-block:: python From 44ff6844a15e3614a81c1814de6e6556d920260a Mon Sep 17 00:00:00 2001 From: Sid Mohan <61345237+sidmohan0@users.noreply.github.com> Date: Mon, 28 Sep 2026 16:34:50 -0700 Subject: [PATCH 36/37] chore: prepare guarded 4.9.0 release workflow and notes --- .github/workflows/release.yml | 136 ++++++++++++++++++++++++++++--- AGENTS.md | 4 +- CHANGELOG.MD | 12 +-- README.md | 10 ++- RELEASE_NOTES_4.9.0.md | 91 +++++++++++++++++++++ docs/cli.rst | 4 +- docs/getting-started.rst | 17 ++-- docs/important-concepts.rst | 4 +- docs/index.rst | 15 ++-- docs/migration-4.9.md | 34 ++++++-- docs/python-sdk.rst | 17 ++-- docs/roadmap.rst | 2 +- docs/v5-compatibility-matrix.rst | 2 +- 13 files changed, 285 insertions(+), 63 deletions(-) create mode 100644 RELEASE_NOTES_4.9.0.md diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 30bd23b4..75fa4590 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -21,6 +21,11 @@ on: required: false default: false type: boolean + dry_run_ref: + description: "Candidate ref to validate (allowed only with dry_run)" + required: false + default: "" + type: string force_build: description: "Force build even if no changes" required: false @@ -43,10 +48,18 @@ jobs: release_type: ${{ steps.resolve.outputs.release_type }} has_changes: ${{ steps.changes.outputs.has_changes }} target_branch: ${{ steps.resolve.outputs.target_branch }} + source_sha: ${{ steps.source.outputs.sha }} steps: - name: Resolve release type id: resolve + env: + DRY_RUN: ${{ inputs.dry_run }} + DRY_RUN_REF: ${{ inputs.dry_run_ref }} run: | + if [ -n "$DRY_RUN_REF" ] && [ "$DRY_RUN" != "true" ]; then + echo "dry_run_ref is allowed only with dry_run=true" + exit 1 + fi if [ "${{ github.event_name }}" = "workflow_dispatch" ]; then TYPE="${{ inputs.release_type }}" elif [ "${{ github.event.schedule }}" = "0 2 * * 4" ]; then @@ -61,14 +74,21 @@ jobs: BRANCH="dev" fi - echo "release_type=$TYPE" >> "$GITHUB_OUTPUT" - echo "target_branch=$BRANCH" >> "$GITHUB_OUTPUT" + { + echo "release_type=$TYPE" + echo "target_branch=$BRANCH" + echo "checkout_ref=${DRY_RUN_REF:-$BRANCH}" + } >> "$GITHUB_OUTPUT" echo "Release type: $TYPE from $BRANCH" - uses: actions/checkout@v6 with: fetch-depth: 0 - ref: ${{ steps.resolve.outputs.target_branch }} + ref: ${{ steps.resolve.outputs.checkout_ref }} + + - name: Freeze source commit + id: source + run: echo "sha=$(git rev-parse HEAD)" >> "$GITHUB_OUTPUT" - name: Check for changes id: changes @@ -110,7 +130,7 @@ jobs: - uses: actions/checkout@v6 with: fetch-depth: 0 - ref: ${{ needs.determine-release.outputs.target_branch }} + ref: ${{ needs.determine-release.outputs.source_sha }} - name: Set up Python ${{ matrix.python-version }} uses: actions/setup-python@v6 @@ -131,9 +151,21 @@ jobs: python -m spacy download en_core_web_lg datafog download-model urchade/gliner_multi_pii-v1 --engine gliner - - name: Run tests with segfault protection + - name: Run release tests + env: + OMP_NUM_THREADS: "1" + MKL_NUM_THREADS: "1" + OPENBLAS_NUM_THREADS: "1" run: | - python run_tests.py tests/ --ignore=tests/test_gliner_annotator.py --cov-report=xml --cov-config=.coveragerc + python -m pytest tests/ -m "not slow" \ + --ignore=tests/test_detection_accuracy.py \ + --ignore=tests/test_image_service.py \ + --ignore=tests/test_ocr_integration.py \ + --ignore=tests/test_spark_integration.py \ + --cov=datafog --cov-report=xml --cov-config=.coveragerc + + - name: Run detection accuracy corpus + run: python -m pytest tests/test_detection_accuracy.py -v --tb=short - name: Run performance validation run: | @@ -148,7 +180,7 @@ jobs: - uses: actions/checkout@v6 with: fetch-depth: 0 - ref: ${{ needs.determine-release.outputs.target_branch }} + ref: ${{ needs.determine-release.outputs.source_sha }} - name: Set up Python 3.14 uses: actions/setup-python@v6 @@ -171,8 +203,49 @@ jobs: --ignore=tests/test_spark_integration.py \ --ignore=tests/test_text_service_integration.py + rust-bridge: + needs: determine-release + if: needs.determine-release.outputs.has_changes == 'true' + runs-on: ${{ matrix.os }} + strategy: + fail-fast: false + matrix: + include: + - os: ubuntu-latest + python-version: "3.10" + - os: ubuntu-latest + python-version: "3.14" + - os: macos-latest + python-version: "3.12" + - os: windows-latest + python-version: "3.12" + steps: + - uses: actions/checkout@v6 + with: + ref: ${{ needs.determine-release.outputs.source_sha }} + - uses: actions/setup-python@v6 + with: + python-version: ${{ matrix.python-version }} + - name: Build and install wheel with published Core + shell: bash + run: | + python -m pip install build + python -m build --wheel + python -c "import glob, subprocess, sys; wheel = glob.glob('dist/*.whl')[0]; subprocess.check_call([sys.executable, '-m', 'pip', 'install', wheel + '[test,cli,rust]'])" + python -m pip check + - name: Verify installed wheel + run: python -I scripts/check_rust_install.py + - name: Test binding integration and compatibility + run: python -m pytest tests/test_contract_481.py tests/test_rust_backend.py tests/test_api_bridge_49.py tests/test_rust_contract.py tests/test_core_capability_adapter.py tests/test_core04_integration.py -q + - name: Record parity + run: python -m tests.rust_contract --output rust-parity.json + - uses: actions/upload-artifact@v4 + with: + name: release-rust-parity-${{ matrix.os }}-${{ matrix.python-version }} + path: rust-parity.json + publish: - needs: [determine-release, test, python314-core] + needs: [determine-release, test, python314-core, rust-bridge] runs-on: ubuntu-latest outputs: version: ${{ steps.version.outputs.version }} @@ -180,7 +253,7 @@ jobs: - uses: actions/checkout@v6 with: fetch-depth: 0 - ref: ${{ needs.determine-release.outputs.target_branch }} + ref: ${{ needs.determine-release.outputs.source_sha }} token: ${{ secrets.GH_PAT }} - name: Set up Python @@ -194,12 +267,15 @@ jobs: pip install build twine bump2version - name: Configure git + if: inputs.dry_run != true run: | git config --local user.email "action@github.com" git config --local user.name "GitHub Action" - name: Generate version id: version + env: + VERSION_OVERRIDE: ${{ inputs.version_override }} run: | set -e git fetch --tags @@ -214,13 +290,17 @@ jobs: # Strip any pre-release suffix to get base version BASE=$(echo "$CURRENT" | sed -E 's/(a|b)[0-9]+([.][0-9A-Za-z]+)?$//') - if [ -n "${{ inputs.version_override }}" ]; then - BASE="${{ inputs.version_override }}" + if [ -n "$VERSION_OVERRIDE" ]; then + BASE="$VERSION_OVERRIDE" if echo "$BASE" | grep -Eq '(a|b)[0-9]+([.][0-9A-Za-z]+)?$'; then echo "version_override must be a stable base version like 4.4.0, not a prerelease" exit 1 fi fi + if ! echo "$BASE" | grep -Eq '^[0-9]+\.[0-9]+\.[0-9]+$'; then + echo "Release base must have numeric major.minor.patch format" + exit 1 + fi echo "Base version: $BASE" if [ "$TYPE" = "alpha" ]; then @@ -250,12 +330,16 @@ jobs: fi - name: Generate changelog + env: + VERSION: ${{ steps.version.outputs.version }} run: | TYPE="${{ needs.determine-release.outputs.release_type }}" if [ "$TYPE" = "alpha" ]; then python scripts/generate_changelog.py --alpha --output RELEASE_CHANGELOG.md elif [ "$TYPE" = "beta" ]; then python scripts/generate_changelog.py --beta --output RELEASE_CHANGELOG.md + elif [ -f "RELEASE_NOTES_${VERSION}.md" ]; then + cp "RELEASE_NOTES_${VERSION}.md" RELEASE_CHANGELOG.md else python scripts/generate_changelog.py --output RELEASE_CHANGELOG.md fi @@ -264,6 +348,34 @@ jobs: run: | python -m build python scripts/check_wheel_size.py + python -m twine check dist/* + + - name: Verify final versioned wheel with published Core + run: | + python -c "import glob, subprocess, sys; wheel = glob.glob('dist/*.whl')[0]; subprocess.check_call([sys.executable, '-m', 'pip', 'install', wheel + '[rust]'])" + python -m pip check + python -I scripts/check_rust_install.py + + - name: Save release artifacts for review + uses: actions/upload-artifact@v4 + with: + name: release-${{ steps.version.outputs.version }}-${{ needs.determine-release.outputs.source_sha }} + path: | + dist/* + RELEASE_CHANGELOG.md + if-no-files-found: error + + - name: Verify release branch still matches tested source + if: inputs.dry_run != true + env: + RELEASE_BRANCH: ${{ needs.determine-release.outputs.target_branch }} + TESTED_SHA: ${{ needs.determine-release.outputs.source_sha }} + run: | + git fetch origin "$RELEASE_BRANCH" + if [ "$(git rev-parse FETCH_HEAD)" != "$TESTED_SHA" ]; then + echo "Release branch changed during validation; rerun against its new head" + exit 1 + fi - name: Publish to PyPI if: inputs.dry_run != true @@ -284,7 +396,7 @@ jobs: git add datafog/__about__.py setup.py git commit -m "chore: bump version to $VERSION [skip ci]" || echo "No version changes to commit" - git push origin "$BRANCH" + git push origin "HEAD:$BRANCH" git tag -a "v$VERSION" -m "Release $VERSION" git push origin "v$VERSION" diff --git a/AGENTS.md b/AGENTS.md index 1a174106..13439692 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -13,9 +13,9 @@ ## Current Project Status -**Stable version: 4.8.0** +**Stable version: 4.9.0** -**Development version: 4.8.0** +**Development version: 4.9.0** **Next major target: 5.0.0** diff --git a/CHANGELOG.MD b/CHANGELOG.MD index b5327ede..576597a3 100644 --- a/CHANGELOG.MD +++ b/CHANGELOG.MD @@ -1,14 +1,13 @@ # ChangeLog -## [Unreleased] +## [4.9.0] #### Added - Experimental `datafog[rust]` detection with keyword-only `backend="rust"` - on scan/redact entry points; Python remains the default. The follow-up requires - `datafog-core>=0.4.0,<0.5` and capability contract 1. Core 0.4.0 is published - and registry-wheel validation passed; DataFog Python remains unreleased. - ML composition is unchanged. + on scan/redact entry points; Python remains the default. Requires + `datafog-core>=0.4.0,<0.5` and capability contract 1. The base installation + does not install or import Core. ML composition is unchanged. - Capability-driven entity and locale discovery, German locale activation, explicit UUID activation, and forward-compatible finding labels. Core's structured-only PERSON is rejected for explicit Rust text selection. @@ -42,7 +41,7 @@ - **Redaction token numbering now follows document order**: `redact()` previously assigned per-type counters while replacing spans right-to-left, - so with multiple entities of the same type the *last* occurrence received + so with multiple entities of the same type the _last_ occurrence received `_1` (two emails redacted as `[EMAIL_2] ... [EMAIL_1]`). Tokens for the `token` and `pseudonymize` strategies are now numbered left-to-right, so the first occurrence is always `_1`. This changes redacted output text for @@ -347,6 +346,7 @@ opt-in. No changes to the core library or its dependencies. #### Migration Guide For users upgrading from v4.1.1: + - All existing functionality remains unchanged - To use GLiNER: `pip install datafog[nlp-advanced]` - Smart cascading: `TextService(engine="smart")` for best balance diff --git a/README.md b/README.md index 799fd689..4fd354a3 100644 --- a/README.md +++ b/README.md @@ -188,7 +188,7 @@ scan/redact helpers, or guardrail helpers. model. - A Java runtime is required by PySpark. -The upcoming 4.9 release deprecates OCR and Spark with visible use-time +DataFog 4.9.0 deprecates OCR and Spark with visible use-time warnings; their APIs and extras will be removed in 5.0. They remain functional in 4.9. Users who need these features can stay on the final 4.x release. See the [4.9 migration guide](docs/migration-4.9.md) for the transition plan. @@ -261,9 +261,11 @@ The [4.9 migration guide](docs/migration-4.9.md) explains opt-in Rust detection, the native `datafog.v5` preview, and the revised 5.0 retirement schedule for `detect`/`process`, OCR, and Spark. The Python detector remains the default. -The unreleased Rust adapter requires Core `>=0.4.0,<0.5` and capability contract 1. -Core 0.4.0 is available on PyPI; install the development checkout with -`python -m pip install -e ".[rust]"` to evaluate this unreleased Python adapter. +The experimental Rust adapter in 4.9.0 requires Core `>=0.4.0,<0.5` and capability +contract 1. Upgrade with `python -m pip install --upgrade "datafog[rust]==4.9.0"` +to evaluate it. Base-only users can install `datafog==4.9.0` without Core; +installing the Rust extra does not change the default backend. See the +[4.9.0 release notes](RELEASE_NOTES_4.9.0.md). Entity labels, locales, and activation settings come from the installed Core, allowing compatible releases to add detectors without a Python update. German detection is opt-in through locale or entity selection; UUID is opt-in diff --git a/RELEASE_NOTES_4.9.0.md b/RELEASE_NOTES_4.9.0.md new file mode 100644 index 00000000..8e97ebd4 --- /dev/null +++ b/RELEASE_NOTES_4.9.0.md @@ -0,0 +1,91 @@ +# DataFog Python 4.9.0 + +4.9.0 bridges the existing Python API and DataFog Core. **Python detection remains +the default.** Existing imports, result classes, and legacy redaction strategies +continue to work. Rust detection and the native API preview are experimental, +explicit opt-ins. + +## Upgrade + +```bash +# Base package: no native dependency is automatically installed. +python -m pip install --upgrade "datafog==4.9.0" + +# Optional Rust backend and native API preview. +python -m pip install --upgrade "datafog[rust]==4.9.0" +``` + +The Rust extra requires `datafog-core>=0.4.0,<0.5` and capability contract `1`. +Installing it does not select Rust automatically. The `all` extra does not +include Rust; request `datafog[all,rust]==4.9.0` if both are needed. Lock your Core +version when detection output must be reproducible across installations. + +```python +import datafog + +result = datafog.redact("Contact jane@example.com", engine="regex", backend="rust") +assert result.redacted_text == "Contact [EMAIL_1]" +``` + +Rust is supported only with `engine="regex"`. Missing dependencies, unsupported +configurations, and native failures raise errors without silently falling back. +The CLI, `DataFog`, `TextService`, ML engines, and existing integrations retain +their current detection paths. + +## What changes + +- `datafog.compat.v4` exposes existing scan/redact APIs and result classes with + the same class identity as the established Python API. +- `datafog.v5` exposes native Core types and operations. Scanning returns + `list[Finding]`; native transformations return `TransformResult`, use Core's + strategies, and do not promise legacy numbered tokens or plaintext mappings. +- Rust detector labels, supported locales, and opt-in settings come from Core's + capability metadata. Core 0.4.0 supports the seven German detectors through + German locales or explicit entity selection. UUID remains opt-in. Core's + structured-only `PERSON` is rejected for explicit Rust text selection. +- Core 0.4.0 adds JWT, private-key, contextual US routing-number, and contextual + NPI detection. Python allowlists and legacy transformations remain in the + compatibility adapter. + +## Known differences + +Rust is not advertised as universally equivalent to the Python detector. On the +frozen 111-case baseline there are 77 exact matches, two reviewed detector +differences, one validation-message difference, and 31 cases outside backend +scope. Core rejects invalid-checksum cards and alphanumeric-embedded SSNs in the +reviewed differing cases. Unsupported locale validation uses Core capabilities. + +**Explicit NPI selection has a compatibility limitation:** when Core reports both +`PHONE` and `NPI` for the same span, legacy overlap handling keeps `PHONE` before +entity filtering. Consequently `entity_types=["NPI"]` can return no entities. +Default legacy redaction still protects that span as a phone. Native +`datafog.v5.scan()` retains NPI, and native transformation can select it. Choose +the native API when retaining NPI identity is required. + +The adapter resolves overlaps before entity filtering. A later compatible Core +release can introduce a finding that wins an overlap and suppresses a previously +selected label. The dependency range promises API compatibility, not identical +detection or selection results. For reproducible behavior, pin both packages: + +```bash +python -m pip install "datafog[rust]==4.9.0" "datafog-core==0.4.0" +``` + +Use native findings and native transformation entity selection when preserving +specific overlapping labels is required. + +## Deprecations for 5.0 + +`detect()` and `process()` still work in 4.9 but now warn of removal in 5.0. This +revises the earlier promise to retain them throughout 5.x. Migrate to scan/redact +and review transformation differences, particularly older placeholder and hash +formats. + +OCR, Donut, Tesseract, image services, Spark, and distributed helpers remain +functional in 4.9 with visible use-time warnings. Their removal is planned for +5.0. Users needing them can remain on the final 4.x release; this does not promise +indefinite maintenance or introduce successor packages. spaCy and GLiNER are not +removed by this release. + +See the [migration guide](https://github.com/DataFog/datafog-python/blob/v4.9.0/docs/migration-4.9.md) for API examples, compatibility +boundaries, parity evidence, and reproducible verification commands. diff --git a/docs/cli.rst b/docs/cli.rst index e27f4d72..b07a080a 100644 --- a/docs/cli.rst +++ b/docs/cli.rst @@ -8,7 +8,7 @@ The main entrypoint for the CLI is through the DataFog client file, defined in : We use Typer to build the CLI, with each command defined as a separate function. Core text commands such as ``scan-text``, ``redact-text``, ``replace-text``, -and ``hash-text`` are the primary CLI path. The unreleased 4.9 bridge leaves text commands on their +and ``hash-text`` are the primary CLI path. The 4.9.0 bridge leaves text commands on their existing Python detection paths; the opt-in Rust backend is a Python API option, not a new CLI flag. Install ``datafog[cli]`` for the command-line dependencies. @@ -26,7 +26,7 @@ commands. Spark support is also deprecated in 4.9 for removal in 5.0. Install ``datafog[distributed]`` when using ``SparkService`` during 4.x. Users needing OCR/Spark after the cutover can remain on the final 4.x release. See :doc:`optional-surfaces` and the -:download:`unreleased 4.9 migration guide `. +:download:`4.9 migration guide `. German locale support --------------------- diff --git a/docs/getting-started.rst b/docs/getting-started.rst index 74234da6..11753c41 100644 --- a/docs/getting-started.rst +++ b/docs/getting-started.rst @@ -45,19 +45,20 @@ Optional extras are explicit: - ``pip install "datafog[all]"`` - You are developing or deliberately want every optional surface. -Unreleased 4.9 bridge -===================== +4.9.0 migration bridge +====================== -The following APIs are development previews, not a claim that 4.9 is published. -From a checkout containing the 4.9 implementation, install the explicit Rust -extra to evaluate them: +The native API preview and Rust backend are experimental additions in 4.9.0. +Upgrade the base package with ``python -m pip install --upgrade datafog==4.9.0`` +to keep using Python detection without a native dependency. To evaluate Rust, +install the optional extra explicitly: .. code-block:: bash - python -m pip install -e ".[rust]" + python -m pip install --upgrade "datafog[rust]==4.9.0" The extra requires ``datafog-core>=0.4.0,<0.5`` and capability contract 1. -Core 0.4.0 is available on PyPI; DataFog Python 4.9 remains unreleased. The extra +The extra does not change the default Python backend, and the existing ``all`` extra does not include Rust. To opt in: @@ -162,7 +163,7 @@ The CLI core path is text-first: datafog hash-text "Contact jane@example.com" datafog redact-text "Steuer-ID 12345678901" --locale de -Image commands are optional and scheduled for deprecation in 4.9 and removal +Image commands are optional, deprecated in 4.9, and scheduled for removal in 5.0. They remain functional in 4.9. Install ``datafog[ocr]`` for local OCR and ``datafog[web,ocr]`` when the CLI needs to download image inputs. diff --git a/docs/important-concepts.rst b/docs/important-concepts.rst index a1655593..239471bc 100644 --- a/docs/important-concepts.rst +++ b/docs/important-concepts.rst @@ -9,7 +9,7 @@ Overview Data Models ^^^^^^^^^^^ Existing models support legacy PII annotation and optional OCR analysis. -The unreleased 4.9 bridge retains them and adds ``datafog.compat.v4`` for the +The 4.9.0 bridge retains them and adds ``datafog.compat.v4`` for the legacy scan/redact result classes (``Entity``, ``ScanResult``, ``RedactResult``). ``datafog.v5`` separately previews native Core ``Finding`` and ``TransformResult`` objects; these have different fields and transformation semantics. See @@ -25,7 +25,7 @@ objects; these have different fields and transformation semantics. See Processors ^^^^^^^^^^^ Text processors remain available. OCR processors below are deprecated in the -unreleased 4.9 bridge and scheduled for removal in 5.0: +4.9.0 bridge and scheduled for removal in 5.0: * SpacyAnnotator Text annotation with spaCy diff --git a/docs/index.rst b/docs/index.rst index 44f94e63..df61a539 100644 --- a/docs/index.rst +++ b/docs/index.rst @@ -11,14 +11,13 @@ Start with :doc:`getting-started` if you want the shortest route from install to scanning text. The roadmap and historical planning pages remain available, but the live user docs are the first path for current text APIs. -4.9 development preview -======================= +4.9.0 migration bridge +====================== -.. note:: - - The 4.9 bridge described here is unreleased development work. These pages do - not announce a published 4.9 package. A normal PyPI install does not imply - availability of the new APIs below. +DataFog 4.9.0 is the migration bridge to the planned Rust-backed 5.0 API. +Upgrade the base package with ``python -m pip install --upgrade datafog==4.9.0``; +the native dependency remains optional. The ``rust`` extra requires +``datafog-core>=0.4.0,<0.5`` and capability contract version 1. 4.9 preserves existing imports, result classes, Python detection defaults, and redaction behavior. It adds explicit experimental Rust detection through @@ -26,7 +25,7 @@ redaction behavior. It adds explicit experimental Rust detection through Core schema preview. See :doc:`python-sdk` and the :download:`complete 4.9 migration guide `. -The planned 4.9 release deprecates ``detect()``/``process()``, OCR, and Spark for +DataFog 4.9.0 deprecates ``detect()``/``process()``, OCR, and Spark for removal in 5.0. The earlier promise to retain ``detect()``/``process()`` throughout 5.x is revised. OCR/Spark remain functional in 4.9; users needing them after the cutover can remain on the final 4.x release. See :doc:`optional-surfaces`. diff --git a/docs/migration-4.9.md b/docs/migration-4.9.md index dff739ae..3bf2b900 100644 --- a/docs/migration-4.9.md +++ b/docs/migration-4.9.md @@ -1,18 +1,21 @@ -# Migrating incrementally with DataFog 4.9 +# Migrating incrementally with DataFog 4.9.0 -> **Unreleased:** This guide describes the upcoming 4.9 bridge. Until it is -> published, install the development checkout with `python -m pip install -e -".[rust]"`. Core 0.4.0 is available on PyPI. A normal released DataFog Python -> install does not include this follow-up. - -4.9 is a bridge to the Rust-backed 5.0 API. The default Python detector, existing +4.9.0 is a bridge to the Rust-backed 5.0 API. The default Python detector, existing imports, result objects, and redaction strategies continue to work. The optional Rust backend and native API preview are experimental and explicitly selected. -## Opt into Rust detection +## Upgrade and opt into Rust detection + +Upgrade the base package while retaining Python detection and no native dependency: ```bash -pip install "datafog[rust]" +python -m pip install --upgrade "datafog==4.9.0" +``` + +Install the optional native dependency explicitly to evaluate Rust detection: + +```bash +python -m pip install --upgrade "datafog[rust]==4.9.0" ``` The extra requires `datafog-core>=0.4.0,<0.5` and capability contract version 1. @@ -204,3 +207,16 @@ for the short payload, 39.50 versus 7.58 microseconds for mixed PII, and 121.91 versus 5.33 milliseconds for the large sparse payload (Python versus Rust). Fresh-process import plus first scan was approximately 81 milliseconds for both. These are local measurements, not release performance guarantees. + +## Detector upgrades and overlapping labels + +Capability discovery allows new Core labels to pass through the adapter without +a Python inventory update; it does not guarantee unchanged selected results. +The legacy adapter resolves overlaps before applying `entity_types`. A new +detector can therefore suppress a previously selected label, leaving an explicit +selection empty. The NPI/PHONE case above is one concrete example. + +Pin `datafog-core==0.4.0` alongside `datafog==4.9.0` for reproducibility against +this release's validated detector set. Test detector upgrades against your own +selection policies. Use `datafog.v5.scan()` to retain native candidates and +native transformation entity selection when overlapping label identity matters. diff --git a/docs/python-sdk.rst b/docs/python-sdk.rst index e126c94d..2a8a2ddf 100644 --- a/docs/python-sdk.rst +++ b/docs/python-sdk.rst @@ -27,11 +27,11 @@ available for existing users. ``TextService(engine="regex")`` is the dependency-light service path; ``spacy``, ``gliner``, ``smart``, OCR, and Spark surfaces require their explicit extras. -4.9 compatibility and Core preview (unreleased) ------------------------------------------------ +4.9.0 compatibility and Core preview +------------------------------------ -These additions describe development work for 4.9; they do not indicate a -published release. Existing top-level ``scan``/``redact`` functions retain their +DataFog 4.9.0 provides an experimental Rust backend and native schema preview. +Existing top-level ``scan``/``redact`` functions retain their result shapes and use the Python backend by default. They delegate through the new facade, whose classes are the same objects as the established result types: @@ -44,9 +44,10 @@ continue to work at the top level in 4.9 but warn of removal in 5.0, revising th previous promise to retain them throughout 5.x. Moving from ``process`` to ``redact`` can change old placeholder and hash output; compare results explicitly. -The unreleased adapter requires ``datafog-core>=0.4.0,<0.5``. Core 0.4.0 is -available on PyPI. Install ``.[rust]`` from the development checkout to evaluate -Rust detection; DataFog Python 4.9 itself is not yet released: +The adapter requires ``datafog-core>=0.4.0,<0.5``. Install with +``python -m pip install --upgrade "datafog[rust]==4.9.0"`` to evaluate Rust. +A base install of ``datafog==4.9.0`` has no native dependency; installing the +extra does not change the default Python backend: .. code-block:: python @@ -155,7 +156,7 @@ OCR and Spark remain available as optional surfaces throughout 4.9: * Use ``datafog[distributed,nlp]`` plus an installed spaCy model for Spark PII UDF helpers. -The unreleased 4.9 bridge deprecates OCR and Spark for removal in 5.0. Use-time +DataFog 4.9.0 deprecates OCR and Spark for removal in 5.0. Use-time ``FutureWarning`` notices are visible under normal Python warning filters. Users needing these features can remain on the final 4.x release; this migration does not introduce successor packages or promise indefinite maintenance. See diff --git a/docs/roadmap.rst b/docs/roadmap.rst index a4aee1f8..3b7fd12c 100644 --- a/docs/roadmap.rst +++ b/docs/roadmap.rst @@ -4,7 +4,7 @@ Release Roadmap .. note:: - This earlier planning document is retained for context. The upcoming 4.9 + This earlier planning document is retained for context. The 4.9.0 bridge revises its compatibility commitments: ``detect``/``process``, OCR, and Spark are deprecated in 4.9 and removed in 5.0. The former promise to retain the shims through 5.x no longer applies. See the diff --git a/docs/v5-compatibility-matrix.rst b/docs/v5-compatibility-matrix.rst index ac29c378..b953729d 100644 --- a/docs/v5-compatibility-matrix.rst +++ b/docs/v5-compatibility-matrix.rst @@ -4,7 +4,7 @@ v5 Compatibility Matrix .. note:: - This earlier planning document is retained for context. The upcoming 4.9 + This earlier planning document is retained for context. The 4.9.0 bridge revises its compatibility commitments: ``detect``/``process``, OCR, and Spark are deprecated in 4.9 and removed in 5.0. The former promise to retain the shims through 5.x no longer applies. See the From e31e9e5dd97581622515a97e4421dc558828e706 Mon Sep 17 00:00:00 2001 From: Sid Mohan <61345237+sidmohan0@users.noreply.github.com> Date: Mon, 28 Sep 2026 16:38:02 -0700 Subject: [PATCH 37/37] fix: attribute release commits to the official Actions bot --- .github/workflows/release.yml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 75fa4590..77e7aedd 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -269,8 +269,8 @@ jobs: - name: Configure git if: inputs.dry_run != true run: | - git config --local user.email "action@github.com" - git config --local user.name "GitHub Action" + git config --local user.email "41898282+github-actions[bot]@users.noreply.github.com" + git config --local user.name "github-actions[bot]" - name: Generate version id: version