From 8542f4e4b10dd543aae3cef61b0bec581b7f36c2 Mon Sep 17 00:00:00 2001 From: Quinten Steenhuis Date: Wed, 30 Sep 2026 14:14:57 -0400 Subject: [PATCH 1/8] Prepare filing PDFs and require document previews --- .dockerignore | 2 +- .gitignore | 3 + Dockerfile | 7 + .../issue-113/corpus-comparison.json | 318 ++++++++++++++++++ docs/developer-notes/issue-113/validation.md | 171 ++++++++++ docs/docs/admin/configuration.md | 50 +++ docs/docs/admin/deployment.md | 5 +- efile_app/.env.example | 7 + efile_app/efile/config_text_strings.py | 5 + .../commands/expire_unclaimed_handoffs.py | 11 +- .../0029_document_preparation_and_preview.py | 34 ++ efile_app/efile/models.py | 13 + .../efile/services/document_preparation.py | 250 ++++++++++++++ efile_app/efile/services/document_previews.py | 23 ++ efile_app/efile/services/document_uploads.py | 90 ++--- efile_app/efile/services/draft_urls.py | 1 + efile_app/efile/services/drafts.py | 11 + efile_app/efile/services/handoff.py | 2 + efile_app/efile/settings_base.py | 6 + .../efile/static/config/states/vermont.yaml | 12 + .../efile/static/css/document-preview.css | 57 ++++ efile_app/efile/static/js/document-preview.js | 101 ++++++ efile_app/efile/static/js/upload-documents.js | 6 +- efile_app/efile/static/js/waiver-upload.js | 6 +- .../efile/components/document_preview.html | 50 +++ .../templates/efile/document_checklist.html | 5 +- .../templates/efile/extraction_review.html | 3 + .../efile/templates/efile/handoff_review.html | 15 +- .../templates/efile/organize_documents.html | 1 + efile_app/efile/templates/efile/payment.html | 7 +- .../templates/efile/preview_documents.html | 50 +++ efile_app/efile/templates/efile/review.html | 1 + .../templates/efile/upload_documents.html | 21 +- .../efile/templates/efile/workflow_base.html | 7 + efile_app/efile/tests/pdf_helpers.py | 85 +++++ efile_app/efile/tests/test_ai_opt_out.py | 3 +- .../efile/tests/test_document_extractions.py | 2 +- .../efile/tests/test_document_preparation.py | 145 ++++++++ .../test_document_preparation_browser.py | 116 +++++++ .../efile/tests/test_document_previews.py | 160 +++++++++ .../tests/test_end_to_end_new_flow_states.py | 13 + efile_app/efile/tests/test_handoff.py | 4 +- .../efile/tests/test_reorganized_start.py | 5 +- .../efile/tests/test_waiver_documents.py | 16 +- efile_app/efile/tests/test_workflow.py | 1 + efile_app/efile/urls.py | 3 + efile_app/efile/utils/ui_text.py | 10 + efile_app/efile/views/document_previews.py | 100 ++++++ efile_app/efile/views/extraction_review.py | 5 + efile_app/efile/views/handoff.py | 40 ++- efile_app/efile/views/payment.py | 1 + efile_app/efile/views/review.py | 6 +- efile_app/efile/views/submission.py | 17 +- efile_app/efile/views/upload_documents.py | 25 +- efile_app/efile/views/waiver_documents.py | 30 +- efile_app/efile/workflow.py | 2 + efile_app/eslint.config.mjs | 2 +- efile_app/package-lock.json | 280 ++++++++++++++- efile_app/package.json | 6 +- efile_app/pytest.ini | 3 + efile_app/scripts/copy-pdfjs.mjs | 20 ++ efile_app/stylelint.config.mjs | 1 + .../tests/document-preparation-browser.js | 191 +++++++++++ package-lock.json | 6 + testing/validate_document_preparation.py | 136 ++++++++ 65 files changed, 2681 insertions(+), 104 deletions(-) create mode 100644 docs/developer-notes/issue-113/corpus-comparison.json create mode 100644 docs/developer-notes/issue-113/validation.md create mode 100644 efile_app/efile/migrations/0029_document_preparation_and_preview.py create mode 100644 efile_app/efile/services/document_preparation.py create mode 100644 efile_app/efile/services/document_previews.py create mode 100644 efile_app/efile/static/css/document-preview.css create mode 100644 efile_app/efile/static/js/document-preview.js create mode 100644 efile_app/efile/templates/efile/components/document_preview.html create mode 100644 efile_app/efile/templates/efile/preview_documents.html create mode 100644 efile_app/efile/tests/pdf_helpers.py create mode 100644 efile_app/efile/tests/test_document_preparation.py create mode 100644 efile_app/efile/tests/test_document_preparation_browser.py create mode 100644 efile_app/efile/tests/test_document_previews.py create mode 100644 efile_app/efile/views/document_previews.py create mode 100644 efile_app/scripts/copy-pdfjs.mjs create mode 100644 efile_app/tests/document-preparation-browser.js create mode 100644 package-lock.json create mode 100644 testing/validate_document_preparation.py diff --git a/.dockerignore b/.dockerignore index 3769a417..e5fcc204 100644 --- a/.dockerignore +++ b/.dockerignore @@ -40,7 +40,7 @@ node_modules/ **/node_modules/ benchmarking/ court_forms/ -package-lock.json +/package-lock.json # Local DBs *.sqlite3 diff --git a/.gitignore b/.gitignore index 96256622..aa3833fa 100644 --- a/.gitignore +++ b/.gitignore @@ -184,3 +184,6 @@ court_forms/ # Derived lab-notebook visual review pages benchmarking/promptfoo/lab-notebook/studies/2026-08-27-deterministic-form-identifier-scan/illinois-form-code-review/ + +# Generated by npm ci and the Docker asset stage. +efile_app/efile/static/vendor/pdfjs/ diff --git a/Dockerfile b/Dockerfile index c1e32475..06c2707f 100644 --- a/Dockerfile +++ b/Dockerfile @@ -1,4 +1,10 @@ # syntax=docker/dockerfile:1.7 +FROM node:22-slim AS pdfjs-assets +WORKDIR /assets +COPY efile_app/package.json efile_app/package-lock.json ./ +COPY efile_app/scripts/copy-pdfjs.mjs ./scripts/copy-pdfjs.mjs +RUN npm ci --omit=dev + FROM python:3.12-slim AS base ENV PYTHONDONTWRITEBYTECODE=1 \ @@ -24,6 +30,7 @@ RUN uv sync --frozen --no-install-project # Copy the rest of the source code into /app COPY . /app +COPY --from=pdfjs-assets /assets/efile/static/vendor/pdfjs /app/efile_app/efile/static/vendor/pdfjs # Install the project itself (editable-like install) RUN uv sync --frozen diff --git a/docs/developer-notes/issue-113/corpus-comparison.json b/docs/developer-notes/issue-113/corpus-comparison.json new file mode 100644 index 00000000..3f3dc9e0 --- /dev/null +++ b/docs/developer-notes/issue-113/corpus-comparison.json @@ -0,0 +1,318 @@ +[ + { + "source": "docassemble-ALDashboard/docassemble/ALDashboard/test/civil_docketing_statement_polished_repaired.pdf", + "sha256": "110986860fb8016e2ac5f6d83ad7f421cd21eccf75f79184838eff93704e5f38", + "input_pages": 3, + "input_fields": 59, + "filled_fields": 3, + "multiline_fields": 4, + "raw-gotenberg": { + "accepted": true, + "pages": 3, + "remaining_fields": 0, + "text_characters": 3611, + "tagged": false + }, + "litefile": { + "accepted": true, + "pages": 3, + "remaining_fields": 0, + "text_characters": 3611, + "tagged": false + }, + "pdftk": { + "accepted": true, + "pages": 3, + "remaining_fields": 0, + "text_characters": 3611, + "tagged": false + } + }, + { + "source": "docassemble-ALWeaver/build/lib/docassemble/ALWeaver/data/sources/test_civil_docketing_statement.pdf", + "sha256": "74751dadb930ed97d66243596108d54207ce249c83a309467500b23f541e58c3", + "input_pages": 3, + "input_fields": 60, + "filled_fields": 3, + "multiline_fields": 6, + "raw-gotenberg": { + "accepted": true, + "pages": 3, + "remaining_fields": 0, + "text_characters": 3607, + "tagged": false + }, + "litefile": { + "accepted": true, + "pages": 3, + "remaining_fields": 0, + "text_characters": 3607, + "tagged": false + }, + "pdftk": { + "accepted": true, + "pages": 3, + "remaining_fields": 0, + "text_characters": 3607, + "tagged": false + } + }, + { + "source": "docassemble-ALWeaver/docassemble/ALWeaver/test/test_option_groups.pdf", + "sha256": "498a9feccb3d135385b4fbb59755075c99cc280e014e22eb077746d3ae5956dc", + "input_pages": 1, + "input_fields": 6, + "filled_fields": 5, + "multiline_fields": 0, + "raw-gotenberg": { + "accepted": true, + "pages": 1, + "remaining_fields": 0, + "text_characters": 0, + "tagged": false + }, + "litefile": { + "accepted": true, + "pages": 1, + "remaining_fields": 0, + "text_characters": 0, + "tagged": false + }, + "pdftk": { + "accepted": false, + "error_type": "CalledProcessError" + } + }, + { + "source": "docassemble-ALWeaver/docassemble/ALWeaver/test/test_push_button.pdf", + "sha256": "4e5d1b4fbc42099ea12b37c6956a80208ae79a5d4ce599a928b54c16fdba4d69", + "input_pages": 1, + "input_fields": 2, + "filled_fields": 1, + "multiline_fields": 0, + "raw-gotenberg": { + "accepted": true, + "pages": 1, + "remaining_fields": 0, + "text_characters": 0, + "tagged": false + }, + "litefile": { + "accepted": true, + "pages": 1, + "remaining_fields": 0, + "text_characters": 0, + "tagged": false + }, + "pdftk": { + "accepted": true, + "pages": 1, + "remaining_fields": 0, + "text_characters": 0, + "tagged": false + } + }, + { + "source": "docassemble-AppearanceEfile/docassemble/AppearanceEfile/data/templates/appearance_acrobat_pro_edited.pdf", + "sha256": "8b98f685ae4a5c2ebb35cfa673ae6520e3945fc0471915e515ee3fbb8ac61d5c", + "input_pages": 3, + "input_fields": 70, + "filled_fields": 2, + "multiline_fields": 2, + "raw-gotenberg": { + "accepted": true, + "pages": 3, + "remaining_fields": 0, + "text_characters": 6132, + "tagged": true + }, + "litefile": { + "accepted": true, + "pages": 3, + "remaining_fields": 0, + "text_characters": 6132, + "tagged": true + }, + "pdftk": { + "accepted": true, + "pages": 3, + "remaining_fields": 0, + "text_characters": 6131, + "tagged": true + } + }, + { + "source": "docassemble-AppearanceEfile/docassemble/AppearanceEfile/data/templates/appearance_edited.pdf", + "sha256": "4a75bda4130a6f941ea96b1fd7d329ce820cac09d4f13d4d75b4cff94b93c5c7", + "input_pages": 3, + "input_fields": 46, + "filled_fields": 3, + "multiline_fields": 2, + "raw-gotenberg": { + "accepted": true, + "pages": 3, + "remaining_fields": 0, + "text_characters": 7616, + "tagged": true + }, + "litefile": { + "accepted": true, + "pages": 3, + "remaining_fields": 0, + "text_characters": 7618, + "tagged": true + }, + "pdftk": { + "accepted": true, + "pages": 3, + "remaining_fields": 0, + "text_characters": 7616, + "tagged": true + } + }, + { + "source": "docassemble-CLAGuardianship/docassemble/CLAGuardianship/data/templates/affidavit_disclosing_care_or_custody_old.pdf", + "sha256": "88cd00883b53724eac7b562ddc1ccabcef96a54785fdafd74dfb39f526e6cf27", + "input_pages": 2, + "input_fields": 76, + "filled_fields": 1, + "multiline_fields": 0, + "raw-gotenberg": { + "accepted": true, + "pages": 2, + "remaining_fields": 0, + "text_characters": 5902, + "tagged": false + }, + "litefile": { + "accepted": true, + "pages": 2, + "remaining_fields": 0, + "text_characters": 5902, + "tagged": false + }, + "pdftk": { + "accepted": true, + "pages": 2, + "remaining_fields": 0, + "text_characters": 5902, + "tagged": false + } + }, + { + "source": "docassemble-MLHDivorceAndCustody/docassemble/MLHDivorceAndCustody/data/templates/child_support_services_dhs1201d.pdf", + "sha256": "2c9b742e183e955f5ddcf095abce5d442880fc1cba14c1b0a3257523e31406f2", + "input_pages": 1, + "input_fields": 44, + "filled_fields": 3, + "multiline_fields": 0, + "raw-gotenberg": { + "accepted": true, + "pages": 1, + "remaining_fields": 0, + "text_characters": 3790, + "tagged": true + }, + "litefile": { + "accepted": true, + "pages": 1, + "remaining_fields": 0, + "text_characters": 3790, + "tagged": true + }, + "pdftk": { + "accepted": true, + "pages": 1, + "remaining_fields": 0, + "text_characters": 3790, + "tagged": true + } + }, + { + "source": "docassemble-MLHDivorceAndCustody/docassemble/MLHDivorceAndCustody/data/templates/confidential_case_inventory_mc21.pdf", + "sha256": "2f02a121e63c11c1406399b57cb7b9060c1a3b17a8cbfe8adf4db3f7e9be8748", + "input_pages": 1, + "input_fields": 48, + "filled_fields": 10, + "multiline_fields": 0, + "raw-gotenberg": { + "accepted": true, + "pages": 1, + "remaining_fields": 0, + "text_characters": 1954, + "tagged": true + }, + "litefile": { + "accepted": true, + "pages": 1, + "remaining_fields": 0, + "text_characters": 1955, + "tagged": true + }, + "pdftk": { + "accepted": true, + "pages": 1, + "remaining_fields": 0, + "text_characters": 1954, + "tagged": true + } + }, + { + "source": "docassemble-MLHDivorceAndCustody/docassemble/MLHDivorceAndCustody/data/templates/uniform_spousal_support_order_foc10b.pdf", + "sha256": "4540a473fcfbf179cffd3ec144bf428be3a87afc0252c1fceee78cc4a27f0cbd", + "input_pages": 2, + "input_fields": 61, + "filled_fields": 1, + "multiline_fields": 7, + "raw-gotenberg": { + "accepted": true, + "pages": 2, + "remaining_fields": 0, + "text_characters": 4320, + "tagged": true + }, + "litefile": { + "accepted": true, + "pages": 2, + "remaining_fields": 0, + "text_characters": 4322, + "tagged": true + }, + "pdftk": { + "accepted": true, + "pages": 2, + "remaining_fields": 0, + "text_characters": 4320, + "tagged": true + } + }, + { + "source": "docassemble-USCISApplications/docassemble/USCISApplications/data/templates/eoir-33_coa.pdf", + "sha256": "ae5bfc58ed65e4e1bc9826bbd841ec3c7224972a6bb560108f878f06253863c0", + "input_pages": 2, + "input_fields": 26, + "filled_fields": 1, + "multiline_fields": 1, + "raw-gotenberg": { + "accepted": true, + "pages": 2, + "remaining_fields": 0, + "text_characters": 6959, + "tagged": true + }, + "litefile": { + "accepted": true, + "pages": 2, + "remaining_fields": 0, + "text_characters": 6957, + "tagged": true + }, + "pdftk": { + "accepted": true, + "pages": 2, + "remaining_fields": 0, + "text_characters": 6960, + "tagged": true + } + } +] diff --git a/docs/developer-notes/issue-113/validation.md b/docs/developer-notes/issue-113/validation.md new file mode 100644 index 00000000..2edc8e18 --- /dev/null +++ b/docs/developer-notes/issue-113/validation.md @@ -0,0 +1,171 @@ +# Document preparation and preview validation + +Validated September 30, 2026 for [LITEFile issue #113](https://github.com/SuffolkLITLab/LITEFile/issues/113). +Branch: `feature/document-preparation-preview`. +Validation gist: https://gist.github.com/nonprofittechy/54857d2ed0b841a486dbf7a30b1ad915. + +## Result and engine choice + +Use Gotenberg for Word conversion and PDF flattening, with the existing `pypdf` +dependency repairing missing or stale text appearance streams first. PDFtk is +used only in the comparison harness; it is not a runtime dependency. + +The local scan found 12 PDFs with populated form values, comprising 11 distinct +files after SHA-256 deduplication and 22 pages. These come from ALWeaver, +ALDashboard, AppearanceEfile, MLHDivorceAndCustody, USCISApplications and +CLAGuardianship repositories. The corpus includes template defaults, whitespace +values and checked controls; supplemental synthetic values exercise multiline +text, signatures, Unicode and complete packets. It should continue to grow as +more representative completed filings become available. + +| Method | Accepted local specimens | Fields remaining in accepted outputs | +| --- | --- | --- | +| Raw Gotenberg flatten route | 11 / 11 | 0 | +| LITEFile preparation pipeline | 11 / 11 | 0 | +| PDFtk Java 3.3.3 `flatten` | 10 / 11 | 0 | + +PDFtk rejected ALWeaver's option-group fixture. Full provenance, SHA-256 hashes, +page counts, text counts and results are in +[corpus-comparison.json](https://github.com/SuffolkLITLab/LITEFile/blob/feature/document-preparation-preview/docs/developer-notes/issue-113/corpus-comparison.json). +Source documents and filled values stay local; the published report contains +metadata and synthetic screenshots. + +Text extraction initially made raw Gotenberg look sufficient. Raster inspection +revealed that its appearance generation placed three multiline answers on one +line when `/AP` was missing. This matches QPDF's documented limits on appearance +generation. LITEFile creates the text appearance with `pypdf`, clears +`NeedAppearances`, then lets Gotenberg flatten that appearance. Existing usable +appearances remain intact. The synthetic reproducer shows the difference: + +![Raw Gotenberg joins the three lines](https://raw.githubusercontent.com/SuffolkLITLab/LITEFile/feature/document-preparation-preview/docs/developer-notes/issue-113/screenshots/12-raw-gotenberg-multiline.png) + +![LITEFile preserves three separate lines](https://raw.githubusercontent.com/SuffolkLITLab/LITEFile/feature/document-preparation-preview/docs/developer-notes/issue-113/screenshots/13-repaired-multiline.png) + +An EOIR missing-appearance stress case containing accented names demonstrated +another engine limitation. The prepared result retained form fields, so LITEFile +rejected it with a printed-PDF recovery instruction. The automated checks also +reject lost filled text, changed page counts and unreadable responses. Preview +and filer confirmation remain necessary for layout, fonts, clipping, checkbox +and signature fidelity. This work does not establish universal PDF compatibility. + +## Court forms and packet checks + +The current [Vermont File & Serve FAQ](https://www.vtcourts.gov/about-vermont-judiciary/electronic-access/electronic-filing/faqs) +requires form-fillable PDFs to be saved as flat files before filing. Vermont's +YAML explicitly enables flattening and supplies plain-language upload guidance +with that source. Other jurisdictions can set `flatten_pdf_forms: false` and +provide their own guidance through `ui_text`. + +Downloaded the official [Small Claims Answer, form 100-00126, July 2025](https://www.vtcourts.gov/sites/default/files/documents/100-00126%20%E2%80%93%20Small%20Claims%20Answer_0.pdf) +and filled it with synthetic names, a synthetic docket, a selected checkbox, +a three-line answer and a typed `/s/` signature. Checked the rasterized filing +copy on both pages. The checkbox, names, answer lines and signature remained +visible. Appended two synthetic exhibit pages and verified the prepared packet +has four pages, no form fields, preserved signature text and an intact last +exhibit page. No court filing or fee request was made for these documents. + +![Vermont answer retains the checkbox and multiline response](https://raw.githubusercontent.com/SuffolkLITLab/LITEFile/feature/document-preparation-preview/docs/developer-notes/issue-113/screenshots/09-vt-small-claims-answer.png) + +![Vermont answer retains its typed signature](https://raw.githubusercontent.com/SuffolkLITLab/LITEFile/feature/document-preparation-preview/docs/developer-notes/issue-113/screenshots/10-vt-signature.png) + +## Browser validation + +Ran Chromium against Django's live test server with real Gotenberg conversion, +real PDF.js rendering and an in-memory S3 test double. Documents and account +identity were synthetic. This validates app behavior without uploading test +files to production storage or contacting a court. The separate PDF corpus +comparison calls the configured Gotenberg service and writes local artifacts. + +Checked the following: + +- Selecting and uploading a two-page PDF with missing multiline appearances and a two-page DOCX together. +- Loading actual filing bytes through the authenticated preview endpoint. +- Rendering both PDFs, moving to page two, zooming and retaining selectable text layers. +- Keeping the originals separately and providing original and filing-copy download links. +- Requiring a confirmation checkbox for every document before continuing. +- Rendering expandable PDFs in the organize and fees screens, with assertions that each expected page loaded. +- A 390-pixel mobile viewport with no horizontal page overflow. +- Invalid-PDF upload guidance with the existing files retained. +- A simulated 503 from document storage, followed by successful retry after reopening the preview. +- Zero browser page errors and zero Axe violations in the preview screen. + +The initial browser run caught a compatibility problem in PDF.js 6's modern +bundle on the installed Chromium. The shipped assets now use PDF.js's official +legacy build, including its compatibility polyfills. A second run caught +repeated page-region landmark labels across documents; those labels now identify +both the document and the page. The final run passes. A final evidence review also caught a redirected organize +page; the fixture now supplies synthetic case data and asserts the page URL and +heading before capturing that screenshot. Court choices and payment accounts +are stubbed, and the backend fee estimate is stubbed. + +![Preview step with the actual multiline PDF](https://raw.githubusercontent.com/SuffolkLITLab/LITEFile/feature/document-preparation-preview/docs/developer-notes/issue-113/screenshots/02-multiline-preview.png) + +![Word filing copy rendered in the preview](https://raw.githubusercontent.com/SuffolkLITLab/LITEFile/feature/document-preparation-preview/docs/developer-notes/issue-113/screenshots/04-word-preview.png) + +![Expandable preview while organizing documents](https://raw.githubusercontent.com/SuffolkLITLab/LITEFile/feature/document-preparation-preview/docs/developer-notes/issue-113/screenshots/05-organize-preview.png) + +![Expandable preview while choosing payment](https://raw.githubusercontent.com/SuffolkLITLab/LITEFile/feature/document-preparation-preview/docs/developer-notes/issue-113/screenshots/14-fees-preview.png) + +![Mobile preview](https://raw.githubusercontent.com/SuffolkLITLab/LITEFile/feature/document-preparation-preview/docs/developer-notes/issue-113/screenshots/06-mobile-preview.png) + +[All screenshots and the Axe result](https://github.com/SuffolkLITLab/LITEFile/tree/feature/document-preparation-preview/docs/developer-notes/issue-113/screenshots) +include the second-page view, selection screen, invalid upload, preview failure +and final packet exhibit. + +## Automated validation + +| Check | Result | +| --- | --- | +| `uv run pytest -q` | 1,215 passed; 1 opt-in browser test skipped | +| Opt-in real Gotenberg and Chromium test | 1 passed | +| `npm run test:unit` | 64 passed | +| `uv run ruff check .` and formatting | Passed | +| `uv run ty check` | Passed | +| `npm run lint:js` and Prettier check | Passed | +| New Django template lint and format | Passed | +| Stylelint | 0 errors; 19 existing warnings in other stylesheets | +| Bandit | Passed after removing a production assertion | +| `manage.py makemigrations --check --dry-run` | No changes detected | +| `docker build -t litefile:issue-113 .` | Passed; generated and collected PDF.js assets | +| Docusaurus `npm run build` | Passed | + +Regression coverage includes complete filing flows in Vermont, Illinois and +Massachusetts; later fee-waiver uploads; interview handoffs and replacement +PDFs; private preview ownership and draft isolation; original downloads; +unchanged PDFs; damaged and unsupported inputs; service timeout; oversized +output; conservative loss checks; batch rollback and orphan cleanup; server-side +approval; stale preview fingerprints; and original/review metadata preservation +when legacy clients rebuild supporting rows. + +The existing npm dependency tree reports four audit findings in development/test +dependencies. No finding names the new PDF.js package. Those existing dependencies +were left at their locked versions to keep this feature's dependency changes +focused. Accessibility testing checks the preview controls and generated tagged +Word PDF structure; it does not certify every filing PDF as PDF/UA compliant. + +## Reproduction + +From `efile_app`, configure Gotenberg credentials and run: + +```bash +uv sync --group dev +npm ci +uv run pytest -q +npm run test:unit +uv run python ../testing/validate_document_preparation.py \ + --output /tmp/litefile-pdf-validation --render +DOCUMENT_PREPARATION_BROWSER_TESTS=1 \ +DOCUMENT_PREPARATION_EVIDENCE_DIR=/tmp/litefile-browser-evidence \ +uv run pytest -q -s efile/tests/test_document_preparation_browser.py +``` + +The corpus script defaults to `~/docassemble-*`, accepts repeatable `--source` +paths, and writes source hashes and engine results without filled values. It +requires PDFtk for comparison and Poppler for `--render`; the application requires +neither. The browser test requires a Playwright Chromium install. For an existing +local browser, set `PLAYWRIGHT_CHROMIUM_EXECUTABLE` to its executable path. + +Relevant engine documentation: +[Gotenberg flatten route](https://gotenberg.dev/docs/manipulate-pdfs/flatten-pdfs), +[Gotenberg Word conversion](https://gotenberg.dev/docs/convert-with-libreoffice/convert-to-pdf), +and [QPDF appearance generation](https://qpdf.readthedocs.io/en/stable/cli.html#option-generate-appearances). diff --git a/docs/docs/admin/configuration.md b/docs/docs/admin/configuration.md index 9ddd25d6..f2498970 100644 --- a/docs/docs/admin/configuration.md +++ b/docs/docs/admin/configuration.md @@ -35,6 +35,10 @@ LITEFile follows [Twelve-Factor App](https://12factor.net/) principles, configur | `AWS_SESSION_TOKEN` | No | `""` | AWS session token when using temporary credentials. | | `AWS_S3_REGION_NAME` | No | `us-east-1` | AWS region where the S3 bucket is hosted. | | `DJANGO_LOG_LEVEL` | No | `DEBUG` (Dev) / `INFO` (Prod) | Logging verbosity for the `efile` application logger. | +| `GOTENBERG_URL` | Optional | `""` | Base URL for the Gotenberg document conversion API (e.g. `https://gotenberg-dev.fly.dev`). | +| `GOTENBERG_USERNAME` | Optional | `""` | HTTP basic auth username for the Gotenberg service. | +| `GOTENBERG_PASSWORD` | Optional | `""` | HTTP basic auth password for the Gotenberg service. | +| `DOCUMENT_PREPARATION_TIMEOUT_SECONDS` | No | `45` | Timeout for a conversion or flattening request. | --- @@ -56,3 +60,49 @@ AWS_S3_REGION_NAME="us-east-1" # AI Extraction (Optional) OPENAI_API_KEY="sk-..." ``` + + +## Document preparation and previews + +Configure Gotenberg 8.16 or newer for Word conversion and PDF form flattening. +The service must support `/forms/libreoffice/convert` and `/forms/pdfengines/flatten`. +Use HTTPS and service credentials when connecting to a remote instance. Install +fonts used by your forms in Gotenberg: missing fonts can change pagination or layout. + +LITEFile keeps an unchanged PDF byte for byte. When form fields need locking, it +preserves existing appearance streams and repairs missing or stale text appearances +with `pypdf` before Gotenberg flattens the fields. It rejects unreadable, encrypted, +XFA, digitally certificate-signed PDFs that would need flattening, and results with +missing pages, remaining fields, or lost filled-in text. Filers can upload a printed +PDF copy instead. These checks do not guarantee visual fidelity; every new filing +copy must be previewed and confirmed before submission. + +Word conversion requests tagged PDF output and lossless images. It does not +rasterize the document or certify accessibility conformance. Flattening can change +accessibility tags, links, or annotations. The private original is retained separately +from the filing PDF and is available for download. Only the filing copy reaches the +court. Removing a document or expiring an unclaimed handoff cleans up both private +copies when another draft does not reference them. + +State YAML can override the default policy: + +```yaml +document_preparation: + flatten_pdf_forms: false +text: + upload_documents: + preparation_help_unflattened: >- + LITEFile converts Word documents to PDF. PDF form fields are kept as uploaded. + Check the filing PDFs before continuing. +``` + +The default is to lock interactive fields. Vermont explicitly enables this policy +and links to the court's preparation instructions. PDFs without form widgets are +left unchanged, including already flattened and remediated documents. + +PDF.js is pinned in `efile_app/package-lock.json` and served from the application, +including its worker, fonts, and character maps. Run `npm ci` in `efile_app` before +local previews; its install script copies these assets. The Docker build generates +the same assets in a separate Node stage. Document bytes come from an authenticated, +draft-scoped endpoint with `Cache-Control: private, no-store`, so previewing does +not require public storage URLs or S3 CORS configuration. diff --git a/docs/docs/admin/deployment.md b/docs/docs/admin/deployment.md index 47661767..88d9fd4c 100644 --- a/docs/docs/admin/deployment.md +++ b/docs/docs/admin/deployment.md @@ -99,7 +99,10 @@ fly secrets set \ AWS_SECRET_ACCESS_KEY="..." \ AWS_S3_BUCKET_NAME="litefile-production-documents" \ AWS_S3_REGION_NAME="us-east-1" \ - OPENAI_API_KEY="sk-..." + OPENAI_API_KEY="sk-..." \ + GOTENBERG_URL="https://..." \ + GOTENBERG_USERNAME="..." \ + GOTENBERG_PASSWORD="..." ``` --- diff --git a/efile_app/.env.example b/efile_app/.env.example index 484cf2c2..ec02abc6 100644 --- a/efile_app/.env.example +++ b/efile_app/.env.example @@ -39,3 +39,10 @@ MAX_FILE_SIZE = 10 * 1024 * 1024 # 10MB ALLOWED_FILE_TYPES = ['.pdf', '.doc', '.docx'] OPENAI_API_KEY = "..." OPENAI_BASE_URL = "https://api.openai.com/v1/" + +# Gotenberg document conversion service +GOTENBERG_URL = "https://gotenberg-dev.fly.dev" +GOTENBERG_USERNAME = "your-gotenberg-username" +GOTENBERG_PASSWORD = "your-gotenberg-password" + +DOCUMENT_PREPARATION_TIMEOUT_SECONDS = 45 diff --git a/efile_app/efile/config_text_strings.py b/efile_app/efile/config_text_strings.py index da9aa63e..e0e4a96a 100644 --- a/efile_app/efile/config_text_strings.py +++ b/efile_app/efile/config_text_strings.py @@ -50,4 +50,9 @@ "terms.starting_document_example", "complaint", ), + # Translators: Preparation guidance when flatten_pdf_forms is enabled. May link to state-specific requirements. + pgettext_lazy( + "upload_documents.preparation_help", + "Vermont requires filled-in PDF forms to have their answers locked in place (sometimes called flattening). LITEFile prepares that filing copy for you, so you do not need to flatten it yourself. We also convert Word documents to PDF. Check every page of the filing PDFs before continuing. [Vermont's PDF preparation instructions](https://www.vtcourts.gov/about-vermont-judiciary/electronic-access/electronic-filing/faqs).", + ), ] diff --git a/efile_app/efile/management/commands/expire_unclaimed_handoffs.py b/efile_app/efile/management/commands/expire_unclaimed_handoffs.py index 40b679e0..68e76669 100644 --- a/efile_app/efile/management/commands/expire_unclaimed_handoffs.py +++ b/efile_app/efile/management/commands/expire_unclaimed_handoffs.py @@ -4,9 +4,11 @@ from django.core.management.base import BaseCommand, CommandError from django.db import transaction +from django.db.models import Q from django.utils import timezone from efile.models import FilingDocument, FilingDraft, InterviewHandoff +from efile.services.document_previews import document_storage_keys from efile.utils.s3_upload_handler import S3UploadHandler @@ -36,8 +38,13 @@ def handle(self, *args, **options): draft = FilingDraft.objects.select_for_update().filter(pk=draft_id, user__isnull=True).first() if draft is None: continue - for key in draft.documents.exclude(s3_key="").values_list("s3_key", flat=True): - if not FilingDocument.objects.filter(s3_key=key).exclude(draft=draft).exists(): + keys = {key for doc in draft.documents.all() for key in document_storage_keys(doc)} + for key in keys: + if ( + not FilingDocument.objects.filter(Q(s3_key=key) | Q(original_s3_key=key)) + .exclude(draft=draft) + .exists() + ): result = handler.delete_file(key) if not result.get("success"): raise CommandError( diff --git a/efile_app/efile/migrations/0029_document_preparation_and_preview.py b/efile_app/efile/migrations/0029_document_preparation_and_preview.py new file mode 100644 index 00000000..2f72d6d6 --- /dev/null +++ b/efile_app/efile/migrations/0029_document_preparation_and_preview.py @@ -0,0 +1,34 @@ +# Generated by Django 5.2.17 on 2026-09-30 17:30 + +import efile.workflow +from django.db import migrations, models + + +class Migration(migrations.Migration): + + dependencies = [ + ('efile', '0028_extraction_claim_leases'), + ] + + operations = [ + migrations.AddField( + model_name='filingdocument', + name='original_s3_key', + field=models.CharField(blank=True, max_length=1024), + ), + migrations.AddField( + model_name='filingdocument', + name='preparation', + field=models.CharField(blank=True, choices=[('unchanged', 'Original PDF'), ('converted', 'Converted from Word'), ('flattened', 'Form fields locked'), ('converted_flattened', 'Converted and fields locked')], max_length=30), + ), + migrations.AddField( + model_name='filingdocument', + name='preparation_reviewed_at', + field=models.DateTimeField(blank=True, null=True), + ), + migrations.AlterField( + model_name='filingdraft', + name='current_step', + field=models.CharField(choices=[('options', 'Options'), ('filing_path', 'Start'), ('upload_documents', 'Upload documents'), ('preview_documents', 'Preview documents'), ('extraction_review', 'Confirm filing'), ('case_lookup', 'Find your case'), ('case_confirmation', 'Confirm your case'), ('document_checklist', 'Check documents'), ('organize_documents', 'Organize documents'), ('your_information', 'Your information'), ('parties', 'People in this filing'), ('party_details', 'Person details'), ('case_questions', 'Case questions'), ('payment', 'Fees'), ('review', 'Review'), ('confirmation', 'Confirmation')], default=efile.workflow.WorkflowStepKey['OPTIONS'], max_length=64), + ), + ] diff --git a/efile_app/efile/models.py b/efile_app/efile/models.py index bcbf4db8..1f433cb1 100644 --- a/efile_app/efile/models.py +++ b/efile_app/efile/models.py @@ -356,6 +356,19 @@ class Role(models.TextChoices): content_type = models.CharField(max_length=255, blank=True) s3_key = models.CharField(max_length=1024, blank=True) public_url = models.URLField(max_length=2048, blank=True) + # Originals are private recovery copies; only s3_key is sent to the court. + original_s3_key = models.CharField(max_length=1024, blank=True) + preparation = models.CharField( + max_length=30, + blank=True, + choices=[ + ("unchanged", "Original PDF"), + ("converted", "Converted from Word"), + ("flattened", "Form fields locked"), + ("converted_flattened", "Converted and fields locked"), + ], + ) + preparation_reviewed_at = models.DateTimeField(null=True, blank=True) filing_type_code = models.CharField(max_length=100, blank=True) filing_type_name = models.CharField(max_length=255, blank=True) diff --git a/efile_app/efile/services/document_preparation.py b/efile_app/efile/services/document_preparation.py new file mode 100644 index 00000000..9beca24e --- /dev/null +++ b/efile_app/efile/services/document_preparation.py @@ -0,0 +1,250 @@ +"""Prepare filing copies without replacing or rasterizing the original upload.""" + +from __future__ import annotations + +import io +import logging +import time +import zipfile +from dataclasses import dataclass +from pathlib import Path + +import requests +from django.conf import settings +from django.core.files.uploadedfile import SimpleUploadedFile +from pypdf import PdfReader, PdfWriter +from pypdf.generic import DictionaryObject + +from efile.utils.config_loader import config_loader + +logger = logging.getLogger(__name__) + + +class PreparationError(ValueError): + """An actionable preparation failure safe to show to the filer.""" + + +@dataclass(frozen=True) +class PreparedDocument: + content: bytes + filename: str + operation: str + + +def requires_flattening(jurisdiction): + config = config_loader.load_jurisdiction_config(jurisdiction) + return config.get("document_preparation", {}).get("flatten_pdf_forms", True) + + +def inspect_pdf(content): + try: + if not content.startswith(b"%PDF-"): + raise ValueError("Not a PDF") + reader = PdfReader(io.BytesIO(content)) + if reader.is_encrypted or not reader.pages: + raise ValueError("Encrypted or empty PDF") + root = reader.trailer["/Root"] + if not isinstance(root, DictionaryObject): + raise ValueError("Invalid PDF catalog") + form = root.get("/AcroForm") + if form and form.get_object().get("/XFA"): + raise PreparationError( + "This PDF uses an unsupported form format. Save a printed PDF copy and upload it again." + ) + # Force page and annotation parsing before accepting a filing copy. + widgets = [ + ref.get_object() + for page in reader.pages + for ref in page.get("/Annots", []) + if ref.get_object().get("/Subtype") == "/Widget" + ] + return reader, widgets + except PreparationError: + raise + except Exception as error: + raise PreparationError("This PDF could not be read. Upload an unlocked, readable PDF and try again.") from error + + +def _word_format(content, suffix): + if suffix == ".doc": + if not content.startswith(b"\xd0\xcf\x11\xe0\xa1\xb1\x1a\xe1"): + raise PreparationError("This file is not a Word document. Save it as DOCX or PDF and upload it again.") + return + try: + with zipfile.ZipFile(io.BytesIO(content)) as archive: + entries = archive.infolist() + if len(entries) > 10000 or sum(entry.file_size for entry in entries) > 100 * 1024 * 1024: + raise ValueError("Expanded document too large") + if not {"[Content_Types].xml", "word/document.xml"}.issubset(archive.namelist()): + raise ValueError("Not DOCX") + except (ValueError, zipfile.BadZipFile) as error: + raise PreparationError( + "This DOCX could not be read. Save a new Word or PDF copy and upload it again." + ) from error + + +def _gotenberg(content, suffix, route, data=None): + base_url = settings.GOTENBERG_URL.rstrip("/") + if not base_url: + raise PreparationError( + "Document preparation is unavailable. Upload a PDF with the form fields already locked, or try again later." + ) + limit = settings.MAX_FILE_SIZE + deadline = time.monotonic() + settings.DOCUMENT_PREPARATION_TIMEOUT_SECONDS + try: + with requests.post( + f"{base_url}{route}", + auth=(settings.GOTENBERG_USERNAME, settings.GOTENBERG_PASSWORD), + # Do not send user filenames to the conversion service. + files={"files": (f"document{suffix}", content, "application/octet-stream")}, + data=data or {}, + timeout=(5, settings.DOCUMENT_PREPARATION_TIMEOUT_SECONDS), + allow_redirects=False, + stream=True, + ) as response: + if response.status_code != 200: + raise PreparationError("We could not prepare this document. Upload a PDF copy or try uploading again.") + result = bytearray() + for chunk in response.iter_content(64 * 1024): + result.extend(chunk) + if len(result) > limit: + raise PreparationError( + "The prepared PDF exceeds 10 MB. Split or reduce the document and upload it again." + ) + if time.monotonic() > deadline: + raise requests.Timeout() + return bytes(result) + except requests.RequestException as error: + logger.warning("Document preparation service unavailable (%s)", type(error).__name__) + raise PreparationError( + "Document preparation timed out or is unavailable. Try again, or upload a PDF copy with its form fields locked." + ) from error + + +def _repair_text_appearances(content, source, widgets): + """QPDF cannot generate multiline appearances. Use pypdf for stale/missing text APs. + + Keep existing correct appearance streams (including embedded fonts) intact. + Turning off NeedAppearances after repair stops QPDF from replacing them. + """ + form = source.trailer["/Root"].get("/AcroForm") + needs_appearances = bool(form and getattr(form.get_object().get("/NeedAppearances"), "value", False)) + missing = False + for widget in widgets: + field = widget.get("/Parent", widget).get_object() + if field.get("/FT") == "/Tx" and (not widget.get("/AP") or not widget["/AP"].get("/N")): + missing = True + if not (needs_appearances or missing): + return content + values = { + name: str(field.get("/V") or "") + for name, field in (source.get_fields() or {}).items() + if field.get("/FT") == "/Tx" + } + try: + writer = PdfWriter(clone_from=source) + writer.update_page_form_field_values(None, values, auto_regenerate=False) + output = io.BytesIO() + writer.write(output) + return output.getvalue() + except Exception as error: + raise PreparationError( + "This PDF's filled-in answers could not be rendered safely. Save a printed PDF copy and upload it again." + ) from error + + +def _flatten(content): + source, widgets = inspect_pdf(content) + if not widgets: + return content, False + fields = source.get_fields() or {} + if any(field.get("/FT") == "/Sig" and field.get("/V") for field in fields.values()): + raise PreparationError( + "This PDF has a digital certificate signature. Upload a filing copy with a visible signature instead; locking its fields would invalidate the certificate." + ) + content = _repair_text_appearances(content, source, widgets) + result = _gotenberg(content, ".pdf", "/forms/pdfengines/flatten") + output, remaining = inspect_pdf(result) + if remaining or output.get_fields() or len(output.pages) != len(source.pages): + raise PreparationError( + "We could not safely lock this PDF's form fields. Save a printed PDF copy and upload it again." + ) + # Engines can return 200 while dropping filled text. This is a conservative + # check, not a guarantee of visual fidelity; the filer still previews it. + visible_text = " ".join(" ".join(page.extract_text() or "" for page in output.pages).split()) + for field in fields.values(): + if field.get("/FT") == "/Tx": + value = str(field.get("/V") or "") + if any(" ".join(line.split()) not in visible_text for line in value.splitlines() if line.strip()): + raise PreparationError( + "Some filled-in text could not be preserved. Save a printed PDF copy and upload it again." + ) + return result, True + + +def prepare_document(uploaded_file, jurisdiction): + suffix = Path(uploaded_file.name).suffix.lower() + if suffix not in settings.ALLOWED_FILE_TYPES: + raise PreparationError("Choose a PDF, DOCX, or Word document.") + uploaded_file.seek(0) + content = uploaded_file.read(settings.MAX_FILE_SIZE + 1) + uploaded_file.seek(0) + if not content or len(content) > settings.MAX_FILE_SIZE: + raise PreparationError("Choose a nonempty document no larger than 10 MB.") + operation = "unchanged" + filename = uploaded_file.name + if suffix in {".doc", ".docx"}: + _word_format(content, suffix) + content = _gotenberg( + content, + suffix, + "/forms/libreoffice/convert", + {"exportFormFields": "false", "pdfua": "true", "losslessImageCompression": "true"}, + ) + filename = f"{Path(filename).stem}.pdf" + operation = "converted" + inspect_pdf(content) + if requires_flattening(jurisdiction): + content, flattened = _flatten(content) + if flattened: + operation = "converted_flattened" if operation == "converted" else "flattened" + return PreparedDocument(content, filename, operation) + + +def store_prepared_document(handler, uploaded_file, jurisdiction, role, *, keys, metadata=None): + """Store both copies. The caller cleans ``keys`` on any later failure.""" + prepared = prepare_document(uploaded_file, jurisdiction) + original_key = "" + if prepared.operation != "unchanged": + uploaded_file.seek(0) + original = handler.upload_file(uploaded_file, file_type="original", metadata=metadata) + if not original.get("success"): + raise PreparationError("The original could not be saved. Try uploading again.") + original_key = original["key"] + keys.append(original_key) + filing = SimpleUploadedFile(prepared.filename, prepared.content, content_type="application/pdf") + result = handler.upload_file(filing, file_type=role, metadata=metadata) + if not result.get("success"): + raise PreparationError("The filing copy could not be saved. Try uploading again.") + keys.append(result["key"]) + return { + "name": prepared.filename[:255], + "original_filename": uploaded_file.name[:255], + "original_s3_key": original_key, + "preparation": prepared.operation, + "preparation_reviewed_at": None, + "size": len(prepared.content), + "content_type": "application/pdf", + "s3_key": result["key"], + "public_url": handler.get_public_url(result["key"]), + } + + +def cleanup_uploads(handler, keys): + for key in keys: + try: + result = handler.delete_file(key) + if not result.get("success"): + logger.warning("Could not remove an uncommitted document upload") + except Exception: + logger.exception("Could not remove an uncommitted document upload") diff --git a/efile_app/efile/services/document_previews.py b/efile_app/efile/services/document_previews.py new file mode 100644 index 00000000..ba99cd9e --- /dev/null +++ b/efile_app/efile/services/document_previews.py @@ -0,0 +1,23 @@ +"""Server-owned preview acknowledgements tied to the current filing bytes.""" + +import hashlib +import json + + +def unreviewed_documents(draft): + return draft.documents.exclude(preparation="").filter(preparation_reviewed_at__isnull=True) + + +def preview_fingerprint(documents): + return hashlib.sha256(json.dumps(sorted((doc.pk, doc.s3_key) for doc in documents)).encode()).hexdigest() + + +def require_document_previews(draft): + if unreviewed_documents(draft).exists(): + raise ValueError( + "Preview your uploaded PDFs and confirm that their pages and signatures are correct before submitting." + ) + + +def document_storage_keys(document): + return list(dict.fromkeys(key for key in (document.s3_key, document.original_s3_key) if key)) diff --git a/efile_app/efile/services/document_uploads.py b/efile_app/efile/services/document_uploads.py index 372176e5..39ae476d 100644 --- a/efile_app/efile/services/document_uploads.py +++ b/efile_app/efile/services/document_uploads.py @@ -1,52 +1,56 @@ -from efile.models import FilingDocument +from django.db import transaction +from django.db.models import Max + +from efile.models import FilingDocument, FilingDraft from efile.services.document_extractions import queue_document_extraction -from efile.services.drafts import read_upload_data, write_upload_data +from efile.services.document_preparation import cleanup_uploads, store_prepared_document +from efile.services.drafts import read_upload_data +from efile.services.fee_quotes import invalidate_fee_quote from efile.utils.s3_upload_handler import S3UploadHandler from efile.workflow import WorkflowStepKey def upload_files(draft, uploaded_files, jurisdiction, *, current_step=WorkflowStepKey.UPLOAD_DOCUMENTS): - """Upload PDFs immediately and queue lead analysis outside the request.""" - + """Prepare and store a whole batch, then queue analysis of the filing copy.""" handler = S3UploadHandler() if not handler._ensure_initialized(): raise ValueError("Document storage is not configured. Please try again later.") - - current = read_upload_data(draft) - files = current.setdefault("files", {}) - supporting = list(files.get("supporting", [])) - found_lead = False - - for uploaded_file in uploaded_files: - validation = handler.validate_file(uploaded_file, max_size_mb=10, allowed_types=[".pdf"]) - if not validation["valid"]: - raise ValueError(f"{uploaded_file.name}: {validation['error']}") - - is_lead = not files.get("lead") and not found_lead - role = FilingDocument.Role.LEAD if is_lead else FilingDocument.Role.SUPPORTING - - uploaded_file.seek(0) - result = handler.upload_file(uploaded_file, file_type=role) - if not result["success"]: - raise ValueError(result.get("error", f"Could not upload {uploaded_file.name}.")) - - file_data = { - "name": uploaded_file.name, - "size": uploaded_file.size, - "type": uploaded_file.content_type, - "url": handler.get_public_url(result["key"]), - "s3_key": result["key"], - } - if is_lead: - files["lead"] = file_data - found_lead = True - current["guesses"] = {} - else: - supporting.append(file_data) - - files["supporting"] = supporting - write_upload_data(draft, current, current_step=current_step) - if found_lead: - lead = FilingDocument.objects.get(draft=draft, role=FilingDocument.Role.LEAD) - queue_document_extraction(lead) - return current + keys = [] + try: + # Prepare the entire batch before changing the durable draft. + prepared = [] + for file in uploaded_files: + try: + prepared.append(store_prepared_document(handler, file, jurisdiction, "document", keys=keys)) + except ValueError as error: + raise ValueError(f"{file.name}: {error}") from error + with transaction.atomic(): + draft = FilingDraft.objects.select_for_update().get(pk=draft.pk) + if draft.status not in {FilingDraft.Status.DRAFT, FilingDraft.Status.ERROR}: + raise ValueError("This filing is no longer available to edit.") + has_lead = draft.documents.filter(role=FilingDocument.Role.LEAD).exists() + highest = draft.documents.filter(role=FilingDocument.Role.SUPPORTING).aggregate(order=Max("sort_order"))[ + "order" + ] + order = 0 if highest is None else highest + 1 + for values in prepared: + is_lead = not has_lead + document = FilingDocument.objects.create( + draft=draft, + role=FilingDocument.Role.LEAD if is_lead else FilingDocument.Role.SUPPORTING, + sort_order=0 if is_lead else order, + **values, + ) + if is_lead: + has_lead = True + draft.extracted_guesses = {} + queue_document_extraction(document) + else: + order += 1 + draft.current_step = str(current_step) + invalidate_fee_quote(draft, save=False) + draft.save() + except Exception: + cleanup_uploads(handler, keys) + raise + return read_upload_data(draft) diff --git a/efile_app/efile/services/draft_urls.py b/efile_app/efile/services/draft_urls.py index 224aff78..215fc203 100644 --- a/efile_app/efile/services/draft_urls.py +++ b/efile_app/efile/services/draft_urls.py @@ -8,6 +8,7 @@ { "filing_path", "upload_documents", + "preview_documents", "document_extraction_status", "extraction_review", "case_lookup", diff --git a/efile_app/efile/services/drafts.py b/efile_app/efile/services/drafts.py index c34f37a0..94ba7659 100644 --- a/efile_app/efile/services/drafts.py +++ b/efile_app/efile/services/drafts.py @@ -494,12 +494,23 @@ def write_upload_data( # Handoff provenance and pending corrections are keyed by row id, so # they follow the file to its rebuilt row the same way. previous_ids = {document.s3_key: document.pk for document in previous} + preparation_metadata = { + document.s3_key: { + field: getattr(document, field) + for field in ("original_filename", "original_s3_key", "preparation", "preparation_reviewed_at") + } + for document in previous + } FilingDocument.objects.filter(draft=draft, role=FilingDocument.Role.SUPPORTING).delete() for index, file_obj in enumerate(supporting_files): config = supporting_configs[index] if index < len(supporting_configs) else {} _upsert_document(draft, FilingDocument.Role.SUPPORTING, index, file_obj or {}, config or {}) moved = {} for document in FilingDocument.objects.filter(draft=draft, role=FilingDocument.Role.SUPPORTING): + if metadata := preparation_metadata.get(document.s3_key): + for field, value in metadata.items(): + setattr(document, field, value) + document.save(update_fields=[*metadata, "updated_at"]) item_id = claimed_items.get(document.s3_key, "") if item_id: document.checklist_item_id = item_id diff --git a/efile_app/efile/services/handoff.py b/efile_app/efile/services/handoff.py index 76d69986..ba3b348d 100644 --- a/efile_app/efile/services/handoff.py +++ b/efile_app/efile/services/handoff.py @@ -326,6 +326,8 @@ def populate(draft, payload, uploads): content_type="application/pdf", s3_key=uploaded["key"], public_url=uploaded["url"], + original_s3_key=uploaded.get("original_s3_key", ""), + preparation=uploaded.get("preparation", ""), ) order[document["role"]] += 1 record(draft, f"documents.{row.pk}", "source_suggestion", document) diff --git a/efile_app/efile/settings_base.py b/efile_app/efile/settings_base.py index e18522cd..cacedf92 100644 --- a/efile_app/efile/settings_base.py +++ b/efile_app/efile/settings_base.py @@ -146,6 +146,12 @@ DOCUMENT_EXTRACTION_MEMORY_MB = int(os.getenv("DOCUMENT_EXTRACTION_MEMORY_MB", "768")) MAX_FILE_SIZE = 10 * 1024 * 1024 # 10MB ALLOWED_FILE_TYPES = [".pdf", ".doc", ".docx"] + +# Gotenberg document conversion service +GOTENBERG_URL = os.getenv("GOTENBERG_URL", "") +GOTENBERG_USERNAME = os.getenv("GOTENBERG_USERNAME", "") +GOTENBERG_PASSWORD = os.getenv("GOTENBERG_PASSWORD", "") +DOCUMENT_PREPARATION_TIMEOUT_SECONDS = int(os.getenv("DOCUMENT_PREPARATION_TIMEOUT_SECONDS", "45")) # Analyze only the front of a filing. Exhibits and discovery can make a PDF # hundreds of pages long, while the caption and filing details normally appear # near the beginning. diff --git a/efile_app/efile/static/config/states/vermont.yaml b/efile_app/efile/static/config/states/vermont.yaml index 65b1ece2..21ee2600 100644 --- a/efile_app/efile/static/config/states/vermont.yaml +++ b/efile_app/efile/static/config/states/vermont.yaml @@ -35,7 +35,19 @@ state: # Words and sentences this jurisdiction says differently. Keys, defaults, and # what each one is for are in efile/utils/ui_text.py; anything not listed here # uses the default English wording. +# Vermont requires form-fillable PDFs to be flattened before filing. +# https://www.vtcourts.gov/about-vermont-judiciary/electronic-access/electronic-filing/faqs +document_preparation: + flatten_pdf_forms: true + text: + upload_documents: + preparation_help: >- + Vermont requires filled-in PDF forms to have their answers locked in place + (sometimes called flattening). LITEFile prepares that filing copy for you, + so you do not need to flatten it yourself. We also convert Word documents + to PDF. Check every page of the filing PDFs before continuing. + [Vermont's PDF preparation instructions](https://www.vtcourts.gov/about-vermont-judiciary/electronic-access/electronic-filing/faqs). terms: # Vermont courts call the document that opens a case a complaint. starting_document_example: "complaint" diff --git a/efile_app/efile/static/css/document-preview.css b/efile_app/efile/static/css/document-preview.css new file mode 100644 index 00000000..b4715718 --- /dev/null +++ b/efile_app/efile/static/css/document-preview.css @@ -0,0 +1,57 @@ +.document-preview { + border: 1px solid #cbd5e1; + border-radius: 0.5rem; + margin-block: 0.75rem; + background: #fff; +} + +.document-preview>summary { + padding: 0.75rem; + color: #163b62; + cursor: pointer; + overflow-wrap: anywhere; +} + +.document-preview__body { + padding: 0.75rem; +} + +.document-preview__toolbar { + display: flex; + gap: 0.5rem; + align-items: center; + flex-wrap: wrap; + margin-block-end: 0.75rem; +} + +.document-preview__toolbar[hidden] { + display: none; +} + +.document-preview__toolbar input { + width: 4rem; +} + +.document-preview__frame { + position: relative; + height: min(70vh, 50rem); + min-height: 15rem; +} + +.document-preview__viewport { + position: absolute; + inset: 0; + overflow: auto; + background: #e2e8f0; +} + +.document-preview :focus-visible { + outline: 3px solid #163b62; + outline-offset: 3px; +} + +@media (forced-colors: active) { + .document-preview :focus-visible { + outline-color: Highlight; + } +} \ No newline at end of file diff --git a/efile_app/efile/static/js/document-preview.js b/efile_app/efile/static/js/document-preview.js new file mode 100644 index 00000000..778a1e79 --- /dev/null +++ b/efile_app/efile/static/js/document-preview.js @@ -0,0 +1,101 @@ +/* PDF.js is served locally; document bytes stay on the authenticated app origin. */ +(() => { + const assetScript = document.querySelector("script[data-pdf-library]"); + let modules; + + function loadModules() { + if (!modules) { + modules = import(assetScript.dataset.pdfLibrary).then(async (pdfjs) => { + pdfjs.GlobalWorkerOptions.workerSrc = assetScript.dataset.pdfWorker; + globalThis.pdfjsLib = pdfjs; + const components = await import(assetScript.dataset.pdfViewer); + return { + pdfjs, + components + }; + }); + } + return modules; + } + + async function openPreview(details) { + if (!details.open || details.dataset.loaded) return; + details.dataset.loaded = "true"; + const status = details.querySelector("[data-pdf-status]"); + status.textContent = gettext("Loading PDF…"); + let task; + try { + const { + pdfjs, + components + } = await loadModules(); + const container = details.querySelector("[data-pdf-url]"); + const eventBus = new components.EventBus(); + const linkService = new components.PDFLinkService({ + eventBus + }); + const viewer = new components.PDFViewer({ + container, + eventBus, + linkService, + textLayerMode: 1, + annotationMode: pdfjs.AnnotationMode.ENABLE, + enableScripting: false, + }); + linkService.setViewer(viewer); + task = pdfjs.getDocument({ + url: container.dataset.pdfUrl, + isEvalSupported: false, + cMapUrl: assetScript.dataset.pdfResources + "cmaps/", + cMapPacked: true, + standardFontDataUrl: assetScript.dataset.pdfResources + "standard_fonts/", + wasmUrl: assetScript.dataset.pdfResources + "wasm/", + disableRange: true, + disableAutoFetch: true, + }); + const pdf = await task.promise; + viewer.setDocument(pdf); + linkService.setDocument(pdf); + const pageInput = details.querySelector("[data-pdf-page]"); + pageInput.max = pdf.numPages; + details.querySelector("[data-pdf-page-count]").textContent = interpolate(gettext("of %s"), [pdf.numPages]); + eventBus.on("pagesinit", () => { + viewer.currentScaleValue = "page-width"; + for (let index = 0; index < pdf.numPages; index++) { + const pageRegion = viewer.getPageView(index).div; + pageRegion.removeAttribute("data-l10n-id"); + pageRegion.removeAttribute("data-l10n-args"); + pageRegion.setAttribute("aria-label", interpolate(gettext("%s, page %s"), [container.getAttribute("aria-label"), index + 1])); + } + }); + eventBus.on("pagerendered", (event) => { + status.textContent = event.error ? gettext("This page could not be rendered. Download the PDF to check it.") : ""; + if (!event.error) details.dataset.rendered = "true"; + }); + eventBus.on("pagechanging", (event) => { + pageInput.value = event.pageNumber; + }); + details.querySelector(".document-preview__toolbar").hidden = false; + pageInput.addEventListener("change", () => { + const page = Number(pageInput.value); + if (Number.isInteger(page) && page >= 1 && page <= pdf.numPages) viewer.currentPageNumber = page; + }); + details.querySelector("[data-pdf-previous]").addEventListener("click", () => viewer.previousPage()); + details.querySelector("[data-pdf-next]").addEventListener("click", () => viewer.nextPage()); + details.querySelector("[data-pdf-zoom-in]").addEventListener("click", () => viewer.increaseScale()); + details.querySelector("[data-pdf-zoom-out]").addEventListener("click", () => viewer.decreaseScale()); + details.addEventListener("toggle", () => { + if (details.open) viewer.update(); + }); + } catch (error) { + console.warn("PDF preview failed", error.message); + if (task) await task.destroy(); + modules = undefined; + delete details.dataset.loaded; + status.textContent = gettext("The PDF preview could not load. Download the filing PDF to check it, or close this panel and open it to retry."); + } + } + document.querySelectorAll("[data-pdf-preview]").forEach((details) => { + details.addEventListener("toggle", () => openPreview(details)); + }); +})(); \ No newline at end of file diff --git a/efile_app/efile/static/js/upload-documents.js b/efile_app/efile/static/js/upload-documents.js index b2db8d60..0802ae63 100644 --- a/efile_app/efile/static/js/upload-documents.js +++ b/efile_app/efile/static/js/upload-documents.js @@ -134,7 +134,7 @@ const fileCountLabel = selectedFiles.size === 1 ? "file" : "files"; dropZone.querySelector("strong").textContent = selectedFiles.size ? `${selectedFiles.size} ${fileCountLabel} selected` : - "Choose PDFs or drag them here"; + "Choose PDFs or Word documents, or drag them here"; } function addFiles(files) { @@ -172,8 +172,8 @@ errorBox.hidden = true; state.hidden = false; uploadButton.disabled = true; - stateTitle.textContent = "Uploading your documents…"; - stateDetail.textContent = "Keep this page open while the files upload."; + stateTitle.textContent = "Uploading and preparing your documents…"; + stateDetail.textContent = "Keep this page open while we prepare the filing PDFs."; try { const response = await fetch(window.location.href, { diff --git a/efile_app/efile/static/js/waiver-upload.js b/efile_app/efile/static/js/waiver-upload.js index 6d201ef9..a32bc705 100644 --- a/efile_app/efile/static/js/waiver-upload.js +++ b/efile_app/efile/static/js/waiver-upload.js @@ -60,7 +60,7 @@ document.addEventListener("DOMContentLoaded", () => { upload.addEventListener("click", async () => { if (uploading) return; if (!file.files.length || !confidentiality.value) { - status.textContent = gettext("Choose a PDF and a confidentiality setting."); + status.textContent = gettext("Choose a PDF or Word document and a confidentiality setting."); (!file.files.length ? file : confidentiality).focus(); return; } @@ -83,6 +83,10 @@ document.addEventListener("DOMContentLoaded", () => { }); const data = await response.json(); if (!response.ok || !data.success) throw new Error(data.error || gettext("The upload failed. Try again.")); + if (data.preview_url) { + window.location.assign(window.withFilingDraft(data.preview_url)); + return; + } document.getElementById("fee-inputs-token").textContent = JSON.stringify(data.fee_inputs_token); document.getElementById("waiver-upload-required").hidden = true; const confirmation = document.getElementById("waiver-upload-confirmation"); diff --git a/efile_app/efile/templates/efile/components/document_preview.html b/efile_app/efile/templates/efile/components/document_preview.html new file mode 100644 index 00000000..726e1129 --- /dev/null +++ b/efile_app/efile/templates/efile/components/document_preview.html @@ -0,0 +1,50 @@ +{% load i18n %} +
+ {% translate "View PDF:" %} {{ document.name|default:document.original_filename }} +
+

+ {% translate "Download filing PDF" %} + {% if document.original_s3_key %} + · {% translate "Download original upload" %} + {% endif %} +

+ +

+
+
+
+
+
+
+
diff --git a/efile_app/efile/templates/efile/document_checklist.html b/efile_app/efile/templates/efile/document_checklist.html index 9cf14e88..04161040 100644 --- a/efile_app/efile/templates/efile/document_checklist.html +++ b/efile_app/efile/templates/efile/document_checklist.html @@ -119,7 +119,7 @@

{% translate "Your document plan" %}

{% if documents %} @@ -194,6 +194,7 @@

{% translate "Files you have added" %}

{% translate "Added" %} + {% include "efile/components/document_preview.html" %} {% endfor %} @@ -39,3 +40,15 @@

Documents

Review and submit

{% endif %} {% endblock public_content %} +{% block extra_css %} + + +{% endblock extra_css %} +{% block extra_js %} + + +{% endblock extra_js %} diff --git a/efile_app/efile/templates/efile/organize_documents.html b/efile_app/efile/templates/efile/organize_documents.html index f7ad40c8..b59a855f 100644 --- a/efile_app/efile/templates/efile/organize_documents.html +++ b/efile_app/efile/templates/efile/organize_documents.html @@ -139,6 +139,7 @@

{% translate "Organize your documents" %}

+ {% include "efile/components/document_preview.html" %} {% endfor %}
diff --git a/efile_app/efile/templates/efile/payment.html b/efile_app/efile/templates/efile/payment.html index ebc9363b..0ee2b715 100644 --- a/efile_app/efile/templates/efile/payment.html +++ b/efile_app/efile/templates/efile/payment.html @@ -11,6 +11,9 @@
{% translate "Fees" %}

{% translate "Choose how to pay court fees" %}

+ {% for document in documents %} + {% include "efile/components/document_preview.html" with document=document %} + {% endfor %}
@@ -116,10 +119,10 @@

{% translate "Estimated fees before a waiver" %}

- + +
+ + +{% endblock workflow_content %} diff --git a/efile_app/efile/templates/efile/review.html b/efile_app/efile/templates/efile/review.html index 91143b78..272f4498 100644 --- a/efile_app/efile/templates/efile/review.html +++ b/efile_app/efile/templates/efile/review.html @@ -100,6 +100,7 @@

{% translate "Documents" %}

{% endif %} + {% include "efile/components/document_preview.html" %} {% endfor %} diff --git a/efile_app/efile/templates/efile/upload_documents.html b/efile_app/efile/templates/efile/upload_documents.html index aeddbaee..8383cb69 100644 --- a/efile_app/efile/templates/efile/upload_documents.html +++ b/efile_app/efile/templates/efile/upload_documents.html @@ -1,6 +1,7 @@ {% extends "efile/workflow_base.html" %} {% load static %} {% load i18n %} +{% load ui_text %} {% block title %} {% translate "Upload your documents" %} {% endblock title %} @@ -35,10 +36,17 @@

{% translate "Upload your court documents" %}

{% endif %}
- {% translate "PDF only" %} + {% translate "PDF or Word" %} {% translate "10 MB per file" %} {% translate "Text must be readable" %}
+

+ {% if flatten_pdf_forms %} + {% ui_text "upload_documents.preparation_help" %} + {% else %} + {% ui_text "upload_documents.preparation_help_unflattened" %} + {% endif %} +

@@ -47,10 +55,10 @@

{% translate "Upload your court documents" %}

- {% translate "Choose PDFs or drag them here" %} + {% translate "Choose PDFs or Word documents, or drag them here" %} {% translate "You can select more than one file." %}
{% translate "Your documents" %} {% translate "Remove" %} + {% include "efile/components/document_preview.html" %} {% endfor %} {% else %}
@@ -210,10 +219,10 @@

{% translate "Your documents" %}

goes. See efile.workflow.get_visible_workflow. {% endcomment %} {% translate "Back" %} - {% translate "Review what we found" %} + href="{% url 'preview_documents' jurisdiction %}" + {% if not has_lead_document %}aria-disabled="true" tabindex="-1"{% endif %}>{% translate "Preview your PDFs" %}
{% endblock workflow_content %} diff --git a/efile_app/efile/templates/efile/workflow_base.html b/efile_app/efile/templates/efile/workflow_base.html index 951ccd28..b70819b4 100644 --- a/efile_app/efile/templates/efile/workflow_base.html +++ b/efile_app/efile/templates/efile/workflow_base.html @@ -20,6 +20,8 @@ + + {% block extra_css %} {% endblock extra_css %} @@ -39,6 +41,11 @@ + {% block extra_js %} {% endblock extra_js %} diff --git a/efile_app/efile/tests/pdf_helpers.py b/efile_app/efile/tests/pdf_helpers.py new file mode 100644 index 00000000..28589099 --- /dev/null +++ b/efile_app/efile/tests/pdf_helpers.py @@ -0,0 +1,85 @@ +"""Small valid documents for upload tests, without optional PDF libraries.""" + +import io +import zipfile +from typing import Any, cast +from xml.sax.saxutils import escape + +from pypdf import PdfWriter +from pypdf.generic import ( + ArrayObject, + BooleanObject, + DecodedStreamObject, + DictionaryObject, + FloatObject, + NameObject, + NumberObject, + TextStringObject, +) + + +def pdf_bytes(text="Synthetic filing document", *, pages=1, form_value=None, missing_appearance=False): + writer = PdfWriter() + font = DictionaryObject( + { + NameObject("/Type"): NameObject("/Font"), + NameObject("/Subtype"): NameObject("/Type1"), + NameObject("/BaseFont"): NameObject("/Helvetica"), + } + ) + fonts = DictionaryObject({NameObject("/Helv"): writer._add_object(font)}) + for index in range(pages): + page = writer.add_blank_page(width=612, height=792) + page[NameObject("/Resources")] = DictionaryObject({NameObject("/Font"): fonts}) + stream = DecodedStreamObject() + escaped = text.replace("\\", "\\\\").replace("(", "\\(").replace(")", "\\)") + stream.set_data(f"BT /Helv 14 Tf 50 730 Td ({escaped} - page {index + 1}) Tj ET".encode("latin-1")) + page[NameObject("/Contents")] = writer._add_object(stream) + if form_value is not None: + widget = DictionaryObject( + { + NameObject("/Type"): NameObject("/Annot"), + NameObject("/Subtype"): NameObject("/Widget"), + NameObject("/FT"): NameObject("/Tx"), + NameObject("/T"): TextStringObject("answers"), + NameObject("/V"): TextStringObject(form_value), + NameObject("/Ff"): NumberObject(4096), + NameObject("/F"): NumberObject(4), + NameObject("/Rect"): ArrayObject([FloatObject(n) for n in [50, 500, 500, 650]]), + NameObject("/DA"): TextStringObject("/Helv 12 Tf 0 g"), + } + ) + ref = writer._add_object(widget) + writer.pages[0][NameObject("/Annots")] = ArrayObject([ref]) + writer._root_object[NameObject("/AcroForm")] = DictionaryObject( + { + NameObject("/Fields"): ArrayObject([ref]), + NameObject("/DR"): DictionaryObject({NameObject("/Font"): fonts}), + NameObject("/DA"): TextStringObject("/Helv 12 Tf 0 g"), + } + ) + writer.update_page_form_field_values(None, {"answers": form_value}, auto_regenerate=False) + if missing_appearance: + widget.pop("/AP", None) + cast(Any, writer._root_object["/AcroForm"])[NameObject("/NeedAppearances")] = BooleanObject(True) + output = io.BytesIO() + writer.write(output) + return output.getvalue() + + +def docx_bytes(text="Synthetic Word filing"): + output = io.BytesIO() + with zipfile.ZipFile(output, "w") as archive: + archive.writestr( + "[Content_Types].xml", + '', + ) + archive.writestr( + "_rels/.rels", + '', + ) + archive.writestr( + "word/document.xml", + f'{escape(text)}/s/ Alex ExampleSecond page exhibit. José García.', + ) + return output.getvalue() diff --git a/efile_app/efile/tests/test_ai_opt_out.py b/efile_app/efile/tests/test_ai_opt_out.py index 83789795..bef03123 100644 --- a/efile_app/efile/tests/test_ai_opt_out.py +++ b/efile_app/efile/tests/test_ai_opt_out.py @@ -24,6 +24,7 @@ ) from efile.services.drafts import create_draft from efile.services.taxonomy_classification import HierarchicalDocumentClassifier +from efile.tests.pdf_helpers import pdf_bytes SYNTHETIC_PDFS = Path(__file__).resolve().parents[3] / "benchmarking/synthetic/filled_pdfs/flattened" @@ -156,7 +157,7 @@ def test_uploading_saves_the_choice_before_analysis_is_queued(client, opted_out_ handler.validate_file.return_value = {"valid": True} handler.upload_file.return_value = {"success": True, "key": "lead.pdf"} handler.get_public_url.return_value = "https://example.com/lead.pdf" - lead = SimpleUploadedFile("complaint.pdf", b"%PDF lead", content_type="application/pdf") + lead = SimpleUploadedFile("complaint.pdf", pdf_bytes(), content_type="application/pdf") with patch("efile.services.document_uploads.S3UploadHandler", return_value=handler): response = client.post(upload_url(), {"documents": [lead], "ai_opt_out": "yes"}) diff --git a/efile_app/efile/tests/test_document_extractions.py b/efile_app/efile/tests/test_document_extractions.py index c4ca7340..df208579 100644 --- a/efile_app/efile/tests/test_document_extractions.py +++ b/efile_app/efile/tests/test_document_extractions.py @@ -232,7 +232,7 @@ def test_status_endpoint_reports_when_review_is_ready(client, extraction_draft): "ready": True, "pages_analyzed": 20, "total_pages": 30, - "review_url": reverse("extraction_review", kwargs={"jurisdiction": "illinois"}), + "review_url": reverse("preview_documents", kwargs={"jurisdiction": "illinois"}), } diff --git a/efile_app/efile/tests/test_document_preparation.py b/efile_app/efile/tests/test_document_preparation.py new file mode 100644 index 00000000..1f0708ee --- /dev/null +++ b/efile_app/efile/tests/test_document_preparation.py @@ -0,0 +1,145 @@ +import io +from typing import Any, cast +from unittest.mock import MagicMock, patch + +import pytest +import requests +from django.core.files.uploadedfile import SimpleUploadedFile +from django.test import override_settings +from pypdf import PdfReader + +from efile.services.document_preparation import PreparationError, prepare_document, store_prepared_document +from efile.tests.pdf_helpers import docx_bytes, pdf_bytes + + +def upload(data, name="form.pdf"): + return SimpleUploadedFile(name, data) + + +def service_response(data, status=200): + response = MagicMock() + response.__enter__.return_value = response + response.status_code = status + response.iter_content.return_value = [data] + return response + + +def test_static_pdf_is_byte_identical_even_when_service_is_unavailable(): + content = pdf_bytes(pages=2) + with override_settings(GOTENBERG_URL=""), patch("efile.services.document_preparation.requests.post") as post: + prepared = prepare_document(upload(content), "vermont") + assert prepared.content == content + assert prepared.operation == "unchanged" + post.assert_not_called() + + +def test_missing_multiline_appearances_are_repaired_before_gotenberg(): + content = pdf_bytes(form_value="First line\nSecond line\nThird line", missing_appearance=True) + output = pdf_bytes("First line Second line Third line") + with patch("efile.services.document_preparation.requests.post", return_value=service_response(output)) as post: + prepared = prepare_document(upload(content), "vermont") + input_bytes = post.call_args.kwargs["files"]["files"][1] + reader = PdfReader(io.BytesIO(input_bytes)) + form = cast(Any, reader.trailer["/Root"])["/AcroForm"] + assert form["/NeedAppearances"].value is False + appearance = cast(Any, reader.pages[0]["/Annots"])[0].get_object()["/AP"]["/N"].get_data() + assert b"First line" in appearance and b"Second line" in appearance and b"Third line" in appearance + assert prepared.operation == "flattened" + assert prepared.content == output + + +def test_existing_appearances_are_preserved(): + content = pdf_bytes(form_value="First line\nSecond line") + with patch( + "efile.services.document_preparation.requests.post", + return_value=service_response(pdf_bytes("First line Second line")), + ) as post: + prepare_document(upload(content), "vermont") + assert post.call_args.kwargs["files"]["files"][1] == content + + +@pytest.mark.parametrize( + "output", + [ + pdf_bytes("Only first line"), + pdf_bytes("First line Second line", pages=2), + pdf_bytes(form_value="First line\nSecond line"), + b"bad gateway", + ], +) +def test_invalid_or_lossy_service_output_is_never_accepted(output): + with patch("efile.services.document_preparation.requests.post", return_value=service_response(output)): + with pytest.raises(PreparationError): + prepare_document(upload(pdf_bytes(form_value="First line\nSecond line")), "vermont") + + +def test_word_uses_tagged_lossless_pdf_and_keeps_original(): + handler = MagicMock() + handler.upload_file.side_effect = [ + {"success": True, "key": "original.docx"}, + {"success": True, "key": "filing.pdf"}, + ] + handler.get_public_url.return_value = "https://storage.example/filing.pdf" + keys = [] + original = docx_bytes() + with patch( + "efile.services.document_preparation.requests.post", return_value=service_response(pdf_bytes(pages=2)) + ) as post: + result = store_prepared_document(handler, upload(original, "brief.DOCX"), "vermont", "lead", keys=keys) + assert result["original_filename"] == "brief.DOCX" + assert result["name"] == "brief.pdf" + assert result["s3_key"] == "filing.pdf" + assert result["original_s3_key"] == "original.docx" + assert result["preparation_reviewed_at"] is None + assert result["content_type"] == "application/pdf" + assert keys == ["original.docx", "filing.pdf"] + assert post.call_args.args[0].endswith("/forms/libreoffice/convert") + assert post.call_args.kwargs["data"]["pdfua"] == "true" + assert post.call_args.kwargs["data"]["exportFormFields"] == "false" + assert post.call_args.kwargs["allow_redirects"] is False + + +@pytest.mark.parametrize( + "data,name", + [ + (b"fake PDF", "fake.pdf"), + (b"not a zip", "bad.docx"), + (b"not a Word file", "bad.doc"), + (b"", "empty.pdf"), + (b"html", "web.html"), + ], +) +def test_bad_uploads_fail_before_outbound_conversion(data, name): + with patch("efile.services.document_preparation.requests.post") as post: + with pytest.raises(PreparationError): + prepare_document(upload(data, name), "vermont") + post.assert_not_called() + + +@pytest.mark.parametrize("error", [requests.Timeout(), requests.ConnectionError()]) +def test_service_failure_has_retry_guidance_without_upstream_details(error): + with patch("efile.services.document_preparation.requests.post", side_effect=error): + with pytest.raises(PreparationError, match="Try again"): + prepare_document(upload(docx_bytes(), "brief.docx"), "vermont") + + +def test_jurisdiction_can_keep_interactive_pdf_byte_identical(): + content = pdf_bytes(form_value="Keep these fields") + with ( + patch( + "efile.services.document_preparation.config_loader.load_jurisdiction_config", + return_value={"document_preparation": {"flatten_pdf_forms": False}}, + ), + patch("efile.services.document_preparation.requests.post") as post, + ): + assert prepare_document(upload(content), "illinois").content == content + post.assert_not_called() + + +def test_oversize_prepared_output_is_rejected(): + with ( + override_settings(MAX_FILE_SIZE=4096), + patch("efile.services.document_preparation.requests.post", return_value=service_response(b"x" * 4097)), + ): + with pytest.raises(PreparationError, match="exceeds"): + prepare_document(upload(docx_bytes(), "brief.docx"), "vermont") diff --git a/efile_app/efile/tests/test_document_preparation_browser.py b/efile_app/efile/tests/test_document_preparation_browser.py new file mode 100644 index 00000000..2bfc00a0 --- /dev/null +++ b/efile_app/efile/tests/test_document_preparation_browser.py @@ -0,0 +1,116 @@ +"""Opt-in real Gotenberg + Chromium validation. Uses only synthetic documents. + +Run with DOCUMENT_PREPARATION_BROWSER_TESTS=1 uv run pytest -s ... +S3 is an in-memory test double; conversion and PDF.js are real. +""" + +import io +import json +import os +import subprocess +from pathlib import Path +from typing import Any, cast +from unittest.mock import MagicMock, patch + +import pytest +from django.conf import settings +from django.urls import reverse +from pypdf import PdfReader + +from efile.models import FilingDraft, FilingParty +from efile.tests.pdf_helpers import docx_bytes, pdf_bytes +from efile.tests.test_document_extractions import authorize + + +@pytest.mark.integration +@pytest.mark.django_db(transaction=True) +@pytest.mark.skipif( + not os.getenv("DOCUMENT_PREPARATION_BROWSER_TESTS"), reason="Opt-in: requires Gotenberg and Chromium" +) +def test_real_conversion_and_browser_previews(live_server, client, django_user_model, tmp_path): + user = django_user_model.objects.create_user(username="synthetic-preview", tyler_jurisdiction="vermont") + draft = FilingDraft.objects.create( + user=user, + jurisdiction="vermont", + workflow_version=2, + ai_assistance_opted_out=True, + court_code="synthetic-court", + case_type_code="synthetic-case", + ) + FilingParty.objects.create( + draft=draft, role="filer", is_filing_party=True, first_name="Jordan", last_name="Example" + ) + authorize(client, draft) + objects = {} + handler = MagicMock() + handler.bucket_name = "synthetic" + + def store(file, **kwargs): + key = f"{kwargs['file_type']}/{len(objects)}" + objects[key] = file.read() + return {"success": True, "key": key} + + handler.upload_file.side_effect = store + handler.get_public_url.side_effect = lambda key: f"https://synthetic.invalid/{key}" + handler.s3_client.get_object.side_effect = lambda **kw: {"Body": io.BytesIO(objects[kw["Key"]])} + paths = [] + for name, content in [ + ( + "multiline.pdf", + pdf_bytes(form_value="First line\nSecond line\nThird line", missing_appearance=True, pages=2), + ), + ("word-filing.docx", docx_bytes("Synthetic court filing from Word")), + ]: + file = tmp_path / name + file.write_bytes(content) + paths.append(str(file)) + evidence = Path(os.getenv("DOCUMENT_PREPARATION_EVIDENCE_DIR", str(tmp_path / "evidence"))) + config = tmp_path / "browser.json" + config.write_text( + json.dumps( + { + "baseUrl": live_server.url, + "cookie": client.cookies[settings.SESSION_COOKIE_NAME].value, + "files": paths, + "evidence": str(evidence), + **{ + key: reverse(view, kwargs={"jurisdiction": "vermont"}) + f"?draft={draft.pk}" + for key, view in [ + ("uploadUrl", "upload_documents"), + ("previewUrl", "preview_documents"), + ("organizeUrl", "organize_documents"), + ("paymentUrl", "payment"), + ] + }, + } + ) + ) + with ( + patch("efile.services.document_uploads.S3UploadHandler", return_value=handler), + patch("efile.views.document_previews.S3UploadHandler", return_value=handler), + patch("efile.services.document_uploads.queue_document_extraction"), + patch("efile.services.people.get_party_types", return_value=[]), + patch("efile.views.payment.estimate_fees", return_value={}), + ): + result = subprocess.run( + ["node", "tests/document-preparation-browser.js", str(config)], + cwd=settings.BASE_DIR, + capture_output=True, + text=True, + timeout=180, + ) + print(result.stdout) + assert result.returncode == 0, result.stdout + result.stderr + assert draft.documents.count() == 2 + for doc in draft.documents.all(): + assert doc.original_s3_key + assert doc.preparation_reviewed_at is not None + reader = PdfReader(io.BytesIO(objects[doc.s3_key])) + assert len(reader.pages) == 2 + assert not reader.get_fields() + text = " ".join(page.extract_text() for page in reader.pages) + if doc.preparation == "flattened": + assert all(line in text for line in ["First line", "Second line", "Third line"]) + else: + assert "Synthetic court filing from Word" in text + assert cast(Any, reader.trailer["/Root"]).get("/StructTreeRoot") diff --git a/efile_app/efile/tests/test_document_previews.py b/efile_app/efile/tests/test_document_previews.py new file mode 100644 index 00000000..882d2e5b --- /dev/null +++ b/efile_app/efile/tests/test_document_previews.py @@ -0,0 +1,160 @@ +import io +from unittest.mock import MagicMock, patch + +import pytest +from django.core.files.uploadedfile import SimpleUploadedFile +from django.urls import reverse +from django.utils import timezone + +from efile.models import FilingDocument, FilingDraft +from efile.services.document_previews import preview_fingerprint +from efile.services.document_uploads import upload_files +from efile.services.drafts import read_upload_data, write_upload_data +from efile.tests.pdf_helpers import pdf_bytes +from efile.tests.test_document_extractions import authorize + +pytestmark = pytest.mark.django_db + + +@pytest.fixture +def preview_draft(client, django_user_model): + user = django_user_model.objects.create_user(username="preview-filer", tyler_jurisdiction="vermont") + draft = FilingDraft.objects.create(user=user, jurisdiction="vermont", workflow_version=2) + authorize(client, draft) + FilingDocument.objects.create( + draft=draft, + role="lead", + name="filing.pdf", + s3_key="private/filing.pdf", + original_s3_key="private/original.docx", + original_filename="original.docx", + preparation="converted", + ) + return draft + + +def url(name, draft, **kwargs): + return reverse(name, kwargs={"jurisdiction": draft.jurisdiction, **kwargs}) + f"?draft={draft.pk}" + + +def approve(client, draft, **overrides): + documents = list(draft.documents.all()) + return client.post( + url("preview_documents", draft), + { + "preview_fingerprint": preview_fingerprint(documents), + "reviewed_document": [str(doc.pk) for doc in documents], + **overrides, + }, + ) + + +def test_review_is_required_and_requires_each_current_document(client, preview_draft): + doc = preview_draft.documents.get() + redirect = client.get(url("extraction_review", preview_draft)) + assert redirect.status_code == 302 and "preview-documents" in redirect.url + assert approve(client, preview_draft, reviewed_document=[]).status_code == 200 + doc.refresh_from_db() + assert doc.preparation_reviewed_at is None + assert approve(client, preview_draft, preview_fingerprint="stale").status_code == 200 + assert approve(client, preview_draft).status_code == 302 + doc.refresh_from_db() + assert doc.preparation_reviewed_at is not None + assert client.get(url("extraction_review", preview_draft)).status_code == 200 + + +def test_original_and_filing_bytes_are_separate_and_not_cached(client, preview_draft): + doc = preview_draft.documents.get() + content = pdf_bytes() + handler = MagicMock() + handler.bucket_name = "private" + handler.s3_client.get_object.side_effect = lambda **kw: { + "Body": io.BytesIO(content if kw["Key"] == doc.s3_key else b"original Word bytes") + } + with patch("efile.views.document_previews.S3UploadHandler", return_value=handler): + filing = client.get(url("document_content", preview_draft, document_id=doc.pk)) + assert filing.status_code == 200 + assert b"".join(filing.streaming_content) == content + assert filing["Content-Type"] == "application/pdf" + assert filing["Cache-Control"] == "private, no-store" + original = client.get(url("document_content", preview_draft, document_id=doc.pk) + "&original=1") + assert b"".join(original.streaming_content) == b"original Word bytes" + assert original["Content-Disposition"].startswith("attachment;") + + +def test_preview_cannot_read_another_users_or_another_drafts_document(client, preview_draft, django_user_model): + doc = preview_draft.documents.get() + other = FilingDraft.objects.create(user=preview_draft.user, jurisdiction="vermont") + with patch("efile.views.document_previews.S3UploadHandler") as storage: + response = client.get(url("document_content", other, document_id=doc.pk)) + assert response.status_code == 404 + other_user = django_user_model.objects.create_user(username="other-filer", tyler_jurisdiction="vermont") + client.force_login(other_user) + assert client.get(url("document_content", preview_draft, document_id=doc.pk)).status_code in {403, 409} + storage.assert_not_called() + + +def test_previews_require_sign_in(client, preview_draft): + doc = preview_draft.documents.get() + client.logout() + response = client.get(reverse("document_content", kwargs={"jurisdiction": "vermont", "document_id": doc.pk})) + assert response.status_code == 401 + + +def test_submission_cannot_bypass_preview(client, preview_draft): + with patch("efile.views.submission.forward_final_filing") as forward: + response = client.post(reverse("submit_final_filing"), data="{}", content_type="application/json") + assert response.status_code == 412 + assert "Preview" in response.json()["error"] + forward.assert_not_called() + preview_draft.refresh_from_db() + assert preview_draft.status == FilingDraft.Status.DRAFT + + +def test_later_uploads_are_all_or_nothing_and_leave_existing_approvals(preview_draft): + doc = preview_draft.documents.get() + doc.preparation_reviewed_at = timezone.now() + doc.save() + handler = MagicMock() + handler.upload_file.return_value = {"success": True, "key": "new.pdf"} + handler.get_public_url.return_value = "https://s3.example/new.pdf" + with patch("efile.services.document_uploads.S3UploadHandler", return_value=handler): + with pytest.raises(ValueError, match="could not be read"): + upload_files( + preview_draft, + [SimpleUploadedFile("valid.pdf", pdf_bytes()), SimpleUploadedFile("bad.pdf", b"invalid")], + "vermont", + ) + assert preview_draft.documents.count() == 1 + doc.refresh_from_db() + assert doc.preparation_reviewed_at is not None + handler.delete_file.assert_called_once_with("new.pdf") + + +def test_original_and_approval_survive_supporting_row_rebuilds(preview_draft): + supporting = FilingDocument.objects.create( + draft=preview_draft, + role="supporting", + name="support.pdf", + original_filename="support.docx", + s3_key="support.pdf", + original_s3_key="support.docx", + preparation="converted", + preparation_reviewed_at=timezone.now(), + ) + wire = read_upload_data(preview_draft) + # Browser-supplied preparation approval and original key must be ignored. + wire["files"]["supporting"][0].update(preparation_reviewed_at="forged", original_s3_key="forged") + write_upload_data(preview_draft, wire) + saved = preview_draft.documents.get(role="supporting") + assert saved.original_s3_key == "support.docx" + assert saved.original_filename == "support.docx" + assert saved.preparation_reviewed_at == supporting.preparation_reviewed_at + + +def test_preview_return_destinations_are_restricted(client, preview_draft): + response = approve(client, preview_draft, return_to="https://attacker.example") + assert "extraction-review" in response.url + assert "attacker" not in response.url + response = approve(client, preview_draft, return_to="payment") + assert "/payment/" in response.url diff --git a/efile_app/efile/tests/test_end_to_end_new_flow_states.py b/efile_app/efile/tests/test_end_to_end_new_flow_states.py index 58193b5a..59cc1d7d 100644 --- a/efile_app/efile/tests/test_end_to_end_new_flow_states.py +++ b/efile_app/efile/tests/test_end_to_end_new_flow_states.py @@ -230,6 +230,19 @@ def fake_download(_key, destination): assert status_resp2.json()["ready"] is True assert status_resp2.json()["status"] == "complete" + # The uploaded filing copies must be checked before extracted case details. + preview_url = reverse("preview_documents", kwargs={"jurisdiction": jurisdiction}) + preview = client.get(preview_url) + assert preview.status_code == 200 + approved = client.post( + preview_url, + { + "preview_fingerprint": preview.context["preview_fingerprint"], + "reviewed_document": [str(doc.pk) for doc in draft.documents.all()], + }, + ) + assert approved.status_code == 302 + # 3. Step: extraction-review review_page = client.get(reverse("extraction_review", kwargs={"jurisdiction": jurisdiction})) assert review_page.status_code == 200 diff --git a/efile_app/efile/tests/test_handoff.py b/efile_app/efile/tests/test_handoff.py index 884e8250..ea36fe72 100644 --- a/efile_app/efile/tests/test_handoff.py +++ b/efile_app/efile/tests/test_handoff.py @@ -8,9 +8,10 @@ from efile.models import FilingDraft, InterviewHandoff from efile.services.handoff import HandoffError, create_correction, effective_hints, resolve_metadata, unique_match +from efile.tests.pdf_helpers import pdf_bytes pytestmark = pytest.mark.django_db -PDF = b"%PDF-1.4\nsynthetic test document" +PDF = pdf_bytes() @pytest.fixture @@ -75,6 +76,7 @@ def storage(): "filename": "complaint.pdf", "size": len(PDF), } + mocked.return_value.get_public_url.return_value = "https://s3.example/signed" yield mocked.return_value diff --git a/efile_app/efile/tests/test_reorganized_start.py b/efile_app/efile/tests/test_reorganized_start.py index 331e0dc7..a2b81ac3 100644 --- a/efile_app/efile/tests/test_reorganized_start.py +++ b/efile_app/efile/tests/test_reorganized_start.py @@ -7,6 +7,7 @@ from efile.models import DocumentExtraction, FilingDocument, FilingDraft from efile.services.current_drafts import CURRENT_DRAFT_SESSION_KEY +from efile.tests.pdf_helpers import pdf_bytes from efile.workflow import ExistingCase, WorkflowStepKey @@ -51,8 +52,8 @@ def test_upload_documents_persists_files_and_queues_analysis(client, reorganized {"success": True, "key": "supporting.pdf"}, ] handler.get_public_url.side_effect = ["https://example.com/lead.pdf", "https://example.com/supporting.pdf"] - lead = SimpleUploadedFile("petition.pdf", b"%PDF lead", content_type="application/pdf") - supporting = SimpleUploadedFile("exhibit.pdf", b"%PDF exhibit", content_type="application/pdf") + lead = SimpleUploadedFile("petition.pdf", pdf_bytes(), content_type="application/pdf") + supporting = SimpleUploadedFile("exhibit.pdf", pdf_bytes(), content_type="application/pdf") with patch("efile.services.document_uploads.S3UploadHandler", return_value=handler): response = client.post( diff --git a/efile_app/efile/tests/test_waiver_documents.py b/efile_app/efile/tests/test_waiver_documents.py index 67bedeb2..202df2eb 100644 --- a/efile_app/efile/tests/test_waiver_documents.py +++ b/efile_app/efile/tests/test_waiver_documents.py @@ -8,6 +8,7 @@ from efile.models import FilingDocument from efile.services.fee_quotes import fee_inputs_token from efile.services.waiver_documents import WAIVER_TYPE, has_waiver_document, waiver_filing_types +from efile.tests.pdf_helpers import pdf_bytes from efile.tests.test_review_submit_flow import submission_draft as _submission_draft payment_draft = _submission_draft @@ -101,7 +102,7 @@ def upload_data(draft, **changes): "filing_type": "waiver", "document_type": "private", "fee_inputs_token": fee_inputs_token(draft), - "document": SimpleUploadedFile("waiver.pdf", b"%PDF-test", content_type="application/pdf"), + "document": SimpleUploadedFile("waiver.pdf", pdf_bytes(), content_type="application/pdf"), **changes, } @@ -229,6 +230,17 @@ def test_payment_upload_stays_with_displayed_draft_through_review(client, paymen result = client.post(payment_url, {"selected_payment_account": "wv"}) assert result.status_code == 302 assert f"draft={payment_draft.pk}" in result.url + preview = client.get(uploaded.json()["preview_url"] + f"&draft={payment_draft.pk}") + assert preview.status_code == 200 + approved = client.post( + uploaded.json()["preview_url"] + f"&draft={payment_draft.pk}", + { + "preview_fingerprint": preview.context["preview_fingerprint"], + "reviewed_document": [str(doc.pk) for doc in payment_draft.documents.all()], + "return_to": "review", + }, + ) + assert approved.status_code == 302 with patch("efile.views.review.get_case_questions", return_value=[]): review = client.get(result.url) assert review.status_code == 200 @@ -260,7 +272,7 @@ def test_general_upload_preserves_identical_names_and_separate_storage_keys(paym {"success": True, "key": "first-copy.pdf"}, {"success": True, "key": "second-copy.pdf"}, ] - files = [SimpleUploadedFile("appearance.pdf", b"%PDF-test", content_type="application/pdf") for _ in range(2)] + files = [SimpleUploadedFile("appearance.pdf", pdf_bytes(), content_type="application/pdf") for _ in range(2)] with patch("efile.services.document_uploads.S3UploadHandler", return_value=storage): upload_files(payment_draft, files, "illinois") assert payment_draft.documents.count() == 3 diff --git a/efile_app/efile/tests/test_workflow.py b/efile_app/efile/tests/test_workflow.py index 3bd3dc73..a734e512 100644 --- a/efile_app/efile/tests/test_workflow.py +++ b/efile_app/efile/tests/test_workflow.py @@ -49,6 +49,7 @@ def test_target_workflow_declares_every_reorganized_screen(): WorkflowStepKey.OPTIONS, WorkflowStepKey.FILING_PATH, WorkflowStepKey.UPLOAD_DOCUMENTS, + WorkflowStepKey.PREVIEW_DOCUMENTS, WorkflowStepKey.EXTRACTION_REVIEW, WorkflowStepKey.CASE_LOOKUP, WorkflowStepKey.CASE_CONFIRMATION, diff --git a/efile_app/efile/urls.py b/efile_app/efile/urls.py index a2879c7e..38af9f1c 100644 --- a/efile_app/efile/urls.py +++ b/efile_app/efile/urls.py @@ -13,6 +13,7 @@ from .views.choose_jurisdiction import change_jurisdiction, choose_jurisdiction from .views.confirmation import filing_confirmation from .views.document_checklist import document_checklist +from .views.document_previews import document_content, preview_documents from .views.draft_views import get_current_draft_view, start_filing, start_filing_from_plan from .views.extraction_review import extraction_review from .views.filing_path import filing_path @@ -94,6 +95,8 @@ def jurisdiction_homepage(request, jurisdiction): path("jurisdiction//filing-path/", filing_path, name="filing_path"), path("jurisdiction//waiver-documents/", waiver_documents, name="waiver_documents"), path("jurisdiction//upload-documents/", upload_documents, name="upload_documents"), + path("jurisdiction//preview-documents/", preview_documents, name="preview_documents"), + path("jurisdiction//documents//content/", document_content, name="document_content"), path( "jurisdiction//document-extraction-status/", document_extraction_status, diff --git a/efile_app/efile/utils/ui_text.py b/efile_app/efile/utils/ui_text.py index 48f750f9..01e57834 100644 --- a/efile_app/efile/utils/ui_text.py +++ b/efile_app/efile/utils/ui_text.py @@ -91,6 +91,16 @@ class UIString: UI_STRINGS: dict[str, UIString] = { + "upload_documents.preparation_help": UIString( + default="You do not need to lock PDF form fields yourself. LITEFile makes a filing copy with the filled-in answers locked in place (sometimes called flattening), and converts Word documents to PDF. Check the filing PDFs on the next page before continuing.", + description="Preparation guidance when flatten_pdf_forms is enabled. May link to state-specific requirements.", + links=True, + ), + "upload_documents.preparation_help_unflattened": UIString( + default="LITEFile converts Word documents to PDF. For this jurisdiction, PDF form fields are kept as uploaded. Check the filing PDFs on the next page before continuing.", + description="Preparation guidance when flatten_pdf_forms is disabled.", + links=True, + ), # -- Terms --------------------------------------------------------------- # Short nouns. Keep them lowercase and in the middle of a sentence: they are # interpolated into the passages below, which capitalize for themselves. diff --git a/efile_app/efile/views/document_previews.py b/efile_app/efile/views/document_previews.py new file mode 100644 index 00000000..a4d7b3d4 --- /dev/null +++ b/efile_app/efile/views/document_previews.py @@ -0,0 +1,100 @@ +import io +import logging + +from botocore.exceptions import BotoCoreError, ClientError +from django.conf import settings +from django.db import transaction +from django.http import FileResponse, Http404, HttpResponse, HttpResponseBase +from django.shortcuts import redirect, render +from django.utils import timezone +from django.views.decorators.http import require_http_methods + +from efile.api.suffolk_api_views import get_tyler_token +from efile.models import FilingDocument, FilingDraft +from efile.services.current_drafts import ensure_current_draft, get_current_draft +from efile.services.document_previews import preview_fingerprint +from efile.utils.s3_upload_handler import S3UploadHandler +from efile.workflow import WorkflowStepKey, get_workflow_context + +logger = logging.getLogger(__name__) + + +@require_http_methods(["GET", "POST"]) +def preview_documents(request, jurisdiction): + if not request.user.is_authenticated or not get_tyler_token(request, jurisdiction): + return redirect("efile_login", jurisdiction=jurisdiction) + draft = ensure_current_draft(request, jurisdiction, current_step=WorkflowStepKey.PREVIEW_DOCUMENTS) + if not FilingDocument.objects.filter(draft=draft).exists(): + return redirect("upload_documents", jurisdiction=jurisdiction) + # These destinations are server-defined, including documents added later + # from the checklist or fees screen. Never use an arbitrary return URL. + destinations = {"review": "case_review", "payment": "payment", "document_checklist": "document_checklist"} + return_to = request.POST.get("return_to") or request.GET.get("return_to", "") + error = "" + if request.method == "POST": + with transaction.atomic(): + draft = FilingDraft.objects.select_for_update().get(pk=draft.pk) + documents = list(FilingDocument.objects.filter(draft=draft).order_by("role", "sort_order", "pk")) + acknowledged = set(request.POST.getlist("reviewed_document")) + if draft.status not in {FilingDraft.Status.DRAFT, FilingDraft.Status.ERROR}: + return HttpResponse("This filing is no longer available to edit.", status=409) + if request.POST.get("preview_fingerprint") != preview_fingerprint(documents): + error = "Your documents changed. Preview the current copies before continuing." + elif any(str(doc.pk) not in acknowledged for doc in documents): + error = "Confirm that you checked each PDF before continuing." + else: + FilingDocument.objects.filter(draft=draft).update(preparation_reviewed_at=timezone.now()) + return redirect(destinations.get(return_to, "extraction_review"), jurisdiction=jurisdiction) + documents = list(FilingDocument.objects.filter(draft=draft).order_by("role", "sort_order", "pk")) + context = { + "documents": documents, + "preview_fingerprint": preview_fingerprint(documents), + "preview_error": error, + "return_to": return_to, + "is_logged_in": True, + } + context.update(get_workflow_context(WorkflowStepKey.PREVIEW_DOCUMENTS, jurisdiction, draft)) + return render(request, "efile/preview_documents.html", context) + + +@require_http_methods(["GET"]) +def document_content(request, jurisdiction, document_id) -> HttpResponseBase: + """Serve current private bytes to PDF.js without exposing or trusting URLs.""" + if not request.user.is_authenticated or not get_tyler_token(request, jurisdiction): + return HttpResponse("Sign in again to view this document.", status=401) + draft = get_current_draft(request, jurisdiction=jurisdiction, resume_latest=False) + document = FilingDocument.objects.filter(draft=draft, pk=document_id).first() if draft else None + if document is None: + raise Http404 + original = request.GET.get("original") == "1" + key = (document.original_s3_key or document.s3_key) if original else document.s3_key + if not key: + raise Http404 + handler = S3UploadHandler() + if not handler._ensure_initialized() or handler.s3_client is None: + return HttpResponse("Document storage is unavailable. Please try again.", status=503) + try: + result = handler.s3_client.get_object(Bucket=handler.bucket_name, Key=key) + body = result["Body"] + try: + content = body.read(settings.MAX_FILE_SIZE + 1) + finally: + body.close() + if len(content) > settings.MAX_FILE_SIZE: + return HttpResponse("This document is too large to preview.", status=413) + except (BotoCoreError, ClientError): + logger.warning("Could not load document %s for preview", document.pk) + return HttpResponse("We could not load this document. Please try again.", status=503) + filename = document.original_filename if original else (document.name or document.original_filename) + is_pdf = content.startswith(b"%PDF-") + if not original and not is_pdf: + return HttpResponse("This document is not a readable PDF. Replace it before continuing.", status=422) + response = FileResponse( + io.BytesIO(content), + as_attachment=original or request.GET.get("download") == "1", + filename=filename, + content_type="application/pdf" if is_pdf else "application/octet-stream", + ) + response["Cache-Control"] = "private, no-store" + response["X-Content-Type-Options"] = "nosniff" + return response diff --git a/efile_app/efile/views/extraction_review.py b/efile_app/efile/views/extraction_review.py index 66733dfe..f6043994 100644 --- a/efile_app/efile/views/extraction_review.py +++ b/efile_app/efile/views/extraction_review.py @@ -8,6 +8,7 @@ from efile.services.current_drafts import ensure_current_draft from efile.services.document_checklists import resolve_filer_roles from efile.services.document_extractions import extraction_for_document +from efile.services.document_previews import unreviewed_documents from efile.services.drafts import draft_snapshot, write_case_data from efile.services.extracted_parties import review_rows, save_reviewed_parties from efile.services.extraction_fields import display_extracted_fields, document_summary_details @@ -108,6 +109,9 @@ def extraction_review(request, jurisdiction): messages.error(request, "Upload at least one document before reviewing the filing.") return redirect("upload_documents", jurisdiction=jurisdiction) + if unreviewed_documents(draft).exists(): + return redirect("preview_documents", jurisdiction=jurisdiction) + lead = FilingDocument.objects.filter(draft=draft, role=FilingDocument.Role.LEAD).first() extraction = extraction_for_document(lead) if lead else None if extraction is not None and extraction.status in { @@ -274,6 +278,7 @@ def classified(level, key): ) context = { "is_logged_in": True, + "lead_document": lead, "filing_draft": draft_snapshot(draft), "has_guesses": needs_acknowledgement, "document_summary_details": summary_details, diff --git a/efile_app/efile/views/handoff.py b/efile_app/efile/views/handoff.py index 89bca63e..0ffdc94e 100644 --- a/efile_app/efile/views/handoff.py +++ b/efile_app/efile/views/handoff.py @@ -17,6 +17,7 @@ from efile.api.suffolk_api_views import get_tyler_token from efile.models import FilingDraft, HandoffDocumentUpdate, InterviewHandoff from efile.services.current_drafts import attach_current_draft +from efile.services.document_preparation import store_prepared_document from efile.services.draft_urls import draft_url from efile.services.fee_quotes import invalidate_fee_quote from efile.services.filings import describe_filing_detail, fetch_filing_detail @@ -78,12 +79,23 @@ def _response(request, receipt, *, created=False): def _upload(payload, files, handler, keys): uploads = {} for document in payload.get("documents", []): - result = handler.upload_file( - files[document["id"]], file_type=document["role"], metadata={"sha256": document["sha256"]} - ) - if not result.get("success"): - raise HandoffError("Document storage is unavailable. Retry the same handoff.", status=503) - keys.append(result["key"]) + try: + prepared = store_prepared_document( + handler, + files[document["id"]], + payload["jurisdiction"], + document["role"], + keys=keys, + metadata={"sha256": document["sha256"]}, + ) + except ValueError as error: + raise HandoffError(str(error), status=503) from error + result = { + **prepared, + "key": prepared["s3_key"], + "url": prepared["public_url"], + "filename": prepared["original_filename"], + } uploads[document["id"]] = result return uploads @@ -390,7 +402,21 @@ def replace_documents(request): row.public_url = uploaded["url"] row.original_filename = uploaded["filename"] row.size = uploaded["size"] - row.save(update_fields=["s3_key", "public_url", "original_filename", "size", "updated_at"]) + row.original_s3_key = uploaded["original_s3_key"] + row.preparation = uploaded["preparation"] + row.preparation_reviewed_at = None + row.save( + update_fields=[ + "s3_key", + "public_url", + "original_filename", + "size", + "original_s3_key", + "preparation", + "preparation_reviewed_at", + "updated_at", + ] + ) record( draft, f"documents.{row.pk}", diff --git a/efile_app/efile/views/payment.py b/efile_app/efile/views/payment.py index 767b103b..85da4c36 100644 --- a/efile_app/efile/views/payment.py +++ b/efile_app/efile/views/payment.py @@ -75,6 +75,7 @@ def efile_payment(request, jurisdiction): return redirect(get_step_url(WorkflowStepKey.REVIEW, jurisdiction)) context = { + "documents": FilingDocument.objects.filter(draft=draft).order_by("role", "sort_order", "pk"), "waiver_upload_url": draft_url(reverse("waiver_documents", kwargs={"jurisdiction": jurisdiction}), draft.pk), "has_waiver_document": has_waiver_document(draft), "is_logged_in": True, diff --git a/efile_app/efile/views/review.py b/efile_app/efile/views/review.py index 88a69264..c2ad5a7e 100644 --- a/efile_app/efile/views/review.py +++ b/efile_app/efile/views/review.py @@ -5,13 +5,14 @@ from efile.models import FilingDocument, FilingParty from efile.services.current_drafts import ensure_current_draft from efile.services.disclaimers import disclaimer_context +from efile.services.document_previews import unreviewed_documents from efile.services.drafts import draft_snapshot, read_case_data, read_upload_data from efile.services.extracted_parties import party_display_name from efile.services.fee_quotes import fee_inputs_token, fee_quote_summary from efile.services.filing_plans import documents_missing_from_envelope from efile.services.people import get_case_questions -from ..workflow import WorkflowStepKey, get_workflow_context +from ..workflow import WorkflowStepKey, get_step_url, get_workflow_context def _matches_extracted_value(current, extracted, exact=False): @@ -45,6 +46,9 @@ def case_review(request, jurisdiction): current_step=WorkflowStepKey.REVIEW, workflow_version=2, ) + if unreviewed_documents(draft).exists(): + return redirect(get_step_url(WorkflowStepKey.PREVIEW_DOCUMENTS, jurisdiction) + "?return_to=review") + if not draft.selected_payment_account_id: messages.error(request, "Choose a payment method before reviewing your filing.") return redirect("payment", jurisdiction=jurisdiction) diff --git a/efile_app/efile/views/submission.py b/efile_app/efile/views/submission.py index 3a69f51c..193b7e0e 100644 --- a/efile_app/efile/views/submission.py +++ b/efile_app/efile/views/submission.py @@ -10,6 +10,7 @@ from efile.models import FilingDraft from efile.services.current_drafts import clear_current_draft, get_current_draft from efile.services.disclaimers import validate_acceptance +from efile.services.document_previews import require_document_previews from efile.services.fee_quotes import fee_quote_is_usable from efile.services.filing_plans import mark_attached_items_filed from efile.services.submission_errors import PRE_SUBMIT_ERROR_CODES, SubmissionErrorCode @@ -26,6 +27,7 @@ _CLAIMABLE_STATUSES = (FilingDraft.Status.DRAFT,) +@transaction.atomic def _claim_for_submission(draft: FilingDraft, acceptance: dict) -> bool: """Atomically move a DRAFT into SUBMITTING, recording the accepted court requirements. @@ -33,6 +35,10 @@ def _claim_for_submission(draft: FilingDraft, acceptance: dict) -> bool: each forward to the external API. A draft already SUBMITTING or ERROR is not reclaimed here -- those are not safe to retry automatically. """ + locked = FilingDraft.objects.select_for_update().get(pk=draft.pk) + if locked.status not in _CLAIMABLE_STATUSES: + return False + require_document_previews(locked) claimed = FilingDraft.objects.filter(pk=draft.pk, status__in=_CLAIMABLE_STATUSES).update( status=FilingDraft.Status.SUBMITTING, disclaimer_acceptance=acceptance, @@ -97,6 +103,11 @@ def submit_final_filing(request): status=400, ) + try: + require_document_previews(draft) + except ValueError as error: + return JsonResponse({"success": False, "error": str(error)}, status=412) + # The filer agreed to a total on Review. If anything that prices the filing # changed since, that total is not the one they would be charged, so the # page has to show the new one before anything reaches the court. @@ -117,7 +128,11 @@ def submit_final_filing(request): return JsonResponse({"success": False, "error": str(error)}, status=400) # Claim the draft before forwarding so a concurrent request can't file twice. - if not _claim_for_submission(draft, acceptance): + try: + claimed = _claim_for_submission(draft, acceptance) + except ValueError as error: + return JsonResponse({"success": False, "error": str(error)}, status=412) + if not claimed: return JsonResponse( {"success": False, "error": "This filing can't be submitted again automatically."}, status=409, diff --git a/efile_app/efile/views/upload_documents.py b/efile_app/efile/views/upload_documents.py index a9eab70b..e6adaa7d 100644 --- a/efile_app/efile/views/upload_documents.py +++ b/efile_app/efile/views/upload_documents.py @@ -2,6 +2,7 @@ from django.conf import settings from django.db import transaction +from django.db.models import Q from django.http import JsonResponse from django.shortcuts import redirect, render from django.views.decorators.http import require_http_methods @@ -10,6 +11,8 @@ from efile.models import DocumentExtraction, FilingDocument, FilingDraft from efile.services.current_drafts import ensure_current_draft from efile.services.document_extractions import extraction_for_document, queue_document_extraction +from efile.services.document_preparation import requires_flattening +from efile.services.document_previews import document_storage_keys from efile.services.document_uploads import upload_files from efile.services.drafts import draft_snapshot, read_upload_data from efile.utils.config_loader import config_loader @@ -105,7 +108,7 @@ def upload_documents(request, jurisdiction): if document is None: return JsonResponse({"success": False, "error": "Document not found."}, status=404) removed_lead = document.role == FilingDocument.Role.LEAD - s3_key = document.s3_key + storage_keys = document_storage_keys(document) promote_document = None other_documents = FilingDocument.objects.filter(draft=draft).exclude(pk=document.pk) if document.role == FilingDocument.Role.LEAD: @@ -122,17 +125,22 @@ def upload_documents(request, jurisdiction): if removed_lead and draft.extracted_guesses: draft.extracted_guesses = {} draft.save(update_fields=["extracted_guesses", "updated_at"]) - if s3_key: + if storage_keys: handler = S3UploadHandler() if handler._ensure_initialized(): - deletion = handler.delete_file(s3_key) - if not deletion.get("success"): - logger.warning("Could not delete removed draft document %s from storage", s3_key) + for key in storage_keys: + if FilingDocument.objects.filter(Q(s3_key=key) | Q(original_s3_key=key)).exists(): + continue + deletion = handler.delete_file(key) + if not deletion.get("success"): + logger.warning("Could not delete removed draft document from storage") return JsonResponse({"success": True}) uploaded_files = request.FILES.getlist("documents") if not uploaded_files: - return JsonResponse({"success": False, "error": "Choose at least one PDF to upload."}, status=400) + return JsonResponse( + {"success": False, "error": "Choose at least one PDF or Word document to upload."}, status=400 + ) # Saved before the upload, because uploading the lead queues the # analysis that this choice decides the shape of. opted_out = _opted_out(request) @@ -148,7 +156,7 @@ def upload_documents(request, jurisdiction): return JsonResponse( { "success": True, - "redirect_url": get_step_url(WorkflowStepKey.EXTRACTION_REVIEW, jurisdiction), + "redirect_url": get_step_url(WorkflowStepKey.PREVIEW_DOCUMENTS, jurisdiction), "document_count": FilingDocument.objects.filter(draft=draft).count(), "extraction_pending": FilingDocument.objects.filter( draft=draft, @@ -175,6 +183,7 @@ def upload_documents(request, jurisdiction): "extraction_pending": extraction is not None and extraction.status in {DocumentExtraction.Status.PENDING, DocumentExtraction.Status.PROCESSING}, "max_extraction_pages": settings.DOCUMENT_EXTRACTION_MAX_PAGES, + "flatten_pdf_forms": requires_flattening(jurisdiction), "upload_data": upload_data, "ai_opted_out": draft.ai_assistance_opted_out, "account_ai_opted_out": request.user.ai_assistance_opted_out, @@ -211,6 +220,6 @@ def document_extraction_status(request, jurisdiction): "ready": extraction.status in {DocumentExtraction.Status.COMPLETE, DocumentExtraction.Status.FAILED}, "pages_analyzed": extraction.pages_analyzed, "total_pages": extraction.total_pages, - "review_url": get_step_url(WorkflowStepKey.EXTRACTION_REVIEW, jurisdiction), + "review_url": get_step_url(WorkflowStepKey.PREVIEW_DOCUMENTS, jurisdiction), } ) diff --git a/efile_app/efile/views/waiver_documents.py b/efile_app/efile/views/waiver_documents.py index b576d725..c4ce0b12 100644 --- a/efile_app/efile/views/waiver_documents.py +++ b/efile_app/efile/views/waiver_documents.py @@ -6,10 +6,12 @@ from efile.api.suffolk_api_views import get_tyler_token from efile.models import FilingDocument, FilingDraft from efile.services.current_drafts import explicit_draft_id, get_current_draft +from efile.services.document_preparation import cleanup_uploads, store_prepared_document from efile.services.drafts import ACTIVE_DRAFT_STATUSES from efile.services.fee_quotes import fee_inputs_token, invalidate_fee_quote from efile.services.waiver_documents import waiver_document_choices, waiver_filing_types from efile.utils.s3_upload_handler import S3UploadHandler +from efile.workflow import WorkflowStepKey, get_step_url @require_http_methods(["GET", "POST"]) @@ -21,6 +23,8 @@ def waiver_documents(request, jurisdiction): draft = get_current_draft(request, jurisdiction=jurisdiction) if draft is None or draft.status not in ACTIVE_DRAFT_STATUSES or not draft.court_code: return JsonResponse({"error": "This filing is not available to edit."}, status=409) + keys = [] + handler = S3UploadHandler() try: options = waiver_filing_types(draft) data = request.POST if request.method == "POST" else request.GET @@ -38,10 +42,9 @@ def waiver_documents(request, jurisdiction): raise ValueError("Choose a confidentiality setting for this document.") files = request.FILES.getlist("document") if len(files) != 1: - raise ValueError("Choose one PDF to upload.") + raise ValueError("Choose one PDF or Word document to upload.") file = files[0] - handler = S3UploadHandler() - validation = handler.validate_file(file, max_size_mb=10, allowed_types=[".pdf"]) + validation = handler.validate_file(file, max_size_mb=10, allowed_types=[".pdf", ".doc", ".docx"]) if not validation["valid"]: raise ValueError(validation["error"]) if not handler._ensure_initialized(): @@ -54,10 +57,7 @@ def waiver_documents(request, jurisdiction): ) if not draft.documents.filter(role=FilingDocument.Role.LEAD).exists(): raise ValueError("Add your main document before adding a fee waiver.") - file.seek(0) - uploaded = handler.upload_file(file, file_type=FilingDocument.Role.SUPPORTING) - if not uploaded.get("success"): - raise ValueError("The upload failed. Try again.") + prepared = store_prepared_document(handler, file, jurisdiction, FilingDocument.Role.SUPPORTING, keys=keys) highest = draft.documents.filter(role=FilingDocument.Role.SUPPORTING).aggregate(order=Max("sort_order"))[ "order" ] @@ -65,12 +65,7 @@ def waiver_documents(request, jurisdiction): draft=draft, role=FilingDocument.Role.SUPPORTING, sort_order=0 if highest is None else highest + 1, - name=file.name[:255], - original_filename=file.name[:255], - size=file.size, - content_type=file.content_type, - s3_key=uploaded["key"], - public_url=handler.get_public_url(uploaded["key"]), + **prepared, filing_type_code=code, filing_type_name=selected["name"], filing_requires_amount_in_controversy=str(selected.get("amountincontroversy", "")).casefold() @@ -81,6 +76,13 @@ def waiver_documents(request, jurisdiction): filing_component_name=component["name"], ) invalidate_fee_quote(draft) - return JsonResponse({"success": True, "fee_inputs_token": fee_inputs_token(draft)}) + return JsonResponse( + { + "success": True, + "fee_inputs_token": fee_inputs_token(draft), + "preview_url": get_step_url(WorkflowStepKey.PREVIEW_DOCUMENTS, jurisdiction) + "?return_to=payment", + } + ) except ValueError as error: + cleanup_uploads(handler, keys) return JsonResponse({"error": str(error)}, status=400) diff --git a/efile_app/efile/workflow.py b/efile_app/efile/workflow.py index f2000bba..e94324cc 100644 --- a/efile_app/efile/workflow.py +++ b/efile_app/efile/workflow.py @@ -84,6 +84,7 @@ class WorkflowStepKey(StrEnum): OPTIONS = "options" FILING_PATH = "filing_path" UPLOAD_DOCUMENTS = "upload_documents" + PREVIEW_DOCUMENTS = "preview_documents" EXTRACTION_REVIEW = "extraction_review" CASE_LOOKUP = "case_lookup" CASE_CONFIRMATION = "case_confirmation" @@ -118,6 +119,7 @@ class WorkflowStep: WorkflowStepKey.FILING_PATH, pgettext_lazy("workflow stage", "Start"), "filing_path", WorkflowStage.FILING ), WorkflowStep(WorkflowStepKey.UPLOAD_DOCUMENTS, _("Upload documents"), "upload_documents", WorkflowStage.UPLOAD), + WorkflowStep(WorkflowStepKey.PREVIEW_DOCUMENTS, _("Preview documents"), "preview_documents", WorkflowStage.UPLOAD), WorkflowStep( WorkflowStepKey.EXTRACTION_REVIEW, _("Confirm filing"), diff --git a/efile_app/eslint.config.mjs b/efile_app/eslint.config.mjs index be1a4f5b..17a53be8 100644 --- a/efile_app/eslint.config.mjs +++ b/efile_app/eslint.config.mjs @@ -15,7 +15,7 @@ const sharedGlobals = { export default [ { - ignores: [".venv/**", "node_modules/**", "playwright-report/**", "test-results/**"] + ignores: [".venv/**", "node_modules/**", "efile/static/vendor/**", "playwright-report/**", "test-results/**"] }, eslint.configs.recommended, { diff --git a/efile_app/package-lock.json b/efile_app/package-lock.json index e7cc2c3a..5649ca4c 100644 --- a/efile_app/package-lock.json +++ b/efile_app/package-lock.json @@ -10,7 +10,8 @@ "license": "ISC", "dependencies": { "@playwright/test": "^1.55.0", - "dotenv": "^17.2.1" + "dotenv": "^17.2.1", + "pdfjs-dist": "6.3.289" }, "devDependencies": { "@axe-core/playwright": "^4.13.0", @@ -486,6 +487,271 @@ "dev": true, "license": "MIT" }, + "node_modules/@napi-rs/canvas": { + "version": "1.0.9", + "resolved": "https://registry.npmjs.org/@napi-rs/canvas/-/canvas-1.0.9.tgz", + "integrity": "sha512-QviPdJImDi/jMAvBfqaw+19BndMd/sizXVW3NnpMd3VJGz++QXkOHcP9kWR/smHG0hNjHeyuHFyrx/5lD0oNcQ==", + "license": "MIT", + "optional": true, + "workspaces": [ + "e2e/*" + ], + "engines": { + "node": ">= 10" + }, + "funding": { + "type": "github", + "url": "https://github.com/sponsors/Brooooooklyn" + }, + "optionalDependencies": { + "@napi-rs/canvas-android-arm64": "1.0.9", + "@napi-rs/canvas-darwin-arm64": "1.0.9", + "@napi-rs/canvas-darwin-x64": "1.0.9", + "@napi-rs/canvas-linux-arm-gnueabihf": "1.0.9", + "@napi-rs/canvas-linux-arm64-gnu": "1.0.9", + "@napi-rs/canvas-linux-arm64-musl": "1.0.9", + "@napi-rs/canvas-linux-riscv64-gnu": "1.0.9", + "@napi-rs/canvas-linux-x64-gnu": "1.0.9", + "@napi-rs/canvas-linux-x64-musl": "1.0.9", + "@napi-rs/canvas-win32-arm64-msvc": "1.0.9", + "@napi-rs/canvas-win32-x64-msvc": "1.0.9" + } + }, + "node_modules/@napi-rs/canvas-android-arm64": { + "version": "1.0.9", + "resolved": "https://registry.npmjs.org/@napi-rs/canvas-android-arm64/-/canvas-android-arm64-1.0.9.tgz", + "integrity": "sha512-4LGXk2/0HVzE29K8SzML5WubgCp++B1FH3qgl35XmSZE+lLdr6P9VRQEnZ0MCLZMTSuJP41yyhtoiVNEuvrTIA==", + "cpu": [ + "arm64" + ], + "license": "MIT", + "optional": true, + "os": [ + "android" + ], + "engines": { + "node": ">= 10" + }, + "funding": { + "type": "github", + "url": "https://github.com/sponsors/Brooooooklyn" + } + }, + "node_modules/@napi-rs/canvas-darwin-arm64": { + "version": "1.0.9", + "resolved": "https://registry.npmjs.org/@napi-rs/canvas-darwin-arm64/-/canvas-darwin-arm64-1.0.9.tgz", + "integrity": "sha512-YNdfLBzY0W/Pep9fo2L6RmoNlNksnn05LRnX66W63R3ij58S25QOTcjdtEt2v8+PnCESzqZsYzUo+QPeIR44NA==", + "cpu": [ + "arm64" + ], + "license": "MIT", + "optional": true, + "os": [ + "darwin" + ], + "engines": { + "node": ">= 10" + }, + "funding": { + "type": "github", + "url": "https://github.com/sponsors/Brooooooklyn" + } + }, + "node_modules/@napi-rs/canvas-darwin-x64": { + "version": "1.0.9", + "resolved": "https://registry.npmjs.org/@napi-rs/canvas-darwin-x64/-/canvas-darwin-x64-1.0.9.tgz", + "integrity": "sha512-ceZQSknTEcy3dOXoekv59LTCkXjvnLsq+VW5PeNNDHEPQbRS5Ervkm1EaDa7WLAjiYWMLuSQTRTHao1dEX4prg==", + "cpu": [ + "x64" + ], + "license": "MIT", + "optional": true, + "os": [ + "darwin" + ], + "engines": { + "node": ">= 10" + }, + "funding": { + "type": "github", + "url": "https://github.com/sponsors/Brooooooklyn" + } + }, + "node_modules/@napi-rs/canvas-linux-arm-gnueabihf": { + "version": "1.0.9", + "resolved": "https://registry.npmjs.org/@napi-rs/canvas-linux-arm-gnueabihf/-/canvas-linux-arm-gnueabihf-1.0.9.tgz", + "integrity": "sha512-XhfI0Wwv4llhd6nnWDtY3kQKjq0r+y1i91PlJlJI24ag2U9WrnwbG1qS3+fDLEyouwEVFchiKkTEHozK+5iUNA==", + "cpu": [ + "arm" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">= 10" + }, + "funding": { + "type": "github", + "url": "https://github.com/sponsors/Brooooooklyn" + } + }, + "node_modules/@napi-rs/canvas-linux-arm64-gnu": { + "version": "1.0.9", + "resolved": "https://registry.npmjs.org/@napi-rs/canvas-linux-arm64-gnu/-/canvas-linux-arm64-gnu-1.0.9.tgz", + "integrity": "sha512-012oiYtKaE7i9oxc8q7nraT7kDOpLcaCmFLzVe9Ty34RHDdoDzbWLrVh827CNxYh/EADX1eSikA3ymLjo/nNuw==", + "cpu": [ + "arm64" + ], + "libc": [ + "glibc" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">= 10" + }, + "funding": { + "type": "github", + "url": "https://github.com/sponsors/Brooooooklyn" + } + }, + "node_modules/@napi-rs/canvas-linux-arm64-musl": { + "version": "1.0.9", + "resolved": "https://registry.npmjs.org/@napi-rs/canvas-linux-arm64-musl/-/canvas-linux-arm64-musl-1.0.9.tgz", + "integrity": "sha512-Ls5UWYFFn63casTZEczbeyEg3vRDRkv9lscuGwfchtY5yLLQhgOB8SN4YGxmfJ5vTaBwZ2YUxBS3NtmjJmFXdA==", + "cpu": [ + "arm64" + ], + "libc": [ + "musl" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">= 10" + }, + "funding": { + "type": "github", + "url": "https://github.com/sponsors/Brooooooklyn" + } + }, + "node_modules/@napi-rs/canvas-linux-riscv64-gnu": { + "version": "1.0.9", + "resolved": "https://registry.npmjs.org/@napi-rs/canvas-linux-riscv64-gnu/-/canvas-linux-riscv64-gnu-1.0.9.tgz", + "integrity": "sha512-hLKEGxV7ZiRHqndePTokgDMdBlo/rDfzg7P4p4QIv9pUhuYobnu3R2NIFLCRghG0nwfo+s2sw+c1xZFeCmEAsw==", + "cpu": [ + "riscv64" + ], + "libc": [ + "glibc" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">= 10" + }, + "funding": { + "type": "github", + "url": "https://github.com/sponsors/Brooooooklyn" + } + }, + "node_modules/@napi-rs/canvas-linux-x64-gnu": { + "version": "1.0.9", + "resolved": "https://registry.npmjs.org/@napi-rs/canvas-linux-x64-gnu/-/canvas-linux-x64-gnu-1.0.9.tgz", + "integrity": "sha512-6kaz3w0QMy77PDWk6rJ1ksIihdad3qzEyX2o2oGT8GwCaypfT5mhjr8buOO5hstyLxcWXDScuz56RsINLtBPIQ==", + "cpu": [ + "x64" + ], + "libc": [ + "glibc" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">= 10" + }, + "funding": { + "type": "github", + "url": "https://github.com/sponsors/Brooooooklyn" + } + }, + "node_modules/@napi-rs/canvas-linux-x64-musl": { + "version": "1.0.9", + "resolved": "https://registry.npmjs.org/@napi-rs/canvas-linux-x64-musl/-/canvas-linux-x64-musl-1.0.9.tgz", + "integrity": "sha512-xrGvmS3v55hmZ86ls/kBLVNMUTYio3f6Ik0DireemG994VfPAwiA3ZXA0Uf1bByctkB3NQ1Sfb+H5bkdUnnzfQ==", + "cpu": [ + "x64" + ], + "libc": [ + "musl" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">= 10" + }, + "funding": { + "type": "github", + "url": "https://github.com/sponsors/Brooooooklyn" + } + }, + "node_modules/@napi-rs/canvas-win32-arm64-msvc": { + "version": "1.0.9", + "resolved": "https://registry.npmjs.org/@napi-rs/canvas-win32-arm64-msvc/-/canvas-win32-arm64-msvc-1.0.9.tgz", + "integrity": "sha512-yjmVS3ArZeRVCP7jqbPq4rpZa/BhTeI7ELE2XqJg3snICQBDevLZyArxswHkiTnT34KRic33/4fLirrHI+SY8A==", + "cpu": [ + "arm64" + ], + "license": "MIT", + "optional": true, + "os": [ + "win32" + ], + "engines": { + "node": ">= 10" + }, + "funding": { + "type": "github", + "url": "https://github.com/sponsors/Brooooooklyn" + } + }, + "node_modules/@napi-rs/canvas-win32-x64-msvc": { + "version": "1.0.9", + "resolved": "https://registry.npmjs.org/@napi-rs/canvas-win32-x64-msvc/-/canvas-win32-x64-msvc-1.0.9.tgz", + "integrity": "sha512-QlSYQdMQslB81nlABo9wNfQ6npFhE7/O+saCZdqVGueGanyRk4jCogD5EwQenfP3kIq9e+mm6GreQBjX5MrA8g==", + "cpu": [ + "x64" + ], + "license": "MIT", + "optional": true, + "os": [ + "win32" + ], + "engines": { + "node": ">= 10" + }, + "funding": { + "type": "github", + "url": "https://github.com/sponsors/Brooooooklyn" + } + }, "node_modules/@nodelib/fs.scandir": { "version": "2.1.5", "resolved": "https://registry.npmjs.org/@nodelib/fs.scandir/-/fs.scandir-2.1.5.tgz", @@ -1940,6 +2206,18 @@ "node": ">=8" } }, + "node_modules/pdfjs-dist": { + "version": "6.3.289", + "resolved": "https://registry.npmjs.org/pdfjs-dist/-/pdfjs-dist-6.3.289.tgz", + "integrity": "sha512-ZHjSVpDa3D6izMq8/04lvkhkATUmL9px6ChPaXc1k6nU2Mrhlg1/7F0bdUqCwUjw3NsPTfPZsMDUU6ZIcRaeQw==", + "license": "Apache-2.0", + "engines": { + "node": ">=22.13.0 || >=24" + }, + "optionalDependencies": { + "@napi-rs/canvas": "^1.0.0" + } + }, "node_modules/picocolors": { "version": "1.1.1", "resolved": "https://registry.npmjs.org/picocolors/-/picocolors-1.1.1.tgz", diff --git a/efile_app/package.json b/efile_app/package.json index 77f0c505..7f9bfd44 100644 --- a/efile_app/package.json +++ b/efile_app/package.json @@ -13,7 +13,8 @@ "format:js:check": "prettier --check eslint.config.mjs", "lint:css": "stylelint 'efile/**/*.css'", "test:a11y": "playwright test --config=playwright.a11y.config.js", - "test:confirm-case": "playwright test --config=playwright.confirm-case.config.js" + "test:confirm-case": "playwright test --config=playwright.confirm-case.config.js", + "postinstall": "node scripts/copy-pdfjs.mjs" }, "keywords": [], "author": "", @@ -21,7 +22,8 @@ "description": "", "dependencies": { "@playwright/test": "^1.55.0", - "dotenv": "^17.2.1" + "dotenv": "^17.2.1", + "pdfjs-dist": "6.3.289" }, "devDependencies": { "@axe-core/playwright": "^4.13.0", diff --git a/efile_app/pytest.ini b/efile_app/pytest.ini index cce66000..7986eb97 100644 --- a/efile_app/pytest.ini +++ b/efile_app/pytest.ini @@ -3,3 +3,6 @@ DJANGO_SETTINGS_MODULE = efile.settings python_files = tests.py test_*.py *_tests.py addopts = --reuse-db --nomigrations testpaths = efile + +markers = + integration: tests requiring external conversion/browser dependencies diff --git a/efile_app/scripts/copy-pdfjs.mjs b/efile_app/scripts/copy-pdfjs.mjs new file mode 100644 index 00000000..f8d454ef --- /dev/null +++ b/efile_app/scripts/copy-pdfjs.mjs @@ -0,0 +1,20 @@ +// Keep all PDF.js assets on our origin, including fonts for older court forms. +import { copyFileSync, cpSync, mkdirSync } from "node:fs"; +import { fileURLToPath } from "node:url"; +import path from "node:path"; + +const root = path.resolve(path.dirname(fileURLToPath(import.meta.url)), ".."); +const source = path.join(root, "node_modules/pdfjs-dist"); +const target = path.join(root, "efile/static/vendor/pdfjs"); +mkdirSync(target, { recursive: true }); +for (const name of ["pdf.min.mjs", "pdf.worker.min.mjs"]) { + copyFileSync(path.join(source, "legacy/build", name), path.join(target, name)); +} +for (const name of ["pdf_viewer.mjs", "pdf_viewer.css"]) { + copyFileSync(path.join(source, "legacy/web", name), path.join(target, name)); +} +for (const name of ["cmaps", "standard_fonts", "wasm"]) { + cpSync(path.join(source, name), path.join(target, name), { recursive: true }); +} +cpSync(path.join(source, "web/images"), path.join(target, "images"), { recursive: true }); +copyFileSync(path.join(source, "LICENSE"), path.join(target, "LICENSE")); diff --git a/efile_app/stylelint.config.mjs b/efile_app/stylelint.config.mjs index d7ad795d..62a53dd9 100644 --- a/efile_app/stylelint.config.mjs +++ b/efile_app/stylelint.config.mjs @@ -1,4 +1,5 @@ export default { + ignoreFiles: ['efile/static/vendor/**'], plugins: ['stylelint-plugin-defensive-css'], rules: { // Keep this deliberately small at first: these rules cover failures diff --git a/efile_app/tests/document-preparation-browser.js b/efile_app/tests/document-preparation-browser.js new file mode 100644 index 00000000..6297b44e --- /dev/null +++ b/efile_app/tests/document-preparation-browser.js @@ -0,0 +1,191 @@ +/* Browser validation invoked by the opt-in Django integration test. */ +const assert = require("node:assert/strict"); +const fs = require("node:fs"); +const path = require("node:path"); +const { + chromium +} = require("@playwright/test"); +const AxeBuilder = require("@axe-core/playwright").default; + +async function main() { + const config = JSON.parse(fs.readFileSync(process.argv[2], "utf8")); + const browser = await chromium.launch({ + executablePath: process.env.PLAYWRIGHT_CHROMIUM_EXECUTABLE || undefined + }); + const context = await browser.newContext({ + viewport: { + width: 1280, + height: 1000 + } + }); + await context.addCookies([{ + name: "sessionid", + value: config.cookie, + url: config.baseUrl + }]); + const page = await context.newPage(); + // Court choices and payment accounts are synthetic; no court API calls. + await page.route(/\/api\/(?:dropdowns\/|payment-(?:accounts|account-types)\/)/, (route) => + route.fulfill({ + json: { + success: true, + data: [] + } + }) + ); + const errors = []; + page.on("console", (message) => { + if (["warning", "error"].includes(message.type())) console.log(message.text()); + }); + page.on("pageerror", (error) => errors.push(error.message)); + fs.mkdirSync(config.evidence, { + recursive: true + }); + const screenshot = (name) => + page.screenshot({ + path: path.join(config.evidence, name), + fullPage: true + }); + try { + await page.goto(config.baseUrl + config.uploadUrl); + await page.locator("#documents-input").setInputFiles(config.files); + await screenshot("01-upload-selection.png"); + await page.locator("#upload-button").click(); + await page.waitForResponse( + (response) => response.url().includes("upload-documents") && response.request().method() === "POST" + ); + await page.locator("#continue-to-analysis:not(.disabled)").waitFor(); + await page.waitForFunction(() => document.querySelectorAll(".document-row").length === 2); + await page.locator("#continue-to-analysis").click(); + await page.waitForURL(/preview-documents/); + const previews = page.locator("[data-pdf-preview]"); + assert.equal(await previews.count(), 2); + await previews.first().locator("summary").click(); + await previews.first().locator("canvas").first().waitFor(); + await page.waitForFunction(() => document.querySelector("[data-pdf-preview]").dataset.rendered === "true"); + assert.match(await previews.first().innerText(), /of 2/); + await screenshot("02-multiline-preview.png"); + await previews.first().locator("[data-pdf-next]").click(); + assert.equal(await previews.first().locator("[data-pdf-page]").inputValue(), "2"); + await previews.first().locator("[data-pdf-zoom-in]").click(); + await screenshot("03-second-page.png"); + await previews.nth(1).locator("summary").click(); + await page.waitForFunction( + () => document.querySelectorAll("[data-pdf-preview]")[1].dataset.rendered === "true" + ); + await screenshot("04-word-preview.png"); + const accessibility = await new AxeBuilder({ + page + }) + .include(".workflow-card") + .analyze(); + const serious = accessibility.violations.filter((item) => ["serious", "critical"].includes(item.impact)); + fs.writeFileSync( + path.join(config.evidence, "accessibility.json"), + JSON.stringify({ + violations: accessibility.violations, + passes: accessibility.passes.map((item) => item.id) + }, + null, + 2 + ) + ); + assert.equal( + accessibility.violations.length, + 0, + JSON.stringify( + serious.map((item) => ({ + id: item.id, + nodes: item.nodes.map((node) => node.target) + })) + ) + ); + await page + .getByRole("button", { + name: "Continue", + exact: true + }) + .click(); + assert.match(page.url(), /preview-documents/); + await page.locator('input[name="reviewed_document"]').evaluateAll((nodes) => + nodes.forEach((node) => { + node.checked = true; + }) + ); + await page + .getByRole("button", { + name: "Continue", + exact: true + }) + .click(); + await page.waitForURL(/extraction-review/); + await page.goto(config.baseUrl + config.organizeUrl); + assert.match(page.url(), /organize-documents/); + assert.equal(await page.locator("h1").innerText(), "Organize your documents"); + await page.locator("[data-pdf-preview]").first().locator("summary").click(); + await page.waitForFunction(() => document.querySelector("[data-pdf-preview]").dataset.rendered === "true"); + await screenshot("05-organize-preview.png"); + await page.goto(config.baseUrl + config.paymentUrl); + assert.match(page.url(), /\/payment\//); + assert.equal(await page.locator("h1").innerText(), "Choose how to pay court fees"); + await page.locator("[data-pdf-preview]").first().locator("summary").click(); + await page.waitForFunction(() => document.querySelector("[data-pdf-preview]").dataset.rendered === "true"); + await screenshot("14-fees-preview.png"); + await page.setViewportSize({ + width: 390, + height: 844 + }); + await page.goto(config.baseUrl + config.previewUrl); + await page.locator("[data-pdf-preview]").first().locator("summary").click(); + await page.waitForFunction(() => document.querySelector("[data-pdf-preview]").dataset.rendered === "true"); + await screenshot("06-mobile-preview.png"); + assert.equal(await page.evaluate(() => document.documentElement.scrollWidth <= window.innerWidth), true); + await page.setViewportSize({ + width: 1280, + height: 1000 + }); + await page.goto(config.baseUrl + config.uploadUrl); + await page.locator("#documents-input").setInputFiles({ + name: "invalid.pdf", + mimeType: "application/pdf", + buffer: Buffer.from("not a PDF") + }); + await page.locator("#upload-button").click(); + await page.locator("#upload-error:not([hidden])").waitFor(); + await screenshot("07-recoverable-error.png"); + assert.match(await page.locator("#upload-error").innerText(), /could not be read/); + // A storage failure must leave a usable download/retry explanation. + await page.route("**/documents/*/content/**", (route) => + route.fulfill({ + status: 503, + body: "Unavailable" + }) + ); + await page.goto(config.baseUrl + config.previewUrl); + await page.locator("[data-pdf-preview]").first().locator("summary").click(); + await page + .getByText("The PDF preview could not load.", { + exact: false + }) + .waitFor(); + await screenshot("08-preview-failure.png"); + await page.unroute("**/documents/*/content/**"); + await page.locator("[data-pdf-preview]").first().locator("summary").click(); + await page.locator("[data-pdf-preview]").first().locator("summary").click(); + await page.waitForFunction(() => document.querySelector("[data-pdf-preview]").dataset.rendered === "true"); + assert.deepEqual(errors, []); + console.log( + "Browser validation passed: upload, real PDF.js pages, navigation, zoom, Word preview, acknowledgement, organize, fees, mobile, errors/retry, and Axe." + ); + } catch (error) { + await screenshot("browser-failure.png"); + throw error; + } finally { + await browser.close(); + } +} + +main().catch((error) => { + console.error(error); + process.exitCode = 1; +}); \ No newline at end of file diff --git a/package-lock.json b/package-lock.json new file mode 100644 index 00000000..c9a3f0e3 --- /dev/null +++ b/package-lock.json @@ -0,0 +1,6 @@ +{ + "name": "LITEFile", + "lockfileVersion": 3, + "requires": true, + "packages": {} +} diff --git a/testing/validate_document_preparation.py b/testing/validate_document_preparation.py new file mode 100644 index 00000000..74438d08 --- /dev/null +++ b/testing/validate_document_preparation.py @@ -0,0 +1,136 @@ +"""Compare raw Gotenberg, LITEFile and PDFtk using local filled PDF forms. + +From efile_app: uv run python ../testing/validate_document_preparation.py --output /tmp/pdf-validation +Rendered documents remain local. Publish only synthetic examples. +""" + +import argparse +import hashlib +import io +import json +import os +import subprocess +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "efile_app")) +os.environ.setdefault("DJANGO_SETTINGS_MODULE", "efile.settings_dev") +import django # noqa: E402 + +django.setup() +from django.core.files.uploadedfile import SimpleUploadedFile # noqa: E402 +from pypdf import PdfReader # noqa: E402 + +from efile.services.document_preparation import _gotenberg, prepare_document # noqa: E402 + + +def summary(content): + reader = PdfReader(io.BytesIO(content)) + return { + "pages": len(reader.pages), + "remaining_fields": len(reader.get_fields() or {}), + "text_characters": sum(len(page.extract_text() or "") for page in reader.pages), + "tagged": bool(reader.trailer["/Root"].get("/StructTreeRoot")), + } + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--output", type=Path, required=True) + parser.add_argument( + "--source", + type=Path, + action="append", + help="PDF or repository directory; defaults to ~/docassemble-*.", + ) + parser.add_argument("--render", action="store_true") + args = parser.parse_args() + args.output.mkdir(parents=True, exist_ok=True) + seen, report = set(), [] + for root in args.source or sorted(Path.home().glob("docassemble-*")): + for path in [root] if root.is_file() else sorted(root.rglob("*.pdf")): + if ".git" in path.parts: + continue + content = path.read_bytes() + digest = hashlib.sha256(content).hexdigest() + if digest in seen: + continue + try: + reader = PdfReader(io.BytesIO(content)) + fields = reader.get_fields() or {} + filled = [field for field in fields.values() if field.get("/V")] + if not filled: + continue + except Exception: + continue + seen.add(digest) + folder = args.output / f"{len(report):02}" + folder.mkdir(exist_ok=True) + row = { + "source": str(path.relative_to(root.parent)), + "sha256": digest, + "input_pages": len(reader.pages), + "input_fields": len(fields), + "filled_fields": len(filled), + "multiline_fields": sum( + bool(int(field.get("/Ff", 0)) & 4096) for field in fields.values() + ), + } + for engine in ["raw-gotenberg", "litefile", "pdftk"]: + try: + target = folder / f"{engine}.pdf" + if engine == "raw-gotenberg": + output = _gotenberg( + content, ".pdf", "/forms/pdfengines/flatten" + ) + elif engine == "litefile": + output = prepare_document( + SimpleUploadedFile("input.pdf", content), "vermont" + ).content + else: + subprocess.run( + ["pdftk", str(path), "output", str(target), "flatten"], + capture_output=True, + check=True, + timeout=45, + ) + output = target.read_bytes() + target.write_bytes(output) + row[engine] = {"accepted": True, **summary(output)} + if args.render: + subprocess.run( + [ + "pdftoppm", + "-f", + "1", + "-singlefile", + "-scale-to", + "1400", + "-png", + str(target), + str(folder / engine), + ], + capture_output=True, + check=True, + timeout=45, + ) + except Exception as error: + row[engine] = { + "accepted": False, + "error_type": type(error).__name__, + } + report.append(row) + print( + row["source"], + { + engine: row[engine]["accepted"] + for engine in ["raw-gotenberg", "litefile", "pdftk"] + }, + flush=True, + ) + (args.output / "comparison.json").write_text(json.dumps(report, indent=2) + "\n") + print(f"Compared {len(report)} distinct filled PDFs.") + + +if __name__ == "__main__": + main() From c98f21e8e17a89f4f9360737b4905937e91e31d1 Mon Sep 17 00:00:00 2001 From: Quinten Steenhuis Date: Wed, 30 Sep 2026 14:22:12 -0400 Subject: [PATCH 2/8] Update virtualenv to clear the dependency audit --- docs/developer-notes/issue-113/validation.md | 7 ++++++ efile_app/uv.lock | 25 +++++++++++++++----- 2 files changed, 26 insertions(+), 6 deletions(-) diff --git a/docs/developer-notes/issue-113/validation.md b/docs/developer-notes/issue-113/validation.md index 2edc8e18..3ce3abff 100644 --- a/docs/developer-notes/issue-113/validation.md +++ b/docs/developer-notes/issue-113/validation.md @@ -125,6 +125,7 @@ and final packet exhibit. | New Django template lint and format | Passed | | Stylelint | 0 errors; 19 existing warnings in other stylesheets | | Bandit | Passed after removing a production assertion | +| `uv run pip-audit --local --skip-editable` | No known vulnerabilities found | | `manage.py makemigrations --check --dry-run` | No changes detected | | `docker build -t litefile:issue-113 .` | Passed; generated and collected PDF.js assets | | Docusaurus `npm run build` | Passed | @@ -143,6 +144,12 @@ were left at their locked versions to keep this feature's dependency changes focused. Accessibility testing checks the preview controls and generated tagged Word PDF structure; it does not certify every filing PDF as PDF/UA compliant. +The first GitHub dependency audit found four advisories in the pre-existing +`virtualenv` 20.36.1 development dependency. Updated the lockfile to virtualenv +21.7.13 and its required dependencies, then verified the audit passes and that +it creates a working virtual environment. The relevant fixes are documented in +[virtualenv's release history](https://virtualenv.pypa.io/en/latest/changelog.html#v21-7-13-2026-09-18). + ## Reproduction From `efile_app`, configure Gotenberg credentials and run: diff --git a/efile_app/uv.lock b/efile_app/uv.lock index 866c4536..a837f9d9 100644 --- a/efile_app/uv.lock +++ b/efile_app/uv.lock @@ -681,11 +681,11 @@ wheels = [ [[package]] name = "filelock" -version = "3.20.3" +version = "3.32.7" source = { registry = "https://pypi.org/simple" } -sdist = { url = "https://files.pythonhosted.org/packages/1d/65/ce7f1b70157833bf3cb851b556a37d4547ceafc158aa9b34b36782f23696/filelock-3.20.3.tar.gz", hash = "sha256:18c57ee915c7ec61cff0ecf7f0f869936c7c30191bb0cf406f1341778d0834e1", size = 19485, upload-time = "2026-01-09T17:55:05.421Z" } +sdist = { url = "https://files.pythonhosted.org/packages/0f/59/e19834834cb01a32febfbb0f8a23a9088088f5d45991824ff2bc3b5e8acb/filelock-3.32.7.tar.gz", hash = "sha256:37b8a3d9811b0f9aef7e5ec5c71bb320de52df51e6ca9bcd6f5ad81187660da7", size = 225154, upload-time = "2026-09-16T00:24:20.907Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/b5/36/7fb70f04bf00bc646cd5bb45aa9eddb15e19437a28b8fb2b4a5249fac770/filelock-3.20.3-py3-none-any.whl", hash = "sha256:4b0dda527ee31078689fc205ec4f1c1bf7d56cf88b6dc9426c4f230e46c2dce1", size = 16701, upload-time = "2026-01-09T17:55:04.334Z" }, + { url = "https://files.pythonhosted.org/packages/15/df/31098c5aeb4d966b553641472bd55fcf5fdfac953549894b8a765ba44e91/filelock-3.32.7-py3-none-any.whl", hash = "sha256:65ff0d0190ea42038b32bda4b77834fb05be2cad4c5b9b01aa4dfb3614536e52", size = 100157, upload-time = "2026-09-16T00:24:19.543Z" }, ] [[package]] @@ -1776,6 +1776,18 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/ec/57/56b9bcc3c9c6a792fcbaf139543cee77261f3651ca9da0c93f5c1221264b/python_dateutil-2.9.0.post0-py2.py3-none-any.whl", hash = "sha256:a8b2bc7bffae282281c8140a97d3aa9c14da0b136dfe83f850eea9a5f7470427", size = 229892, upload-time = "2024-03-01T18:36:18.57Z" }, ] +[[package]] +name = "python-discovery" +version = "1.6.1" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "filelock" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/0c/57/250bd238b966cece44328235eb85290045d059265fdaf7527a3a958123db/python_discovery-1.6.1.tar.gz", hash = "sha256:cf87d3627dfb4412437fdd5b13eae402607722998d21567993aedbc59b23c15e", size = 84338, upload-time = "2026-09-18T01:31:53.971Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/16/7d/e9ffbadfbf89c93848412d04594135c4ae8c1d37d9e053b9c3ed718fabc4/python_discovery-1.6.1-py3-none-any.whl", hash = "sha256:d43fcdef879fe795352bd13ccf8d185ba5a9f86f36cfcd00529f596e737442b3", size = 38664, upload-time = "2026-09-18T01:31:52.448Z" }, +] + [[package]] name = "python-dotenv" version = "1.2.3" @@ -2334,16 +2346,17 @@ wheels = [ [[package]] name = "virtualenv" -version = "20.36.1" +version = "21.7.13" source = { registry = "https://pypi.org/simple" } dependencies = [ { name = "distlib" }, { name = "filelock" }, { name = "platformdirs" }, + { name = "python-discovery" }, ] -sdist = { url = "https://files.pythonhosted.org/packages/aa/a3/4d310fa5f00863544e1d0f4de93bddec248499ccf97d4791bc3122c9d4f3/virtualenv-20.36.1.tar.gz", hash = "sha256:8befb5c81842c641f8ee658481e42641c68b5eab3521d8e092d18320902466ba", size = 6032239, upload-time = "2026-01-09T18:21:01.296Z" } +sdist = { url = "https://files.pythonhosted.org/packages/13/50/c9b84eb106d0db420b9878dac0a386c5791726f127ce20b06897f6e3a1e9/virtualenv-21.7.13.tar.gz", hash = "sha256:0355558b6f33619aab31347e43643b0ebc97f61ea3acf617b2b69e1f8a843d11", size = 5360306, upload-time = "2026-09-18T04:35:49.354Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/6a/2a/dc2228b2888f51192c7dc766106cd475f1b768c10caaf9727659726f7391/virtualenv-20.36.1-py3-none-any.whl", hash = "sha256:575a8d6b124ef88f6f51d56d656132389f961062a9177016a50e4f507bbcc19f", size = 6008258, upload-time = "2026-01-09T18:20:59.425Z" }, + { url = "https://files.pythonhosted.org/packages/9a/ce/e74453531b0c49a58d0e8a4f9bab4495705859fee4a4f7de27d58f4a791a/virtualenv-21.7.13-py3-none-any.whl", hash = "sha256:1bea5af7463f59c4719db48fe739579a2a4f569c96f26c086edda85c96da9f59", size = 5328680, upload-time = "2026-09-18T04:35:47.194Z" }, ] [[package]] From 60d8014cef011adda6abd202032b6af853e46e51 Mon Sep 17 00:00:00 2001 From: Quinten Steenhuis Date: Wed, 30 Sep 2026 14:54:57 -0400 Subject: [PATCH 3/8] Close preview bypasses and harden document replacement cleanup --- docs/developer-notes/issue-113/validation.md | 33 ++++- docs/docs/admin/configuration.md | 4 +- .../efile/services/document_preparation.py | 78 ++++++++-- efile_app/efile/services/document_previews.py | 13 +- efile_app/efile/services/document_uploads.py | 54 ++++++- efile_app/efile/services/drafts.py | 14 ++ efile_app/efile/services/handoff.py | 16 +- .../efile/components/document_preview.html | 2 +- .../templates/efile/preview_documents.html | 11 +- efile_app/efile/tests/helpers.py | 11 ++ efile_app/efile/tests/test_ai_opt_out.py | 9 +- efile_app/efile/tests/test_court_selection.py | 3 +- .../tests/test_current_draft_selection.py | 3 +- .../efile/tests/test_document_extractions.py | 23 +-- efile_app/efile/tests/test_document_prep.py | 9 +- .../efile/tests/test_document_preparation.py | 51 +++++++ .../efile/tests/test_document_previews.py | 139 ++++++++++++++++++ efile_app/efile/tests/test_durable_drafts.py | 6 +- .../tests/test_end_to_end_new_flow_states.py | 2 +- .../efile/tests/test_extracted_parties.py | 3 +- .../efile/tests/test_extraction_claims.py | 3 +- efile_app/efile/tests/test_fee_quotes.py | 6 +- efile_app/efile/tests/test_filer_role.py | 3 +- .../efile/tests/test_filing_integrity.py | 12 +- .../efile/tests/test_filing_on_behalf.py | 3 +- .../tests/test_filing_path_navigation.py | 7 +- .../efile/tests/test_filing_plan_actions.py | 5 +- efile_app/efile/tests/test_filing_plans.py | 3 +- efile_app/efile/tests/test_handoff.py | 69 +++++++++ efile_app/efile/tests/test_my_drafts.py | 3 +- .../efile/tests/test_parties_save_role.py | 3 +- .../tests/test_party_address_requirements.py | 3 +- efile_app/efile/tests/test_people_flow.py | 11 +- .../efile/tests/test_reorganized_start.py | 21 +-- .../efile/tests/test_review_submit_flow.py | 3 +- efile_app/efile/tests/test_ui_text.py | 5 +- .../efile/tests/test_waiver_documents.py | 25 ++++ efile_app/efile/tests/tests.py | 18 +-- efile_app/efile/views/document_previews.py | 23 ++- efile_app/efile/views/handoff.py | 34 ++++- efile_app/efile/views/waiver_documents.py | 3 + 41 files changed, 630 insertions(+), 117 deletions(-) diff --git a/docs/developer-notes/issue-113/validation.md b/docs/developer-notes/issue-113/validation.md index 3ce3abff..aa11cc24 100644 --- a/docs/developer-notes/issue-113/validation.md +++ b/docs/developer-notes/issue-113/validation.md @@ -116,8 +116,10 @@ and final packet exhibit. | Check | Result | | --- | --- | -| `uv run pytest -q` | 1,215 passed; 1 opt-in browser test skipped | +| `uv run pytest -q` | 1,230 passed; 1 opt-in browser test skipped | | Opt-in real Gotenberg and Chromium test | 1 passed | +| Conversion and preview tests with Gotenberg environment variables cleared | Passed; final full suite also runs with these variables cleared | +| GitHub accessibility workflow | Passed | | `npm run test:unit` | 64 passed | | `uv run ruff check .` and formatting | Passed | | `uv run ty check` | Passed | @@ -150,6 +152,35 @@ The first GitHub dependency audit found four advisories in the pre-existing it creates a working virtual environment. The relevant fixes are documented in [virtualenv's release history](https://virtualenv.pypa.io/en/latest/changelog.html#v21-7-13-2026-09-18). +CI also exposed mocked conversion tests that depended on a locally configured +Gotenberg URL. The test fixture now sets a synthetic service URL explicitly; +The conversion and preview tests pass with all Gotenberg environment variables +cleared. The opt-in integration test still uses the real configured service. + +## Regression review + +Addressed the follow-up review with tests that start from unprepared or stale +server state rather than relying on already reviewed fixtures: + +| Concern | Behavior checked | +| --- | --- | +| Lead storage-key replacement | Clears preparation, original metadata and approval; deletes superseded copies after commit while keeping shared references | +| Legacy supporting uploads | Prepares stored DOCX/PDF bytes on preview, rejects acknowledgement before preparation, and blocks direct submission | +| Stale preview fingerprint | Includes preparation, original key, size and document update time in addition to row and filing key | +| Indirect annotation arrays | Resolves and validates the array before iterating; accepts the indirect-array regression specimen | +| Handoff preparation errors | Permanent input/conversion rejection returns 422; storage and conversion-service outages return 503 | +| Extraction timing | Registers the job creation with `transaction.on_commit`; a rolled-back transaction does not queue extraction | +| Waiver write failure | Database errors after storing original and filing copies remove both staged objects | +| Handoff replacement cleanup | Deletes old objects after commit, preserving objects referenced by another draft | +| Special text-field appearances | Skips literal-value matching for comb, password, formatted, rich-text and hidden/no-view fields; retains structural and ordinary visible-text checks | + +The existing extraction queue is a database record, so creating it inside the +original atomic transaction was already isolated from worker reads and rollback. +The callback now makes the commit boundary explicit. Downstream-flow fixtures +explicitly represent prepared and acknowledged documents; the new bypass tests +keep their documents unprepared. A concurrency unit test also now mocks court +choices instead of intermittently depending on a live court response. + ## Reproduction From `efile_app`, configure Gotenberg credentials and run: diff --git a/docs/docs/admin/configuration.md b/docs/docs/admin/configuration.md index f2498970..b641d656 100644 --- a/docs/docs/admin/configuration.md +++ b/docs/docs/admin/configuration.md @@ -81,7 +81,9 @@ Word conversion requests tagged PDF output and lossless images. It does not rasterize the document or certify accessibility conformance. Flattening can change accessibility tags, links, or annotations. The private original is retained separately from the filing PDF and is available for download. Only the filing copy reaches the -court. Removing a document or expiring an unclaimed handoff cleans up both private +court. Existing editable drafts without preparation metadata are prepared when +the filer opens the preview step. Missing stored uploads must be replaced; legacy +clients cannot bypass preparation or preview approval. Removing a document or expiring an unclaimed handoff cleans up both private copies when another draft does not reference them. State YAML can override the default policy: diff --git a/efile_app/efile/services/document_preparation.py b/efile_app/efile/services/document_preparation.py index 9beca24e..18ee107d 100644 --- a/efile_app/efile/services/document_preparation.py +++ b/efile_app/efile/services/document_preparation.py @@ -13,7 +13,7 @@ from django.conf import settings from django.core.files.uploadedfile import SimpleUploadedFile from pypdf import PdfReader, PdfWriter -from pypdf.generic import DictionaryObject +from pypdf.generic import ArrayObject, DictionaryObject from efile.utils.config_loader import config_loader @@ -21,7 +21,11 @@ class PreparationError(ValueError): - """An actionable preparation failure safe to show to the filer.""" + """An actionable document failure safe to show to the filer.""" + + +class PreparationUnavailable(PreparationError): + """A temporary service/storage failure for which retry is appropriate.""" @dataclass(frozen=True) @@ -52,12 +56,19 @@ def inspect_pdf(content): "This PDF uses an unsupported form format. Save a printed PDF copy and upload it again." ) # Force page and annotation parsing before accepting a filing copy. - widgets = [ - ref.get_object() - for page in reader.pages - for ref in page.get("/Annots", []) - if ref.get_object().get("/Subtype") == "/Widget" - ] + widgets = [] + for page in reader.pages: + if "/Annots" not in page: + continue + annotations = page["/Annots"].get_object() + if not isinstance(annotations, ArrayObject): + raise ValueError("Invalid annotation array") + for ref in annotations: + annotation = ref.get_object() + if not isinstance(annotation, DictionaryObject): + raise ValueError("Invalid annotation") + if annotation.get("/Subtype") == "/Widget": + widgets.append(annotation) return reader, widgets except PreparationError: raise @@ -86,7 +97,7 @@ def _word_format(content, suffix): def _gotenberg(content, suffix, route, data=None): base_url = settings.GOTENBERG_URL.rstrip("/") if not base_url: - raise PreparationError( + raise PreparationUnavailable( "Document preparation is unavailable. Upload a PDF with the form fields already locked, or try again later." ) limit = settings.MAX_FILE_SIZE @@ -102,6 +113,8 @@ def _gotenberg(content, suffix, route, data=None): allow_redirects=False, stream=True, ) as response: + if response.status_code >= 500 or response.status_code in {401, 403, 429}: + raise PreparationUnavailable("Document preparation is unavailable. Please try again later.") if response.status_code != 200: raise PreparationError("We could not prepare this document. Upload a PDF copy or try uploading again.") result = bytearray() @@ -116,7 +129,7 @@ def _gotenberg(content, suffix, route, data=None): return bytes(result) except requests.RequestException as error: logger.warning("Document preparation service unavailable (%s)", type(error).__name__) - raise PreparationError( + raise PreparationUnavailable( "Document preparation timed out or is unavailable. Try again, or upload a PDF copy with its form fields locked." ) from error @@ -172,8 +185,26 @@ def _flatten(content): # Engines can return 200 while dropping filled text. This is a conservative # check, not a guarantee of visual fidelity; the filer still previews it. visible_text = " ".join(" ".join(page.extract_text() or "" for page in output.pages).split()) - for field in fields.values(): - if field.get("/FT") == "/Tx": + checked_fields = set() + for widget in widgets: + field = widget.get("/Parent", widget).get_object() + name = field.get("/T") + flags = int(field.get("/Ff", 0)) + annotation_flags = int(widget.get("/F", 0)) + rect = widget.get("/Rect", [0, 0, 0, 0]) + # Hidden/no-view widgets, passwords, combs, rich/formatted fields can + # legitimately have a different appearance from their stored /V. + plain_visible = ( + field.get("/FT") == "/Tx" + and not (flags & ((1 << 13) | (1 << 24) | (1 << 25))) + and not (annotation_flags & (1 | 2 | 32)) + and not field.get("/AA") + and not widget.get("/AA") + and rect[0] != rect[2] + and rect[1] != rect[3] + ) + if plain_visible and name not in checked_fields: + checked_fields.add(name) value = str(field.get("/V") or "") if any(" ".join(line.split()) not in visible_text for line in value.splitlines() if line.strip()): raise PreparationError( @@ -219,13 +250,13 @@ def store_prepared_document(handler, uploaded_file, jurisdiction, role, *, keys, uploaded_file.seek(0) original = handler.upload_file(uploaded_file, file_type="original", metadata=metadata) if not original.get("success"): - raise PreparationError("The original could not be saved. Try uploading again.") + raise PreparationUnavailable("The original could not be saved. Try uploading again.") original_key = original["key"] keys.append(original_key) filing = SimpleUploadedFile(prepared.filename, prepared.content, content_type="application/pdf") result = handler.upload_file(filing, file_type=role, metadata=metadata) if not result.get("success"): - raise PreparationError("The filing copy could not be saved. Try uploading again.") + raise PreparationUnavailable("The filing copy could not be saved. Try uploading again.") keys.append(result["key"]) return { "name": prepared.filename[:255], @@ -248,3 +279,22 @@ def cleanup_uploads(handler, keys): logger.warning("Could not remove an uncommitted document upload") except Exception: logger.exception("Could not remove an uncommitted document upload") + + +def cleanup_unreferenced_uploads(keys, handler=None): + """Remove superseded copies after commit, preserving cross-draft references.""" + from django.db.models import Q + + from efile.models import FilingDocument + from efile.utils.s3_upload_handler import S3UploadHandler + + unused = [ + key + for key in dict.fromkeys(keys) + if key and not FilingDocument.objects.filter(Q(s3_key=key) | Q(original_s3_key=key)).exists() + ] + if not unused: + return + handler = handler or S3UploadHandler() + if handler._ensure_initialized(): + cleanup_uploads(handler, unused) diff --git a/efile_app/efile/services/document_previews.py b/efile_app/efile/services/document_previews.py index ba99cd9e..527a28d0 100644 --- a/efile_app/efile/services/document_previews.py +++ b/efile_app/efile/services/document_previews.py @@ -5,15 +5,22 @@ def unreviewed_documents(draft): - return draft.documents.exclude(preparation="").filter(preparation_reviewed_at__isnull=True) + return draft.documents.filter(preparation_reviewed_at__isnull=True) def preview_fingerprint(documents): - return hashlib.sha256(json.dumps(sorted((doc.pk, doc.s3_key) for doc in documents)).encode()).hexdigest() + return hashlib.sha256( + json.dumps( + sorted( + (doc.pk, doc.s3_key, doc.original_s3_key, doc.preparation, doc.size, str(doc.updated_at)) + for doc in documents + ) + ).encode() + ).hexdigest() def require_document_previews(draft): - if unreviewed_documents(draft).exists(): + if draft.documents.filter(preparation="").exists() or unreviewed_documents(draft).exists(): raise ValueError( "Preview your uploaded PDFs and confirm that their pages and signatures are correct before submitting." ) diff --git a/efile_app/efile/services/document_uploads.py b/efile_app/efile/services/document_uploads.py index 39ae476d..1530e810 100644 --- a/efile_app/efile/services/document_uploads.py +++ b/efile_app/efile/services/document_uploads.py @@ -1,9 +1,19 @@ +from functools import partial + +from django.conf import settings +from django.core.files.uploadedfile import SimpleUploadedFile from django.db import transaction from django.db.models import Max from efile.models import FilingDocument, FilingDraft from efile.services.document_extractions import queue_document_extraction -from efile.services.document_preparation import cleanup_uploads, store_prepared_document +from efile.services.document_preparation import ( + PreparationError, + PreparationUnavailable, + cleanup_unreferenced_uploads, + cleanup_uploads, + store_prepared_document, +) from efile.services.drafts import read_upload_data from efile.services.fee_quotes import invalidate_fee_quote from efile.utils.s3_upload_handler import S3UploadHandler @@ -44,7 +54,7 @@ def upload_files(draft, uploaded_files, jurisdiction, *, current_step=WorkflowSt if is_lead: has_lead = True draft.extracted_guesses = {} - queue_document_extraction(document) + transaction.on_commit(partial(queue_document_extraction, document), robust=True) else: order += 1 draft.current_step = str(current_step) @@ -54,3 +64,43 @@ def upload_files(draft, uploaded_files, jurisdiction, *, current_step=WorkflowSt cleanup_uploads(handler, keys) raise return read_upload_data(draft) + + +def prepare_stored_documents(draft, handler): + """Bring legacy stored uploads through the same preparation and review gate.""" + keys = [] + try: + with transaction.atomic(): + locked = FilingDraft.objects.select_for_update().get(pk=draft.pk) + documents = list(locked.documents.filter(preparation="")) + if not documents: + return + if locked.status not in {FilingDraft.Status.DRAFT, FilingDraft.Status.ERROR}: + raise PreparationError("This filing is no longer available to edit.") + if not handler._ensure_initialized() or handler.s3_client is None: + raise PreparationUnavailable("Document storage is unavailable. Please try again later.") + old_keys = [] + for document in documents: + if not document.s3_key: + raise PreparationError("The stored upload is unavailable. Replace this document before continuing.") + response = handler.s3_client.get_object(Bucket=handler.bucket_name, Key=document.s3_key) + body = response["Body"] + try: + content = body.read(settings.MAX_FILE_SIZE + 1) + finally: + body.close() + old_keys.extend([document.s3_key, document.original_s3_key]) + file = SimpleUploadedFile(document.original_filename or document.name or "document.pdf", content) + prepared = store_prepared_document(handler, file, draft.jurisdiction, document.role, keys=keys) + for field, value in prepared.items(): + setattr(document, field, value) + document.save() + if document.role == FilingDocument.Role.LEAD: + locked.extracted_guesses = {} + transaction.on_commit(partial(queue_document_extraction, document), robust=True) + invalidate_fee_quote(locked, save=False) + locked.save() + transaction.on_commit(partial(cleanup_unreferenced_uploads, old_keys, handler), robust=True) + except Exception: + cleanup_uploads(handler, keys) + raise diff --git a/efile_app/efile/services/drafts.py b/efile_app/efile/services/drafts.py index 94ba7659..3ca46035 100644 --- a/efile_app/efile/services/drafts.py +++ b/efile_app/efile/services/drafts.py @@ -13,6 +13,7 @@ from django.db.models import QuerySet from efile.models import FilingDocument, FilingDraft, FilingParty +from efile.services.document_preparation import cleanup_unreferenced_uploads from efile.workflow import WorkflowStepKey, legacy_existing_case_value, normalize_existing_case ACTIVE_DRAFT_STATUSES = (FilingDraft.Status.DRAFT, FilingDraft.Status.ERROR) @@ -388,6 +389,11 @@ def _positive_int(value: Any) -> int | None: def _apply_document(doc: FilingDocument, file_obj: dict[str, Any], config: dict[str, Any]) -> None: + if "s3_key" in file_obj and _as_str(file_obj.get("s3_key")) != doc.s3_key: + doc.original_filename = "" + doc.original_s3_key = "" + doc.preparation = "" + doc.preparation_reviewed_at = None if "name" in file_obj: doc.name = _as_str(file_obj.get("name")) if not doc.original_filename: @@ -445,8 +451,11 @@ def _upsert_document( config: dict[str, Any], ) -> None: doc, _created = FilingDocument.objects.get_or_create(draft=draft, role=role, sort_order=sort_order) + old_keys = [doc.s3_key, doc.original_s3_key] _apply_document(doc, file_obj, config) doc.save() + if old_keys[0] != doc.s3_key: + transaction.on_commit(lambda: cleanup_unreferenced_uploads(old_keys), robust=True) @transaction.atomic @@ -458,6 +467,9 @@ def write_upload_data( ) -> FilingDraft: """Persist a (possibly partial) upload_data blob into FilingDocument rows.""" + locked = FilingDraft.objects.select_for_update().get(pk=draft.pk) + if locked.status not in ACTIVE_DRAFT_STATUSES: + raise ValueError("This filing is no longer available to edit.") data = dict(upload_data or {}) update_fields: list[str] = [] @@ -517,6 +529,8 @@ def write_upload_data( document.save(update_fields=["checklist_item_id", "updated_at"]) if document.s3_key in previous_ids: moved[previous_ids[document.s3_key]] = document.pk + old_keys = [key for document in previous for key in (document.s3_key, document.original_s3_key)] + transaction.on_commit(lambda: cleanup_unreferenced_uploads(old_keys), robust=True) if moved: from efile.services.handoff import carry_document_paths diff --git a/efile_app/efile/services/handoff.py b/efile_app/efile/services/handoff.py index ba3b348d..a0cd1516 100644 --- a/efile_app/efile/services/handoff.py +++ b/efile_app/efile/services/handoff.py @@ -200,7 +200,7 @@ def validate_payload(payload, source_config, files, *, require_lead=True): raise HandoffError("return_url must use an allowed HTTPS origin.") documents = payload.get("documents", []) if not isinstance(documents, list) or len(documents) > 20: - raise HandoffError("documents must be a list of up to 20 PDFs.") + raise HandoffError("documents must be a list of up to 20 documents.") ids = set() leads = 0 for document in documents: @@ -216,21 +216,19 @@ def validate_payload(payload, source_config, files, *, require_lead=True): _hints(document, "document") uploaded = files.get(key) if uploaded is None: - raise HandoffError(f"Upload the PDF for document {key}.") + raise HandoffError(f"Upload the PDF or Word file for document {key}.") if uploaded.size > MAX_DOCUMENT_BYTES: - raise HandoffError("Each PDF must be at most 10 MB.") + raise HandoffError("Each document must be at most 10 MB.") digest = hashlib.sha256() - prefix = uploaded.read(5) - uploaded.seek(0) - if prefix != b"%PDF-": - raise HandoffError("Only PDF documents are accepted.") + # Content validation and PDF/Word preparation share the app upload + # pipeline, which distinguishes unfixable input from service outages. for chunk in uploaded.chunks(): digest.update(chunk) uploaded.seek(0) if digest.hexdigest() != document.get("sha256"): raise HandoffError(f"Document hash mismatch: {key}.") if leads > 1 or (require_lead and documents and leads != 1): - raise HandoffError("A document bundle needs exactly one lead PDF.") + raise HandoffError("A document bundle needs exactly one lead document.") if set(files) != ids or any(len(files.getlist(key)) != 1 for key in files): raise HandoffError("Upload each declared document exactly once.") _validate_filing_hint_overrides(payload, ids) @@ -320,7 +318,7 @@ def populate(draft, payload, uploads): draft=draft, role=document["role"], sort_order=order[document["role"]], - name=document.get("form_name") or uploaded["filename"], + name=document.get("form_name") or uploaded.get("name", uploaded["filename"]), original_filename=uploaded["filename"], size=uploaded["size"], content_type="application/pdf", diff --git a/efile_app/efile/templates/efile/components/document_preview.html b/efile_app/efile/templates/efile/components/document_preview.html index 726e1129..c2d6a9da 100644 --- a/efile_app/efile/templates/efile/components/document_preview.html +++ b/efile_app/efile/templates/efile/components/document_preview.html @@ -5,7 +5,7 @@

{% translate "Download filing PDF" %} - {% if document.original_s3_key %} + {% if document.original_s3_key or not document.preparation %} · {% translate "Download original upload" %} {% endif %}

diff --git a/efile_app/efile/templates/efile/preview_documents.html b/efile_app/efile/templates/efile/preview_documents.html index 91e80f8f..7dd235c2 100644 --- a/efile_app/efile/templates/efile/preview_documents.html +++ b/efile_app/efile/templates/efile/preview_documents.html @@ -20,7 +20,9 @@

{% translate "Check your filing PDFs" %}

{{ document.name|default:document.original_filename }}

- {% if document.preparation == "converted" %}{% translate "Converted from Word to PDF. Original:" %} {{ document.original_filename }} + {% if not document.preparation %} + {% translate "This upload still needs preparation. Reload this page to try again, or replace the document." %} + {% elif document.preparation == "converted" %}{% translate "Converted from Word to PDF. Original:" %} {{ document.original_filename }} {% elif document.preparation == "flattened" %} {% translate "PDF form fields locked for filing." %} {% elif document.preparation == "converted_flattened" %} @@ -35,7 +37,8 @@

{{ document.name|default:document.original_filename }}

type="checkbox" name="reviewed_document" value="{{ document.pk }}" - required /> + required + {% if not document.preparation %}disabled{% endif %} /> {% blocktranslate with name=document.name %}I checked every page of {{ name }} and the document is correct.{% endblocktranslate %}
@@ -43,7 +46,9 @@

{{ document.name|default:document.original_filename }}

{% translate "Replace or remove documents" %} - +
diff --git a/efile_app/efile/tests/helpers.py b/efile_app/efile/tests/helpers.py index 2cd5ff9e..602d8d67 100644 --- a/efile_app/efile/tests/helpers.py +++ b/efile_app/efile/tests/helpers.py @@ -10,3 +10,14 @@ def accepted_submission(draft, **fields): "confirm_submission": True, **fields, } + + +def reviewed_document(**fields): + """A previously prepared and acknowledged document for downstream flow tests.""" + from django.utils import timezone + + from efile.models import FilingDocument + + fields.setdefault("preparation", "unchanged") + fields.setdefault("preparation_reviewed_at", timezone.now()) + return FilingDocument.objects.create(**fields) diff --git a/efile_app/efile/tests/test_ai_opt_out.py b/efile_app/efile/tests/test_ai_opt_out.py index bef03123..b8c61f9a 100644 --- a/efile_app/efile/tests/test_ai_opt_out.py +++ b/efile_app/efile/tests/test_ai_opt_out.py @@ -24,6 +24,7 @@ ) from efile.services.drafts import create_draft from efile.services.taxonomy_classification import HierarchicalDocumentClassifier +from efile.tests.helpers import reviewed_document from efile.tests.pdf_helpers import pdf_bytes SYNTHETIC_PDFS = Path(__file__).resolve().parents[3] / "benchmarking/synthetic/filled_pdfs/flattened" @@ -103,7 +104,7 @@ def test_keyword_analysis_identifies_a_form_without_calling_a_model(): @pytest.mark.django_db def test_worker_reads_an_opted_out_document_with_keywords_only(opted_out_draft): - document = FilingDocument.objects.create( + document = reviewed_document( draft=opted_out_draft, role=FilingDocument.Role.LEAD, name="MA-03.pdf", @@ -148,7 +149,7 @@ def test_upload_page_offers_the_choice_and_says_what_still_happens(client, opted assert "AI is off for this filing." in page -@pytest.mark.django_db +@pytest.mark.django_db(transaction=True) def test_uploading_saves_the_choice_before_analysis_is_queued(client, opted_out_draft): opted_out_draft.ai_assistance_opted_out = False opted_out_draft.save(update_fields=["ai_assistance_opted_out", "updated_at"]) @@ -172,7 +173,7 @@ def test_uploading_saves_the_choice_before_analysis_is_queued(client, opted_out_ @pytest.mark.django_db def test_changing_the_choice_after_upload_drops_the_old_reading_and_re_runs(client, opted_out_draft): - document = FilingDocument.objects.create( + document = reviewed_document( draft=opted_out_draft, role=FilingDocument.Role.LEAD, name="complaint.pdf", @@ -222,7 +223,7 @@ def test_choosing_before_any_upload_saves_without_queueing_anything(client, opte @pytest.mark.django_db def test_review_screen_credits_the_keyword_search_rather_than_a_reading(client, opted_out_draft): - FilingDocument.objects.create( + reviewed_document( draft=opted_out_draft, role=FilingDocument.Role.LEAD, name="complaint.pdf", diff --git a/efile_app/efile/tests/test_court_selection.py b/efile_app/efile/tests/test_court_selection.py index 0a70d6c4..e748fdde 100644 --- a/efile_app/efile/tests/test_court_selection.py +++ b/efile_app/efile/tests/test_court_selection.py @@ -21,6 +21,7 @@ is_non_filing_court, selector_config, ) +from efile.tests.helpers import reviewed_document ILLINOIS_COURTS = [ {"value": "TSUPCRT", "text": "Supreme Court of Illinois"}, @@ -435,7 +436,7 @@ def draft(self, client, django_user_model): user = django_user_model.objects.create_user(username="court-user", tyler_jurisdiction="illinois") draft = FilingDraft.objects.create(user=user, jurisdiction="illinois", workflow_version=2) - FilingDocument.objects.create(draft=draft, role=FilingDocument.Role.LEAD, name="petition.pdf") + reviewed_document(draft=draft, role=FilingDocument.Role.LEAD, name="petition.pdf") client.force_login(user) session = client.session session[CURRENT_DRAFT_SESSION_KEY] = draft.pk diff --git a/efile_app/efile/tests/test_current_draft_selection.py b/efile_app/efile/tests/test_current_draft_selection.py index c7248712..e69447a7 100644 --- a/efile_app/efile/tests/test_current_draft_selection.py +++ b/efile_app/efile/tests/test_current_draft_selection.py @@ -12,6 +12,7 @@ from efile.models import FilingDocument, FilingDraft from efile.services.current_drafts import CURRENT_DRAFT_SESSION_KEY from efile.services.filing_plans import ensure_plan_for_draft +from efile.tests.helpers import reviewed_document from efile.workflow import ExistingCase, WorkflowStepKey OPTIONS_URL = reverse("efile_options", kwargs={"jurisdiction": "illinois"}) @@ -39,7 +40,7 @@ def last_months_filing(user): case_category_name="Miscellaneous", case_type_name="Name Change", ) - FilingDocument.objects.create( + reviewed_document( draft=draft, role=FilingDocument.Role.LEAD, sort_order=0, diff --git a/efile_app/efile/tests/test_document_extractions.py b/efile_app/efile/tests/test_document_extractions.py index df208579..c94c3a39 100644 --- a/efile_app/efile/tests/test_document_extractions.py +++ b/efile_app/efile/tests/test_document_extractions.py @@ -17,6 +17,7 @@ ) from efile.services.extraction_fields import normalize_document_evidence, normalize_extracted_fields from efile.services.taxonomy_classification import ClassificationRun, HierarchicalDocumentClassifier +from efile.tests.helpers import reviewed_document from efile.workflow import ExistingCase @@ -72,7 +73,7 @@ def test_normalization_canonicalizes_common_party_name_keys(normalizer): @pytest.mark.django_db def test_worker_caps_pages_and_persists_the_complete_payload(extraction_draft): - document = FilingDocument.objects.create( + document = reviewed_document( draft=extraction_draft, role=FilingDocument.Role.LEAD, name="petition.pdf", @@ -124,7 +125,7 @@ def test_worker_reads_a_real_uploaded_pdf_before_classification(extraction_draft """Do not let the standard worker test regress to a fully mocked document.""" source_pdf = Path(__file__).resolve().parents[3] / "benchmarking/synthetic/filled_pdfs/flattened/MA-01.pdf" assert source_pdf.is_file() - document = FilingDocument.objects.create( + document = reviewed_document( draft=extraction_draft, role=FilingDocument.Role.LEAD, name="MA-01.pdf", @@ -195,7 +196,7 @@ def classify_source(_classifier, jurisdiction, evidence, source_text): @pytest.mark.django_db def test_review_waits_for_background_analysis(client, extraction_draft): authorize(client, extraction_draft) - document = FilingDocument.objects.create( + document = reviewed_document( draft=extraction_draft, role=FilingDocument.Role.LEAD, name="petition.pdf", @@ -211,7 +212,7 @@ def test_review_waits_for_background_analysis(client, extraction_draft): @pytest.mark.django_db def test_status_endpoint_reports_when_review_is_ready(client, extraction_draft): authorize(client, extraction_draft) - document = FilingDocument.objects.create( + document = reviewed_document( draft=extraction_draft, role=FilingDocument.Role.LEAD, name="petition.pdf", @@ -243,7 +244,7 @@ def test_review_shows_only_the_document_summary_and_fields_the_filer_can_edit(cl values it does not use (like amounts) are not shown at all.""" authorize(client, extraction_draft) - FilingDocument.objects.create( + reviewed_document( draft=extraction_draft, role=FilingDocument.Role.LEAD, name="petition.pdf", @@ -305,7 +306,7 @@ def _post_review(client, jurisdiction="illinois", **fields): @pytest.mark.django_db def test_missing_acknowledgement_shows_an_error_beside_the_checkbox_and_keeps_edits(client, extraction_draft): authorize(client, extraction_draft) - FilingDocument.objects.create(draft=extraction_draft, role=FilingDocument.Role.LEAD, name="petition.pdf") + reviewed_document(draft=extraction_draft, role=FilingDocument.Role.LEAD, name="petition.pdf") extraction_draft.extracted_guesses = {"document title": "Complaint", "case title": "Old title"} extraction_draft.save(update_fields=["extracted_guesses", "updated_at"]) @@ -333,7 +334,7 @@ def test_missing_acknowledgement_shows_an_error_beside_the_checkbox_and_keeps_ed @pytest.mark.django_db def test_acknowledging_after_the_error_lets_the_filer_continue(client, extraction_draft): authorize(client, extraction_draft) - FilingDocument.objects.create(draft=extraction_draft, role=FilingDocument.Role.LEAD, name="petition.pdf") + reviewed_document(draft=extraction_draft, role=FilingDocument.Role.LEAD, name="petition.pdf") extraction_draft.extracted_guesses = {"document title": "Complaint"} extraction_draft.save(update_fields=["extracted_guesses", "updated_at"]) @@ -358,7 +359,7 @@ def test_acknowledging_after_the_error_lets_the_filer_continue(client, extractio ) def test_no_acknowledgement_is_demanded_when_none_is_shown(client, extraction_draft, extracted_guesses): authorize(client, extraction_draft) - FilingDocument.objects.create(draft=extraction_draft, role=FilingDocument.Role.LEAD, name="petition.pdf") + reviewed_document(draft=extraction_draft, role=FilingDocument.Role.LEAD, name="petition.pdf") extraction_draft.extracted_guesses = extracted_guesses extraction_draft.save(update_fields=["extracted_guesses", "updated_at"]) @@ -373,7 +374,7 @@ def test_no_acknowledgement_is_demanded_when_none_is_shown(client, extraction_dr def test_vermont_review_uses_court_form_terms_and_neutral_copy(client, django_user_model): user = django_user_model.objects.create_user(username="vermont-reviewer", tyler_jurisdiction="vermont") draft = FilingDraft.objects.create(user=user, jurisdiction="vermont", workflow_version=2) - FilingDocument.objects.create(draft=draft, role=FilingDocument.Role.LEAD, name="complaint.pdf") + reviewed_document(draft=draft, role=FilingDocument.Role.LEAD, name="complaint.pdf") authorize(client, draft) content = client.get(reverse("extraction_review", kwargs={"jurisdiction": "vermont"})).content.decode() @@ -389,7 +390,7 @@ def test_vermont_review_uses_court_form_terms_and_neutral_copy(client, django_us @pytest.mark.django_db def test_failed_extraction_allows_review_with_neutral_failure_copy(client, extraction_draft): authorize(client, extraction_draft) - document = FilingDocument.objects.create( + document = reviewed_document( draft=extraction_draft, role=FilingDocument.Role.LEAD, name="petition.pdf", @@ -418,7 +419,7 @@ def test_management_command_runs_once_with_no_jobs(): def test_management_command_processes_and_retries_failures(extraction_draft): from django.core.management import call_command - document = FilingDocument.objects.create( + document = reviewed_document( draft=extraction_draft, role=FilingDocument.Role.LEAD, name="petition.pdf", diff --git a/efile_app/efile/tests/test_document_prep.py b/efile_app/efile/tests/test_document_prep.py index b880bf94..3e1b633c 100644 --- a/efile_app/efile/tests/test_document_prep.py +++ b/efile_app/efile/tests/test_document_prep.py @@ -7,6 +7,7 @@ from efile.models import FilingDocument, FilingDraft from efile.services.current_drafts import CURRENT_DRAFT_SESSION_KEY from efile.services.drafts import read_upload_data +from efile.tests.helpers import reviewed_document from efile.workflow import ExistingCase, WorkflowStepKey @@ -31,7 +32,7 @@ def document_draft(client, django_user_model): case_type_code="200", current_step=WorkflowStepKey.DOCUMENT_CHECKLIST, ) - FilingDocument.objects.create( + reviewed_document( draft=draft, role=FilingDocument.Role.LEAD, sort_order=0, @@ -231,7 +232,7 @@ def test_organize_shows_no_radio_choice_for_a_single_document(client, document_d def test_organize_shows_radio_choice_for_multiple_documents(client, document_draft): document_draft.document_checklist_acknowledged = True document_draft.save(update_fields=["document_checklist_acknowledged", "updated_at"]) - FilingDocument.objects.create( + reviewed_document( draft=document_draft, role=FilingDocument.Role.SUPPORTING, sort_order=0, @@ -248,13 +249,13 @@ def test_organize_shows_radio_choice_for_multiple_documents(client, document_dra @pytest.mark.django_db def test_organize_saves_details_and_supporting_order(client, document_draft): - first = FilingDocument.objects.create( + first = reviewed_document( draft=document_draft, role=FilingDocument.Role.SUPPORTING, sort_order=0, name="first.pdf", ) - second = FilingDocument.objects.create( + second = reviewed_document( draft=document_draft, role=FilingDocument.Role.SUPPORTING, sort_order=1, diff --git a/efile_app/efile/tests/test_document_preparation.py b/efile_app/efile/tests/test_document_preparation.py index 1f0708ee..6a31e159 100644 --- a/efile_app/efile/tests/test_document_preparation.py +++ b/efile_app/efile/tests/test_document_preparation.py @@ -12,6 +12,14 @@ from efile.tests.pdf_helpers import docx_bytes, pdf_bytes +@pytest.fixture(autouse=True) +def configured_conversion_service(settings): + # Mocked HTTP tests must not depend on the developer's .env. + settings.GOTENBERG_URL = "https://conversion.example.invalid" + settings.GOTENBERG_USERNAME = "" + settings.GOTENBERG_PASSWORD = "" + + def upload(data, name="form.pdf"): return SimpleUploadedFile(name, data) @@ -143,3 +151,46 @@ def test_oversize_prepared_output_is_rejected(): ): with pytest.raises(PreparationError, match="exceeds"): prepare_document(upload(docx_bytes(), "brief.docx"), "vermont") + + +def test_indirect_annotations_are_resolved_before_inspection(): + from pypdf import PdfWriter + from pypdf.generic import NameObject + + writer = PdfWriter(clone_from=PdfReader(io.BytesIO(pdf_bytes(form_value="Filled answer")))) + writer.pages[0][NameObject("/Annots")] = writer._add_object(writer.pages[0]["/Annots"]) + output = io.BytesIO() + writer.write(output) + with patch( + "efile.services.document_preparation.requests.post", return_value=service_response(pdf_bytes("Filled answer")) + ): + assert prepare_document(upload(output.getvalue()), "vermont").operation == "flattened" + + +@pytest.mark.parametrize("kind", ["comb", "password", "hidden", "formatted"]) +def test_special_field_appearances_do_not_require_literal_value_text(kind): + from pypdf import PdfWriter + from pypdf.generic import DictionaryObject, NameObject, NumberObject, TextStringObject + + writer = PdfWriter(clone_from=PdfReader(io.BytesIO(pdf_bytes(form_value="123456")))) + widget = cast(Any, writer.pages[0]["/Annots"])[0].get_object() + if kind == "comb": + widget[NameObject("/Ff")] = NumberObject(1 << 24) + elif kind == "password": + widget[NameObject("/Ff")] = NumberObject(1 << 13) + elif kind == "hidden": + widget[NameObject("/F")] = NumberObject(2) + else: + widget[NameObject("/AA")] = DictionaryObject( + { + NameObject("/F"): DictionaryObject( + {NameObject("/S"): NameObject("/JavaScript"), NameObject("/JS"): TextStringObject("formatNumber()")} + ) + } + ) + output = io.BytesIO() + writer.write(output) + with patch( + "efile.services.document_preparation.requests.post", return_value=service_response(pdf_bytes("12/34/56")) + ): + assert prepare_document(upload(output.getvalue()), "vermont").operation == "flattened" diff --git a/efile_app/efile/tests/test_document_previews.py b/efile_app/efile/tests/test_document_previews.py index 882d2e5b..b58590e7 100644 --- a/efile_app/efile/tests/test_document_previews.py +++ b/efile_app/efile/tests/test_document_previews.py @@ -158,3 +158,142 @@ def test_preview_return_destinations_are_restricted(client, preview_draft): assert "attacker" not in response.url response = approve(client, preview_draft, return_to="payment") assert "/payment/" in response.url + + +def test_lead_key_swap_resets_approval_and_cleans_only_unused_copies(preview_draft, django_capture_on_commit_callbacks): + lead = preview_draft.documents.get() + lead.preparation_reviewed_at = timezone.now() + lead.save() + old_key, original_key = lead.s3_key, lead.original_s3_key + shared = FilingDraft.objects.create(user=preview_draft.user, jurisdiction="vermont") + FilingDocument.objects.create(draft=shared, role="lead", s3_key=old_key) + handler = MagicMock() + with ( + patch("efile.utils.s3_upload_handler.S3UploadHandler", return_value=handler), + django_capture_on_commit_callbacks(execute=True), + ): + write_upload_data(preview_draft, {"files": {"lead": {"s3_key": "replacement.pdf", "name": "new.pdf"}}}) + lead.refresh_from_db() + assert lead.preparation_reviewed_at is None + assert lead.preparation == "" + assert lead.original_s3_key == "" + assert lead.original_filename == "new.pdf" + handler.delete_file.assert_called_once_with(original_key) + + +def test_legacy_word_support_is_prepared_before_it_can_be_approved(client, preview_draft): + from efile.tests.pdf_helpers import docx_bytes + from efile.tests.test_document_preparation import service_response + + write_upload_data(preview_draft, {"files": {"supporting": [{"s3_key": "legacy.docx", "name": "legacy.docx"}]}}) + supporting = preview_draft.documents.get(role="supporting") + assert approve(client, preview_draft).status_code == 200 + supporting.refresh_from_db() + assert supporting.preparation_reviewed_at is None + handler = MagicMock() + handler.bucket_name = "private" + handler.get_public_url.return_value = "https://synthetic.invalid/prepared.pdf" + handler.s3_client.get_object.return_value = {"Body": io.BytesIO(docx_bytes())} + handler.upload_file.side_effect = [ + {"success": True, "key": "retained.docx"}, + {"success": True, "key": "prepared.pdf"}, + ] + with ( + patch("efile.views.document_previews.S3UploadHandler", return_value=handler), + patch("efile.services.document_preparation.requests.post", return_value=service_response(pdf_bytes())), + patch("efile.services.document_preparation.settings.GOTENBERG_URL", "https://synthetic.invalid"), + ): + assert client.get(url("preview_documents", preview_draft)).status_code == 200 + supporting.refresh_from_db() + assert supporting.preparation == "converted" + assert supporting.s3_key == "prepared.pdf" + assert supporting.original_s3_key == "retained.docx" + assert supporting.preparation_reviewed_at is None + assert approve(client, preview_draft).status_code == 302 + + +def test_legacy_rows_and_changed_preparation_cannot_bypass_submission(client, preview_draft): + doc = preview_draft.documents.get() + doc.preparation_reviewed_at = timezone.now() + doc.save() + fingerprint = preview_fingerprint([doc]) + doc.original_s3_key = "new-original" + doc.save() + assert preview_fingerprint([doc]) != fingerprint + assert approve(client, preview_draft, preview_fingerprint=fingerprint).status_code == 200 + doc.preparation = "" + doc.save() + with patch("efile.views.submission.forward_final_filing") as forward: + assert ( + client.post(reverse("submit_final_filing"), data="{}", content_type="application/json").status_code == 412 + ) + forward.assert_not_called() + + +def test_extraction_callback_is_discarded_when_outer_transaction_rolls_back(preview_draft): + from django.db import transaction + + preview_draft.documents.all().delete() + handler = MagicMock() + handler.upload_file.return_value = {"success": True, "key": "new.pdf"} + handler.get_public_url.return_value = "https://synthetic.invalid/new.pdf" + with ( + patch("efile.services.document_uploads.S3UploadHandler", return_value=handler), + patch("efile.services.document_uploads.queue_document_extraction") as queue, + ): + with pytest.raises(RuntimeError): + with transaction.atomic(): + upload_files(preview_draft, [SimpleUploadedFile("new.pdf", pdf_bytes())], "vermont") + queue.assert_not_called() + raise RuntimeError("Rollback") + queue.assert_not_called() + assert not preview_draft.documents.exists() + + +def test_legacy_pdf_is_flattened_and_cannot_be_acknowledged_after_preparation_failure(client, preview_draft): + from efile.tests.test_document_preparation import service_response + + doc = preview_draft.documents.get() + doc.preparation = "" + doc.original_s3_key = "" + doc.save() + source = pdf_bytes(form_value="Stored answer") + handler = MagicMock() + handler.bucket_name = "private" + handler.get_public_url.return_value = "https://synthetic.invalid/prepared.pdf" + handler.s3_client.get_object.side_effect = lambda **kwargs: {"Body": io.BytesIO(source)} + handler.upload_file.side_effect = [{"success": True, "key": "retained.pdf"}, {"success": True, "key": "flat.pdf"}] + doc.original_filename = "source.pdf" + doc.save() + with ( + patch("efile.views.document_previews.S3UploadHandler", return_value=handler), + patch("efile.services.document_preparation.settings.GOTENBERG_URL", "https://synthetic.invalid"), + patch( + "efile.services.document_preparation.requests.post", + return_value=service_response(pdf_bytes("Stored answer")), + ), + ): + assert client.get(url("preview_documents", preview_draft)).status_code == 200 + doc.refresh_from_db() + assert doc.preparation == "flattened" + assert doc.s3_key == "flat.pdf" + assert doc.original_s3_key == "retained.pdf" + assert doc.preparation_reviewed_at is None + # A structurally valid response that drops text remains blocked. + doc.preparation = "" + doc.preparation_reviewed_at = None + doc.save() + with ( + patch("efile.views.document_previews.S3UploadHandler", return_value=handler), + patch("efile.services.document_preparation.settings.GOTENBERG_URL", "https://synthetic.invalid"), + patch( + "efile.services.document_preparation.requests.post", + return_value=service_response(pdf_bytes("Dropped answer")), + ), + ): + response = client.get(url("preview_documents", preview_draft)) + assert response.status_code == 422 + assert b"filled-in text could not be preserved" in response.content + assert approve(client, preview_draft).status_code == 200 + doc.refresh_from_db() + assert doc.preparation_reviewed_at is None diff --git a/efile_app/efile/tests/test_durable_drafts.py b/efile_app/efile/tests/test_durable_drafts.py index 4f9c8052..ba2b3fc3 100644 --- a/efile_app/efile/tests/test_durable_drafts.py +++ b/efile_app/efile/tests/test_durable_drafts.py @@ -2,6 +2,7 @@ import pytest from django.urls import reverse +from django.utils import timezone from efile.models import FilingDocument, FilingDraft, FilingParty, FilingPlan from efile.services.current_drafts import CURRENT_DRAFT_SESSION_KEY, get_current_draft @@ -14,7 +15,7 @@ ) from efile.services.efsp_payload import PayloadValidationError from efile.services.fee_quotes import record_fee_quote -from efile.tests.helpers import accepted_submission +from efile.tests.helpers import accepted_submission, reviewed_document from efile.workflow import WorkflowStepKey, get_workflow_step_choices @@ -40,6 +41,7 @@ def _prepare_submission(client, draft, jurisdiction="illinois"): """Populate the draft (the source of truth) and session so submit can run.""" write_case_data(draft, {"court": "cook:cd"}) write_upload_data(draft, {"files": {"lead": {"url": "https://example.com/petition.pdf"}}}) + draft.documents.update(preparation="unchanged", preparation_reviewed_at=timezone.now()) # Submission is only offered against a quote that still prices the filing. record_fee_quote(draft, "0.00", []) session = client.session @@ -526,7 +528,7 @@ def fake_post(*_args, **_kwargs): jurisdiction="illinois", current_step=WorkflowStepKey.REVIEW, ) - FilingDocument.objects.create( + reviewed_document( draft=draft, role=FilingDocument.Role.LEAD, sort_order=0, diff --git a/efile_app/efile/tests/test_end_to_end_new_flow_states.py b/efile_app/efile/tests/test_end_to_end_new_flow_states.py index 59cc1d7d..3f0c2454 100644 --- a/efile_app/efile/tests/test_end_to_end_new_flow_states.py +++ b/efile_app/efile/tests/test_end_to_end_new_flow_states.py @@ -42,7 +42,7 @@ def make_dummy_pdf(): return buf.getvalue() -@pytest.mark.django_db +@pytest.mark.django_db(transaction=True) @pytest.mark.parametrize( ( "jurisdiction", diff --git a/efile_app/efile/tests/test_extracted_parties.py b/efile_app/efile/tests/test_extracted_parties.py index 3e3f840e..3c7b8e33 100644 --- a/efile_app/efile/tests/test_extracted_parties.py +++ b/efile_app/efile/tests/test_extracted_parties.py @@ -22,6 +22,7 @@ guess_filer_party_type, match_party_type, ) +from efile.tests.helpers import reviewed_document from efile.workflow import ExistingCase, WorkflowStepKey PARTY_TYPES = [ @@ -58,7 +59,7 @@ def review_draft(db, django_user_model): current_step=WorkflowStepKey.EXTRACTION_REVIEW, extracted_guesses=dict(GUESSES), ) - FilingDocument.objects.create(draft=draft, role=FilingDocument.Role.LEAD, name="complaint.pdf") + reviewed_document(draft=draft, role=FilingDocument.Role.LEAD, name="complaint.pdf") return draft diff --git a/efile_app/efile/tests/test_extraction_claims.py b/efile_app/efile/tests/test_extraction_claims.py index cee10bb1..0fe2628b 100644 --- a/efile_app/efile/tests/test_extraction_claims.py +++ b/efile_app/efile/tests/test_extraction_claims.py @@ -19,13 +19,14 @@ record_extraction_failure, renew_extraction_lease, ) +from efile.tests.helpers import reviewed_document @pytest.fixture def lead(db, django_user_model): user = django_user_model.objects.create_user(username="claim-test") draft = FilingDraft.objects.create(user=user, jurisdiction="illinois") - return FilingDocument.objects.create(draft=draft, role=FilingDocument.Role.LEAD, name="lead.pdf", s3_key="lead.pdf") + return reviewed_document(draft=draft, role=FilingDocument.Role.LEAD, name="lead.pdf", s3_key="lead.pdf") def expire(job): diff --git a/efile_app/efile/tests/test_fee_quotes.py b/efile_app/efile/tests/test_fee_quotes.py index fb42bdf6..b0a3d130 100644 --- a/efile_app/efile/tests/test_fee_quotes.py +++ b/efile_app/efile/tests/test_fee_quotes.py @@ -17,7 +17,7 @@ quote_from_efsp_response, record_fee_quote, ) -from efile.tests.helpers import accepted_submission +from efile.tests.helpers import accepted_submission, reviewed_document from efile.workflow import WorkflowStepKey REVIEW_URL = reverse("case_review", kwargs={"jurisdiction": "illinois"}) @@ -70,7 +70,7 @@ def draft(client, django_user_model): selected_payment_account_name="Card ending in 4242", selected_payment_account_type="CC", ) - FilingDocument.objects.create( + reviewed_document( draft=draft, role=FilingDocument.Role.LEAD, sort_order=0, @@ -186,7 +186,7 @@ def _lead(draft): "optional service added": lambda draft: FilingDocument.objects.filter(pk=_lead(draft).pk).update( requested_optional_services=["certified-copy", "service-by-mail"] ), - "document added": lambda draft: FilingDocument.objects.create( + "document added": lambda draft: reviewed_document( draft=draft, role=FilingDocument.Role.SUPPORTING, sort_order=0, filing_type_code="exhibit" ), "party added": lambda draft: FilingParty.objects.create( diff --git a/efile_app/efile/tests/test_filer_role.py b/efile_app/efile/tests/test_filer_role.py index 1c0a7117..322ca2dc 100644 --- a/efile_app/efile/tests/test_filer_role.py +++ b/efile_app/efile/tests/test_filer_role.py @@ -12,6 +12,7 @@ from efile.services.current_drafts import CURRENT_DRAFT_SESSION_KEY from efile.services.filing_plans import create_draft_from_plan, ensure_plan_for_draft, set_checklist_progress from efile.services.people import guess_filer_party_type +from efile.tests.helpers import reviewed_document from efile.workflow import ExistingCase, WorkflowStepKey CHECKLIST_URL = reverse("document_checklist", kwargs={"jurisdiction": "illinois"}) @@ -45,7 +46,7 @@ def draft(user): current_step=WorkflowStepKey.DOCUMENT_CHECKLIST, **EVICTION_CASE, ) - FilingDocument.objects.create( + reviewed_document( draft=draft, role=FilingDocument.Role.LEAD, sort_order=0, diff --git a/efile_app/efile/tests/test_filing_integrity.py b/efile_app/efile/tests/test_filing_integrity.py index 2c59c6a7..eb247b19 100644 --- a/efile_app/efile/tests/test_filing_integrity.py +++ b/efile_app/efile/tests/test_filing_integrity.py @@ -7,7 +7,7 @@ from django.test import Client from django.urls import reverse -from efile.models import FilingDocument, FilingDraft, FilingParty +from efile.models import FilingDraft, FilingParty from efile.services.current_drafts import CURRENT_DRAFT_SESSION_KEY from efile.services.drafts import read_case_data, write_case_data from efile.services.extracted_parties import review_rows, save_reviewed_parties @@ -17,7 +17,7 @@ normalize_extracted_fields, ) from efile.services.fee_quotes import record_fee_quote -from efile.tests.helpers import accepted_submission +from efile.tests.helpers import accepted_submission, reviewed_document @pytest.mark.parametrize( @@ -87,7 +87,7 @@ def test_family_form_children_remain_evidence_until_explicitly_added(draft): def test_old_extraction_placeholders_are_not_prefilled_or_displayed(draft): draft.extracted_guesses = {"case title": "unknown", "docket number": "N/A", "form revision": "Not provided"} draft.save() - FilingDocument.objects.create(draft=draft, role="lead", name="petition.pdf") + reviewed_document(draft=draft, role="lead", name="petition.pdf") response = signed_in(draft.user).get(route("extraction_review", draft)) assert response.status_code == 200 assert response.context["document_summary_details"] == [] @@ -181,7 +181,7 @@ def test_backfill_repairs_existing_submitted_draft_summary(draft): from django.apps import apps from django.db import connection - FilingDocument.objects.create(draft=draft, role="lead", filing_type_code="27959", filing_type_name="Complaint") + reviewed_document(draft=draft, role="lead", filing_type_code="27959", filing_type_name="Complaint") FilingDraft.objects.filter(pk=draft.pk).update( status=FilingDraft.Status.SUBMITTED, filing_type_code="", filing_type_name="" ) @@ -192,9 +192,7 @@ def test_backfill_repairs_existing_submitted_draft_summary(draft): def test_primary_type_tracks_edits_clearing_and_lead_deletion(draft): - lead = FilingDocument.objects.create( - draft=draft, role="lead", filing_type_code="27959", filing_type_name="Complaint" - ) + lead = reviewed_document(draft=draft, role="lead", filing_type_code="27959", filing_type_name="Complaint") stale_draft = FilingDraft.objects.get(pk=draft.pk) lead.filing_type_code = "123" lead.filing_type_name = "Petition" diff --git a/efile_app/efile/tests/test_filing_on_behalf.py b/efile_app/efile/tests/test_filing_on_behalf.py index 8dcd5fd7..d76b13fb 100644 --- a/efile_app/efile/tests/test_filing_on_behalf.py +++ b/efile_app/efile/tests/test_filing_on_behalf.py @@ -20,6 +20,7 @@ from efile.services.current_drafts import CURRENT_DRAFT_SESSION_KEY from efile.services.drafts import read_case_data from efile.services.people import NOT_A_PARTY, absorb_filer_duplicates, filing_parties, party_is_complete +from efile.tests.helpers import reviewed_document from efile.workflow import ExistingCase, WorkflowStepKey PARTIES_URL = reverse("parties", kwargs={"jurisdiction": "illinois"}) @@ -50,7 +51,7 @@ def draft(client, django_user_model): current_step=WorkflowStepKey.PARTIES, document_checklist_acknowledged=True, ) - FilingDocument.objects.create( + reviewed_document( draft=draft, role=FilingDocument.Role.LEAD, sort_order=0, diff --git a/efile_app/efile/tests/test_filing_path_navigation.py b/efile_app/efile/tests/test_filing_path_navigation.py index 88003363..bff0a136 100644 --- a/efile_app/efile/tests/test_filing_path_navigation.py +++ b/efile_app/efile/tests/test_filing_path_navigation.py @@ -19,6 +19,7 @@ change_filing_path, filing_path_conflict, ) +from efile.tests.helpers import reviewed_document from efile.workflow import ExistingCase, WorkflowStepKey, get_resume_step_url, get_visible_workflow J = {"jurisdiction": "illinois"} @@ -69,7 +70,7 @@ def back_link(content): def lead_with_evidence(draft, *, phase, title="Answer", filing_type="answer-code"): - lead = FilingDocument.objects.create( + lead = reviewed_document( draft=draft, role=FilingDocument.Role.LEAD, name="answer.pdf", @@ -373,7 +374,7 @@ def test_the_conflict_note_on_a_resent_form_describes_the_answer_submitted(signe @pytest.mark.django_db def test_change_filing_path_leaves_a_first_answer_and_a_same_answer_alone(user): draft = FilingDraft.objects.create(user=user, jurisdiction="illinois", quoted_fee_total="10.00") - FilingDocument.objects.create(draft=draft, role=FilingDocument.Role.LEAD, filing_type_code="x") + reviewed_document(draft=draft, role=FilingDocument.Role.LEAD, filing_type_code="x") first = change_filing_path(draft, ExistingCase.NEW) assert first.changed and not first.switched @@ -386,7 +387,7 @@ def test_change_filing_path_leaves_a_first_answer_and_a_same_answer_alone(user): @pytest.mark.django_db def test_unsure_to_new_keeps_filing_types_since_the_court_lists_are_the_same(user): draft = FilingDraft.objects.create(user=user, jurisdiction="illinois", existing_case=ExistingCase.UNSURE) - FilingDocument.objects.create(draft=draft, role=FilingDocument.Role.LEAD, filing_type_code="complaint") + reviewed_document(draft=draft, role=FilingDocument.Role.LEAD, filing_type_code="complaint") change = change_filing_path(draft, ExistingCase.NEW) diff --git a/efile_app/efile/tests/test_filing_plan_actions.py b/efile_app/efile/tests/test_filing_plan_actions.py index 9f521f8d..3cfa4dc9 100644 --- a/efile_app/efile/tests/test_filing_plan_actions.py +++ b/efile_app/efile/tests/test_filing_plan_actions.py @@ -20,6 +20,7 @@ set_checklist_answers, set_checklist_progress, ) +from efile.tests.helpers import reviewed_document from efile.workflow import ExistingCase, WorkflowStepKey CHECKLIST_URL = reverse("document_checklist", kwargs={"jurisdiction": "illinois"}) @@ -58,7 +59,7 @@ def draft(user): case_type_code="78346", case_type_name="Name Change", ) - FilingDocument.objects.create( + reviewed_document( draft=draft, role=FilingDocument.Role.LEAD, sort_order=0, @@ -77,7 +78,7 @@ def signed_in(client, user, draft): def a_supporting_document(draft, name="fee-waiver.pdf"): - return FilingDocument.objects.create( + return reviewed_document( draft=draft, role=FilingDocument.Role.SUPPORTING, sort_order=FilingDocument.objects.filter(draft=draft, role=FilingDocument.Role.SUPPORTING).count(), diff --git a/efile_app/efile/tests/test_filing_plans.py b/efile_app/efile/tests/test_filing_plans.py index afb56556..8b9a7f07 100644 --- a/efile_app/efile/tests/test_filing_plans.py +++ b/efile_app/efile/tests/test_filing_plans.py @@ -16,6 +16,7 @@ set_checklist_answers, set_checklist_progress, ) +from efile.tests.helpers import reviewed_document from efile.workflow import ExistingCase, WorkflowStepKey @@ -39,7 +40,7 @@ def make_draft(user, **overrides): } fields.update(overrides) draft = FilingDraft.objects.create(user=user, **fields) - FilingDocument.objects.create( + reviewed_document( draft=draft, role=FilingDocument.Role.LEAD, sort_order=0, diff --git a/efile_app/efile/tests/test_handoff.py b/efile_app/efile/tests/test_handoff.py index ea36fe72..9fd63b53 100644 --- a/efile_app/efile/tests/test_handoff.py +++ b/efile_app/efile/tests/test_handoff.py @@ -8,6 +8,7 @@ from efile.models import FilingDraft, InterviewHandoff from efile.services.handoff import HandoffError, create_correction, effective_hints, resolve_metadata, unique_match +from efile.tests.helpers import reviewed_document from efile.tests.pdf_helpers import pdf_bytes pytestmark = pytest.mark.django_db @@ -696,3 +697,71 @@ def test_handoff_review_shows_case_identity_only_once_the_case_is_found( assert ("Doe v. Doe" in content) is shown assert ("24-FA-00123" in content) is shown + + +def test_invalid_pdf_requires_replacement_instead_of_retry(client, source, payload, storage): + invalid = b"%PDF- broken" + payload["documents"][0]["sha256"] = hashlib.sha256(invalid).hexdigest() + response = send(client, source, payload, invalid) + assert response.status_code == 422 + assert not FilingDraft.objects.exists() + storage.upload_file.assert_not_called() + + +def test_replacement_cleans_old_copies_and_preserves_shared_filing( + client, source, payload, storage, django_user_model, django_capture_on_commit_callbacks +): + from urllib.parse import parse_qs, urlsplit + + from django.utils import timezone + + send(client, source, payload) + draft = FilingDraft.objects.get() + draft.user = login(client, django_user_model) + draft.save() + document = draft.documents.get() + document.original_s3_key = "old-original.docx" + document.preparation_reviewed_at = timezone.now() + document.save() + other = FilingDraft.objects.create(user=draft.user, jurisdiction="vermont") + reviewed_document(draft=other, role="lead", s3_key=document.s3_key) + token = parse_qs(urlsplit(client.post(reverse("return_to_interview", args=[draft.pk])).url).query)[ + "litefile_correction" + ][0] + payload["idempotency_key"] = "replace-cleanup" + storage.upload_file.return_value = {"success": True, "key": "replacement.pdf"} + with django_capture_on_commit_callbacks(execute=True): + response = client.post( + reverse("handoff_replace_documents"), + {"payload": json.dumps(payload), "complaint": SimpleUploadedFile("complaint.pdf", PDF)}, + **source, + HTTP_X_LITEFILE_CORRECTION=token, + ) + assert response.status_code == 200 + document.refresh_from_db() + assert document.s3_key == "replacement.pdf" + assert document.preparation_reviewed_at is None + storage.delete_file.assert_called_once_with("old-original.docx") + + +@pytest.mark.parametrize("status,expected", [(400, 422), (503, 503)]) +def test_handoff_distinguishes_conversion_rejection_from_service_outage( + client, source, payload, storage, status, expected +): + from efile.tests.pdf_helpers import docx_bytes + from efile.tests.test_document_preparation import service_response + + data = docx_bytes() + payload["documents"][0]["sha256"] = hashlib.sha256(data).hexdigest() + with ( + patch("efile.services.document_preparation.settings.GOTENBERG_URL", "https://synthetic.invalid"), + patch("efile.services.document_preparation.requests.post", return_value=service_response(b"", status=status)), + ): + response = client.post( + reverse("external_handoff"), + {"payload": json.dumps(payload), "complaint": SimpleUploadedFile("complaint.docx", data)}, + **source, + ) + assert response.status_code == expected + assert not FilingDraft.objects.exists() + storage.upload_file.assert_not_called() diff --git a/efile_app/efile/tests/test_my_drafts.py b/efile_app/efile/tests/test_my_drafts.py index d8e1a1df..b80968f7 100644 --- a/efile_app/efile/tests/test_my_drafts.py +++ b/efile_app/efile/tests/test_my_drafts.py @@ -10,6 +10,7 @@ from efile.models import FilingDocument, FilingDraft, FilingPlan from efile.services.current_drafts import CURRENT_DRAFT_SESSION_KEY +from efile.tests.helpers import reviewed_document from efile.workflow import ExistingCase, WorkflowStepKey DRAFTS_URL = reverse("my_drafts", kwargs={"jurisdiction": "illinois"}) @@ -81,7 +82,7 @@ def test_resuming_from_the_list_opens_that_draft_and_not_the_newest(client, user def test_throwing_a_draft_away_takes_it_out_of_every_list(client, user): sign_in(client, user) draft = make_draft(user, case_title="Started by mistake") - FilingDocument.objects.create(draft=draft, role=FilingDocument.Role.LEAD, sort_order=0, name="petition.pdf") + reviewed_document(draft=draft, role=FilingDocument.Role.LEAD, sort_order=0, name="petition.pdf") session = client.session session[CURRENT_DRAFT_SESSION_KEY] = draft.pk session.save() diff --git a/efile_app/efile/tests/test_parties_save_role.py b/efile_app/efile/tests/test_parties_save_role.py index 8a2528df..2b0f5cf7 100644 --- a/efile_app/efile/tests/test_parties_save_role.py +++ b/efile_app/efile/tests/test_parties_save_role.py @@ -16,6 +16,7 @@ from efile.models import FilingDocument, FilingDraft, FilingParty from efile.services.current_drafts import CURRENT_DRAFT_SESSION_KEY from efile.services.people import NOT_A_PARTY +from efile.tests.helpers import reviewed_document from efile.workflow import ExistingCase, WorkflowStepKey PARTIES_URL = reverse("parties", kwargs={"jurisdiction": "illinois"}) @@ -50,7 +51,7 @@ def draft(client, django_user_model): current_step=WorkflowStepKey.PARTIES, document_checklist_acknowledged=True, ) - FilingDocument.objects.create( + reviewed_document( draft=draft, role=FilingDocument.Role.LEAD, sort_order=0, diff --git a/efile_app/efile/tests/test_party_address_requirements.py b/efile_app/efile/tests/test_party_address_requirements.py index 84128ca0..b80ed246 100644 --- a/efile_app/efile/tests/test_party_address_requirements.py +++ b/efile_app/efile/tests/test_party_address_requirements.py @@ -5,6 +5,7 @@ from efile.models import FilingDocument, FilingDraft, FilingParty from efile.services.party_requirements import party_address_requirement from efile.services.people import party_is_complete +from efile.tests.helpers import reviewed_document @pytest.fixture @@ -71,7 +72,7 @@ def test_layered_config_can_require_address_by_party_filing_or_service(draft, ru last_name="Lee", ) if document_values: - FilingDocument.objects.create(draft=draft, role=FilingDocument.Role.LEAD, **document_values) + reviewed_document(draft=draft, role=FilingDocument.Role.LEAD, **document_values) with ( patch( diff --git a/efile_app/efile/tests/test_people_flow.py b/efile_app/efile/tests/test_people_flow.py index e07a9d01..3704892a 100644 --- a/efile_app/efile/tests/test_people_flow.py +++ b/efile_app/efile/tests/test_people_flow.py @@ -9,6 +9,7 @@ from efile.services.drafts import read_case_data from efile.services.party_requirements import AddressRequirement from efile.services.people import guess_filer_party_type +from efile.tests.helpers import reviewed_document from efile.workflow import ExistingCase, WorkflowStepKey PARTY_TYPES = [ @@ -112,7 +113,7 @@ def test_guess_filer_party_type_suggests_the_initiator_for_a_new_case(people_dra def test_guess_filer_party_type_suggests_the_respondent_for_an_answer(people_draft): people_draft.existing_case = ExistingCase.EXISTING people_draft.save(update_fields=["existing_case", "updated_at"]) - FilingDocument.objects.create( + reviewed_document( draft=people_draft, role=FilingDocument.Role.LEAD, name="answer.pdf", @@ -510,7 +511,7 @@ def test_case_questions_asks_for_amount_in_controversy_with_no_other_questions(c """The early "nothing to ask, skip ahead" exit used to fire even when a document's filing type required an amount in controversy, since it only checked the config-driven questions list.""" - FilingDocument.objects.create( + reviewed_document( draft=people_draft, role=FilingDocument.Role.LEAD, name="petition.pdf", @@ -527,7 +528,7 @@ def test_case_questions_asks_for_amount_in_controversy_with_no_other_questions(c @pytest.mark.django_db def test_case_questions_saves_a_valid_amount_in_controversy(client, people_draft): - FilingDocument.objects.create( + reviewed_document( draft=people_draft, role=FilingDocument.Role.LEAD, name="petition.pdf", @@ -549,7 +550,7 @@ def test_case_questions_saves_a_valid_amount_in_controversy(client, people_draft @pytest.mark.django_db def test_case_questions_rejects_a_missing_or_invalid_amount(client, people_draft): - FilingDocument.objects.create( + reviewed_document( draft=people_draft, role=FilingDocument.Role.LEAD, name="petition.pdf", @@ -599,7 +600,7 @@ def test_parties_routes_to_case_questions_when_amount_in_controversy_is_needed(c state="IL", zip_code="60602", ) - FilingDocument.objects.create( + reviewed_document( draft=people_draft, role=FilingDocument.Role.LEAD, name="petition.pdf", diff --git a/efile_app/efile/tests/test_reorganized_start.py b/efile_app/efile/tests/test_reorganized_start.py index a2b81ac3..df1287f5 100644 --- a/efile_app/efile/tests/test_reorganized_start.py +++ b/efile_app/efile/tests/test_reorganized_start.py @@ -7,6 +7,7 @@ from efile.models import DocumentExtraction, FilingDocument, FilingDraft from efile.services.current_drafts import CURRENT_DRAFT_SESSION_KEY +from efile.tests.helpers import reviewed_document from efile.tests.pdf_helpers import pdf_bytes from efile.workflow import ExistingCase, WorkflowStepKey @@ -42,7 +43,7 @@ def test_filing_path_saves_normalized_branch(client, reorganized_draft): assert reorganized_draft.current_step == WorkflowStepKey.UPLOAD_DOCUMENTS -@pytest.mark.django_db +@pytest.mark.django_db(transaction=True) def test_upload_documents_persists_files_and_queues_analysis(client, reorganized_draft): handler = MagicMock() handler._ensure_initialized.return_value = True @@ -74,14 +75,14 @@ def test_upload_documents_persists_files_and_queues_analysis(client, reorganized @pytest.mark.django_db def test_removing_analyzed_document_cleans_storage_and_stale_guesses(client, reorganized_draft): - lead = FilingDocument.objects.create( + lead = reviewed_document( draft=reorganized_draft, role=FilingDocument.Role.LEAD, sort_order=0, name="petition.pdf", s3_key="lead/petition.pdf", ) - supporting = FilingDocument.objects.create( + supporting = reviewed_document( draft=reorganized_draft, role=FilingDocument.Role.SUPPORTING, sort_order=0, @@ -111,7 +112,7 @@ def test_removing_analyzed_document_cleans_storage_and_stale_guesses(client, reo @pytest.mark.django_db def test_extraction_review_branches_new_case_to_checklist(client, reorganized_draft): - FilingDocument.objects.create( + reviewed_document( draft=reorganized_draft, role=FilingDocument.Role.LEAD, name="petition.pdf", @@ -142,7 +143,7 @@ def test_extraction_review_branches_new_case_to_checklist(client, reorganized_dr @pytest.mark.django_db def test_extraction_review_returns_to_review_when_edited_from_there(client, reorganized_draft): - FilingDocument.objects.create( + reviewed_document( draft=reorganized_draft, role=FilingDocument.Role.LEAD, name="petition.pdf", @@ -170,7 +171,7 @@ def test_extraction_review_returns_to_review_when_edited_from_there(client, reor @pytest.mark.django_db def test_extraction_review_new_case_requires_matched_court_and_type(client, reorganized_draft): - FilingDocument.objects.create( + reviewed_document( draft=reorganized_draft, role=FilingDocument.Role.LEAD, name="petition.pdf", @@ -195,7 +196,7 @@ def test_extraction_review_new_case_requires_matched_court_and_type(client, reor @pytest.mark.django_db def test_extraction_review_requires_a_case_path(client, reorganized_draft): - FilingDocument.objects.create( + reviewed_document( draft=reorganized_draft, role=FilingDocument.Role.LEAD, name="petition.pdf", @@ -217,7 +218,7 @@ def test_extraction_review_does_not_offer_case_number_or_title_for_a_new_case(cl reorganized_draft.existing_case = ExistingCase.NEW reorganized_draft.extracted_guesses = {"case title": "Rivera v. Example", "docket number": "2024-L-1"} reorganized_draft.save(update_fields=["existing_case", "extracted_guesses"]) - FilingDocument.objects.create( + reviewed_document( draft=reorganized_draft, role=FilingDocument.Role.LEAD, name="petition.pdf", @@ -237,7 +238,7 @@ def test_extraction_review_does_not_offer_case_number_or_title_for_a_new_case(cl def test_extraction_review_asks_existing_case_for_its_number_only(client, reorganized_draft): reorganized_draft.existing_case = ExistingCase.EXISTING reorganized_draft.save(update_fields=["existing_case"]) - FilingDocument.objects.create(draft=reorganized_draft, role=FilingDocument.Role.LEAD, name="motion.pdf") + reviewed_document(draft=reorganized_draft, role=FilingDocument.Role.LEAD, name="motion.pdf") content = client.get(reverse("extraction_review", kwargs={"jurisdiction": "illinois"})).content.decode() @@ -262,7 +263,7 @@ def test_extraction_review_saves_case_identity_only_for_existing_cases( reorganized_draft.case_title = "Court's own title" reorganized_draft.docket_number = "stale" reorganized_draft.save(update_fields=["case_title", "docket_number"]) - FilingDocument.objects.create(draft=reorganized_draft, role=FilingDocument.Role.LEAD, name="petition.pdf") + reviewed_document(draft=reorganized_draft, role=FilingDocument.Role.LEAD, name="petition.pdf") response = client.post( reverse("extraction_review", kwargs={"jurisdiction": "illinois"}), diff --git a/efile_app/efile/tests/test_review_submit_flow.py b/efile_app/efile/tests/test_review_submit_flow.py index ea6e4525..6f0b2379 100644 --- a/efile_app/efile/tests/test_review_submit_flow.py +++ b/efile_app/efile/tests/test_review_submit_flow.py @@ -4,6 +4,7 @@ from efile.models import FilingDocument, FilingDraft, FilingParty from efile.services.current_drafts import CURRENT_DRAFT_SESSION_KEY from efile.services.fee_quotes import record_fee_quote +from efile.tests.helpers import reviewed_document from efile.workflow import WorkflowStepKey @@ -24,7 +25,7 @@ def submission_draft(client, django_user_model): case_type_name="Contract", document_checklist_acknowledged=True, ) - FilingDocument.objects.create( + reviewed_document( draft=draft, role=FilingDocument.Role.LEAD, sort_order=0, diff --git a/efile_app/efile/tests/test_ui_text.py b/efile_app/efile/tests/test_ui_text.py index f8392bcf..a8119d5c 100644 --- a/efile_app/efile/tests/test_ui_text.py +++ b/efile_app/efile/tests/test_ui_text.py @@ -8,6 +8,7 @@ from efile.checks import configured_ui_text_keys_are_known from efile.models import FilingDocument, FilingDraft from efile.services.current_drafts import CURRENT_DRAFT_SESSION_KEY +from efile.tests.helpers import reviewed_document from efile.utils.ui_text import UI_STRINGS, config_overrides, get_html, get_text, get_texts from efile.workflow import ExistingCase, WorkflowStepKey @@ -228,8 +229,8 @@ def organize_draft(client, django_user_model, jurisdiction): current_step=WorkflowStepKey.ORGANIZE_DOCUMENTS, document_checklist_acknowledged=True, ) - FilingDocument.objects.create(draft=draft, role=FilingDocument.Role.LEAD, sort_order=0, name="filing.pdf") - FilingDocument.objects.create(draft=draft, role=FilingDocument.Role.SUPPORTING, sort_order=0, name="exhibit.pdf") + reviewed_document(draft=draft, role=FilingDocument.Role.LEAD, sort_order=0, name="filing.pdf") + reviewed_document(draft=draft, role=FilingDocument.Role.SUPPORTING, sort_order=0, name="exhibit.pdf") client.force_login(user) session = client.session session[CURRENT_DRAFT_SESSION_KEY] = draft.pk diff --git a/efile_app/efile/tests/test_waiver_documents.py b/efile_app/efile/tests/test_waiver_documents.py index 202df2eb..fc48c027 100644 --- a/efile_app/efile/tests/test_waiver_documents.py +++ b/efile_app/efile/tests/test_waiver_documents.py @@ -283,3 +283,28 @@ def test_general_upload_preserves_identical_names_and_separate_storage_keys(paym } assert list(payment_draft.documents.values_list("name", flat=True)) == ["appearance.pdf"] * 3 assert len(read_upload_data(payment_draft)["files"]["supporting"]) == 2 + + +def test_database_failure_cleans_uploaded_original_and_filing(client, payment_draft, storage): + from django.db import DatabaseError + + from efile.tests.pdf_helpers import docx_bytes + from efile.tests.test_document_preparation import service_response + + storage.upload_file.side_effect = [ + {"success": True, "key": "original.docx"}, + {"success": True, "key": "filing.pdf"}, + ] + with ( + patch("efile.services.waiver_documents._codes", side_effect=codes), + patch("efile.services.document_preparation.settings.GOTENBERG_URL", "https://synthetic.invalid"), + patch("efile.services.document_preparation.requests.post", return_value=service_response(pdf_bytes())), + patch("efile.views.waiver_documents.FilingDocument.objects.create", side_effect=DatabaseError("Write failed")), + pytest.raises(DatabaseError), + ): + client.post( + endpoint(payment_draft), + upload_data(payment_draft, document=SimpleUploadedFile("waiver.docx", docx_bytes())), + ) + assert {call.args[0] for call in storage.delete_file.call_args_list} == {"original.docx", "filing.pdf"} + assert payment_draft.documents.count() == 1 diff --git a/efile_app/efile/tests/tests.py b/efile_app/efile/tests/tests.py index 6899d602..3bf64042 100644 --- a/efile_app/efile/tests/tests.py +++ b/efile_app/efile/tests/tests.py @@ -987,16 +987,14 @@ def make_api_call(): ) results.append(response.status_code) - # Create multiple threads to make simultaneous calls - threads = [] - for _ in range(5): - thread = threading.Thread(target=make_api_call) - threads.append(thread) - thread.start() - - # Wait for all threads to complete - for thread in threads: - thread.join() + response = Mock(status_code=200) + response.json.return_value = [{"code": "civil", "name": "Civil"}] + with patch("efile.api.dropdown_views.requests.get", return_value=response): + threads = [threading.Thread(target=make_api_call) for _ in range(5)] + for thread in threads: + thread.start() + for thread in threads: + thread.join() # All calls should succeed assert len(results) == 5 diff --git a/efile_app/efile/views/document_previews.py b/efile_app/efile/views/document_previews.py index a4d7b3d4..29a4018a 100644 --- a/efile_app/efile/views/document_previews.py +++ b/efile_app/efile/views/document_previews.py @@ -12,7 +12,9 @@ from efile.api.suffolk_api_views import get_tyler_token from efile.models import FilingDocument, FilingDraft from efile.services.current_drafts import ensure_current_draft, get_current_draft +from efile.services.document_preparation import PreparationError, PreparationUnavailable from efile.services.document_previews import preview_fingerprint +from efile.services.document_uploads import prepare_stored_documents from efile.utils.s3_upload_handler import S3UploadHandler from efile.workflow import WorkflowStepKey, get_workflow_context @@ -31,6 +33,16 @@ def preview_documents(request, jurisdiction): destinations = {"review": "case_review", "payment": "payment", "document_checklist": "document_checklist"} return_to = request.POST.get("return_to") or request.GET.get("return_to", "") error = "" + status = 200 + if request.method == "GET": + try: + prepare_stored_documents(draft, S3UploadHandler()) + except PreparationUnavailable as exc: + error, status = str(exc), 503 + except PreparationError as exc: + error, status = str(exc), 422 + except (BotoCoreError, ClientError): + error, status = "Document storage is unavailable. Please try again later.", 503 if request.method == "POST": with transaction.atomic(): draft = FilingDraft.objects.select_for_update().get(pk=draft.pk) @@ -38,7 +50,11 @@ def preview_documents(request, jurisdiction): acknowledged = set(request.POST.getlist("reviewed_document")) if draft.status not in {FilingDraft.Status.DRAFT, FilingDraft.Status.ERROR}: return HttpResponse("This filing is no longer available to edit.", status=409) - if request.POST.get("preview_fingerprint") != preview_fingerprint(documents): + if any(not doc.preparation for doc in documents): + error = ( + "These uploads still need preparation. Reload this page or replace the documents before continuing." + ) + elif request.POST.get("preview_fingerprint") != preview_fingerprint(documents): error = "Your documents changed. Preview the current copies before continuing." elif any(str(doc.pk) not in acknowledged for doc in documents): error = "Confirm that you checked each PDF before continuing." @@ -50,11 +66,12 @@ def preview_documents(request, jurisdiction): "documents": documents, "preview_fingerprint": preview_fingerprint(documents), "preview_error": error, + "preparation_pending": any(not doc.preparation for doc in documents), "return_to": return_to, "is_logged_in": True, } context.update(get_workflow_context(WorkflowStepKey.PREVIEW_DOCUMENTS, jurisdiction, draft)) - return render(request, "efile/preview_documents.html", context) + return render(request, "efile/preview_documents.html", context, status=status) @require_http_methods(["GET"]) @@ -89,6 +106,8 @@ def document_content(request, jurisdiction, document_id) -> HttpResponseBase: is_pdf = content.startswith(b"%PDF-") if not original and not is_pdf: return HttpResponse("This document is not a readable PDF. Replace it before continuing.", status=422) + if is_pdf and not original and not filename.lower().endswith(".pdf"): + filename += ".pdf" response = FileResponse( io.BytesIO(content), as_attachment=original or request.GET.get("download") == "1", diff --git a/efile_app/efile/views/handoff.py b/efile_app/efile/views/handoff.py index 0ffdc94e..032956ef 100644 --- a/efile_app/efile/views/handoff.py +++ b/efile_app/efile/views/handoff.py @@ -5,6 +5,7 @@ from urllib.parse import parse_qsl, urlencode, urlsplit, urlunsplit import requests +from botocore.exceptions import BotoCoreError, ClientError from django.conf import settings from django.core import signing from django.db import IntegrityError, transaction @@ -17,7 +18,12 @@ from efile.api.suffolk_api_views import get_tyler_token from efile.models import FilingDraft, HandoffDocumentUpdate, InterviewHandoff from efile.services.current_drafts import attach_current_draft -from efile.services.document_preparation import store_prepared_document +from efile.services.document_preparation import ( + PreparationError, + PreparationUnavailable, + cleanup_unreferenced_uploads, + store_prepared_document, +) from efile.services.draft_urls import draft_url from efile.services.fee_quotes import invalidate_fee_quote from efile.services.filings import describe_filing_detail, fetch_filing_detail @@ -88,8 +94,12 @@ def _upload(payload, files, handler, keys): keys=keys, metadata={"sha256": document["sha256"]}, ) - except ValueError as error: + except (BotoCoreError, ClientError) as error: + raise HandoffError("Document storage is unavailable. Please try again later.", status=503) from error + except PreparationUnavailable as error: raise HandoffError(str(error), status=503) from error + except PreparationError as error: + raise HandoffError(str(error), status=422) from error result = { **prepared, "key": prepared["s3_key"], @@ -145,9 +155,11 @@ def external_handoff(request): # also runs when the filer opens the draft, after choosing a court. return _response(request, receipt, created=True) except HandoffError as exc: - for key in keys: - handler.delete_file(key) + cleanup_unreferenced_uploads(keys, handler) return JsonResponse({"error": str(exc)}, status=exc.status) + except Exception: + cleanup_unreferenced_uploads(keys, handler) + raise def _private(response): @@ -398,6 +410,12 @@ def replace_documents(request): uploads = _upload(payload, request.FILES, handler, keys) for row, doc in updates: uploaded = uploads[doc["id"]] + old_keys = [row.s3_key, row.original_s3_key] + transaction.on_commit( + lambda old_keys=old_keys: cleanup_unreferenced_uploads(old_keys, handler), robust=True + ) + row.name = uploaded["name"] + row.content_type = uploaded["content_type"] row.s3_key = uploaded["key"] row.public_url = uploaded["url"] row.original_filename = uploaded["filename"] @@ -407,6 +425,8 @@ def replace_documents(request): row.preparation_reviewed_at = None row.save( update_fields=[ + "name", + "content_type", "s3_key", "public_url", "original_filename", @@ -445,6 +465,8 @@ def replace_documents(request): } ) except HandoffError as exc: - for key in keys: - handler.delete_file(key) + cleanup_unreferenced_uploads(keys, handler) return JsonResponse({"error": str(exc)}, status=exc.status) + except Exception: + cleanup_unreferenced_uploads(keys, handler) + raise diff --git a/efile_app/efile/views/waiver_documents.py b/efile_app/efile/views/waiver_documents.py index c4ce0b12..f98fcb02 100644 --- a/efile_app/efile/views/waiver_documents.py +++ b/efile_app/efile/views/waiver_documents.py @@ -86,3 +86,6 @@ def waiver_documents(request, jurisdiction): except ValueError as error: cleanup_uploads(handler, keys) return JsonResponse({"error": str(error)}, status=400) + except Exception: + cleanup_uploads(handler, keys) + raise From bc45d38f4e761898396f7640de0686aa08fadde7 Mon Sep 17 00:00:00 2001 From: Quinten Steenhuis Date: Wed, 30 Sep 2026 15:01:57 -0400 Subject: [PATCH 4/8] Mark the accessibility fixture as already reviewed --- docs/developer-notes/issue-113/validation.md | 6 ++++-- .../management/commands/seed_accessibility_session.py | 4 ++++ efile_app/efile/tests/test_document_previews.py | 11 +++++++++++ 3 files changed, 19 insertions(+), 2 deletions(-) diff --git a/docs/developer-notes/issue-113/validation.md b/docs/developer-notes/issue-113/validation.md index aa11cc24..76b48c69 100644 --- a/docs/developer-notes/issue-113/validation.md +++ b/docs/developer-notes/issue-113/validation.md @@ -116,7 +116,7 @@ and final packet exhibit. | Check | Result | | --- | --- | -| `uv run pytest -q` | 1,230 passed; 1 opt-in browser test skipped | +| `uv run pytest -q` | 1,231 passed; 1 opt-in browser test skipped | | Opt-in real Gotenberg and Chromium test | 1 passed | | Conversion and preview tests with Gotenberg environment variables cleared | Passed; final full suite also runs with these variables cleared | | GitHub accessibility workflow | Passed | @@ -178,7 +178,9 @@ The existing extraction queue is a database record, so creating it inside the original atomic transaction was already isolated from worker reads and rollback. The callback now makes the commit boundary explicit. Downstream-flow fixtures explicitly represent prepared and acknowledged documents; the new bypass tests -keep their documents unprepared. A concurrency unit test also now mocks court +keep their documents unprepared. The accessibility seed command also marks its +downstream fixture as prepared and confirmed, with a regression test for this +state. A concurrency unit test also now mocks court choices instead of intermittently depending on a live court response. ## Reproduction diff --git a/efile_app/efile/management/commands/seed_accessibility_session.py b/efile_app/efile/management/commands/seed_accessibility_session.py index ad23de80..8f3b65f8 100644 --- a/efile_app/efile/management/commands/seed_accessibility_session.py +++ b/efile_app/efile/management/commands/seed_accessibility_session.py @@ -6,6 +6,7 @@ from django.conf import settings from django.core.management.base import BaseCommand from django.test import Client +from django.utils import timezone from efile.models import FilingDocument, FilingDraft, FilingParty, FilingPlan from efile.services.current_drafts import CURRENT_DRAFT_SESSION_KEY @@ -68,6 +69,9 @@ def handle(self, *args, **options): role=FilingDocument.Role.LEAD, name="Accessibility complaint.pdf", original_filename="Accessibility complaint.pdf", + # This fixture starts downstream of preparation and confirmation. + preparation="unchanged", + preparation_reviewed_at=timezone.now(), filing_type_code="143132", filing_type_name="Complaint", document_type_code="public", diff --git a/efile_app/efile/tests/test_document_previews.py b/efile_app/efile/tests/test_document_previews.py index b58590e7..c953b635 100644 --- a/efile_app/efile/tests/test_document_previews.py +++ b/efile_app/efile/tests/test_document_previews.py @@ -297,3 +297,14 @@ def test_legacy_pdf_is_flattened_and_cannot_be_acknowledged_after_preparation_fa assert approve(client, preview_draft).status_code == 200 doc.refresh_from_db() assert doc.preparation_reviewed_at is None + + +def test_accessibility_seed_starts_with_a_prepared_acknowledged_document(tmp_path): + from django.core.management import call_command + + from efile.services.document_previews import require_document_previews + + call_command("seed_accessibility_session", output=str(tmp_path / "browser-state.json")) + draft = FilingDraft.objects.get(user__username="accessibility-checker") + require_document_previews(draft) + assert (tmp_path / "browser-state.json").exists() From 9dd037114422937bf39ebfe9a5d948336e61eb53 Mon Sep 17 00:00:00 2001 From: Quinten Steenhuis Date: Wed, 30 Sep 2026 15:34:49 -0400 Subject: [PATCH 5/8] Analyze original PDFs and extract DOCX text locally --- docs/developer-notes/issue-113/validation.md | 40 ++- docs/docs/admin/configuration.md | 8 +- docs/docs/admin/deployment.md | 4 +- docs/docs/partners-courts/ai-customization.md | 2 +- .../efile/services/document_extractions.py | 125 +++++++-- .../efile/services/taxonomy_classification.py | 2 +- efile_app/efile/settings_base.py | 1 + efile_app/efile/static/js/upload-documents.js | 2 +- .../templates/efile/extraction_review.html | 9 +- .../templates/efile/upload_documents.html | 10 +- efile_app/efile/tests/pdf_helpers.py | 2 +- efile_app/efile/tests/test_llms.py | 6 +- .../tests/test_original_document_analysis.py | 262 ++++++++++++++++++ efile_app/efile/utils/llms.py | 7 +- efile_app/efile/views/extraction_review.py | 1 + efile_app/pyproject.toml | 1 + efile_app/uv.lock | 136 +++++++++ 17 files changed, 572 insertions(+), 46 deletions(-) create mode 100644 efile_app/efile/tests/test_original_document_analysis.py diff --git a/docs/developer-notes/issue-113/validation.md b/docs/developer-notes/issue-113/validation.md index 76b48c69..4e16a0a8 100644 --- a/docs/developer-notes/issue-113/validation.md +++ b/docs/developer-notes/issue-113/validation.md @@ -116,7 +116,7 @@ and final packet exhibit. | Check | Result | | --- | --- | -| `uv run pytest -q` | 1,231 passed; 1 opt-in browser test skipped | +| `uv run pytest -q` | 1,240 passed; 1 opt-in browser test skipped | | Opt-in real Gotenberg and Chromium test | 1 passed | | Conversion and preview tests with Gotenberg environment variables cleared | Passed; final full suite also runs with these variables cleared | | GitHub accessibility workflow | Passed | @@ -183,6 +183,44 @@ downstream fixture as prepared and confirmed, with a regression test for this state. A concurrency unit test also now mocks court choices instead of intermittently depending on a live court response. +## Analysis uses the original document + +The filing PDF is used for preview and court submission. Analysis downloads the +preserved original PDF or DOCX. PDFs retain the stored `/V` answers, including +multiline values whose appearance streams are absent. Those values are supplied +alongside the original PDF to the evidence pass and included in locally extracted +classification text. Page limiting preserves the AcroForm and excludes fields +from omitted pages. DOCX text is extracted locally with `docx2python`, including +Unicode, tables, headers and footnotes; the evidence pass receives that text +instead of the converted PDF. Older binary DOC files use the converted PDF. + +DOCX has no reliable page boundaries. `DOCUMENT_EXTRACTION_MAX_TEXT_CHARS` +limits its analysis text to 100,000 characters by default; review flags truncated +text. AI opt-out still runs entirely locally. A storage-key change during analysis +invalidates the outbound permission check and prevents stale results from being +saved; the current source is requeued. + +Validation of this change: + +- 76 focused extraction, worker-claim, AI opt-out and preview tests passed. +- Nine new regression cases cover original DOCX/PDF selection, absent appearance + streams, selected-page form preservation, Word text truncation, binary DOC + fallback, stored values reaching the native model request and surviving a + gateway text fallback, and source replacement during or after analysis. +- A synthetic DOCX containing Unicode, a table, header and footnote was extracted + locally 20 times: median 2.85 ms, maximum 4.38 ms, 148 text characters. +- A separate synthetic Vermont motion passed through the real configured AI and + live taxonomy analysis service in 15.52 seconds. Its docket number was extracted + correctly, with `evidence_input_mode=docx2python_text` and + `source_conversion=docx2python`. This is a smoke measurement, not a speed + comparison against the same document as a PDF. No real filer data was used. +- The real Gotenberg/Chromium browser test passed again in 21.17 seconds and + refreshed the upload screenshots. The Python dependency audit and both the + Docker and documentation builds passed with the new dependency. + +The [docx2python documentation](https://github.com/ShayHill/docx2python) +describes its extraction of body text, tables, headers, footers and notes. + ## Reproduction From `efile_app`, configure Gotenberg credentials and run: diff --git a/docs/docs/admin/configuration.md b/docs/docs/admin/configuration.md index b641d656..c47a9af0 100644 --- a/docs/docs/admin/configuration.md +++ b/docs/docs/admin/configuration.md @@ -26,7 +26,8 @@ LITEFile follows [Twelve-Factor App](https://12factor.net/) principles, configur | `LITEFILE_PROMPTS_DIR` | Optional | Bundled `efile/prompts/` directory | Override path for the versioned LLM prompt catalog. | | `DOCUMENT_EVIDENCE_MODEL` | Optional | First available small model | Exact deployed model used for direct document evidence extraction. | | `DOCUMENT_CLASSIFICATION_MODEL` | Optional | First available medium model | Exact deployed model used for live taxonomy selection. | -| `DOCUMENT_EXTRACTION_MAX_PAGES` | No | `20` | Maximum lead-document pages supplied to the evidence pass. | +| `DOCUMENT_EXTRACTION_MAX_PAGES` | No | `20` | Maximum original PDF pages supplied to the evidence pass; stored form values are preserved. | +| `DOCUMENT_EXTRACTION_MAX_TEXT_CHARS` | No | `100000` | Maximum locally extracted DOCX text characters supplied to analysis. Word files have no reliable page boundaries. | | `DOCUMENT_CLASSIFICATION_SOURCE_PAGES` | No | `3` | Maximum pages converted with MarkItDown and retained as source evidence during taxonomy selection. | | `FORM_CODE_CROSSWALK_PATH` | Optional | Bundled `efile/data/form_code_crosswalk.json` | Override path for exact official-form retrieval hints. | | `AWS_ACCESS_KEY_ID` | Yes (Storage) | `""` | AWS IAM access key for document upload to S3. | @@ -81,7 +82,10 @@ Word conversion requests tagged PDF output and lossless images. It does not rasterize the document or certify accessibility conformance. Flattening can change accessibility tags, links, or annotations. The private original is retained separately from the filing PDF and is available for download. Only the filing copy reaches the -court. Existing editable drafts without preparation metadata are prepared when +court. Analysis reads the original PDF and its stored form values, or text extracted +locally from the original DOCX with `docx2python`. Older binary DOC files use the +converted PDF for analysis. The AI opt-out applies to every format. +Existing editable drafts without preparation metadata are prepared when the filer opens the preview step. Missing stored uploads must be replaced; legacy clients cannot bypass preparation or preview approval. Removing a document or expiring an unclaimed handoff cleans up both private copies when another draft does not reference them. diff --git a/docs/docs/admin/deployment.md b/docs/docs/admin/deployment.md index 88d9fd4c..359fb550 100644 --- a/docs/docs/admin/deployment.md +++ b/docs/docs/admin/deployment.md @@ -86,9 +86,9 @@ primary_region = 'lax' ### Document extraction worker -PDF analysis runs outside the web request in the `extraction_worker` process group. The web process stores the upload and queues a durable database job; the worker downloads the lead PDF from S3 and records the extracted details on the filing draft. Keep at least one worker Machine running so queued documents are analyzed. +Document analysis runs outside the web request in the `extraction_worker` process group. The web process stores the upload and queues a durable database job; the worker downloads the original lead document from S3 and records the extracted details on the filing draft. PDFs retain their stored form values for extraction. DOCX files are read locally with `docx2python`, and their text is supplied to analysis. Older binary DOC files use the converted PDF. Keep at least one worker Machine running so queued documents are analyzed. -By default, LITEFile sends only the first 20 PDF pages for analysis. Set `DOCUMENT_EXTRACTION_MAX_PAGES` to a positive integer to change that cap. `DOCUMENT_EXTRACTION_MAX_ATTEMPTS` controls how many times a failed job is tried before the filer is sent to manual review. +By default, LITEFile sends only the first 20 PDF pages for analysis. Set `DOCUMENT_EXTRACTION_MAX_PAGES` to a positive integer to change that cap. DOCX text is limited to the first 100,000 characters with `DOCUMENT_EXTRACTION_MAX_TEXT_CHARS`; Word files have no reliable page boundaries. Review identifies when either limit omitted part of a document. `DOCUMENT_EXTRACTION_MAX_ATTEMPTS` controls how many times a failed job is tried before the filer is sent to manual review. ### Setting Fly.io production secrets: ```bash diff --git a/docs/docs/partners-courts/ai-customization.md b/docs/docs/partners-courts/ai-customization.md index a3bd2156..99c84223 100644 --- a/docs/docs/partners-courts/ai-customization.md +++ b/docs/docs/partners-courts/ai-customization.md @@ -7,7 +7,7 @@ sidebar_position: 4 # Customizing AI document extraction & prompts WIP -LITEFile includes a staged document-analysis engine that extracts facts from an uploaded court PDF and recommends an exact current court, case category, case type, and filing type for the filer to confirm. +LITEFile includes a staged document-analysis engine that extracts facts from an uploaded court PDF or Word document and recommends an exact current court, case category, case type, and filing type for the filer to confirm. This guide explains how court partners and developers can customize extraction hints, field definitions, model tiers, and private LLM gateways. diff --git a/efile_app/efile/services/document_extractions.py b/efile_app/efile/services/document_extractions.py index 810b803e..ae4785a3 100644 --- a/efile_app/efile/services/document_extractions.py +++ b/efile_app/efile/services/document_extractions.py @@ -1,5 +1,6 @@ """Queue and process durable lead-document extraction jobs.""" +import json import logging import re import uuid @@ -13,6 +14,7 @@ from django.db import connection, transaction from django.db.models import F, Q from django.utils import timezone +from docx2python import docx2python from markitdown import MarkItDown from pypdf import PdfReader, PdfWriter @@ -31,7 +33,7 @@ scan_document_for_form_identifiers, summarize_form_crosswalk_matches, ) -from efile.utils.llms import extract_fields_from_file, get_default_model +from efile.utils.llms import extract_fields_from_file, extract_fields_from_text, get_default_model from efile.utils.prompt_config import prompt_version from efile.utils.s3_upload_handler import S3UploadHandler @@ -43,7 +45,7 @@ class ExtractionSuperseded(Exception): def queue_document_extraction(document): - """Create or reset the one background extraction job for a lead PDF.""" + """Create or reset the one background extraction job for a lead document.""" if document.role != FilingDocument.Role.LEAD: raise ValueError("Only a lead document can be analyzed") job, _created = DocumentExtraction.objects.update_or_create( @@ -88,8 +90,8 @@ def limited_pdf(source_path, max_pages): try: with NamedTemporaryFile(delete=False, suffix=".pdf") as limited_file: writer = PdfWriter() - for page in reader.pages[:max_pages]: - writer.add_page(page) + # append retains the AcroForm and only the widgets on selected pages. + writer.append(reader, pages=(0, max_pages), import_outline=False) writer.write(limited_file) temp_path = limited_file.name yield temp_path, total_pages, pages_analyzed @@ -104,23 +106,66 @@ def _searchable_pdf_text(file_path): return "\f".join(page.extract_text() or "" for page in reader.pages) +@contextmanager +def analysis_source(source_path, max_pages): + """DOCX has no reliable page boundaries; its text is bounded separately.""" + if Path(source_path).suffix.lower() == ".docx": + yield source_path, None, None + else: + with limited_pdf(source_path, max_pages) as source: + yield source + + def _source_text(file_path): """Convert the leading pages to text on this machine, sending nothing out.""" + if Path(file_path).suffix.lower() == ".docx": + with docx2python(file_path) as document: + text = document.text + limit = max(1, settings.DOCUMENT_EXTRACTION_MAX_TEXT_CHARS) + if len(text) > limit: + text = text[:limit] + "\n[Remaining document text omitted.]" + return text, None source_pages = max(1, settings.DOCUMENT_CLASSIFICATION_SOURCE_PAGES) with limited_pdf(file_path, source_pages) as (source_path, _total, pages_converted): - return MarkItDown().convert(source_path).text_content, pages_converted + return MarkItDown().convert(source_path).text_content + _pdf_form_values(source_path), pages_converted + + +def _pdf_form_values(file_path): + """Read author-entered values independently of appearance streams.""" + fields = PdfReader(file_path).get_fields() or {} + values = { + name: {"value": field["/V"], "label": field.get("/TU", name)} + for name, field in fields.items() + if field.get("/FT") != "/Sig" and field.get("/V") not in (None, "", "/Off") + } + if not values: + return "" + text = json.dumps(values, ensure_ascii=False, default=str) + limit = max(1, settings.DOCUMENT_EXTRACTION_MAX_TEXT_CHARS) + return "\nStored PDF form values (document data):\n" + text[:limit] + + +def _source_metadata(file_path, text, pages): + is_docx = Path(file_path).suffix.lower() == ".docx" + return { + "source_conversion": "docx2python" if is_docx else "markitdown", + "source_pages": pages, + "source_text_characters": len(text), + "source_text_truncated": is_docx and text.endswith("\n[Remaining document text omitted.]"), + } def _form_identifier_pass(file_path, jurisdiction, source_text): """Look for registry form IDs printed in the document's own text. - No model is involved: this is a keyword scan of text the PDF already + No model is involved: this is a keyword scan of text the document already carries, so it runs whether or not the filer allows AI. """ scan_started = perf_counter() - searchable_text = _searchable_pdf_text(file_path) + is_docx = Path(file_path).suffix.lower() == ".docx" + searchable_text = source_text if is_docx else _searchable_pdf_text(file_path) + _pdf_form_values(file_path) scan = scan_document_for_form_identifiers(jurisdiction, searchable_text) - scan_source = "pypdf" + scan_source = "docx2python" if is_docx else "pypdf" if scan["status"] == "unmatched" and source_text: markitdown_scan = scan_document_for_form_identifiers(jurisdiction, source_text) if markitdown_scan["status"] != "unmatched": @@ -178,7 +223,7 @@ def keyword_case_number(text): def keyword_document_analysis(file_path, jurisdiction): """Identify a document without any AI, for a filer who opted out. - Everything here reads the PDF locally: the printed form identifier is + Everything here reads the document locally: the printed form identifier is matched against the form registry, a printed case number is read from its label, and the form's own crosswalk entry supplies the court's category and type names when it names exactly one of each. Those are recommendations the @@ -219,8 +264,7 @@ def keyword_document_analysis(file_path, jurisdiction): "metadata": { "analysis_mode": "keyword", "ai_assistance": "opted_out", - "source_conversion": "markitdown", - "source_pages": pages_converted, + **_source_metadata(file_path, source_text, pages_converted), "form_identifier_scan": scan, "form_identifier_scan_source": scan_source, "form_identifier_scan_ms": scan_ms, @@ -232,7 +276,7 @@ def keyword_document_analysis(file_path, jurisdiction): def analyze_document(file_path, jurisdiction, *, use_ai=True, before_outbound=None): - """Run vision evidence extraction, source-text conversion, and live classification. + """Extract evidence from original PDF bytes or DOCX text and classify it. ``use_ai=False`` is the filer's opt-out (issue #104): it takes the keyword path instead, which never sends the document to a model. @@ -253,17 +297,25 @@ def analyze_document(file_path, jurisdiction, *, use_ai=True, before_outbound=No evidence_diagnostics = {} if before_outbound is not None: before_outbound() - evidence = normalize_document_evidence( - extract_fields_from_file( + extraction_kwargs = { + "llm_hint": EXTRACTION_HINTS.get(jurisdiction, EXTRACTION_HINTS["default"]), + "model": evidence_model, + "prompt_name": evidence_prompt, + "prompt_version_name": evidence_version, + } + fields = EXTRACTION_FIELDS.get(jurisdiction, EXTRACTION_FIELDS["default"]) + if Path(file_path).suffix.lower() == ".docx": + evidence_diagnostics["input_mode"] = "docx2python_text" + raw_evidence = extract_fields_from_text(source_text, fields, **extraction_kwargs) + else: + raw_evidence = extract_fields_from_file( file_path, - EXTRACTION_FIELDS.get(jurisdiction, EXTRACTION_FIELDS["default"]), - llm_hint=EXTRACTION_HINTS.get(jurisdiction, EXTRACTION_HINTS["default"]), - model=evidence_model, - prompt_name=evidence_prompt, - prompt_version_name=evidence_version, + fields, diagnostics=evidence_diagnostics, + supplemental_text=_pdf_form_values(file_path), + **extraction_kwargs, ) - ) + evidence = normalize_document_evidence(raw_evidence) ai_form_identifier = evidence.get("form identifier") if form_identifier_scan.get("deterministic"): # The printed identifier found in the source text is stronger than an @@ -287,8 +339,7 @@ def analyze_document(file_path, jurisdiction, *, use_ai=True, before_outbound=No "evidence_prompt_version": evidence_version, "evidence_model": evidence_model, "evidence_input_mode": evidence_diagnostics.get("input_mode", "unknown"), - "source_conversion": "markitdown", - "source_pages": pages_converted, + **_source_metadata(file_path, source_text, pages_converted), "form_identifier_scan": form_identifier_scan, "form_identifier_scan_source": scan_source, "form_identifier_scan_ms": scan_ms, @@ -304,16 +355,23 @@ def process_document_extraction(job_id, claim_token): if job is None: return None document = job.document + filing_key = document.s3_key + original_key = document.original_s3_key + original_suffix = Path(document.original_filename).suffix.lower() + use_original = bool(original_key and original_suffix in {".pdf", ".docx"}) + source_key = original_key if use_original else filing_key + source_suffix = original_suffix if use_original else ".pdf" + source_kind = f"original_{source_suffix[1:]}" if use_original else "filing_pdf" handler = S3UploadHandler() with TemporaryDirectory(prefix="litefile-extraction-") as temp_dir: - source_path = str(Path(temp_dir) / "lead.pdf") - download = handler.download_file(document.s3_key, source_path) + source_path = str(Path(temp_dir) / f"lead{source_suffix}") + download = handler.download_file(source_key, source_path) if not download.get("success"): - raise RuntimeError(download.get("error") or "Could not read the uploaded PDF") + raise RuntimeError(download.get("error") or "Could not read the uploaded document") max_pages = max(1, settings.DOCUMENT_EXTRACTION_MAX_PAGES) - with limited_pdf(source_path, max_pages) as (analysis_path, total_pages, pages_analyzed): + with analysis_source(source_path, max_pages) as (analysis_path, total_pages, pages_analyzed): # Download/parsing may take time. Recheck the claim and preference # before starting analysis that can send the document upstream. if not _current_claim(job_id, claim_token).exists(): @@ -327,6 +385,8 @@ def check_outbound_permission(): .filter( document__draft__ai_assistance_opted_out=False, document__role=FilingDocument.Role.LEAD, + document__s3_key=filing_key, + document__original_s3_key=original_key, ) .exists() ): @@ -341,6 +401,7 @@ def check_outbound_permission(): ) except ExtractionSuperseded: _requeue_changed_preference(job_id, claim_token, opted_out) + _requeue_changed_source(job_id, claim_token, filing_key, original_key) return None # Keep compatibility with extensions that still return the old flat shape. @@ -354,6 +415,7 @@ def check_outbound_permission(): evidence = {} classification = {} metadata = {"pipeline": "legacy-flat-result"} + metadata["analysis_source"] = source_kind with transaction.atomic(): # Match the preference update's lock order: draft, then job. @@ -366,6 +428,9 @@ def check_outbound_permission(): if draft.ai_assistance_opted_out != opted_out: _requeue_changed_preference(job_id, claim_token, opted_out) return None + if not FilingDocument.objects.filter(pk=document.pk, s3_key=filing_key, original_s3_key=original_key).exists(): + queue_document_extraction(job.document) + return None document = job.document # A filer can remove or replace the lead while this worker is running. # Never let the old document overwrite the new lead's extraction. @@ -466,6 +531,14 @@ def renew_extraction_lease(job_id, claim_token): return bool(_current_claim(job_id, claim_token).update(lease_expires_at=timezone.now() + timedelta(minutes=15))) +def _requeue_changed_source(job_id, claim_token, filing_key, original_key): + """Restart analysis if a source was replaced while it was being read.""" + with transaction.atomic(): + job = _current_claim(job_id, claim_token).select_for_update().select_related("document").first() + if job is not None and (job.document.s3_key != filing_key or job.document.original_s3_key != original_key): + queue_document_extraction(job.document) + + def _requeue_changed_preference(job_id, claim_token, opted_out): """Refund a superseded attempt without resetting earlier real failures.""" now = timezone.now() diff --git a/efile_app/efile/services/taxonomy_classification.py b/efile_app/efile/services/taxonomy_classification.py index 7622b753..47fe94e1 100644 --- a/efile_app/efile/services/taxonomy_classification.py +++ b/efile_app/efile/services/taxonomy_classification.py @@ -829,7 +829,7 @@ def _select( "extracted_evidence": evidence, "crosswalk_matches": crosswalk, "crosswalk_constraints": crosswalk_summary, - "source_scope": f"MarkItDown text from the first {settings.DOCUMENT_CLASSIFICATION_SOURCE_PAGES} pages", + "source_scope": "Locally extracted document text, including stored PDF form values when present", }, ) inference = version_config.get("inference", {}) diff --git a/efile_app/efile/settings_base.py b/efile_app/efile/settings_base.py index cacedf92..71bb1273 100644 --- a/efile_app/efile/settings_base.py +++ b/efile_app/efile/settings_base.py @@ -156,6 +156,7 @@ # hundreds of pages long, while the caption and filing details normally appear # near the beginning. DOCUMENT_EXTRACTION_MAX_PAGES = int(os.getenv("DOCUMENT_EXTRACTION_MAX_PAGES", "20")) +DOCUMENT_EXTRACTION_MAX_TEXT_CHARS = int(os.getenv("DOCUMENT_EXTRACTION_MAX_TEXT_CHARS", "100000")) DOCUMENT_EXTRACTION_MAX_ATTEMPTS = int(os.getenv("DOCUMENT_EXTRACTION_MAX_ATTEMPTS", "3")) DOCUMENT_CLASSIFICATION_SOURCE_PAGES = int(os.getenv("DOCUMENT_CLASSIFICATION_SOURCE_PAGES", "3")) DOCUMENT_EVIDENCE_MODEL = os.getenv("DOCUMENT_EVIDENCE_MODEL", "") diff --git a/efile_app/efile/static/js/upload-documents.js b/efile_app/efile/static/js/upload-documents.js index 0802ae63..25e0b647 100644 --- a/efile_app/efile/static/js/upload-documents.js +++ b/efile_app/efile/static/js/upload-documents.js @@ -191,7 +191,7 @@ if (!response.ok || !result.success) throw new Error(result.error || "Upload failed."); stateTitle.textContent = result.extraction_pending ? "Your documents are uploaded" : "Your documents are ready"; let pendingDetail = "Analysis will continue in the background."; - if (aiIsOff()) pendingDetail = "We are checking your PDF's text for a form number, without AI."; + if (aiIsOff()) pendingDetail = "We are checking your document's text for a form number, without AI."; stateDetail.textContent = result.extraction_pending ? pendingDetail : "Review what we found before you continue."; diff --git a/efile_app/efile/templates/efile/extraction_review.html b/efile_app/efile/templates/efile/extraction_review.html index 1af86296..ab1e4ba3 100644 --- a/efile_app/efile/templates/efile/extraction_review.html +++ b/efile_app/efile/templates/efile/extraction_review.html @@ -16,9 +16,9 @@

{% translate "Check what we read from your document" %}

{% if ai_opted_out %} - {% translate "You turned AI off, so we searched the text of your PDF for a form number and a case number instead. These are clues from the document, not answers you gave us, and the search can be wrong. Correct anything that is." %} + {% translate "You turned AI off, so we searched the text of your document for a form number and a case number instead. These are clues from the document, not answers you gave us, and the search can be wrong. Correct anything that is." %} {% else %} - {% translate "These are clues from the PDF, not answers you gave us, and the system can make mistakes. Correct anything that is wrong." %} + {% translate "These are clues from the document, not answers you gave us, and the system can make mistakes. Correct anything that is wrong." %} {% endif %}

@@ -47,6 +47,11 @@

{% translate "The document we read" %}

{% blocktranslate %}This PDF has {{ extraction_total_pages }} pages. We analyzed only the first {{ extraction_pages_analyzed }} pages.{% endblocktranslate %}

{% endif %} + {% if extraction_text_truncated %} +

+ {% translate "We analyzed only the beginning of this Word document. Check the details below and add anything we missed." %} +

+ {% endif %}