diff --git a/.dockerignore b/.dockerignore
index 3769a417..e5fcc204 100644
--- a/.dockerignore
+++ b/.dockerignore
@@ -40,7 +40,7 @@ node_modules/
**/node_modules/
benchmarking/
court_forms/
-package-lock.json
+/package-lock.json
# Local DBs
*.sqlite3
diff --git a/.gitignore b/.gitignore
index 96256622..9c761d50 100644
--- a/.gitignore
+++ b/.gitignore
@@ -184,3 +184,9 @@ court_forms/
# Derived lab-notebook visual review pages
benchmarking/promptfoo/lab-notebook/studies/2026-08-27-deterministic-form-identifier-scan/illinois-form-code-review/
+
+# Generated by npm ci and the Docker asset stage.
+efile_app/efile/static/vendor/pdfjs/
+
+# Store validation screenshots in gists; capture temporary files under /tmp.
+/docs/developer-notes/**/screenshots/
diff --git a/AGENTS.md b/AGENTS.md
index 1259ec54..c72835e5 100644
--- a/AGENTS.md
+++ b/AGENTS.md
@@ -19,6 +19,7 @@ This document provides conventions and context for AI coding assistants working
- **Docusaurus v3**: The documentation site is located in `docs/` targeting `@docusaurus/core` v3.
- **GitHub Pages deployment**: Documentation builds and deploys via GitHub Actions (`.github/workflows/deploy-docs.yml`) to `https://litefile-docs.suffolklitlab.org`.
- **CNAME**: Custom domain is defined in `docs/static/CNAME` (`litefile-docs.suffolklitlab.org`).
+- **Validation screenshots**: Store screenshots in a gist and link to them from notes and PRs. Capture temporary images under `/tmp`. Do not commit validation screenshots or place them in `docs/developer-notes/`.
- **Internal notes**: Internal developer notes, MVP vision briefs, and evaluation notes live in `docs/developer-notes/` and must not be published to the public Docusaurus docs tree (`docs/docs/`).
- **Docker isolation**: `docs/`, `node_modules/`, and `.docusaurus/` are excluded in `.dockerignore` so they are never copied into backend container images.
diff --git a/Dockerfile b/Dockerfile
index c1e32475..06c2707f 100644
--- a/Dockerfile
+++ b/Dockerfile
@@ -1,4 +1,10 @@
# syntax=docker/dockerfile:1.7
+FROM node:22-slim AS pdfjs-assets
+WORKDIR /assets
+COPY efile_app/package.json efile_app/package-lock.json ./
+COPY efile_app/scripts/copy-pdfjs.mjs ./scripts/copy-pdfjs.mjs
+RUN npm ci --omit=dev
+
FROM python:3.12-slim AS base
ENV PYTHONDONTWRITEBYTECODE=1 \
@@ -24,6 +30,7 @@ RUN uv sync --frozen --no-install-project
# Copy the rest of the source code into /app
COPY . /app
+COPY --from=pdfjs-assets /assets/efile/static/vendor/pdfjs /app/efile_app/efile/static/vendor/pdfjs
# Install the project itself (editable-like install)
RUN uv sync --frozen
diff --git a/docs/developer-notes/issue-113/corpus-comparison.json b/docs/developer-notes/issue-113/corpus-comparison.json
new file mode 100644
index 00000000..3f3dc9e0
--- /dev/null
+++ b/docs/developer-notes/issue-113/corpus-comparison.json
@@ -0,0 +1,318 @@
+[
+ {
+ "source": "docassemble-ALDashboard/docassemble/ALDashboard/test/civil_docketing_statement_polished_repaired.pdf",
+ "sha256": "110986860fb8016e2ac5f6d83ad7f421cd21eccf75f79184838eff93704e5f38",
+ "input_pages": 3,
+ "input_fields": 59,
+ "filled_fields": 3,
+ "multiline_fields": 4,
+ "raw-gotenberg": {
+ "accepted": true,
+ "pages": 3,
+ "remaining_fields": 0,
+ "text_characters": 3611,
+ "tagged": false
+ },
+ "litefile": {
+ "accepted": true,
+ "pages": 3,
+ "remaining_fields": 0,
+ "text_characters": 3611,
+ "tagged": false
+ },
+ "pdftk": {
+ "accepted": true,
+ "pages": 3,
+ "remaining_fields": 0,
+ "text_characters": 3611,
+ "tagged": false
+ }
+ },
+ {
+ "source": "docassemble-ALWeaver/build/lib/docassemble/ALWeaver/data/sources/test_civil_docketing_statement.pdf",
+ "sha256": "74751dadb930ed97d66243596108d54207ce249c83a309467500b23f541e58c3",
+ "input_pages": 3,
+ "input_fields": 60,
+ "filled_fields": 3,
+ "multiline_fields": 6,
+ "raw-gotenberg": {
+ "accepted": true,
+ "pages": 3,
+ "remaining_fields": 0,
+ "text_characters": 3607,
+ "tagged": false
+ },
+ "litefile": {
+ "accepted": true,
+ "pages": 3,
+ "remaining_fields": 0,
+ "text_characters": 3607,
+ "tagged": false
+ },
+ "pdftk": {
+ "accepted": true,
+ "pages": 3,
+ "remaining_fields": 0,
+ "text_characters": 3607,
+ "tagged": false
+ }
+ },
+ {
+ "source": "docassemble-ALWeaver/docassemble/ALWeaver/test/test_option_groups.pdf",
+ "sha256": "498a9feccb3d135385b4fbb59755075c99cc280e014e22eb077746d3ae5956dc",
+ "input_pages": 1,
+ "input_fields": 6,
+ "filled_fields": 5,
+ "multiline_fields": 0,
+ "raw-gotenberg": {
+ "accepted": true,
+ "pages": 1,
+ "remaining_fields": 0,
+ "text_characters": 0,
+ "tagged": false
+ },
+ "litefile": {
+ "accepted": true,
+ "pages": 1,
+ "remaining_fields": 0,
+ "text_characters": 0,
+ "tagged": false
+ },
+ "pdftk": {
+ "accepted": false,
+ "error_type": "CalledProcessError"
+ }
+ },
+ {
+ "source": "docassemble-ALWeaver/docassemble/ALWeaver/test/test_push_button.pdf",
+ "sha256": "4e5d1b4fbc42099ea12b37c6956a80208ae79a5d4ce599a928b54c16fdba4d69",
+ "input_pages": 1,
+ "input_fields": 2,
+ "filled_fields": 1,
+ "multiline_fields": 0,
+ "raw-gotenberg": {
+ "accepted": true,
+ "pages": 1,
+ "remaining_fields": 0,
+ "text_characters": 0,
+ "tagged": false
+ },
+ "litefile": {
+ "accepted": true,
+ "pages": 1,
+ "remaining_fields": 0,
+ "text_characters": 0,
+ "tagged": false
+ },
+ "pdftk": {
+ "accepted": true,
+ "pages": 1,
+ "remaining_fields": 0,
+ "text_characters": 0,
+ "tagged": false
+ }
+ },
+ {
+ "source": "docassemble-AppearanceEfile/docassemble/AppearanceEfile/data/templates/appearance_acrobat_pro_edited.pdf",
+ "sha256": "8b98f685ae4a5c2ebb35cfa673ae6520e3945fc0471915e515ee3fbb8ac61d5c",
+ "input_pages": 3,
+ "input_fields": 70,
+ "filled_fields": 2,
+ "multiline_fields": 2,
+ "raw-gotenberg": {
+ "accepted": true,
+ "pages": 3,
+ "remaining_fields": 0,
+ "text_characters": 6132,
+ "tagged": true
+ },
+ "litefile": {
+ "accepted": true,
+ "pages": 3,
+ "remaining_fields": 0,
+ "text_characters": 6132,
+ "tagged": true
+ },
+ "pdftk": {
+ "accepted": true,
+ "pages": 3,
+ "remaining_fields": 0,
+ "text_characters": 6131,
+ "tagged": true
+ }
+ },
+ {
+ "source": "docassemble-AppearanceEfile/docassemble/AppearanceEfile/data/templates/appearance_edited.pdf",
+ "sha256": "4a75bda4130a6f941ea96b1fd7d329ce820cac09d4f13d4d75b4cff94b93c5c7",
+ "input_pages": 3,
+ "input_fields": 46,
+ "filled_fields": 3,
+ "multiline_fields": 2,
+ "raw-gotenberg": {
+ "accepted": true,
+ "pages": 3,
+ "remaining_fields": 0,
+ "text_characters": 7616,
+ "tagged": true
+ },
+ "litefile": {
+ "accepted": true,
+ "pages": 3,
+ "remaining_fields": 0,
+ "text_characters": 7618,
+ "tagged": true
+ },
+ "pdftk": {
+ "accepted": true,
+ "pages": 3,
+ "remaining_fields": 0,
+ "text_characters": 7616,
+ "tagged": true
+ }
+ },
+ {
+ "source": "docassemble-CLAGuardianship/docassemble/CLAGuardianship/data/templates/affidavit_disclosing_care_or_custody_old.pdf",
+ "sha256": "88cd00883b53724eac7b562ddc1ccabcef96a54785fdafd74dfb39f526e6cf27",
+ "input_pages": 2,
+ "input_fields": 76,
+ "filled_fields": 1,
+ "multiline_fields": 0,
+ "raw-gotenberg": {
+ "accepted": true,
+ "pages": 2,
+ "remaining_fields": 0,
+ "text_characters": 5902,
+ "tagged": false
+ },
+ "litefile": {
+ "accepted": true,
+ "pages": 2,
+ "remaining_fields": 0,
+ "text_characters": 5902,
+ "tagged": false
+ },
+ "pdftk": {
+ "accepted": true,
+ "pages": 2,
+ "remaining_fields": 0,
+ "text_characters": 5902,
+ "tagged": false
+ }
+ },
+ {
+ "source": "docassemble-MLHDivorceAndCustody/docassemble/MLHDivorceAndCustody/data/templates/child_support_services_dhs1201d.pdf",
+ "sha256": "2c9b742e183e955f5ddcf095abce5d442880fc1cba14c1b0a3257523e31406f2",
+ "input_pages": 1,
+ "input_fields": 44,
+ "filled_fields": 3,
+ "multiline_fields": 0,
+ "raw-gotenberg": {
+ "accepted": true,
+ "pages": 1,
+ "remaining_fields": 0,
+ "text_characters": 3790,
+ "tagged": true
+ },
+ "litefile": {
+ "accepted": true,
+ "pages": 1,
+ "remaining_fields": 0,
+ "text_characters": 3790,
+ "tagged": true
+ },
+ "pdftk": {
+ "accepted": true,
+ "pages": 1,
+ "remaining_fields": 0,
+ "text_characters": 3790,
+ "tagged": true
+ }
+ },
+ {
+ "source": "docassemble-MLHDivorceAndCustody/docassemble/MLHDivorceAndCustody/data/templates/confidential_case_inventory_mc21.pdf",
+ "sha256": "2f02a121e63c11c1406399b57cb7b9060c1a3b17a8cbfe8adf4db3f7e9be8748",
+ "input_pages": 1,
+ "input_fields": 48,
+ "filled_fields": 10,
+ "multiline_fields": 0,
+ "raw-gotenberg": {
+ "accepted": true,
+ "pages": 1,
+ "remaining_fields": 0,
+ "text_characters": 1954,
+ "tagged": true
+ },
+ "litefile": {
+ "accepted": true,
+ "pages": 1,
+ "remaining_fields": 0,
+ "text_characters": 1955,
+ "tagged": true
+ },
+ "pdftk": {
+ "accepted": true,
+ "pages": 1,
+ "remaining_fields": 0,
+ "text_characters": 1954,
+ "tagged": true
+ }
+ },
+ {
+ "source": "docassemble-MLHDivorceAndCustody/docassemble/MLHDivorceAndCustody/data/templates/uniform_spousal_support_order_foc10b.pdf",
+ "sha256": "4540a473fcfbf179cffd3ec144bf428be3a87afc0252c1fceee78cc4a27f0cbd",
+ "input_pages": 2,
+ "input_fields": 61,
+ "filled_fields": 1,
+ "multiline_fields": 7,
+ "raw-gotenberg": {
+ "accepted": true,
+ "pages": 2,
+ "remaining_fields": 0,
+ "text_characters": 4320,
+ "tagged": true
+ },
+ "litefile": {
+ "accepted": true,
+ "pages": 2,
+ "remaining_fields": 0,
+ "text_characters": 4322,
+ "tagged": true
+ },
+ "pdftk": {
+ "accepted": true,
+ "pages": 2,
+ "remaining_fields": 0,
+ "text_characters": 4320,
+ "tagged": true
+ }
+ },
+ {
+ "source": "docassemble-USCISApplications/docassemble/USCISApplications/data/templates/eoir-33_coa.pdf",
+ "sha256": "ae5bfc58ed65e4e1bc9826bbd841ec3c7224972a6bb560108f878f06253863c0",
+ "input_pages": 2,
+ "input_fields": 26,
+ "filled_fields": 1,
+ "multiline_fields": 1,
+ "raw-gotenberg": {
+ "accepted": true,
+ "pages": 2,
+ "remaining_fields": 0,
+ "text_characters": 6959,
+ "tagged": true
+ },
+ "litefile": {
+ "accepted": true,
+ "pages": 2,
+ "remaining_fields": 0,
+ "text_characters": 6957,
+ "tagged": true
+ },
+ "pdftk": {
+ "accepted": true,
+ "pages": 2,
+ "remaining_fields": 0,
+ "text_characters": 6960,
+ "tagged": true
+ }
+ }
+]
diff --git a/docs/developer-notes/issue-113/validation.md b/docs/developer-notes/issue-113/validation.md
new file mode 100644
index 00000000..784ffa1f
--- /dev/null
+++ b/docs/developer-notes/issue-113/validation.md
@@ -0,0 +1,275 @@
+# Document preparation and preview validation
+
+Validated September 30, 2026 for [LITEFile issue #113](https://github.com/SuffolkLITLab/LITEFile/issues/113).
+Branch: `feature/document-preparation-preview`.
+Validation gist: https://gist.github.com/nonprofittechy/54857d2ed0b841a486dbf7a30b1ad915.
+
+## Result and engine choice
+
+Use Gotenberg for Word conversion and PDF flattening, with the existing `pypdf`
+dependency repairing missing or stale text appearance streams first. PDFtk is
+used only in the comparison harness; it is not a runtime dependency.
+
+The local scan found 12 PDFs with populated form values, comprising 11 distinct
+files after SHA-256 deduplication and 22 pages. These come from ALWeaver,
+ALDashboard, AppearanceEfile, MLHDivorceAndCustody, USCISApplications and
+CLAGuardianship repositories. The corpus includes template defaults, whitespace
+values and checked controls; supplemental synthetic values exercise multiline
+text, signatures, Unicode and complete packets. It should continue to grow as
+more representative completed filings become available.
+
+| Method | Accepted local specimens | Fields remaining in accepted outputs |
+| --- | --- | --- |
+| Raw Gotenberg flatten route | 11 / 11 | 0 |
+| LITEFile preparation pipeline | 11 / 11 | 0 |
+| PDFtk Java 3.3.3 `flatten` | 10 / 11 | 0 |
+
+PDFtk rejected ALWeaver's option-group fixture. Full provenance, SHA-256 hashes,
+page counts, text counts and results are in
+[corpus-comparison.json](https://github.com/SuffolkLITLab/LITEFile/blob/feature/document-preparation-preview/docs/developer-notes/issue-113/corpus-comparison.json).
+Source documents and filled values stay local; the published report contains
+metadata and synthetic screenshots.
+
+Text extraction initially made raw Gotenberg look sufficient. Raster inspection
+revealed that its appearance generation placed three multiline answers on one
+line when `/AP` was missing. This matches QPDF's documented limits on appearance
+generation. LITEFile creates the text appearance with `pypdf`, clears
+`NeedAppearances`, then lets Gotenberg flatten that appearance. Existing usable
+appearances remain intact. The synthetic reproducer shows the difference:
+
+
+
+
+
+An EOIR missing-appearance stress case containing accented names demonstrated
+another engine limitation. The prepared result retained form fields, so LITEFile
+rejected it with a printed-PDF recovery instruction. The automated checks also
+reject lost filled text, changed page counts and unreadable responses. Preview
+and filer review remain necessary for layout, fonts, clipping, checkbox
+and signature fidelity. This work does not establish universal PDF compatibility.
+
+## Court forms and packet checks
+
+The current [Vermont File & Serve FAQ](https://www.vtcourts.gov/about-vermont-judiciary/electronic-access/electronic-filing/faqs)
+requires form-fillable PDFs to be saved as flat files before filing. Vermont's
+YAML explicitly enables flattening and supplies plain-language upload guidance
+with that source. Other jurisdictions can set `flatten_pdf_forms: false` and
+provide their own guidance through `ui_text`.
+
+Downloaded the official [Small Claims Answer, form 100-00126, July 2025](https://www.vtcourts.gov/sites/default/files/documents/100-00126%20%E2%80%93%20Small%20Claims%20Answer_0.pdf)
+and filled it with synthetic names, a synthetic docket, a selected checkbox,
+a three-line answer and a typed `/s/` signature. Checked the rasterized filing
+copy on both pages. The checkbox, names, answer lines and signature remained
+visible. Appended two synthetic exhibit pages and verified the prepared packet
+has four pages, no form fields, preserved signature text and an intact last
+exhibit page. No court filing or fee request was made for these documents.
+
+
+
+
+
+## Browser validation
+
+Ran Chromium against Django's live test server with real Gotenberg conversion,
+real PDF.js rendering and an in-memory S3 test double. Documents and account
+identity were synthetic. This validates app behavior without uploading test
+files to production storage or contacting a court. The separate PDF corpus
+comparison calls the configured Gotenberg service and writes local artifacts.
+
+Checked the following:
+
+- Selecting and uploading a two-page PDF with missing multiline appearances and a two-page DOCX together.
+- Loading actual filing bytes through the authenticated preview endpoint.
+- Rendering both PDFs, moving to page two, zooming and retaining selectable text layers.
+- Keeping the originals separately and providing original and filing-copy download links.
+- Continuing from PDF preview without checkboxes; the server records the current files as reviewed when Continue is pressed.
+- Rendering expandable PDFs in the organize and fees screens, with assertions that each expected page loaded.
+- A 390-pixel mobile viewport with no horizontal page overflow.
+- Invalid-PDF upload guidance with the existing files retained.
+- A simulated 503 from document storage, followed by successful retry after reopening the preview.
+- Zero browser page errors and zero Axe violations in the preview screen.
+
+The initial browser run caught a compatibility problem in PDF.js 6's modern
+bundle on the installed Chromium. The shipped assets now use PDF.js's official
+legacy build, including its compatibility polyfills. A second run caught
+repeated page-region landmark labels across documents; those labels now identify
+both the document and the page. The final run passes. A final evidence review also caught a redirected organize
+page; the fixture now supplies synthetic case data and asserts the page URL and
+heading before capturing that screenshot. Court choices and payment accounts
+are stubbed, and the backend fee estimate is stubbed.
+
+
+
+
+
+
+
+
+
+
+
+[All screenshots and the Axe result](https://gist.github.com/nonprofittechy/54857d2ed0b841a486dbf7a30b1ad915)
+include the second-page view, selection screen, invalid upload, preview failure
+and final packet exhibit.
+
+## Automated validation
+
+| Check | Result |
+| --- | --- |
+| `uv run pytest -q` | 1,240 passed; 1 opt-in browser test skipped |
+| Opt-in real Gotenberg and Chromium test | 1 passed |
+| Conversion and preview tests with Gotenberg environment variables cleared | Passed; final full suite also runs with these variables cleared |
+| GitHub accessibility workflow | Passed |
+| `npm run test:unit` | 64 passed |
+| `uv run ruff check .` and formatting | Passed |
+| `uv run ty check` | Passed |
+| `npm run lint:js` and Prettier check | Passed |
+| New Django template lint and format | Passed |
+| Stylelint | 0 errors; 19 existing warnings in other stylesheets |
+| Bandit | Passed after removing a production assertion |
+| `uv run pip-audit --local --skip-editable` | No known vulnerabilities found |
+| `manage.py makemigrations --check --dry-run` | No changes detected |
+| `docker build -t litefile:issue-113 .` | Passed; generated and collected PDF.js assets |
+| Docusaurus `npm run build` | Passed |
+
+Regression coverage includes complete filing flows in Vermont, Illinois and
+Massachusetts; later fee-waiver uploads; interview handoffs and replacement
+PDFs; private preview ownership and draft isolation; original downloads;
+unchanged PDFs; damaged and unsupported inputs; service timeout; oversized
+output; conservative loss checks; batch rollback and orphan cleanup; server-side
+completion of the preview step; stale preview fingerprints; and original/review metadata preservation
+when legacy clients rebuild supporting rows.
+
+The existing npm dependency tree reports four audit findings in development/test
+dependencies. No finding names the new PDF.js package. Those existing dependencies
+were left at their locked versions to keep this feature's dependency changes
+focused. Accessibility testing checks the preview controls and generated tagged
+Word PDF structure; it does not certify every filing PDF as PDF/UA compliant.
+
+The first GitHub dependency audit found four advisories in the pre-existing
+`virtualenv` 20.36.1 development dependency. Updated the lockfile to virtualenv
+21.7.13 and its required dependencies, then verified the audit passes and that
+it creates a working virtual environment. The relevant fixes are documented in
+[virtualenv's release history](https://virtualenv.pypa.io/en/latest/changelog.html#v21-7-13-2026-09-18).
+
+CI also exposed mocked conversion tests that depended on a locally configured
+Gotenberg URL. The test fixture now sets a synthetic service URL explicitly;
+The conversion and preview tests pass with all Gotenberg environment variables
+cleared. The opt-in integration test still uses the real configured service.
+
+## Regression review
+
+Addressed the follow-up review with tests that start from unprepared or stale
+server state rather than relying on already reviewed fixtures:
+
+| Concern | Behavior checked |
+| --- | --- |
+| Lead storage-key replacement | Clears preparation, original metadata and approval; deletes superseded copies after commit while keeping shared references |
+| Legacy supporting uploads | Prepares stored DOCX/PDF bytes on preview, blocks Continue before preparation, and blocks direct submission |
+| Stale preview fingerprint | Includes preparation, original key, size and document update time in addition to row and filing key |
+| Indirect annotation arrays | Resolves and validates the array before iterating; accepts the indirect-array regression specimen |
+| Handoff preparation errors | Permanent input/conversion rejection returns 422; storage and conversion-service outages return 503 |
+| Extraction timing | Registers the job creation with `transaction.on_commit`; a rolled-back transaction does not queue extraction |
+| Waiver write failure | Database errors after storing original and filing copies remove both staged objects |
+| Handoff replacement cleanup | Deletes old objects after commit, preserving objects referenced by another draft |
+| Special text-field appearances | Skips literal-value matching for comb, password, formatted, rich-text and hidden/no-view fields; retains structural and ordinary visible-text checks |
+
+The existing extraction queue is a database record, so creating it inside the
+original atomic transaction was already isolated from worker reads and rollback.
+The callback now makes the commit boundary explicit. Downstream-flow fixtures
+explicitly represent prepared documents that completed the preview step; the new bypass tests
+keep their documents unprepared. The accessibility seed command also marks its
+downstream fixture as prepared with the preview step complete, with a regression test for this
+state. A concurrency unit test also now mocks court
+choices instead of intermittently depending on a live court response.
+
+## Analysis uses the original document
+
+The filing PDF is used for preview and court submission. Analysis downloads the
+preserved original PDF or DOCX. PDFs retain the stored `/V` answers, including
+multiline values whose appearance streams are absent. Those values are supplied
+alongside the original PDF to the evidence pass and included in locally extracted
+classification text. Page limiting preserves the AcroForm and excludes fields
+from omitted pages. DOCX text is extracted locally with `docx2python`, including
+Unicode, tables, headers and footnotes; the evidence pass receives that text
+instead of the converted PDF. Older binary DOC files use the converted PDF.
+
+DOCX has no reliable page boundaries. `DOCUMENT_EXTRACTION_MAX_TEXT_CHARS`
+limits its analysis text to 100,000 characters by default; review flags truncated
+text. AI opt-out still runs entirely locally. A storage-key change during analysis
+invalidates the outbound permission check and prevents stale results from being
+saved; the current source is requeued.
+
+Validation of this change:
+
+- 76 focused extraction, worker-claim, AI opt-out and preview tests passed.
+- Nine new regression cases cover original DOCX/PDF selection, absent appearance
+ streams, selected-page form preservation, Word text truncation, binary DOC
+ fallback, stored values reaching the native model request and surviving a
+ gateway text fallback, and source replacement during or after analysis.
+- A synthetic DOCX containing Unicode, a table, header and footnote was extracted
+ locally 20 times: median 2.85 ms, maximum 4.38 ms, 148 text characters.
+- A separate synthetic Vermont motion passed through the real configured AI and
+ live taxonomy analysis service in 15.52 seconds. Its docket number was extracted
+ correctly, with `evidence_input_mode=docx2python_text` and
+ `source_conversion=docx2python`. This is a smoke measurement, not a speed
+ comparison against the same document as a PDF. No real filer data was used.
+- The real Gotenberg/Chromium browser test passed again in 21.17 seconds and
+ refreshed the upload screenshots. The Python dependency audit and both the
+ Docker and documentation builds passed with the new dependency.
+
+The [docx2python documentation](https://github.com/ShayHill/docx2python)
+describes its extraction of body text, tables, headers, footers and notes.
+
+## Preview and copy follow-up
+
+PDF preview has no confirmation checkboxes. The screen asks filers to check each
+PDF before continuing. Continue records completion of the preview step for the
+current document list. Stale fingerprints and unprepared files still block it;
+replacing a file requires another preview step. The existing checkbox on
+“Review what we found” is unchanged.
+
+Shortened the upload guidance, download labels, preview instructions and error
+messages. Removed duplicate filenames, per-file conversion explanations and the
+technical upload introduction. Updated the state override and its translation
+source together.
+
+The 89 focused preparation, preview, AI opt-out and configurable-copy tests
+passed. The real Gotenberg/Chromium browser test passed in 14.20 seconds, verifies
+Continue works without preview checkboxes, and reports zero Axe violations and
+zero browser page errors. Refreshed the desktop and mobile screenshots.
+
+## Screenshot storage
+
+All 14 screenshots and the Axe result are stored in the validation gist.
+Screenshot files are removed from the feature branch and its commit history.
+Capture new screenshots in `/tmp`, upload them to the gist, and link to them
+from validation notes and PR descriptions. Do not put screenshots in developer
+notes or commit them to the application repository.
+
+## Reproduction
+
+From `efile_app`, configure Gotenberg credentials and run:
+
+```bash
+uv sync --group dev
+npm ci
+uv run pytest -q
+npm run test:unit
+uv run python ../testing/validate_document_preparation.py \
+ --output /tmp/litefile-pdf-validation --render
+DOCUMENT_PREPARATION_BROWSER_TESTS=1 \
+DOCUMENT_PREPARATION_EVIDENCE_DIR=/tmp/litefile-browser-evidence \
+uv run pytest -q -s efile/tests/test_document_preparation_browser.py
+```
+
+The corpus script defaults to `~/docassemble-*`, accepts repeatable `--source`
+paths, and writes source hashes and engine results without filled values. It
+requires PDFtk for comparison and Poppler for `--render`; the application requires
+neither. The browser test requires a Playwright Chromium install. For an existing
+local browser, set `PLAYWRIGHT_CHROMIUM_EXECUTABLE` to its executable path.
+
+Relevant engine documentation:
+[Gotenberg flatten route](https://gotenberg.dev/docs/manipulate-pdfs/flatten-pdfs),
+[Gotenberg Word conversion](https://gotenberg.dev/docs/convert-with-libreoffice/convert-to-pdf),
+and [QPDF appearance generation](https://qpdf.readthedocs.io/en/stable/cli.html#option-generate-appearances).
diff --git a/docs/docs/admin/configuration.md b/docs/docs/admin/configuration.md
index 9ddd25d6..a5c5318b 100644
--- a/docs/docs/admin/configuration.md
+++ b/docs/docs/admin/configuration.md
@@ -26,7 +26,8 @@ LITEFile follows [Twelve-Factor App](https://12factor.net/) principles, configur
| `LITEFILE_PROMPTS_DIR` | Optional | Bundled `efile/prompts/` directory | Override path for the versioned LLM prompt catalog. |
| `DOCUMENT_EVIDENCE_MODEL` | Optional | First available small model | Exact deployed model used for direct document evidence extraction. |
| `DOCUMENT_CLASSIFICATION_MODEL` | Optional | First available medium model | Exact deployed model used for live taxonomy selection. |
-| `DOCUMENT_EXTRACTION_MAX_PAGES` | No | `20` | Maximum lead-document pages supplied to the evidence pass. |
+| `DOCUMENT_EXTRACTION_MAX_PAGES` | No | `20` | Maximum original PDF pages supplied to the evidence pass; stored form values are preserved. |
+| `DOCUMENT_EXTRACTION_MAX_TEXT_CHARS` | No | `100000` | Maximum locally extracted DOCX text characters supplied to analysis. Word files have no reliable page boundaries. |
| `DOCUMENT_CLASSIFICATION_SOURCE_PAGES` | No | `3` | Maximum pages converted with MarkItDown and retained as source evidence during taxonomy selection. |
| `FORM_CODE_CROSSWALK_PATH` | Optional | Bundled `efile/data/form_code_crosswalk.json` | Override path for exact official-form retrieval hints. |
| `AWS_ACCESS_KEY_ID` | Yes (Storage) | `""` | AWS IAM access key for document upload to S3. |
@@ -35,6 +36,10 @@ LITEFile follows [Twelve-Factor App](https://12factor.net/) principles, configur
| `AWS_SESSION_TOKEN` | No | `""` | AWS session token when using temporary credentials. |
| `AWS_S3_REGION_NAME` | No | `us-east-1` | AWS region where the S3 bucket is hosted. |
| `DJANGO_LOG_LEVEL` | No | `DEBUG` (Dev) / `INFO` (Prod) | Logging verbosity for the `efile` application logger. |
+| `GOTENBERG_URL` | Optional | `""` | Base URL for the Gotenberg document conversion API (e.g. `https://gotenberg-dev.fly.dev`). |
+| `GOTENBERG_USERNAME` | Optional | `""` | HTTP basic auth username for the Gotenberg service. |
+| `GOTENBERG_PASSWORD` | Optional | `""` | HTTP basic auth password for the Gotenberg service. |
+| `DOCUMENT_PREPARATION_TIMEOUT_SECONDS` | No | `45` | Timeout for a conversion or flattening request. |
---
@@ -56,3 +61,55 @@ AWS_S3_REGION_NAME="us-east-1"
# AI Extraction (Optional)
OPENAI_API_KEY="sk-..."
```
+
+
+## Document preparation and previews
+
+Configure Gotenberg 8.16 or newer for Word conversion and PDF form flattening.
+The service must support `/forms/libreoffice/convert` and `/forms/pdfengines/flatten`.
+Use HTTPS and service credentials when connecting to a remote instance. Install
+fonts used by your forms in Gotenberg: missing fonts can change pagination or layout.
+
+LITEFile keeps an unchanged PDF byte for byte. When form fields need locking, it
+preserves existing appearance streams and repairs missing or stale text appearances
+with `pypdf` before Gotenberg flattens the fields. It rejects unreadable, encrypted,
+XFA, digitally certificate-signed PDFs that would need flattening, and results with
+missing pages, remaining fields, or lost filled-in text. Filers can upload a printed
+PDF copy instead. These checks do not guarantee visual fidelity; every new filing
+copy has a preview step before submission. Filers are asked to check every page;
+Continue records that step without requiring a checkbox.
+
+Word conversion requests tagged PDF output and lossless images. It does not
+rasterize the document or certify accessibility conformance. Flattening can change
+accessibility tags, links, or annotations. The private original is retained separately
+from the filing PDF and is available for download. Only the filing copy reaches the
+court. Analysis reads the original PDF and its stored form values, or text extracted
+locally from the original DOCX with `docx2python`. Older binary DOC files use the
+converted PDF for analysis. The AI opt-out applies to every format.
+Existing editable drafts without preparation metadata are prepared when
+the filer opens the preview step. Missing stored uploads must be replaced; legacy
+clients cannot bypass preparation or the preview step. Removing a document or expiring an unclaimed handoff cleans up both private
+copies when another draft does not reference them.
+
+State YAML can override the default policy:
+
+```yaml
+document_preparation:
+ flatten_pdf_forms: false
+text:
+ upload_documents:
+ preparation_help_unflattened: >-
+ LITEFile converts Word documents to PDF. PDF form fields are kept as uploaded.
+ Check the filing PDFs before continuing.
+```
+
+The default is to lock interactive fields. Vermont explicitly enables this policy
+and links to the court's preparation instructions. PDFs without form widgets are
+left unchanged, including already flattened and remediated documents.
+
+PDF.js is pinned in `efile_app/package-lock.json` and served from the application,
+including its worker, fonts, and character maps. Run `npm ci` in `efile_app` before
+local previews; its install script copies these assets. The Docker build generates
+the same assets in a separate Node stage. Document bytes come from an authenticated,
+draft-scoped endpoint with `Cache-Control: private, no-store`, so previewing does
+not require public storage URLs or S3 CORS configuration.
diff --git a/docs/docs/admin/deployment.md b/docs/docs/admin/deployment.md
index 47661767..359fb550 100644
--- a/docs/docs/admin/deployment.md
+++ b/docs/docs/admin/deployment.md
@@ -86,9 +86,9 @@ primary_region = 'lax'
### Document extraction worker
-PDF analysis runs outside the web request in the `extraction_worker` process group. The web process stores the upload and queues a durable database job; the worker downloads the lead PDF from S3 and records the extracted details on the filing draft. Keep at least one worker Machine running so queued documents are analyzed.
+Document analysis runs outside the web request in the `extraction_worker` process group. The web process stores the upload and queues a durable database job; the worker downloads the original lead document from S3 and records the extracted details on the filing draft. PDFs retain their stored form values for extraction. DOCX files are read locally with `docx2python`, and their text is supplied to analysis. Older binary DOC files use the converted PDF. Keep at least one worker Machine running so queued documents are analyzed.
-By default, LITEFile sends only the first 20 PDF pages for analysis. Set `DOCUMENT_EXTRACTION_MAX_PAGES` to a positive integer to change that cap. `DOCUMENT_EXTRACTION_MAX_ATTEMPTS` controls how many times a failed job is tried before the filer is sent to manual review.
+By default, LITEFile sends only the first 20 PDF pages for analysis. Set `DOCUMENT_EXTRACTION_MAX_PAGES` to a positive integer to change that cap. DOCX text is limited to the first 100,000 characters with `DOCUMENT_EXTRACTION_MAX_TEXT_CHARS`; Word files have no reliable page boundaries. Review identifies when either limit omitted part of a document. `DOCUMENT_EXTRACTION_MAX_ATTEMPTS` controls how many times a failed job is tried before the filer is sent to manual review.
### Setting Fly.io production secrets:
```bash
@@ -99,7 +99,10 @@ fly secrets set \
AWS_SECRET_ACCESS_KEY="..." \
AWS_S3_BUCKET_NAME="litefile-production-documents" \
AWS_S3_REGION_NAME="us-east-1" \
- OPENAI_API_KEY="sk-..."
+ OPENAI_API_KEY="sk-..." \
+ GOTENBERG_URL="https://..." \
+ GOTENBERG_USERNAME="..." \
+ GOTENBERG_PASSWORD="..."
```
---
diff --git a/docs/docs/partners-courts/ai-customization.md b/docs/docs/partners-courts/ai-customization.md
index a3bd2156..99c84223 100644
--- a/docs/docs/partners-courts/ai-customization.md
+++ b/docs/docs/partners-courts/ai-customization.md
@@ -7,7 +7,7 @@ sidebar_position: 4
# Customizing AI document extraction & prompts WIP
-LITEFile includes a staged document-analysis engine that extracts facts from an uploaded court PDF and recommends an exact current court, case category, case type, and filing type for the filer to confirm.
+LITEFile includes a staged document-analysis engine that extracts facts from an uploaded court PDF or Word document and recommends an exact current court, case category, case type, and filing type for the filer to confirm.
This guide explains how court partners and developers can customize extraction hints, field definitions, model tiers, and private LLM gateways.
diff --git a/efile_app/.env.example b/efile_app/.env.example
index 484cf2c2..ec02abc6 100644
--- a/efile_app/.env.example
+++ b/efile_app/.env.example
@@ -39,3 +39,10 @@ MAX_FILE_SIZE = 10 * 1024 * 1024 # 10MB
ALLOWED_FILE_TYPES = ['.pdf', '.doc', '.docx']
OPENAI_API_KEY = "..."
OPENAI_BASE_URL = "https://api.openai.com/v1/"
+
+# Gotenberg document conversion service
+GOTENBERG_URL = "https://gotenberg-dev.fly.dev"
+GOTENBERG_USERNAME = "your-gotenberg-username"
+GOTENBERG_PASSWORD = "your-gotenberg-password"
+
+DOCUMENT_PREPARATION_TIMEOUT_SECONDS = 45
diff --git a/efile_app/efile/api/s3_upload.py b/efile_app/efile/api/s3_upload.py
deleted file mode 100644
index 4b0e57de..00000000
--- a/efile_app/efile/api/s3_upload.py
+++ /dev/null
@@ -1,187 +0,0 @@
-import logging
-import uuid
-
-from django.http import JsonResponse
-from django.views.decorators.csrf import csrf_exempt
-from django.views.decorators.http import require_http_methods
-
-from ..utils.s3_upload_handler import S3UploadHandler
-
-logger = logging.getLogger(__name__)
-
-
-@csrf_exempt
-@require_http_methods(["GET", "POST"])
-def test_s3_connection(request):
- """Test S3 connection and bucket access."""
- try:
- # A fresh handler per request, so a credential change takes effect
- # without a restart.
- s3_handler = S3UploadHandler()
-
- # Test S3 connection
- if s3_handler._ensure_initialized():
- # Ensure the client is initialized for type checkers
- if s3_handler.s3_client is None:
- return JsonResponse({"success": False, "error": "S3 client not initialized"}, status=500)
- response = s3_handler.s3_client.list_objects_v2(
- Bucket=s3_handler.bucket_name, Prefix="efile-documents/", MaxKeys=1
- )
-
- return JsonResponse(
- {
- "success": True,
- "message": "S3 connection successful",
- "bucket": s3_handler.bucket_name,
- "region": s3_handler.region_name,
- "objects_exist": "Contents" in response,
- }
- )
- else:
- return JsonResponse({"success": False, "error": "S3 client not initialized - check AWS credentials"})
-
- except Exception as e:
- return JsonResponse({"success": False, "error": f"S3 connection failed: {str(e)}"})
-
-
-@csrf_exempt
-@require_http_methods(["POST"])
-def simple_s3_upload(request):
- """Simple S3 upload that just uploads files and returns URLs."""
- try:
- logger.debug(
- "simple_s3_upload method=%s file_keys=%s post_keys=%s",
- request.method,
- list(request.FILES.keys()),
- list(request.POST.keys()),
- )
-
- # Handle file uploads
- uploaded_files = request.FILES.getlist("documents")
-
- logger.debug("simple_s3_upload found %d files", len(uploaded_files))
-
- if not uploaded_files:
- return JsonResponse({"success": False, "error": "No documents provided."}, status=400)
-
- s3_handler = S3UploadHandler()
-
- if not s3_handler._ensure_initialized():
- return JsonResponse(
- {"success": False, "error": "S3 not configured properly. Check AWS credentials."}, status=500
- )
-
- s3_upload_results = []
-
- # Upload all files to S3
- for i, uploaded_file in enumerate(uploaded_files):
- # Validate file
- validation_result = s3_handler.validate_file(uploaded_file, max_size_mb=10, allowed_types=[".pdf"])
-
- if not validation_result["valid"]:
- return JsonResponse(
- {
- "success": False,
- "error": f"File validation failed for {uploaded_file.name}: {validation_result['error']}",
- },
- status=400,
- )
-
- # Prepare metadata
- file_type = "lead" if i == 0 else "supporting"
- metadata = {
- "file-type": file_type,
- "original-size": str(uploaded_file.size),
- "original-name": uploaded_file.name,
- "upload-session": str(uuid.uuid4())[:8],
- }
-
- # Upload to S3
- upload_result = s3_handler.upload_file(uploaded_file, file_type=file_type, metadata=metadata)
-
- if not upload_result["success"]:
- return JsonResponse(
- {"success": False, "error": f"S3 upload failed for {uploaded_file.name}: {upload_result['error']}"},
- status=500,
- )
-
- s3_upload_results.append(
- {
- "original_name": uploaded_file.name,
- "url": upload_result["url"],
- "public_url": s3_handler.get_public_url(upload_result["key"]),
- "key": upload_result["key"],
- "size": upload_result["size"],
- "type": file_type,
- }
- )
-
- return JsonResponse(
- {
- "success": True,
- "message": f"Successfully uploaded {len(s3_upload_results)} file(s) to S3",
- "files": s3_upload_results,
- }
- )
-
- except Exception as e:
- logger.error(f"Simple S3 upload error: {e}")
- return JsonResponse({"success": False, "error": f"Upload error: {str(e)}"}, status=500)
-
-
-@csrf_exempt
-@require_http_methods(["POST"])
-def mock_s3_upload(request):
- """Mock S3 upload for testing when AWS permissions aren't available."""
- try:
- # Handle file uploads
- uploaded_files = request.FILES.getlist("documents")
-
- if not uploaded_files:
- return JsonResponse({"success": False, "error": "No documents provided."}, status=400)
-
- mock_upload_results = []
-
- # Simulate S3 upload results
- for i, uploaded_file in enumerate(uploaded_files):
- # Validate file type
- if not (uploaded_file.name.lower().endswith(".pdf") or uploaded_file.content_type == "application/pdf"):
- return JsonResponse(
- {"success": False, "error": f"Invalid file type: {uploaded_file.name}. Only PDF files allowed."},
- status=400,
- )
-
- # Simulate file size validation
- max_size = 10 * 1024 * 1024 # 10MB
- if uploaded_file.size > max_size:
- return JsonResponse(
- {"success": False, "error": f"File too large: {uploaded_file.name}. Maximum size is 10MB."},
- status=400,
- )
-
- # Generate mock S3 URLs
- file_id = str(uuid.uuid4())[:8]
- file_type = "lead" if i == 0 else "supporting"
-
- mock_upload_results.append(
- {
- "original_name": uploaded_file.name,
- "url": f"https://litefile-staging.s3.amazonaws.com/efile-documents/{file_type}/{file_id}.pdf",
- "public_url": f"https://litefile-staging.s3.amazonaws.com/efile-documents/{file_type}/{file_id}.pdf",
- "key": f"efile-documents/{file_type}/{file_id}.pdf",
- "size": uploaded_file.size,
- "type": file_type,
- }
- )
-
- return JsonResponse(
- {
- "success": True,
- "message": f"Mock upload: Successfully processed {len(mock_upload_results)} file(s)",
- "files": mock_upload_results,
- }
- )
-
- except Exception as e:
- logger.error(f"Mock S3 upload error: {e}")
- return JsonResponse({"success": False, "error": f"Upload error: {str(e)}"}, status=500)
diff --git a/efile_app/efile/config_text_strings.py b/efile_app/efile/config_text_strings.py
index da9aa63e..c8a7efe8 100644
--- a/efile_app/efile/config_text_strings.py
+++ b/efile_app/efile/config_text_strings.py
@@ -50,4 +50,9 @@
"terms.starting_document_example",
"complaint",
),
+ # Translators: Preparation guidance when flatten_pdf_forms is enabled. May link to state-specific requirements.
+ pgettext_lazy(
+ "upload_documents.preparation_help",
+ "Upload your files as they are. We make PDF copies for court. [Vermont's PDF rules](https://www.vtcourts.gov/about-vermont-judiciary/electronic-access/electronic-filing/faqs).",
+ ),
]
diff --git a/efile_app/efile/management/commands/expire_unclaimed_handoffs.py b/efile_app/efile/management/commands/expire_unclaimed_handoffs.py
index 40b679e0..68e76669 100644
--- a/efile_app/efile/management/commands/expire_unclaimed_handoffs.py
+++ b/efile_app/efile/management/commands/expire_unclaimed_handoffs.py
@@ -4,9 +4,11 @@
from django.core.management.base import BaseCommand, CommandError
from django.db import transaction
+from django.db.models import Q
from django.utils import timezone
from efile.models import FilingDocument, FilingDraft, InterviewHandoff
+from efile.services.document_previews import document_storage_keys
from efile.utils.s3_upload_handler import S3UploadHandler
@@ -36,8 +38,13 @@ def handle(self, *args, **options):
draft = FilingDraft.objects.select_for_update().filter(pk=draft_id, user__isnull=True).first()
if draft is None:
continue
- for key in draft.documents.exclude(s3_key="").values_list("s3_key", flat=True):
- if not FilingDocument.objects.filter(s3_key=key).exclude(draft=draft).exists():
+ keys = {key for doc in draft.documents.all() for key in document_storage_keys(doc)}
+ for key in keys:
+ if (
+ not FilingDocument.objects.filter(Q(s3_key=key) | Q(original_s3_key=key))
+ .exclude(draft=draft)
+ .exists()
+ ):
result = handler.delete_file(key)
if not result.get("success"):
raise CommandError(
diff --git a/efile_app/efile/management/commands/seed_accessibility_session.py b/efile_app/efile/management/commands/seed_accessibility_session.py
index ad23de80..8f3b65f8 100644
--- a/efile_app/efile/management/commands/seed_accessibility_session.py
+++ b/efile_app/efile/management/commands/seed_accessibility_session.py
@@ -6,6 +6,7 @@
from django.conf import settings
from django.core.management.base import BaseCommand
from django.test import Client
+from django.utils import timezone
from efile.models import FilingDocument, FilingDraft, FilingParty, FilingPlan
from efile.services.current_drafts import CURRENT_DRAFT_SESSION_KEY
@@ -68,6 +69,9 @@ def handle(self, *args, **options):
role=FilingDocument.Role.LEAD,
name="Accessibility complaint.pdf",
original_filename="Accessibility complaint.pdf",
+ # This fixture starts downstream of preparation and confirmation.
+ preparation="unchanged",
+ preparation_reviewed_at=timezone.now(),
filing_type_code="143132",
filing_type_name="Complaint",
document_type_code="public",
diff --git a/efile_app/efile/migrations/0029_document_preparation_and_preview.py b/efile_app/efile/migrations/0029_document_preparation_and_preview.py
new file mode 100644
index 00000000..2f72d6d6
--- /dev/null
+++ b/efile_app/efile/migrations/0029_document_preparation_and_preview.py
@@ -0,0 +1,34 @@
+# Generated by Django 5.2.17 on 2026-09-30 17:30
+
+import efile.workflow
+from django.db import migrations, models
+
+
+class Migration(migrations.Migration):
+
+ dependencies = [
+ ('efile', '0028_extraction_claim_leases'),
+ ]
+
+ operations = [
+ migrations.AddField(
+ model_name='filingdocument',
+ name='original_s3_key',
+ field=models.CharField(blank=True, max_length=1024),
+ ),
+ migrations.AddField(
+ model_name='filingdocument',
+ name='preparation',
+ field=models.CharField(blank=True, choices=[('unchanged', 'Original PDF'), ('converted', 'Converted from Word'), ('flattened', 'Form fields locked'), ('converted_flattened', 'Converted and fields locked')], max_length=30),
+ ),
+ migrations.AddField(
+ model_name='filingdocument',
+ name='preparation_reviewed_at',
+ field=models.DateTimeField(blank=True, null=True),
+ ),
+ migrations.AlterField(
+ model_name='filingdraft',
+ name='current_step',
+ field=models.CharField(choices=[('options', 'Options'), ('filing_path', 'Start'), ('upload_documents', 'Upload documents'), ('preview_documents', 'Preview documents'), ('extraction_review', 'Confirm filing'), ('case_lookup', 'Find your case'), ('case_confirmation', 'Confirm your case'), ('document_checklist', 'Check documents'), ('organize_documents', 'Organize documents'), ('your_information', 'Your information'), ('parties', 'People in this filing'), ('party_details', 'Person details'), ('case_questions', 'Case questions'), ('payment', 'Fees'), ('review', 'Review'), ('confirmation', 'Confirmation')], default=efile.workflow.WorkflowStepKey['OPTIONS'], max_length=64),
+ ),
+ ]
diff --git a/efile_app/efile/models.py b/efile_app/efile/models.py
index bcbf4db8..1f433cb1 100644
--- a/efile_app/efile/models.py
+++ b/efile_app/efile/models.py
@@ -356,6 +356,19 @@ class Role(models.TextChoices):
content_type = models.CharField(max_length=255, blank=True)
s3_key = models.CharField(max_length=1024, blank=True)
public_url = models.URLField(max_length=2048, blank=True)
+ # Originals are private recovery copies; only s3_key is sent to the court.
+ original_s3_key = models.CharField(max_length=1024, blank=True)
+ preparation = models.CharField(
+ max_length=30,
+ blank=True,
+ choices=[
+ ("unchanged", "Original PDF"),
+ ("converted", "Converted from Word"),
+ ("flattened", "Form fields locked"),
+ ("converted_flattened", "Converted and fields locked"),
+ ],
+ )
+ preparation_reviewed_at = models.DateTimeField(null=True, blank=True)
filing_type_code = models.CharField(max_length=100, blank=True)
filing_type_name = models.CharField(max_length=255, blank=True)
diff --git a/efile_app/efile/services/document_extractions.py b/efile_app/efile/services/document_extractions.py
index 810b803e..ae4785a3 100644
--- a/efile_app/efile/services/document_extractions.py
+++ b/efile_app/efile/services/document_extractions.py
@@ -1,5 +1,6 @@
"""Queue and process durable lead-document extraction jobs."""
+import json
import logging
import re
import uuid
@@ -13,6 +14,7 @@
from django.db import connection, transaction
from django.db.models import F, Q
from django.utils import timezone
+from docx2python import docx2python
from markitdown import MarkItDown
from pypdf import PdfReader, PdfWriter
@@ -31,7 +33,7 @@
scan_document_for_form_identifiers,
summarize_form_crosswalk_matches,
)
-from efile.utils.llms import extract_fields_from_file, get_default_model
+from efile.utils.llms import extract_fields_from_file, extract_fields_from_text, get_default_model
from efile.utils.prompt_config import prompt_version
from efile.utils.s3_upload_handler import S3UploadHandler
@@ -43,7 +45,7 @@ class ExtractionSuperseded(Exception):
def queue_document_extraction(document):
- """Create or reset the one background extraction job for a lead PDF."""
+ """Create or reset the one background extraction job for a lead document."""
if document.role != FilingDocument.Role.LEAD:
raise ValueError("Only a lead document can be analyzed")
job, _created = DocumentExtraction.objects.update_or_create(
@@ -88,8 +90,8 @@ def limited_pdf(source_path, max_pages):
try:
with NamedTemporaryFile(delete=False, suffix=".pdf") as limited_file:
writer = PdfWriter()
- for page in reader.pages[:max_pages]:
- writer.add_page(page)
+ # append retains the AcroForm and only the widgets on selected pages.
+ writer.append(reader, pages=(0, max_pages), import_outline=False)
writer.write(limited_file)
temp_path = limited_file.name
yield temp_path, total_pages, pages_analyzed
@@ -104,23 +106,66 @@ def _searchable_pdf_text(file_path):
return "\f".join(page.extract_text() or "" for page in reader.pages)
+@contextmanager
+def analysis_source(source_path, max_pages):
+ """DOCX has no reliable page boundaries; its text is bounded separately."""
+ if Path(source_path).suffix.lower() == ".docx":
+ yield source_path, None, None
+ else:
+ with limited_pdf(source_path, max_pages) as source:
+ yield source
+
+
def _source_text(file_path):
"""Convert the leading pages to text on this machine, sending nothing out."""
+ if Path(file_path).suffix.lower() == ".docx":
+ with docx2python(file_path) as document:
+ text = document.text
+ limit = max(1, settings.DOCUMENT_EXTRACTION_MAX_TEXT_CHARS)
+ if len(text) > limit:
+ text = text[:limit] + "\n[Remaining document text omitted.]"
+ return text, None
source_pages = max(1, settings.DOCUMENT_CLASSIFICATION_SOURCE_PAGES)
with limited_pdf(file_path, source_pages) as (source_path, _total, pages_converted):
- return MarkItDown().convert(source_path).text_content, pages_converted
+ return MarkItDown().convert(source_path).text_content + _pdf_form_values(source_path), pages_converted
+
+
+def _pdf_form_values(file_path):
+ """Read author-entered values independently of appearance streams."""
+ fields = PdfReader(file_path).get_fields() or {}
+ values = {
+ name: {"value": field["/V"], "label": field.get("/TU", name)}
+ for name, field in fields.items()
+ if field.get("/FT") != "/Sig" and field.get("/V") not in (None, "", "/Off")
+ }
+ if not values:
+ return ""
+ text = json.dumps(values, ensure_ascii=False, default=str)
+ limit = max(1, settings.DOCUMENT_EXTRACTION_MAX_TEXT_CHARS)
+ return "\nStored PDF form values (document data):\n" + text[:limit]
+
+
+def _source_metadata(file_path, text, pages):
+ is_docx = Path(file_path).suffix.lower() == ".docx"
+ return {
+ "source_conversion": "docx2python" if is_docx else "markitdown",
+ "source_pages": pages,
+ "source_text_characters": len(text),
+ "source_text_truncated": is_docx and text.endswith("\n[Remaining document text omitted.]"),
+ }
def _form_identifier_pass(file_path, jurisdiction, source_text):
"""Look for registry form IDs printed in the document's own text.
- No model is involved: this is a keyword scan of text the PDF already
+ No model is involved: this is a keyword scan of text the document already
carries, so it runs whether or not the filer allows AI.
"""
scan_started = perf_counter()
- searchable_text = _searchable_pdf_text(file_path)
+ is_docx = Path(file_path).suffix.lower() == ".docx"
+ searchable_text = source_text if is_docx else _searchable_pdf_text(file_path) + _pdf_form_values(file_path)
scan = scan_document_for_form_identifiers(jurisdiction, searchable_text)
- scan_source = "pypdf"
+ scan_source = "docx2python" if is_docx else "pypdf"
if scan["status"] == "unmatched" and source_text:
markitdown_scan = scan_document_for_form_identifiers(jurisdiction, source_text)
if markitdown_scan["status"] != "unmatched":
@@ -178,7 +223,7 @@ def keyword_case_number(text):
def keyword_document_analysis(file_path, jurisdiction):
"""Identify a document without any AI, for a filer who opted out.
- Everything here reads the PDF locally: the printed form identifier is
+ Everything here reads the document locally: the printed form identifier is
matched against the form registry, a printed case number is read from its
label, and the form's own crosswalk entry supplies the court's category and
type names when it names exactly one of each. Those are recommendations the
@@ -219,8 +264,7 @@ def keyword_document_analysis(file_path, jurisdiction):
"metadata": {
"analysis_mode": "keyword",
"ai_assistance": "opted_out",
- "source_conversion": "markitdown",
- "source_pages": pages_converted,
+ **_source_metadata(file_path, source_text, pages_converted),
"form_identifier_scan": scan,
"form_identifier_scan_source": scan_source,
"form_identifier_scan_ms": scan_ms,
@@ -232,7 +276,7 @@ def keyword_document_analysis(file_path, jurisdiction):
def analyze_document(file_path, jurisdiction, *, use_ai=True, before_outbound=None):
- """Run vision evidence extraction, source-text conversion, and live classification.
+ """Extract evidence from original PDF bytes or DOCX text and classify it.
``use_ai=False`` is the filer's opt-out (issue #104): it takes the keyword
path instead, which never sends the document to a model.
@@ -253,17 +297,25 @@ def analyze_document(file_path, jurisdiction, *, use_ai=True, before_outbound=No
evidence_diagnostics = {}
if before_outbound is not None:
before_outbound()
- evidence = normalize_document_evidence(
- extract_fields_from_file(
+ extraction_kwargs = {
+ "llm_hint": EXTRACTION_HINTS.get(jurisdiction, EXTRACTION_HINTS["default"]),
+ "model": evidence_model,
+ "prompt_name": evidence_prompt,
+ "prompt_version_name": evidence_version,
+ }
+ fields = EXTRACTION_FIELDS.get(jurisdiction, EXTRACTION_FIELDS["default"])
+ if Path(file_path).suffix.lower() == ".docx":
+ evidence_diagnostics["input_mode"] = "docx2python_text"
+ raw_evidence = extract_fields_from_text(source_text, fields, **extraction_kwargs)
+ else:
+ raw_evidence = extract_fields_from_file(
file_path,
- EXTRACTION_FIELDS.get(jurisdiction, EXTRACTION_FIELDS["default"]),
- llm_hint=EXTRACTION_HINTS.get(jurisdiction, EXTRACTION_HINTS["default"]),
- model=evidence_model,
- prompt_name=evidence_prompt,
- prompt_version_name=evidence_version,
+ fields,
diagnostics=evidence_diagnostics,
+ supplemental_text=_pdf_form_values(file_path),
+ **extraction_kwargs,
)
- )
+ evidence = normalize_document_evidence(raw_evidence)
ai_form_identifier = evidence.get("form identifier")
if form_identifier_scan.get("deterministic"):
# The printed identifier found in the source text is stronger than an
@@ -287,8 +339,7 @@ def analyze_document(file_path, jurisdiction, *, use_ai=True, before_outbound=No
"evidence_prompt_version": evidence_version,
"evidence_model": evidence_model,
"evidence_input_mode": evidence_diagnostics.get("input_mode", "unknown"),
- "source_conversion": "markitdown",
- "source_pages": pages_converted,
+ **_source_metadata(file_path, source_text, pages_converted),
"form_identifier_scan": form_identifier_scan,
"form_identifier_scan_source": scan_source,
"form_identifier_scan_ms": scan_ms,
@@ -304,16 +355,23 @@ def process_document_extraction(job_id, claim_token):
if job is None:
return None
document = job.document
+ filing_key = document.s3_key
+ original_key = document.original_s3_key
+ original_suffix = Path(document.original_filename).suffix.lower()
+ use_original = bool(original_key and original_suffix in {".pdf", ".docx"})
+ source_key = original_key if use_original else filing_key
+ source_suffix = original_suffix if use_original else ".pdf"
+ source_kind = f"original_{source_suffix[1:]}" if use_original else "filing_pdf"
handler = S3UploadHandler()
with TemporaryDirectory(prefix="litefile-extraction-") as temp_dir:
- source_path = str(Path(temp_dir) / "lead.pdf")
- download = handler.download_file(document.s3_key, source_path)
+ source_path = str(Path(temp_dir) / f"lead{source_suffix}")
+ download = handler.download_file(source_key, source_path)
if not download.get("success"):
- raise RuntimeError(download.get("error") or "Could not read the uploaded PDF")
+ raise RuntimeError(download.get("error") or "Could not read the uploaded document")
max_pages = max(1, settings.DOCUMENT_EXTRACTION_MAX_PAGES)
- with limited_pdf(source_path, max_pages) as (analysis_path, total_pages, pages_analyzed):
+ with analysis_source(source_path, max_pages) as (analysis_path, total_pages, pages_analyzed):
# Download/parsing may take time. Recheck the claim and preference
# before starting analysis that can send the document upstream.
if not _current_claim(job_id, claim_token).exists():
@@ -327,6 +385,8 @@ def check_outbound_permission():
.filter(
document__draft__ai_assistance_opted_out=False,
document__role=FilingDocument.Role.LEAD,
+ document__s3_key=filing_key,
+ document__original_s3_key=original_key,
)
.exists()
):
@@ -341,6 +401,7 @@ def check_outbound_permission():
)
except ExtractionSuperseded:
_requeue_changed_preference(job_id, claim_token, opted_out)
+ _requeue_changed_source(job_id, claim_token, filing_key, original_key)
return None
# Keep compatibility with extensions that still return the old flat shape.
@@ -354,6 +415,7 @@ def check_outbound_permission():
evidence = {}
classification = {}
metadata = {"pipeline": "legacy-flat-result"}
+ metadata["analysis_source"] = source_kind
with transaction.atomic():
# Match the preference update's lock order: draft, then job.
@@ -366,6 +428,9 @@ def check_outbound_permission():
if draft.ai_assistance_opted_out != opted_out:
_requeue_changed_preference(job_id, claim_token, opted_out)
return None
+ if not FilingDocument.objects.filter(pk=document.pk, s3_key=filing_key, original_s3_key=original_key).exists():
+ queue_document_extraction(job.document)
+ return None
document = job.document
# A filer can remove or replace the lead while this worker is running.
# Never let the old document overwrite the new lead's extraction.
@@ -466,6 +531,14 @@ def renew_extraction_lease(job_id, claim_token):
return bool(_current_claim(job_id, claim_token).update(lease_expires_at=timezone.now() + timedelta(minutes=15)))
+def _requeue_changed_source(job_id, claim_token, filing_key, original_key):
+ """Restart analysis if a source was replaced while it was being read."""
+ with transaction.atomic():
+ job = _current_claim(job_id, claim_token).select_for_update().select_related("document").first()
+ if job is not None and (job.document.s3_key != filing_key or job.document.original_s3_key != original_key):
+ queue_document_extraction(job.document)
+
+
def _requeue_changed_preference(job_id, claim_token, opted_out):
"""Refund a superseded attempt without resetting earlier real failures."""
now = timezone.now()
diff --git a/efile_app/efile/services/document_preparation.py b/efile_app/efile/services/document_preparation.py
new file mode 100644
index 00000000..2b5b1141
--- /dev/null
+++ b/efile_app/efile/services/document_preparation.py
@@ -0,0 +1,291 @@
+"""Prepare filing copies without replacing or rasterizing the original upload."""
+
+from __future__ import annotations
+
+import io
+import logging
+import time
+import zipfile
+from dataclasses import dataclass
+from pathlib import Path
+
+import requests
+from django.conf import settings
+from django.core.files.uploadedfile import SimpleUploadedFile
+from pypdf import PdfReader, PdfWriter
+from pypdf.generic import ArrayObject, DictionaryObject
+
+from efile.utils.config_loader import config_loader
+
+logger = logging.getLogger(__name__)
+
+
+class PreparationError(ValueError):
+ """An actionable document failure safe to show to the filer."""
+
+
+class PreparationUnavailable(PreparationError):
+ """A temporary service/storage failure for which retry is appropriate."""
+
+
+@dataclass(frozen=True)
+class PreparedDocument:
+ content: bytes
+ filename: str
+ operation: str
+
+
+def requires_flattening(jurisdiction):
+ config = config_loader.load_jurisdiction_config(jurisdiction)
+ return config.get("document_preparation", {}).get("flatten_pdf_forms", True)
+
+
+def inspect_pdf(content):
+ try:
+ if not content.startswith(b"%PDF-"):
+ raise ValueError("Not a PDF")
+ reader = PdfReader(io.BytesIO(content))
+ if reader.is_encrypted or not reader.pages:
+ raise ValueError("Encrypted or empty PDF")
+ root = reader.trailer["/Root"]
+ if not isinstance(root, DictionaryObject):
+ raise ValueError("Invalid PDF catalog")
+ form = root.get("/AcroForm")
+ if form and form.get_object().get("/XFA"):
+ raise PreparationError("We cannot use this PDF form. Print it to PDF and upload that copy.")
+ # Force page and annotation parsing before accepting a filing copy.
+ widgets = []
+ for page in reader.pages:
+ if "/Annots" not in page:
+ continue
+ annotations = page["/Annots"].get_object()
+ if not isinstance(annotations, ArrayObject):
+ raise ValueError("Invalid annotation array")
+ for ref in annotations:
+ annotation = ref.get_object()
+ if not isinstance(annotation, DictionaryObject):
+ raise ValueError("Invalid annotation")
+ if annotation.get("/Subtype") == "/Widget":
+ widgets.append(annotation)
+ return reader, widgets
+ except PreparationError:
+ raise
+ except Exception as error:
+ raise PreparationError("This PDF could not be read. Upload a new copy without a password.") from error
+
+
+def _word_format(content, suffix):
+ if suffix == ".doc":
+ if not content.startswith(b"\xd0\xcf\x11\xe0\xa1\xb1\x1a\xe1"):
+ raise PreparationError("Choose a Word file or save this file as a PDF.")
+ return
+ try:
+ with zipfile.ZipFile(io.BytesIO(content)) as archive:
+ entries = archive.infolist()
+ if len(entries) > 10000 or sum(entry.file_size for entry in entries) > 100 * 1024 * 1024:
+ raise ValueError("Expanded document too large")
+ if not {"[Content_Types].xml", "word/document.xml"}.issubset(archive.namelist()):
+ raise ValueError("Not DOCX")
+ except (ValueError, zipfile.BadZipFile) as error:
+ raise PreparationError(
+ "This Word file could not be read. Save a new copy as Word or PDF and upload it."
+ ) from error
+
+
+def _gotenberg(content, suffix, route, data=None):
+ base_url = settings.GOTENBERG_URL.rstrip("/")
+ if not base_url:
+ raise PreparationUnavailable(
+ "We could not prepare your file. Try again later, or print it to PDF and upload that copy."
+ )
+ limit = settings.MAX_FILE_SIZE
+ deadline = time.monotonic() + settings.DOCUMENT_PREPARATION_TIMEOUT_SECONDS
+ try:
+ with requests.post(
+ f"{base_url}{route}",
+ auth=(settings.GOTENBERG_USERNAME, settings.GOTENBERG_PASSWORD),
+ # Do not send user filenames to the conversion service.
+ files={"files": (f"document{suffix}", content, "application/octet-stream")},
+ data=data or {},
+ timeout=(5, settings.DOCUMENT_PREPARATION_TIMEOUT_SECONDS),
+ allow_redirects=False,
+ stream=True,
+ ) as response:
+ if response.status_code >= 500 or response.status_code in {401, 403, 429}:
+ raise PreparationUnavailable("We could not prepare your file. Try again later.")
+ if response.status_code != 200:
+ raise PreparationError("We could not make a PDF. Save your file as a PDF and upload it.")
+ result = bytearray()
+ for chunk in response.iter_content(64 * 1024):
+ result.extend(chunk)
+ if len(result) > limit:
+ raise PreparationError("This PDF is over 10 MB. Split it into smaller files and upload them.")
+ if time.monotonic() > deadline:
+ raise requests.Timeout()
+ return bytes(result)
+ except requests.RequestException as error:
+ logger.warning("Document preparation service unavailable (%s)", type(error).__name__)
+ raise PreparationUnavailable(
+ "We could not prepare your file. Try again later, or print it to PDF and upload that copy."
+ ) from error
+
+
+def _repair_text_appearances(content, source, widgets):
+ """QPDF cannot generate multiline appearances. Use pypdf for stale/missing text APs.
+
+ Keep existing correct appearance streams (including embedded fonts) intact.
+ Turning off NeedAppearances after repair stops QPDF from replacing them.
+ """
+ form = source.trailer["/Root"].get("/AcroForm")
+ needs_appearances = bool(form and getattr(form.get_object().get("/NeedAppearances"), "value", False))
+ missing = False
+ for widget in widgets:
+ field = widget.get("/Parent", widget).get_object()
+ if field.get("/FT") == "/Tx" and (not widget.get("/AP") or not widget["/AP"].get("/N")):
+ missing = True
+ if not (needs_appearances or missing):
+ return content
+ values = {
+ name: str(field.get("/V") or "")
+ for name, field in (source.get_fields() or {}).items()
+ if field.get("/FT") == "/Tx"
+ }
+ try:
+ writer = PdfWriter(clone_from=source)
+ writer.update_page_form_field_values(None, values, auto_regenerate=False)
+ output = io.BytesIO()
+ writer.write(output)
+ return output.getvalue()
+ except Exception as error:
+ raise PreparationError(
+ "This PDF's filled-in answers could not be rendered safely. Save a printed PDF copy and upload it again."
+ ) from error
+
+
+def _flatten(content):
+ source, widgets = inspect_pdf(content)
+ if not widgets:
+ return content, False
+ fields = source.get_fields() or {}
+ if any(field.get("/FT") == "/Sig" and field.get("/V") for field in fields.values()):
+ raise PreparationError("We cannot prepare this signed PDF. Print it to PDF and upload that copy.")
+ content = _repair_text_appearances(content, source, widgets)
+ result = _gotenberg(content, ".pdf", "/forms/pdfengines/flatten")
+ output, remaining = inspect_pdf(result)
+ if remaining or output.get_fields() or len(output.pages) != len(source.pages):
+ raise PreparationError("We could not prepare this form. Print it to PDF and upload that copy.")
+ # Engines can return 200 while dropping filled text. This is a conservative
+ # check, not a guarantee of visual fidelity; the filer still previews it.
+ visible_text = " ".join(" ".join(page.extract_text() or "" for page in output.pages).split())
+ checked_fields = set()
+ for widget in widgets:
+ field = widget.get("/Parent", widget).get_object()
+ name = field.get("/T")
+ flags = int(field.get("/Ff", 0))
+ annotation_flags = int(widget.get("/F", 0))
+ rect = widget.get("/Rect", [0, 0, 0, 0])
+ # Hidden/no-view widgets, passwords, combs, rich/formatted fields can
+ # legitimately have a different appearance from their stored /V.
+ plain_visible = (
+ field.get("/FT") == "/Tx"
+ and not (flags & ((1 << 13) | (1 << 24) | (1 << 25)))
+ and not (annotation_flags & (1 | 2 | 32))
+ and not field.get("/AA")
+ and not widget.get("/AA")
+ and rect[0] != rect[2]
+ and rect[1] != rect[3]
+ )
+ if plain_visible and name not in checked_fields:
+ checked_fields.add(name)
+ value = str(field.get("/V") or "")
+ if any(" ".join(line.split()) not in visible_text for line in value.splitlines() if line.strip()):
+ raise PreparationError("Some answers are missing. Print your file to PDF and upload that copy.")
+ return result, True
+
+
+def prepare_document(uploaded_file, jurisdiction):
+ suffix = Path(uploaded_file.name).suffix.lower()
+ if suffix not in settings.ALLOWED_FILE_TYPES:
+ raise PreparationError("Choose a PDF or Word file.")
+ uploaded_file.seek(0)
+ content = uploaded_file.read(settings.MAX_FILE_SIZE + 1)
+ uploaded_file.seek(0)
+ if not content or len(content) > settings.MAX_FILE_SIZE:
+ raise PreparationError("Choose a file that is not empty and is up to 10 MB.")
+ operation = "unchanged"
+ filename = uploaded_file.name
+ if suffix in {".doc", ".docx"}:
+ _word_format(content, suffix)
+ content = _gotenberg(
+ content,
+ suffix,
+ "/forms/libreoffice/convert",
+ {"exportFormFields": "false", "pdfua": "true", "losslessImageCompression": "true"},
+ )
+ filename = f"{Path(filename).stem}.pdf"
+ operation = "converted"
+ if requires_flattening(jurisdiction):
+ content, flattened = _flatten(content)
+ if flattened:
+ operation = "converted_flattened" if operation == "converted" else "flattened"
+ else:
+ inspect_pdf(content)
+ return PreparedDocument(content, filename, operation)
+
+
+def store_prepared_document(handler, uploaded_file, jurisdiction, role, *, keys, metadata=None):
+ """Store both copies. The caller cleans ``keys`` on any later failure."""
+ prepared = prepare_document(uploaded_file, jurisdiction)
+ original_key = ""
+ if prepared.operation != "unchanged":
+ uploaded_file.seek(0)
+ original = handler.upload_file(uploaded_file, file_type="original", metadata=metadata)
+ if not original.get("success"):
+ raise PreparationUnavailable("The original could not be saved. Try uploading again.")
+ original_key = original["key"]
+ keys.append(original_key)
+ filing = SimpleUploadedFile(prepared.filename, prepared.content, content_type="application/pdf")
+ result = handler.upload_file(filing, file_type=role, metadata=metadata)
+ if not result.get("success"):
+ raise PreparationUnavailable("The filing copy could not be saved. Try uploading again.")
+ keys.append(result["key"])
+ return {
+ "name": prepared.filename[:255],
+ "original_filename": uploaded_file.name[:255],
+ "original_s3_key": original_key,
+ "preparation": prepared.operation,
+ "preparation_reviewed_at": None,
+ "size": len(prepared.content),
+ "content_type": "application/pdf",
+ "s3_key": result["key"],
+ "public_url": handler.get_public_url(result["key"]),
+ }
+
+
+def cleanup_uploads(handler, keys):
+ for key in keys:
+ try:
+ result = handler.delete_file(key)
+ if not result.get("success"):
+ logger.warning("Could not remove an uncommitted document upload")
+ except Exception:
+ logger.exception("Could not remove an uncommitted document upload")
+
+
+def cleanup_unreferenced_uploads(keys, handler=None):
+ """Remove superseded copies after commit, preserving cross-draft references."""
+ from django.db.models import Q
+
+ from efile.models import FilingDocument
+ from efile.utils.s3_upload_handler import S3UploadHandler
+
+ unused = [
+ key
+ for key in dict.fromkeys(keys)
+ if key and not FilingDocument.objects.filter(Q(s3_key=key) | Q(original_s3_key=key)).exists()
+ ]
+ if not unused:
+ return
+ handler = handler or S3UploadHandler()
+ if handler._ensure_initialized():
+ cleanup_uploads(handler, unused)
diff --git a/efile_app/efile/services/document_previews.py b/efile_app/efile/services/document_previews.py
new file mode 100644
index 00000000..daf914d3
--- /dev/null
+++ b/efile_app/efile/services/document_previews.py
@@ -0,0 +1,28 @@
+"""Track the preview step for the current filing bytes."""
+
+import hashlib
+import json
+
+
+def unreviewed_documents(draft):
+ return draft.documents.filter(preparation_reviewed_at__isnull=True)
+
+
+def preview_fingerprint(documents):
+ return hashlib.sha256(
+ json.dumps(
+ sorted(
+ (doc.pk, doc.s3_key, doc.original_s3_key, doc.preparation, doc.size, str(doc.updated_at))
+ for doc in documents
+ )
+ ).encode()
+ ).hexdigest()
+
+
+def require_document_previews(draft):
+ if draft.documents.filter(preparation="").exists() or unreviewed_documents(draft).exists():
+ raise ValueError("Review your PDFs before you submit.")
+
+
+def document_storage_keys(document):
+ return list(dict.fromkeys(key for key in (document.s3_key, document.original_s3_key) if key))
diff --git a/efile_app/efile/services/document_uploads.py b/efile_app/efile/services/document_uploads.py
index 372176e5..fccbb51e 100644
--- a/efile_app/efile/services/document_uploads.py
+++ b/efile_app/efile/services/document_uploads.py
@@ -1,52 +1,106 @@
-from efile.models import FilingDocument
+from functools import partial
+
+from django.conf import settings
+from django.core.files.uploadedfile import SimpleUploadedFile
+from django.db import transaction
+from django.db.models import Max
+
+from efile.models import FilingDocument, FilingDraft
from efile.services.document_extractions import queue_document_extraction
-from efile.services.drafts import read_upload_data, write_upload_data
+from efile.services.document_preparation import (
+ PreparationError,
+ PreparationUnavailable,
+ cleanup_unreferenced_uploads,
+ cleanup_uploads,
+ store_prepared_document,
+)
+from efile.services.drafts import ACTIVE_DRAFT_STATUSES, read_upload_data
+from efile.services.fee_quotes import invalidate_fee_quote
from efile.utils.s3_upload_handler import S3UploadHandler
from efile.workflow import WorkflowStepKey
def upload_files(draft, uploaded_files, jurisdiction, *, current_step=WorkflowStepKey.UPLOAD_DOCUMENTS):
- """Upload PDFs immediately and queue lead analysis outside the request."""
-
+ """Prepare and store a whole batch, then queue analysis of the filing copy."""
handler = S3UploadHandler()
if not handler._ensure_initialized():
raise ValueError("Document storage is not configured. Please try again later.")
+ keys = []
+ try:
+ # Prepare the entire batch before changing the durable draft.
+ prepared = []
+ for file in uploaded_files:
+ try:
+ prepared.append(store_prepared_document(handler, file, jurisdiction, "document", keys=keys))
+ except ValueError as error:
+ raise ValueError(f"{file.name}: {error}") from error
+ with transaction.atomic():
+ draft = FilingDraft.objects.select_for_update().get(pk=draft.pk)
+ if draft.status not in ACTIVE_DRAFT_STATUSES:
+ raise ValueError("This filing is no longer available to edit.")
+ has_lead = draft.documents.filter(role=FilingDocument.Role.LEAD).exists()
+ highest = draft.documents.filter(role=FilingDocument.Role.SUPPORTING).aggregate(order=Max("sort_order"))[
+ "order"
+ ]
+ order = 0 if highest is None else highest + 1
+ for values in prepared:
+ is_lead = not has_lead
+ document = FilingDocument.objects.create(
+ draft=draft,
+ role=FilingDocument.Role.LEAD if is_lead else FilingDocument.Role.SUPPORTING,
+ sort_order=0 if is_lead else order,
+ **values,
+ )
+ if is_lead:
+ has_lead = True
+ draft.extracted_guesses = {}
+ transaction.on_commit(partial(queue_document_extraction, document), robust=True)
+ else:
+ order += 1
+ draft.current_step = str(current_step)
+ invalidate_fee_quote(draft, save=False)
+ draft.save()
+ except Exception:
+ cleanup_uploads(handler, keys)
+ raise
+ return read_upload_data(draft)
- current = read_upload_data(draft)
- files = current.setdefault("files", {})
- supporting = list(files.get("supporting", []))
- found_lead = False
-
- for uploaded_file in uploaded_files:
- validation = handler.validate_file(uploaded_file, max_size_mb=10, allowed_types=[".pdf"])
- if not validation["valid"]:
- raise ValueError(f"{uploaded_file.name}: {validation['error']}")
-
- is_lead = not files.get("lead") and not found_lead
- role = FilingDocument.Role.LEAD if is_lead else FilingDocument.Role.SUPPORTING
-
- uploaded_file.seek(0)
- result = handler.upload_file(uploaded_file, file_type=role)
- if not result["success"]:
- raise ValueError(result.get("error", f"Could not upload {uploaded_file.name}."))
-
- file_data = {
- "name": uploaded_file.name,
- "size": uploaded_file.size,
- "type": uploaded_file.content_type,
- "url": handler.get_public_url(result["key"]),
- "s3_key": result["key"],
- }
- if is_lead:
- files["lead"] = file_data
- found_lead = True
- current["guesses"] = {}
- else:
- supporting.append(file_data)
- files["supporting"] = supporting
- write_upload_data(draft, current, current_step=current_step)
- if found_lead:
- lead = FilingDocument.objects.get(draft=draft, role=FilingDocument.Role.LEAD)
- queue_document_extraction(lead)
- return current
+def prepare_stored_documents(draft, handler):
+ """Bring legacy stored uploads through the same preparation and review gate."""
+ keys = []
+ try:
+ with transaction.atomic():
+ locked = FilingDraft.objects.select_for_update().get(pk=draft.pk)
+ documents = list(locked.documents.filter(preparation=""))
+ if not documents:
+ return
+ if locked.status not in ACTIVE_DRAFT_STATUSES:
+ raise PreparationError("This filing is no longer available to edit.")
+ if not handler._ensure_initialized() or handler.s3_client is None:
+ raise PreparationUnavailable("Document storage is unavailable. Please try again later.")
+ old_keys = []
+ for document in documents:
+ if not document.s3_key:
+ raise PreparationError("The stored upload is unavailable. Replace this document before continuing.")
+ response = handler.s3_client.get_object(Bucket=handler.bucket_name, Key=document.s3_key)
+ body = response["Body"]
+ try:
+ content = body.read(settings.MAX_FILE_SIZE + 1)
+ finally:
+ body.close()
+ old_keys.extend([document.s3_key, document.original_s3_key])
+ file = SimpleUploadedFile(document.original_filename or document.name or "document.pdf", content)
+ prepared = store_prepared_document(handler, file, draft.jurisdiction, document.role, keys=keys)
+ for field, value in prepared.items():
+ setattr(document, field, value)
+ document.save()
+ if document.role == FilingDocument.Role.LEAD:
+ locked.extracted_guesses = {}
+ transaction.on_commit(partial(queue_document_extraction, document), robust=True)
+ invalidate_fee_quote(locked, save=False)
+ locked.save()
+ transaction.on_commit(partial(cleanup_unreferenced_uploads, old_keys, handler), robust=True)
+ except Exception:
+ cleanup_uploads(handler, keys)
+ raise
diff --git a/efile_app/efile/services/draft_urls.py b/efile_app/efile/services/draft_urls.py
index 224aff78..215fc203 100644
--- a/efile_app/efile/services/draft_urls.py
+++ b/efile_app/efile/services/draft_urls.py
@@ -8,6 +8,7 @@
{
"filing_path",
"upload_documents",
+ "preview_documents",
"document_extraction_status",
"extraction_review",
"case_lookup",
diff --git a/efile_app/efile/services/drafts.py b/efile_app/efile/services/drafts.py
index c34f37a0..3ca46035 100644
--- a/efile_app/efile/services/drafts.py
+++ b/efile_app/efile/services/drafts.py
@@ -13,6 +13,7 @@
from django.db.models import QuerySet
from efile.models import FilingDocument, FilingDraft, FilingParty
+from efile.services.document_preparation import cleanup_unreferenced_uploads
from efile.workflow import WorkflowStepKey, legacy_existing_case_value, normalize_existing_case
ACTIVE_DRAFT_STATUSES = (FilingDraft.Status.DRAFT, FilingDraft.Status.ERROR)
@@ -388,6 +389,11 @@ def _positive_int(value: Any) -> int | None:
def _apply_document(doc: FilingDocument, file_obj: dict[str, Any], config: dict[str, Any]) -> None:
+ if "s3_key" in file_obj and _as_str(file_obj.get("s3_key")) != doc.s3_key:
+ doc.original_filename = ""
+ doc.original_s3_key = ""
+ doc.preparation = ""
+ doc.preparation_reviewed_at = None
if "name" in file_obj:
doc.name = _as_str(file_obj.get("name"))
if not doc.original_filename:
@@ -445,8 +451,11 @@ def _upsert_document(
config: dict[str, Any],
) -> None:
doc, _created = FilingDocument.objects.get_or_create(draft=draft, role=role, sort_order=sort_order)
+ old_keys = [doc.s3_key, doc.original_s3_key]
_apply_document(doc, file_obj, config)
doc.save()
+ if old_keys[0] != doc.s3_key:
+ transaction.on_commit(lambda: cleanup_unreferenced_uploads(old_keys), robust=True)
@transaction.atomic
@@ -458,6 +467,9 @@ def write_upload_data(
) -> FilingDraft:
"""Persist a (possibly partial) upload_data blob into FilingDocument rows."""
+ locked = FilingDraft.objects.select_for_update().get(pk=draft.pk)
+ if locked.status not in ACTIVE_DRAFT_STATUSES:
+ raise ValueError("This filing is no longer available to edit.")
data = dict(upload_data or {})
update_fields: list[str] = []
@@ -494,18 +506,31 @@ def write_upload_data(
# Handoff provenance and pending corrections are keyed by row id, so
# they follow the file to its rebuilt row the same way.
previous_ids = {document.s3_key: document.pk for document in previous}
+ preparation_metadata = {
+ document.s3_key: {
+ field: getattr(document, field)
+ for field in ("original_filename", "original_s3_key", "preparation", "preparation_reviewed_at")
+ }
+ for document in previous
+ }
FilingDocument.objects.filter(draft=draft, role=FilingDocument.Role.SUPPORTING).delete()
for index, file_obj in enumerate(supporting_files):
config = supporting_configs[index] if index < len(supporting_configs) else {}
_upsert_document(draft, FilingDocument.Role.SUPPORTING, index, file_obj or {}, config or {})
moved = {}
for document in FilingDocument.objects.filter(draft=draft, role=FilingDocument.Role.SUPPORTING):
+ if metadata := preparation_metadata.get(document.s3_key):
+ for field, value in metadata.items():
+ setattr(document, field, value)
+ document.save(update_fields=[*metadata, "updated_at"])
item_id = claimed_items.get(document.s3_key, "")
if item_id:
document.checklist_item_id = item_id
document.save(update_fields=["checklist_item_id", "updated_at"])
if document.s3_key in previous_ids:
moved[previous_ids[document.s3_key]] = document.pk
+ old_keys = [key for document in previous for key in (document.s3_key, document.original_s3_key)]
+ transaction.on_commit(lambda: cleanup_unreferenced_uploads(old_keys), robust=True)
if moved:
from efile.services.handoff import carry_document_paths
diff --git a/efile_app/efile/services/handoff.py b/efile_app/efile/services/handoff.py
index 76d69986..a0cd1516 100644
--- a/efile_app/efile/services/handoff.py
+++ b/efile_app/efile/services/handoff.py
@@ -200,7 +200,7 @@ def validate_payload(payload, source_config, files, *, require_lead=True):
raise HandoffError("return_url must use an allowed HTTPS origin.")
documents = payload.get("documents", [])
if not isinstance(documents, list) or len(documents) > 20:
- raise HandoffError("documents must be a list of up to 20 PDFs.")
+ raise HandoffError("documents must be a list of up to 20 documents.")
ids = set()
leads = 0
for document in documents:
@@ -216,21 +216,19 @@ def validate_payload(payload, source_config, files, *, require_lead=True):
_hints(document, "document")
uploaded = files.get(key)
if uploaded is None:
- raise HandoffError(f"Upload the PDF for document {key}.")
+ raise HandoffError(f"Upload the PDF or Word file for document {key}.")
if uploaded.size > MAX_DOCUMENT_BYTES:
- raise HandoffError("Each PDF must be at most 10 MB.")
+ raise HandoffError("Each document must be at most 10 MB.")
digest = hashlib.sha256()
- prefix = uploaded.read(5)
- uploaded.seek(0)
- if prefix != b"%PDF-":
- raise HandoffError("Only PDF documents are accepted.")
+ # Content validation and PDF/Word preparation share the app upload
+ # pipeline, which distinguishes unfixable input from service outages.
for chunk in uploaded.chunks():
digest.update(chunk)
uploaded.seek(0)
if digest.hexdigest() != document.get("sha256"):
raise HandoffError(f"Document hash mismatch: {key}.")
if leads > 1 or (require_lead and documents and leads != 1):
- raise HandoffError("A document bundle needs exactly one lead PDF.")
+ raise HandoffError("A document bundle needs exactly one lead document.")
if set(files) != ids or any(len(files.getlist(key)) != 1 for key in files):
raise HandoffError("Upload each declared document exactly once.")
_validate_filing_hint_overrides(payload, ids)
@@ -320,12 +318,14 @@ def populate(draft, payload, uploads):
draft=draft,
role=document["role"],
sort_order=order[document["role"]],
- name=document.get("form_name") or uploaded["filename"],
+ name=document.get("form_name") or uploaded.get("name", uploaded["filename"]),
original_filename=uploaded["filename"],
size=uploaded["size"],
content_type="application/pdf",
s3_key=uploaded["key"],
public_url=uploaded["url"],
+ original_s3_key=uploaded.get("original_s3_key", ""),
+ preparation=uploaded.get("preparation", ""),
)
order[document["role"]] += 1
record(draft, f"documents.{row.pk}", "source_suggestion", document)
diff --git a/efile_app/efile/services/taxonomy_classification.py b/efile_app/efile/services/taxonomy_classification.py
index 7622b753..47fe94e1 100644
--- a/efile_app/efile/services/taxonomy_classification.py
+++ b/efile_app/efile/services/taxonomy_classification.py
@@ -829,7 +829,7 @@ def _select(
"extracted_evidence": evidence,
"crosswalk_matches": crosswalk,
"crosswalk_constraints": crosswalk_summary,
- "source_scope": f"MarkItDown text from the first {settings.DOCUMENT_CLASSIFICATION_SOURCE_PAGES} pages",
+ "source_scope": "Locally extracted document text, including stored PDF form values when present",
},
)
inference = version_config.get("inference", {})
diff --git a/efile_app/efile/settings_base.py b/efile_app/efile/settings_base.py
index e18522cd..71bb1273 100644
--- a/efile_app/efile/settings_base.py
+++ b/efile_app/efile/settings_base.py
@@ -146,10 +146,17 @@
DOCUMENT_EXTRACTION_MEMORY_MB = int(os.getenv("DOCUMENT_EXTRACTION_MEMORY_MB", "768"))
MAX_FILE_SIZE = 10 * 1024 * 1024 # 10MB
ALLOWED_FILE_TYPES = [".pdf", ".doc", ".docx"]
+
+# Gotenberg document conversion service
+GOTENBERG_URL = os.getenv("GOTENBERG_URL", "")
+GOTENBERG_USERNAME = os.getenv("GOTENBERG_USERNAME", "")
+GOTENBERG_PASSWORD = os.getenv("GOTENBERG_PASSWORD", "")
+DOCUMENT_PREPARATION_TIMEOUT_SECONDS = int(os.getenv("DOCUMENT_PREPARATION_TIMEOUT_SECONDS", "45"))
# Analyze only the front of a filing. Exhibits and discovery can make a PDF
# hundreds of pages long, while the caption and filing details normally appear
# near the beginning.
DOCUMENT_EXTRACTION_MAX_PAGES = int(os.getenv("DOCUMENT_EXTRACTION_MAX_PAGES", "20"))
+DOCUMENT_EXTRACTION_MAX_TEXT_CHARS = int(os.getenv("DOCUMENT_EXTRACTION_MAX_TEXT_CHARS", "100000"))
DOCUMENT_EXTRACTION_MAX_ATTEMPTS = int(os.getenv("DOCUMENT_EXTRACTION_MAX_ATTEMPTS", "3"))
DOCUMENT_CLASSIFICATION_SOURCE_PAGES = int(os.getenv("DOCUMENT_CLASSIFICATION_SOURCE_PAGES", "3"))
DOCUMENT_EVIDENCE_MODEL = os.getenv("DOCUMENT_EVIDENCE_MODEL", "")
diff --git a/efile_app/efile/static/config/states/vermont.yaml b/efile_app/efile/static/config/states/vermont.yaml
index 65b1ece2..bdb8e023 100644
--- a/efile_app/efile/static/config/states/vermont.yaml
+++ b/efile_app/efile/static/config/states/vermont.yaml
@@ -35,7 +35,16 @@ state:
# Words and sentences this jurisdiction says differently. Keys, defaults, and
# what each one is for are in efile/utils/ui_text.py; anything not listed here
# uses the default English wording.
+# Vermont requires form-fillable PDFs to be flattened before filing.
+# https://www.vtcourts.gov/about-vermont-judiciary/electronic-access/electronic-filing/faqs
+document_preparation:
+ flatten_pdf_forms: true
+
text:
+ upload_documents:
+ preparation_help: >-
+ Upload your files as they are. We make PDF copies for court.
+ [Vermont's PDF rules](https://www.vtcourts.gov/about-vermont-judiciary/electronic-access/electronic-filing/faqs).
terms:
# Vermont courts call the document that opens a case a complaint.
starting_document_example: "complaint"
diff --git a/efile_app/efile/static/css/document-preview.css b/efile_app/efile/static/css/document-preview.css
new file mode 100644
index 00000000..b4715718
--- /dev/null
+++ b/efile_app/efile/static/css/document-preview.css
@@ -0,0 +1,57 @@
+.document-preview {
+ border: 1px solid #cbd5e1;
+ border-radius: 0.5rem;
+ margin-block: 0.75rem;
+ background: #fff;
+}
+
+.document-preview>summary {
+ padding: 0.75rem;
+ color: #163b62;
+ cursor: pointer;
+ overflow-wrap: anywhere;
+}
+
+.document-preview__body {
+ padding: 0.75rem;
+}
+
+.document-preview__toolbar {
+ display: flex;
+ gap: 0.5rem;
+ align-items: center;
+ flex-wrap: wrap;
+ margin-block-end: 0.75rem;
+}
+
+.document-preview__toolbar[hidden] {
+ display: none;
+}
+
+.document-preview__toolbar input {
+ width: 4rem;
+}
+
+.document-preview__frame {
+ position: relative;
+ height: min(70vh, 50rem);
+ min-height: 15rem;
+}
+
+.document-preview__viewport {
+ position: absolute;
+ inset: 0;
+ overflow: auto;
+ background: #e2e8f0;
+}
+
+.document-preview :focus-visible {
+ outline: 3px solid #163b62;
+ outline-offset: 3px;
+}
+
+@media (forced-colors: active) {
+ .document-preview :focus-visible {
+ outline-color: Highlight;
+ }
+}
\ No newline at end of file
diff --git a/efile_app/efile/static/js/document-preview.js b/efile_app/efile/static/js/document-preview.js
new file mode 100644
index 00000000..d47a397c
--- /dev/null
+++ b/efile_app/efile/static/js/document-preview.js
@@ -0,0 +1,101 @@
+/* PDF.js is served locally; document bytes stay on the authenticated app origin. */
+(() => {
+ const assetScript = document.querySelector("script[data-pdf-library]");
+ let modules;
+
+ function loadModules() {
+ if (!modules) {
+ modules = import(assetScript.dataset.pdfLibrary).then(async (pdfjs) => {
+ pdfjs.GlobalWorkerOptions.workerSrc = assetScript.dataset.pdfWorker;
+ globalThis.pdfjsLib = pdfjs;
+ const components = await import(assetScript.dataset.pdfViewer);
+ return {
+ pdfjs,
+ components
+ };
+ });
+ }
+ return modules;
+ }
+
+ async function openPreview(details) {
+ if (!details.open || details.dataset.loaded) return;
+ details.dataset.loaded = "true";
+ const status = details.querySelector("[data-pdf-status]");
+ status.textContent = gettext("Loading PDF…");
+ let task;
+ try {
+ const {
+ pdfjs,
+ components
+ } = await loadModules();
+ const container = details.querySelector("[data-pdf-url]");
+ const eventBus = new components.EventBus();
+ const linkService = new components.PDFLinkService({
+ eventBus
+ });
+ const viewer = new components.PDFViewer({
+ container,
+ eventBus,
+ linkService,
+ textLayerMode: 1,
+ annotationMode: pdfjs.AnnotationMode.ENABLE,
+ enableScripting: false,
+ });
+ linkService.setViewer(viewer);
+ task = pdfjs.getDocument({
+ url: container.dataset.pdfUrl,
+ isEvalSupported: false,
+ cMapUrl: assetScript.dataset.pdfResources + "cmaps/",
+ cMapPacked: true,
+ standardFontDataUrl: assetScript.dataset.pdfResources + "standard_fonts/",
+ wasmUrl: assetScript.dataset.pdfResources + "wasm/",
+ disableRange: true,
+ disableAutoFetch: true,
+ });
+ const pdf = await task.promise;
+ viewer.setDocument(pdf);
+ linkService.setDocument(pdf);
+ const pageInput = details.querySelector("[data-pdf-page]");
+ pageInput.max = pdf.numPages;
+ details.querySelector("[data-pdf-page-count]").textContent = interpolate(gettext("of %s"), [pdf.numPages]);
+ eventBus.on("pagesinit", () => {
+ viewer.currentScaleValue = "page-width";
+ for (let index = 0; index < pdf.numPages; index++) {
+ const pageRegion = viewer.getPageView(index).div;
+ pageRegion.removeAttribute("data-l10n-id");
+ pageRegion.removeAttribute("data-l10n-args");
+ pageRegion.setAttribute("aria-label", interpolate(gettext("%s, page %s"), [container.getAttribute("aria-label"), index + 1]));
+ }
+ });
+ eventBus.on("pagerendered", (event) => {
+ status.textContent = event.error ? gettext("This page did not load. Download the PDF to view it.") : "";
+ if (!event.error) details.dataset.rendered = "true";
+ });
+ eventBus.on("pagechanging", (event) => {
+ pageInput.value = event.pageNumber;
+ });
+ details.querySelector(".document-preview__toolbar").hidden = false;
+ pageInput.addEventListener("change", () => {
+ const page = Number(pageInput.value);
+ if (Number.isInteger(page) && page >= 1 && page <= pdf.numPages) viewer.currentPageNumber = page;
+ });
+ details.querySelector("[data-pdf-previous]").addEventListener("click", () => viewer.previousPage());
+ details.querySelector("[data-pdf-next]").addEventListener("click", () => viewer.nextPage());
+ details.querySelector("[data-pdf-zoom-in]").addEventListener("click", () => viewer.increaseScale());
+ details.querySelector("[data-pdf-zoom-out]").addEventListener("click", () => viewer.decreaseScale());
+ details.addEventListener("toggle", () => {
+ if (details.open) viewer.update();
+ });
+ } catch (error) {
+ console.warn("PDF preview failed", error.message);
+ if (task) await task.destroy();
+ modules = undefined;
+ delete details.dataset.loaded;
+ status.textContent = gettext("The PDF did not load. Download it or close and reopen this view.");
+ }
+ }
+ document.querySelectorAll("[data-pdf-preview]").forEach((details) => {
+ details.addEventListener("toggle", () => openPreview(details));
+ });
+})();
\ No newline at end of file
diff --git a/efile_app/efile/static/js/upload-documents.js b/efile_app/efile/static/js/upload-documents.js
index b2db8d60..c43473c9 100644
--- a/efile_app/efile/static/js/upload-documents.js
+++ b/efile_app/efile/static/js/upload-documents.js
@@ -134,7 +134,7 @@
const fileCountLabel = selectedFiles.size === 1 ? "file" : "files";
dropZone.querySelector("strong").textContent = selectedFiles.size ?
`${selectedFiles.size} ${fileCountLabel} selected` :
- "Choose PDFs or drag them here";
+ "Choose PDFs or Word documents, or drag them here";
}
function addFiles(files) {
@@ -172,8 +172,8 @@
errorBox.hidden = true;
state.hidden = false;
uploadButton.disabled = true;
- stateTitle.textContent = "Uploading your documents…";
- stateDetail.textContent = "Keep this page open while the files upload.";
+ stateTitle.textContent = "Uploading your files…";
+ stateDetail.textContent = "Keep this page open while we make your PDFs.";
try {
const response = await fetch(window.location.href, {
@@ -190,8 +190,8 @@
const result = await response.json();
if (!response.ok || !result.success) throw new Error(result.error || "Upload failed.");
stateTitle.textContent = result.extraction_pending ? "Your documents are uploaded" : "Your documents are ready";
- let pendingDetail = "Analysis will continue in the background.";
- if (aiIsOff()) pendingDetail = "We are checking your PDF's text for a form number, without AI.";
+ let pendingDetail = "You can review your PDFs while we read your first file.";
+ if (aiIsOff()) pendingDetail = "AI is off. We are looking for form and case numbers.";
stateDetail.textContent = result.extraction_pending ?
pendingDetail :
"Review what we found before you continue.";
diff --git a/efile_app/efile/static/js/waiver-upload.js b/efile_app/efile/static/js/waiver-upload.js
index 6d201ef9..a32bc705 100644
--- a/efile_app/efile/static/js/waiver-upload.js
+++ b/efile_app/efile/static/js/waiver-upload.js
@@ -60,7 +60,7 @@ document.addEventListener("DOMContentLoaded", () => {
upload.addEventListener("click", async () => {
if (uploading) return;
if (!file.files.length || !confidentiality.value) {
- status.textContent = gettext("Choose a PDF and a confidentiality setting.");
+ status.textContent = gettext("Choose a PDF or Word document and a confidentiality setting.");
(!file.files.length ? file : confidentiality).focus();
return;
}
@@ -83,6 +83,10 @@ document.addEventListener("DOMContentLoaded", () => {
});
const data = await response.json();
if (!response.ok || !data.success) throw new Error(data.error || gettext("The upload failed. Try again."));
+ if (data.preview_url) {
+ window.location.assign(window.withFilingDraft(data.preview_url));
+ return;
+ }
document.getElementById("fee-inputs-token").textContent = JSON.stringify(data.fee_inputs_token);
document.getElementById("waiver-upload-required").hidden = true;
const confirmation = document.getElementById("waiver-upload-confirmation");
diff --git a/efile_app/efile/templates/efile/components/document_preview.html b/efile_app/efile/templates/efile/components/document_preview.html
new file mode 100644
index 00000000..5639725b
--- /dev/null
+++ b/efile_app/efile/templates/efile/components/document_preview.html
@@ -0,0 +1,50 @@
+{% load i18n %}
+
+ {% translate "Download PDF" %}
+ {% if document.original_s3_key or not document.preparation %}
+ · {% translate "Download original" %}
+ {% endif %}
+ {% translate "View PDF:" %} {{ document.name|default:document.original_filename }}
+