Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
28 changes: 21 additions & 7 deletions efile_app/efile/services/document_extractions.py
Original file line number Diff line number Diff line change
Expand Up @@ -29,7 +29,6 @@
HierarchicalDocumentClassifier,
deterministic_form_identity,
exact_form_crosswalk_matches,
primary_amount_in_controversy,
scan_document_for_form_identifiers,
summarize_form_crosswalk_matches,
)
Expand All @@ -44,6 +43,12 @@ class ExtractionSuperseded(Exception):
"""The requested analysis changed; this is not a processing failure."""


# How long a filer waits on the upload page for the first file to be read.
# After that the filer may go on and enter the case details by hand; the
# analysis keeps running and its guesses are still offered if they arrive.
EXTRACTION_WAIT_LIMIT = timedelta(minutes=2)


def queue_document_extraction(document):
"""Create or reset the one background extraction job for a lead document."""
if document.role != FilingDocument.Role.LEAD:
Expand All @@ -64,11 +69,22 @@ def queue_document_extraction(document):
"error": "",
"started_at": None,
"completed_at": None,
# Restart the filer's wait for a replaced lead document.
"created_at": timezone.now(),
},
)
return job


def extraction_is_waiting(extraction):
"""Whether the filer should still wait for this analysis before reviewing."""
return (
extraction is not None
and extraction.status in {DocumentExtraction.Status.PENDING, DocumentExtraction.Status.PROCESSING}
and extraction.created_at > timezone.now() - EXTRACTION_WAIT_LIMIT
)


def extraction_for_document(document):
try:
return document.extraction
Expand Down Expand Up @@ -441,12 +457,10 @@ def check_outbound_permission():
).exists()
if is_current_lead:
draft.extracted_guesses = guesses
update_fields = ["extracted_guesses", "updated_at"]
amount = primary_amount_in_controversy(evidence)
if amount and not draft.amount_in_controversy:
draft.amount_in_controversy = amount
update_fields.append("amount_in_controversy")
draft.save(update_fields=update_fields)
# A filer may already have continued manually after the wait limit.
# Keep extracted amounts in evidence until the filer confirms them
# on the case-questions form; workers must not change fee inputs.
draft.save(update_fields=["extracted_guesses", "updated_at"])
job.status = DocumentExtraction.Status.COMPLETE
job.total_pages = total_pages
job.pages_analyzed = pages_analyzed
Expand Down
87 changes: 78 additions & 9 deletions efile_app/efile/static/js/upload-documents.js
Original file line number Diff line number Diff line change
Expand Up @@ -33,6 +33,63 @@
const aiRememberChoice = document.getElementById("ai-remember-choice");
const selectedFiles = new Map();
let nextSelectionId = 0;
let uploading = false;
const hasLead = form.dataset.hasLead === "true";
let analysisReady = hasLead && form.dataset.extractionPending !== "true";
// After the server's wait limit the filer may go on without the analysis.
let waitOver = hasLead && form.dataset.waitSeconds === "0";
// The analysis status the panel shows whenever no upload is in progress.
let analysisTitle = stateTitle.textContent;
let analysisDetail = stateDetail.textContent;

function canContinue() {
return !uploading && (analysisReady || waitOver);
}

function showAnalysisStatus(title, detail) {
analysisTitle = title;
analysisDetail = detail;
if (uploading) return;
stateTitle.textContent = title;
stateDetail.textContent = detail;
}

function endWait() {
if (waitOver) return;
waitOver = true;
updateContinue();
if (!analysisReady) {
showAnalysisStatus(analysisTitle, "This is taking longer than usual. You can continue and enter the case details yourself.");
}
}

function updateContinue() {
const disabled = !canContinue();
continueButton.classList.toggle("disabled", disabled);
if (disabled) {
continueButton.setAttribute("aria-disabled", "true");
continueButton.setAttribute("tabindex", "-1");
continueButton.removeAttribute("href");
} else {
continueButton.removeAttribute("aria-disabled");
continueButton.removeAttribute("tabindex");
continueButton.setAttribute("href", continueButton.dataset.continueUrl);
}
}

continueButton.addEventListener("click", (event) => {
if (!canContinue()) event.preventDefault();
});
updateContinue();
// The server decides when the wait is over; its status checks open
// Continue. This timer only matters if those checks keep failing, so a
// broken connection cannot hold the filer here.
let waitTimerDone = false;
if (hasLead && !waitOver) {
window.setTimeout(() => {
waitTimerDone = true;
}, Number(form.dataset.waitSeconds) * 1000);
}
// The "remember this" row is offered only after the filer changes the
// setting, and only for the rest of this page load. Until then the account
// preference is not this request's business, so it is left out of the post.
Expand Down Expand Up @@ -169,6 +226,9 @@

form.addEventListener("submit", async (event) => {
event.preventDefault();
if (uploading) return;
uploading = true;
updateContinue();
errorBox.hidden = true;
state.hidden = false;
uploadButton.disabled = true;
Expand All @@ -190,14 +250,20 @@
const result = await response.json();
if (!response.ok || !result.success) throw new Error(result.error || "Upload failed.");
stateTitle.textContent = result.extraction_pending ? "Your documents are uploaded" : "Your documents are ready";
let pendingDetail = "You can review your PDFs while we read your first file.";
let pendingDetail = "Please wait while we read your first file. You can continue when it is ready.";
if (aiIsOff()) pendingDetail = "AI is off. We are looking for form and case numbers.";
stateDetail.textContent = result.extraction_pending ?
pendingDetail :
"Review what we found before you continue.";
window.setTimeout(() => window.location.reload(), 300);
} catch (error) {
state.hidden = true;
uploading = false;
updateContinue();
// Go back to the first file's analysis status, including any
// result that arrived while this upload was running.
state.hidden = form.dataset.extractionPending !== "true";
stateTitle.textContent = analysisTitle;
stateDetail.textContent = analysisDetail;
errorBox.textContent = error.message;
errorBox.hidden = false;
uploadButton.disabled = false;
Expand Down Expand Up @@ -236,30 +302,33 @@
const result = await response.json();
if (!response.ok || !result.success) throw new Error(result.error || "Could not check document analysis.");
if (!result.ready) {
if (result.wait_seconds === 0) endWait();
window.setTimeout(pollExtraction, 2500);
return;
}

state.querySelector(".spinner-border")?.remove();
let readyTitle = "Document analysis is ready";
if (result.ai_opted_out) readyTitle = "We finished checking your document";
stateTitle.textContent = result.status === "failed" ? "Your document is ready for manual review" : readyTitle;
// Nothing is reviewed on this page: the details are on the next
// one, so say where the checking actually happens.
const nextPageNudge = "Review the information carefully on the next page.";
stateDetail.textContent = result.total_pages > result.pages_analyzed ?
analysisReady = true;
showAnalysisStatus(
result.status === "failed" ? "Your document is ready for manual review" : readyTitle,
result.total_pages > result.pages_analyzed ?
`We read the first ${result.pages_analyzed} of ${result.total_pages} pages. ${nextPageNudge}` :
nextPageNudge;
continueButton.classList.remove("disabled");
continueButton.removeAttribute("aria-disabled");
continueButton.removeAttribute("tabindex");
nextPageNudge
);
updateContinue();
const analyzingPill = document.querySelector(".status-pill--analyzing");
if (analyzingPill) {
analyzingPill.classList.replace("status-pill--analyzing", "status-pill--ready");
analyzingPill.innerHTML = '<i class="fa-solid fa-check" aria-hidden="true"></i> Ready';
}
} catch (error) {
stateDetail.textContent = error.message;
showAnalysisStatus(analysisTitle, error.message);
if (waitTimerDone) endWait();
window.setTimeout(pollExtraction, 5000);
}
}
Expand Down
12 changes: 6 additions & 6 deletions efile_app/efile/templates/efile/upload_documents.html
Original file line number Diff line number Diff line change
Expand Up @@ -47,7 +47,10 @@ <h1>{% translate "Upload your court documents" %}</h1>
</p>
<form id="document-upload-form"
enctype="multipart/form-data"
data-extraction-status-url="{% url 'document_extraction_status' jurisdiction %}">
data-extraction-status-url="{% url 'document_extraction_status' jurisdiction %}"
data-has-lead="{{ has_lead_document|yesno:'true,false' }}"
data-extraction-pending="{{ extraction_pending|yesno:'true,false' }}"
data-wait-seconds="{{ extraction_wait_seconds }}">
{% csrf_token %}
<label class="drop-zone" id="document-drop-zone" for="documents-input">
<input id="documents-input"
Expand Down Expand Up @@ -82,7 +85,7 @@ <h2 id="pending-files-heading">{% translate "Selected files" %}</h2>
</strong>
<span id="upload-state-detail">
{% if extraction_pending %}
{% translate "You can leave this page and come back while we work." %}
{% translate "Please wait while we read your first file. You can continue when it is ready." %}
{% else %}
{% translate "Keep this page open while the files upload." %}
{% endif %}
Expand Down Expand Up @@ -219,10 +222,7 @@ <h2 id="uploaded-heading">{% translate "Your documents" %}</h2>
{% else %}
<a class="btn btn-outline-secondary" href="{{ workflow_previous_url }}"><i class="fa-solid fa-arrow-left" aria-hidden="true"></i> {% translate "Back" %}</a>
{% endif %}
<a class="btn btn-primary {% if not has_lead_document %}disabled{% endif %}"
id="continue-to-analysis"
href="{% url 'preview_documents' jurisdiction %}{% if return_to %}?return_to={{ return_to }}{% endif %}"
{% if not has_lead_document %}aria-disabled="true" tabindex="-1"{% endif %}>{% translate "Preview your PDFs" %} <i class="fa-solid fa-arrow-right" aria-hidden="true"></i></a>
<a class="btn btn-primary {% if not has_lead_document or extraction_wait_seconds %}disabled{% endif %}" id="continue-to-analysis" data-continue-url="{% url 'preview_documents' jurisdiction %}{% if return_to %}?return_to={{ return_to }}{% endif %}" {% if not has_lead_document or extraction_wait_seconds %}aria-disabled="true" tabindex="-1"{% else %}href="{% url 'preview_documents' jurisdiction %}{% if return_to %}?return_to={{ return_to }}{% endif %}"{% endif %}>{% translate "Preview your PDFs" %} <i class="fa-solid fa-arrow-right" aria-hidden="true"></i></a>
</div>
</section>
{% endblock workflow_content %}
Expand Down
103 changes: 102 additions & 1 deletion efile_app/efile/tests/test_document_extractions.py
Original file line number Diff line number Diff line change
@@ -1,22 +1,27 @@
import os
import re
import shutil
from datetime import timedelta
from pathlib import Path
from unittest.mock import MagicMock, patch

import pytest
from django.test import override_settings
from django.urls import reverse
from django.utils import timezone
from pypdf import PdfReader, PdfWriter

from efile.models import DocumentExtraction, FilingDocument, FilingDraft
from efile.services.current_drafts import CURRENT_DRAFT_SESSION_KEY
from efile.services.document_extractions import (
EXTRACTION_WAIT_LIMIT,
claim_next_extraction,
extraction_is_waiting,
process_document_extraction,
queue_document_extraction,
)
from efile.services.extraction_fields import normalize_document_evidence, normalize_extracted_fields
from efile.services.fee_quotes import fee_fingerprint
from efile.services.taxonomy_classification import ClassificationRun, HierarchicalDocumentClassifier
from efile.tests.helpers import reviewed_document
from efile.workflow import ExistingCase
Expand Down Expand Up @@ -191,7 +196,8 @@ def classify_source(_classifier, jurisdiction, evidence, source_text):
assert job.analysis_metadata["form_identifier_scan_source"] == "pypdf"
assert job.analysis_metadata["form_identifier_scan_ms"] < 2000
assert extraction_draft.extracted_guesses["court"] == "Middlesex Probate and Family Court"
assert extraction_draft.amount_in_controversy == "1275"
assert extraction_draft.amount_in_controversy == ""
assert job.evidence["monetary amounts"][0]["amount"] == "1275"


@pytest.mark.django_db
Expand Down Expand Up @@ -232,6 +238,7 @@ def test_status_endpoint_reports_when_review_is_ready(client, extraction_draft):
"status": "complete",
"ai_opted_out": False,
"ready": True,
"wait_seconds": 0,
"pages_analyzed": 20,
"total_pages": 30,
"review_url": reverse("preview_documents", kwargs={"jurisdiction": "illinois"}),
Expand Down Expand Up @@ -456,6 +463,100 @@ def test_management_command_processes_and_retries_failures(extraction_draft):
assert "Document analysis failed" in job.error


@pytest.mark.django_db
@pytest.mark.parametrize("status", [None, "pending", "processing", "complete", "failed"])
def test_upload_continue_waits_for_analysis(client, extraction_draft, status):
authorize(client, extraction_draft)
if status is not None:
document = reviewed_document(draft=extraction_draft, role=FilingDocument.Role.LEAD, name="petition.pdf")
DocumentExtraction.objects.create(document=document, status=status)
response = client.get(reverse("upload_documents", kwargs={"jurisdiction": "illinois"}))
assert response.status_code == 200
content = response.content.decode()
match = re.search(r'<a\b[^>]*id="continue-to-analysis"[^>]*>', content)
assert match is not None
link = match.group()
if status in {"complete", "failed"}:
assert ' href="' in link
assert 'aria-disabled="true"' not in link
else:
assert ' href="' not in link
assert 'aria-disabled="true"' in link
assert "You can leave this page and come back while we work." not in content


def _queued_lead(draft, *, waited):
document = reviewed_document(draft=draft, role=FilingDocument.Role.LEAD, name="petition.pdf")
job = queue_document_extraction(document)
DocumentExtraction.objects.filter(pk=job.pk).update(created_at=timezone.now() - waited)
return document


@pytest.mark.django_db
@pytest.mark.parametrize(("waited", "may_continue"), [(timedelta(0), False), (EXTRACTION_WAIT_LIMIT, True)])
def test_filer_may_go_on_without_analysis_after_the_wait_limit(client, extraction_draft, waited, may_continue):
authorize(client, extraction_draft)
_queued_lead(extraction_draft, waited=waited + timedelta(seconds=1) if may_continue else waited)

page = client.get(reverse("upload_documents", kwargs={"jurisdiction": "illinois"})).content.decode()
link = re.search(r'<a\b[^>]*id="continue-to-analysis"[^>]*>', page)
assert link is not None
assert (' href="' in link.group()) is may_continue
wait = client.get(reverse("document_extraction_status", kwargs={"jurisdiction": "illinois"})).json()
assert wait["ready"] is False
assert (wait["wait_seconds"] == 0) is may_continue
assert 0 <= wait["wait_seconds"] <= EXTRACTION_WAIT_LIMIT.total_seconds()
review = client.get(reverse("extraction_review", kwargs={"jurisdiction": "illinois"}))
assert (review.status_code == 200) is may_continue


@pytest.mark.django_db
def test_replacing_the_lead_restarts_the_wait(extraction_draft):
document = _queued_lead(extraction_draft, waited=EXTRACTION_WAIT_LIMIT * 2)
job = queue_document_extraction(document)
assert job.created_at > timezone.now() - timedelta(seconds=5)


@pytest.mark.django_db
@pytest.mark.parametrize("saved_amount", ["", "500.00"])
def test_late_analysis_keeps_fee_inputs_unchanged(extraction_draft, saved_amount):
_queued_lead(extraction_draft, waited=EXTRACTION_WAIT_LIMIT * 2)
claimed = claim_next_extraction()
assert not extraction_is_waiting(claimed)
extraction_draft.current_step = "review"
extraction_draft.amount_in_controversy = saved_amount
extraction_draft.save(update_fields=["current_step", "amount_in_controversy"])
fingerprint = fee_fingerprint(extraction_draft)

handler = MagicMock()

def download(_key, destination):
writer = PdfWriter()
writer.add_blank_page(width=612, height=792)
with open(destination, "wb") as pdf_file:
writer.write(pdf_file)
return {"success": True}

handler.download_file.side_effect = download
evidence = {"monetary amounts": [{"label": "Amount in controversy", "amount": "1275.00"}]}
with (
patch("efile.services.document_extractions.S3UploadHandler", return_value=handler),
patch(
"efile.services.document_extractions.analyze_document",
return_value={"guesses": {"document title": "Complaint"}, "evidence": evidence},
),
):
process_document_extraction(claimed.pk, claimed.claim_token)

extraction_draft.refresh_from_db()
claimed.refresh_from_db()
assert claimed.status == DocumentExtraction.Status.COMPLETE
assert claimed.evidence == evidence
assert extraction_draft.extracted_guesses == {"document title": "Complaint"}
assert extraction_draft.amount_in_controversy == saved_amount
assert fee_fingerprint(extraction_draft) == fingerprint


@pytest.mark.django_db
@pytest.mark.parametrize(("configured", "expected"), [(None, "2"), ("4", "4")])
def test_worker_children_start_with_few_malloc_arenas(extraction_draft, monkeypatch, configured, expected):
Expand Down
2 changes: 1 addition & 1 deletion efile_app/efile/tests/test_document_previews.py
Original file line number Diff line number Diff line change
Expand Up @@ -324,7 +324,7 @@ def test_change_files_keeps_where_the_filer_came_from(client, preview_draft):
assert 'upload-documents/?return_to=review"' in preview
upload = client.get(url("upload_documents", preview_draft) + "&return_to=review").content.decode()
# Both ways off the upload page lead back toward Review, not the start.
assert upload.count('preview-documents/?return_to=review"') == 2
assert upload.count('href="/jurisdiction/vermont/preview-documents/?return_to=review"') == 2
unknown = client.get(url("upload_documents", preview_draft) + "&return_to=elsewhere").content.decode()
assert "return_to" not in unknown

Expand Down
Loading
Loading