diff --git a/app/database.py b/app/database.py index 48c7d47..b8f4f6a 100644 --- a/app/database.py +++ b/app/database.py @@ -43,7 +43,7 @@ def connect(self): if version in (0, 1, 2, 3, 4, 5, 6, 7, 8): connection.execute("BEGIN IMMEDIATE") version = connection.execute("PRAGMA user_version").fetchone()[0] - if version not in (0, 1, 2, 3, 4, 5, 6, 7, 8, 9): + if version not in (0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10): raise DatabaseError("Unsupported database schema version") if version == 0: for statement in SCHEMA: @@ -93,6 +93,10 @@ def connect(self): for statement in STATEMENT_MIGRATION_9: connection.execute(statement) connection.execute("PRAGMA user_version=9") + if version in (0, 1, 2, 3, 4, 5, 6, 7, 8, 9): + for statement in FAILED_RETRY_SCHEMA: + connection.execute(statement) + connection.execute("PRAGMA user_version=10") connection.commit() with connection: yield connection @@ -116,7 +120,8 @@ def start(self, receipt_id: str, content_type: str, size: int, image_path: str, ) except sqlite3.IntegrityError: duplicate = db.execute( - "SELECT receipt_id FROM receipts WHERE content_sha256=? AND lifecycle_state<>'DELETED'", + "SELECT receipt_id FROM receipts WHERE content_sha256=? " + "AND lifecycle_state<>'DELETED' AND processing_status<>'FAILED'", (content_sha256,), ).fetchone() if duplicate: @@ -133,7 +138,7 @@ def recover_interrupted(self) -> int: db.execute("BEGIN IMMEDIATE") cursor = db.execute( "UPDATE receipts SET processing_status='FAILED', " - "error='Processing was interrupted. Reprocess from saved OCR or delete and upload again.', " + "error='Processing was interrupted. Reprocess from saved OCR or upload the same file again.', " "updated_at=? WHERE processing_status='PROCESSING'", (now(),), ) @@ -307,3 +312,10 @@ def list(self, decision: str | None, processing_status: str | None, "CREATE UNIQUE INDEX exact_receipt_content ON receipts(content_sha256) " "WHERE content_sha256 IS NOT NULL", ) + +FAILED_RETRY_SCHEMA = ( + "DROP INDEX exact_receipt_content", + "CREATE UNIQUE INDEX exact_receipt_content ON receipts(content_sha256) " + "WHERE content_sha256 IS NOT NULL AND lifecycle_state<>'DELETED' " + "AND processing_status<>'FAILED'", +) diff --git a/app/lifecycle.py b/app/lifecycle.py index ba9c268..f4cb27d 100644 --- a/app/lifecycle.py +++ b/app/lifecycle.py @@ -68,7 +68,7 @@ def apply_lifecycle(store, receipt_id: str, request: LifecycleRequest) -> dict: raise ReviewConflict("Only deleted receipts can be restored") if datetime.fromisoformat(row['purge_after']) <= timestamp: raise ReviewConflict("The 30-day restore window has expired") - if row['content_sha256'] and db.execute("SELECT 1 FROM receipts WHERE content_sha256=? AND lifecycle_state<>'DELETED' AND receipt_id<>?", (row['content_sha256'], receipt_id)).fetchone(): + if row['content_sha256'] and row['processing_status'] != 'FAILED' and db.execute("SELECT 1 FROM receipts WHERE content_sha256=? AND lifecycle_state<>'DELETED' AND processing_status<>'FAILED' AND receipt_id<>?", (row['content_sha256'], receipt_id)).fetchone(): raise ReviewConflict("An identical retained receipt exists. Open that record instead") else: if row['lifecycle_state'] != 'ACTIVE': diff --git a/app/reprocessing.py b/app/reprocessing.py index e948ca8..e7c994f 100644 --- a/app/reprocessing.py +++ b/app/reprocessing.py @@ -94,6 +94,14 @@ def finish(store, rid, request, data=None, error=None): row = db.execute('SELECT * FROM receipts WHERE receipt_id=?', (rid,)).fetchone() result = json.loads(attempt['result_json']) changed = row is None or row['lifecycle_state'] != 'ACTIVE' or row['lifecycle_version'] != request.expected_lifecycle_version or record_version(db, rid) != request.expected_record_version + if not changed and data is not None and row['processing_status'] == 'FAILED' and row['content_sha256']: + changed = db.execute( + "SELECT 1 FROM receipts WHERE receipt_id<>? AND content_sha256=? " + "AND lifecycle_state<>'DELETED' AND processing_status<>'FAILED'", + (rid, row['content_sha256']), + ).fetchone() is not None + if changed: + error = 'An identical active receipt exists. Review the newer upload instead.' result.update(status='SUPERSEDED' if changed else 'FAILED' if error else 'SUCCEEDED', extracted_data=data, error=error, finished_at=datetime.now(timezone.utc).isoformat()) db.execute('UPDATE receipt_reprocessing SET status=?, result_json=? WHERE request_id=?', (result['status'], json.dumps(result), result['request_id'])) diff --git a/docs/docker.md b/docs/docker.md index 2ea5b77..1a53b52 100644 --- a/docs/docker.md +++ b/docs/docker.md @@ -14,7 +14,7 @@ remain attached. Paddle's memory needs must be measured on the target machine; a passing health check does not prove that OCR models fit in RAM. -The current application migrates supported SQLite databases through schema v9 +The current application migrates supported SQLite databases through schema v10 on first use. Create the database-and-evidence backup below before replacing an older image; an older app image cannot read the upgraded schema. diff --git a/docs/pdf.md b/docs/pdf.md index 5eb5e5f..b87bf30 100644 --- a/docs/pdf.md +++ b/docs/pdf.md @@ -38,7 +38,7 @@ invoices before upload. - 202: processing completed; `ocr_engine` begins with `pdf:` and identifies native text, OCR, or a per-page mixture. -- 409: exact original bytes were already uploaded; use `existing_receipt_id`. +- 409: exact original bytes belong to another active or voided, non-failed receipt; use `existing_receipt_id`. A failed attempt can be retried without deletion. - 415: media type or `%PDF-` signature did not match. - 422: encrypted, malformed, excessive-page, unsafe-dimension, or textless PDF. - 504: PDF inspection/rendering or OCR exceeded its timeout. diff --git a/docs/receipt-lifecycle.md b/docs/receipt-lifecycle.md index c115d78..61bdf12 100644 --- a/docs/receipt-lifecycle.md +++ b/docs/receipt-lifecycle.md @@ -20,7 +20,7 @@ The authenticated workspace supports **Move to deleted receipts**, **Restore rec Deletion hides records from ordinary history, review queues, dashboard counts/totals and exports. Deleted receipts appear only in their dedicated UI tab or an explicit `state=DELETED` list query. Voided receipts remain searchable in history but are excluded from dashboards and exports (including explicit ID exports). Previously downloaded spreadsheets do not change. -Identical file uploads may be repeated after soft deletion. Restoration is blocked when the hash matches any non-deleted record, including a voided one. Voiding does not release the file hash. Potential-duplicate searches exclude deleted and voided records. +Identical file uploads may be repeated after soft deletion or a failed processing attempt. The failed record remains in history; a successful retry gets a new receipt ID. Restoration of an accepted receipt is blocked when the hash matches another active or voided, non-failed record. Voiding does not release the file hash. Potential-duplicate searches exclude deleted and voided records. ## Retention and operations diff --git a/docs/reprocessing.md b/docs/reprocessing.md index ad8f6c4..f6680ad 100644 --- a/docs/reprocessing.md +++ b/docs/reprocessing.md @@ -6,7 +6,7 @@ checks arithmetic. None silently approves or posts an expense. After deletion, **Undo deletion** restores the record with one click, retaining a restore audit event under the same self-reported name. It uses the normal restore rules: stale changes, expiry and an identical retained receipt prevent restoration. An uncertain response retries the same request UUID. The banner lasts while the receipt detail remains mounted; after leaving, restore from Deleted receipts within 30 days. Voiding cannot be undone through this action. -**Reprocess receipt** runs extraction against saved OCR text. It does not rerun OCR, classification, upload, or payment processing. The confirmation explains the possible AI cost. A name and reason are required. Active pending, failed, approved, auto-filed and amended records with saved OCR are eligible; processing, deleted, voided and rejected records are not. If original OCR was unavailable, upload a new readable file instead (or delete a failed duplicate first). +**Reprocess receipt** runs extraction against saved OCR text. It does not rerun OCR, classification, upload, or payment processing. The confirmation explains the possible AI cost. A name and reason are required. Active pending, failed, approved, auto-filed and amended records with saved OCR are eligible; processing, deleted, voided and rejected records are not. To retry the full OCR, extraction and classification workflow after a failed upload, select **Try again** on the Upload page; it keeps the selected file and purpose. If you already left that page, select the same file again. The retry creates a new receipt ID and keeps the failed attempt visible. Identical active files remain blocked. Each attempt has a durable UUID, start time, model identifier, self-reported reviewer, reason, status and result. Identical retries return the saved result/status without repeating the paid request. Only one extraction can run per receipt. Calls time out after 180 seconds. An interrupted process retains its running marker until its ten-minute lease expires; a new attempt may then begin. The same interrupted UUID is never automatically rerun. Errors shown to users exclude upstream diagnostics. diff --git a/docs/reviews.md b/docs/reviews.md index 178d754..e396555 100644 --- a/docs/reviews.md +++ b/docs/reviews.md @@ -125,7 +125,7 @@ Uploading a new test image may consume gateway credits; reviewing existing recor does not. Model routing can vary, so an ambiguous image is not guaranteed to queue. The review and amendment features first arrived with schema v3; the current -application migrates supported databases transactionally through schema v9. +application migrates supported databases transactionally through schema v10. Previously accepted invalid requests remain visible in review history; use an audited amendment to correct an approved record. Keep invalid test records out of reports. @@ -137,7 +137,7 @@ Review never modifies vendor rules. ## Migration, verification and deployment -Supported older schemas upgrade transactionally through version 9 on first +Supported older schemas upgrade transactionally through version 10 on first database access (including container startup). The historical v3 step added duplicate metadata, amendments and audit protection triggers; later versions added lifecycle, statement, FX and monthly review data. Back up the database diff --git a/docs/user-guide.md b/docs/user-guide.md index 911214e..844ede1 100644 --- a/docs/user-guide.md +++ b/docs/user-guide.md @@ -20,7 +20,7 @@ The sign-in image is a current capture from the deployed page. The dashboard, hi *Upload a receipt* -Upload a JPEG, PNG or bounded PDF through the web workspace. The selected file appears beside the form, and its preview remains visible after processing. Select **Open processed receipt** to inspect extracted fields and any review reasons, or **Upload another** to clear the form for the next file. This local preview is a convenience; the saved receipt detail shows the retained original evidence. Telegram receipts reach the same authenticated backend through the OpenClaw relay. If an API restart interrupts an upload, refresh Receipt history after the service returns: the interrupted record becomes **Failed**. Open it to reprocess from saved OCR text (when available), or move it to Deleted receipts and upload the file again. An active upload cannot be deleted. +Upload a JPEG, PNG or bounded PDF through the web workspace. The selected file appears beside the form, and its preview remains visible after processing. Select **Open processed receipt** to inspect extracted fields and any review reasons, or **Upload another** to clear the form for the next file. This local preview is a convenience; the saved receipt detail shows the retained original evidence. Telegram receipts reach the same authenticated backend through the OpenClaw relay. If processing fails, the Upload page keeps the selected file and business purpose: check the saved result, then select **Try again**. If an API restart interrupts an upload, refresh Receipt history after the service returns; the interrupted record becomes **Failed**, and you can select the same file on the Upload page again. Failed attempts remain visible for inspection and do not block retrying the same file. You can also reprocess saved OCR as a review draft. An active upload cannot be deleted. ![Pending reviews](assets/screenshots/2026-09-23/03-pending-reviews.jpg) diff --git a/frontend/src/App.tsx b/frontend/src/App.tsx index 8a94d98..56f5076 100644 --- a/frontend/src/App.tsx +++ b/frontend/src/App.tsx @@ -961,7 +961,8 @@ function UploadForm({ [otherPurpose, setOtherPurpose] = useState(""), [busy, setBusy] = useState(false), [error, setError] = useState(""), - [failedId, setFailedId] = useState(null); + [failedId, setFailedId] = useState(null), + [retryableFailedId, setRetryableFailedId] = useState(null); const active = useRef(true), submitting = useRef(false); useEffect(() => { @@ -1000,19 +1001,38 @@ function UploadForm({ body.set("receipt", file); body.set("business_purpose", purpose); try { + if (retryableFailedId) { + const previous = await request<{ processing_status: string }>( + `/receipts/${retryableFailedId}`, token, + ); + if (previous.processing_status !== "FAILED") { + setError("Your previous upload is still processing or has completed. Open its saved record before trying again."); + return; + } + } const result = await request<{ receipt_id: string }>( "/receipts/upload", token, { method: "POST", body, timeoutMs: 330000 }, ); - if (active.current) setUploadedId(result.receipt_id); + if (active.current) { + setUploadedId(result.receipt_id); + setRetryableFailedId(null); + } } catch (err) { if (!active.current) return; setError( message(err) + - " Check history before uploading again; processing may already have started.", + (err instanceof ApiError && err.status === 409 + ? " Open the existing record to see why this upload was rejected." + : err instanceof ApiError && err.receiptId + ? " Try again will check the saved status before processing this same file." + : " Check history before trying again; processing may already have started."), ); - if (err instanceof ApiError) setFailedId(err.receiptId); + if (err instanceof ApiError) { + setFailedId(err.receiptId); + setRetryableFailedId(err.status === 409 ? null : err.receiptId); + } } finally { submitting.current = false; if (active.current) setBusy(false); @@ -1036,6 +1056,7 @@ function UploadForm({ onChange={(e) => { const chosen = e.target.files?.[0] || null; setFile(chosen); + setRetryableFailedId(null); setPreviewUrl( chosen && ["image/jpeg", "image/png", "application/pdf"].includes(chosen.type) ? URL.createObjectURL(chosen) @@ -1127,6 +1148,8 @@ function UploadForm({ setOtherPurpose(""); setUploadedId(null); setError(""); + setFailedId(null); + setRetryableFailedId(null); }} > Upload another @@ -1145,7 +1168,7 @@ function UploadForm({ } > {busy ? : } - {busy ? "Processing receipt…" : "Upload and process"} + {busy ? "Processing receipt…" : retryableFailedId ? "Try again" : "Upload and process"} )}

diff --git a/tests/test_database.py b/tests/test_database.py index 9918503..faa96df 100644 --- a/tests/test_database.py +++ b/tests/test_database.py @@ -3,7 +3,7 @@ import pytest -from app.database import DatabaseError, ReceiptStore +from app.database import DatabaseError, DuplicateReceiptError, ReceiptStore def test_concurrent_initialization_and_writes(tmp_path): @@ -72,3 +72,23 @@ def test_failure_retains_ocr(tmp_path): assert result["processing_status"] == "FAILED" assert result["classification"] is None assert store.list(None, "FAILED", 20, 0)["total"] == 1 + + +def test_v9_migration_releases_failed_hash_but_keeps_active_hash_unique(tmp_path): + path = tmp_path / "expenses.db" + store = ReceiptStore(path) + store.start("failed", "image/jpeg", 10, "private/path", None, "same-file") + store.fail("failed", "Original OCR failed") + with sqlite3.connect(path) as db: + db.execute("DROP INDEX exact_receipt_content") + db.execute("CREATE UNIQUE INDEX exact_receipt_content ON receipts(content_sha256) " + "WHERE content_sha256 IS NOT NULL AND lifecycle_state<>'DELETED'") + db.execute("PRAGMA user_version=9") + + store.start("retry", "image/jpeg", 10, "private/path", None, "same-file") + with store.connect() as db: + assert db.execute("PRAGMA user_version").fetchone()[0] == 10 + assert db.execute("SELECT processing_status FROM receipts WHERE receipt_id='failed'").fetchone()[0] == "FAILED" + with pytest.raises(DuplicateReceiptError) as duplicate: + store.start("concurrent", "image/jpeg", 10, "private/path", None, "same-file") + assert duplicate.value.receipt_id == "retry" diff --git a/tests/test_duplicates_amendments.py b/tests/test_duplicates_amendments.py index 3ff70c5..df878e8 100644 --- a/tests/test_duplicates_amendments.py +++ b/tests/test_duplicates_amendments.py @@ -6,6 +6,8 @@ import pytest from app.database import DatabaseError, ReceiptStore +from app.extraction import ExtractionTimeoutError +from app.ocr import OCRTimeoutError from test_api import TEST_KEY, configured_client @@ -52,6 +54,42 @@ def test_exact_duplicate_stops_before_ocr_and_returns_existing_id(monkeypatch, t assert len(list(tmp_path.glob("*.jpg"))) == 1 +def test_failed_receipt_allows_same_file_upload_without_deleting(monkeypatch, tmp_path): + client, ocr, extractor, _ = configured_client(monkeypatch, tmp_path) + extractor.error = ExtractionTimeoutError("private gateway detail") + failed = upload(client) + assert failed.status_code == 504 + failed_id = failed.headers["X-Receipt-ID"] + extractor.error = None + + retried = upload(client) + assert retried.status_code == 202, retried.text + new_id = retried.json()["receipt_id"] + assert new_id != failed_id + assert len(ocr.paths) == 2 + assert client.get(f"/receipts/{failed_id}", headers=HEADERS).json()["processing_status"] == "FAILED" + assert client.get(f"/receipts/{new_id}", headers=HEADERS).json()["processing_status"] == "COMPLETED" + assert upload(client).status_code == 409 + + +def test_failed_ocr_without_retained_file_allows_same_upload(monkeypatch, tmp_path): + client, ocr, _, _ = configured_client(monkeypatch, tmp_path) + original_extract = ocr.extract + + def fail_ocr(path): + raise OCRTimeoutError("OCR service timed out") + + ocr.extract = fail_ocr + failed = upload(client) + assert failed.status_code == 504 + failed_id = failed.headers["X-Receipt-ID"] + ocr.extract = original_extract + retried = upload(client) + assert retried.status_code == 202, retried.text + assert retried.json()["receipt_id"] != failed_id + assert client.get(f"/receipts/{failed_id}", headers=HEADERS).json()["processing_status"] == "FAILED" + + def test_different_file_is_not_blocked_by_vendor_and_amount(monkeypatch, tmp_path): client, *_ = configured_client(monkeypatch, tmp_path) assert upload(client, b"\xff\xd8\xffone").status_code == 202 diff --git a/tests/test_reprocessing.py b/tests/test_reprocessing.py index bf1b13d..3ae88dd 100644 --- a/tests/test_reprocessing.py +++ b/tests/test_reprocessing.py @@ -93,6 +93,32 @@ def test_running_and_stale_attempts_cannot_overwrite_decisions(monkeypatch, tmp_ assert client.post(f'/receipts/{rid}/amendments', headers=HEADERS, json=amendment).status_code == 409 +def test_old_failed_draft_is_superseded_after_new_upload(monkeypatch, tmp_path): + from app.database import ReceiptStore + from app.main import get_receipt_extractor + from test_api import configured_client + from test_duplicates_amendments import upload + import os + from pathlib import Path + client, _, extractor, _ = configured_client(monkeypatch, tmp_path) + from app.extraction import ExtractionTimeoutError + extractor.error = ExtractionTimeoutError('gateway timeout') + first = upload(client) + rid = first.headers['X-Receipt-ID'] + extractor.error = None + store = ReceiptStore(Path(os.environ['DATABASE_PATH'])) + attempt = ReprocessRequest(**request_body()) + started, text = begin(store, rid, attempt, 'test-model') + assert text + newer = upload(client) + assert newer.status_code == 202 + result = finish(store, rid, attempt, newer.json()['extracted_data']) + assert result['status'] == 'SUPERSEDED' + assert 'identical active receipt' in result['error'] + assert store.get(rid)['processing_status'] == 'FAILED' + assert store.get(newer.json()['receipt_id'])['processing_status'] == 'COMPLETED' + + def test_auth_eligibility_collision_and_interrupted_retry(monkeypatch, tmp_path): client, uploaded, _, store = setup_review(monkeypatch, tmp_path) rid = uploaded['receipt_id']; url = f'/receipts/{rid}/reprocess' diff --git a/tests/test_reviews.py b/tests/test_reviews.py index aaea3f3..604affb 100644 --- a/tests/test_reviews.py +++ b/tests/test_reviews.py @@ -152,7 +152,7 @@ def test_existing_v1_migrates_without_changing_evidence(tmp_path): store = ReceiptStore(path) assert store.get('old')['processing_status'] == 'FAILED' with store.connect() as db: - assert db.execute('PRAGMA user_version').fetchone()[0] == 9 + assert db.execute('PRAGMA user_version').fetchone()[0] == 10 columns = {row[1] for row in db.execute("PRAGMA table_info(bank_statements)")} assert {"source_media_type", "extraction_method", "metadata_json", "validation_json", "imported_by"} <= columns assert db.execute('SELECT image_path FROM receipts').fetchone()[0] == 'secret'