Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
18 changes: 15 additions & 3 deletions app/database.py
Original file line number Diff line number Diff line change
Expand Up @@ -43,7 +43,7 @@ def connect(self):
if version in (0, 1, 2, 3, 4, 5, 6, 7, 8):
connection.execute("BEGIN IMMEDIATE")
version = connection.execute("PRAGMA user_version").fetchone()[0]
if version not in (0, 1, 2, 3, 4, 5, 6, 7, 8, 9):
if version not in (0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10):
raise DatabaseError("Unsupported database schema version")
if version == 0:
for statement in SCHEMA:
Expand Down Expand Up @@ -93,6 +93,10 @@ def connect(self):
for statement in STATEMENT_MIGRATION_9:
connection.execute(statement)
connection.execute("PRAGMA user_version=9")
if version in (0, 1, 2, 3, 4, 5, 6, 7, 8, 9):
for statement in FAILED_RETRY_SCHEMA:
connection.execute(statement)
connection.execute("PRAGMA user_version=10")
connection.commit()
with connection:
yield connection
Expand All @@ -116,7 +120,8 @@ def start(self, receipt_id: str, content_type: str, size: int, image_path: str,
)
except sqlite3.IntegrityError:
duplicate = db.execute(
"SELECT receipt_id FROM receipts WHERE content_sha256=? AND lifecycle_state<>'DELETED'",
"SELECT receipt_id FROM receipts WHERE content_sha256=? "
"AND lifecycle_state<>'DELETED' AND processing_status<>'FAILED'",
(content_sha256,),
).fetchone()
if duplicate:
Expand All @@ -133,7 +138,7 @@ def recover_interrupted(self) -> int:
db.execute("BEGIN IMMEDIATE")
cursor = db.execute(
"UPDATE receipts SET processing_status='FAILED', "
"error='Processing was interrupted. Reprocess from saved OCR or delete and upload again.', "
"error='Processing was interrupted. Reprocess from saved OCR or upload the same file again.', "
"updated_at=? WHERE processing_status='PROCESSING'",
(now(),),
)
Expand Down Expand Up @@ -307,3 +312,10 @@ def list(self, decision: str | None, processing_status: str | None,
"CREATE UNIQUE INDEX exact_receipt_content ON receipts(content_sha256) "
"WHERE content_sha256 IS NOT NULL",
)

FAILED_RETRY_SCHEMA = (
"DROP INDEX exact_receipt_content",
"CREATE UNIQUE INDEX exact_receipt_content ON receipts(content_sha256) "
"WHERE content_sha256 IS NOT NULL AND lifecycle_state<>'DELETED' "
"AND processing_status<>'FAILED'",
)
2 changes: 1 addition & 1 deletion app/lifecycle.py
Original file line number Diff line number Diff line change
Expand Up @@ -68,7 +68,7 @@ def apply_lifecycle(store, receipt_id: str, request: LifecycleRequest) -> dict:
raise ReviewConflict("Only deleted receipts can be restored")
if datetime.fromisoformat(row['purge_after']) <= timestamp:
raise ReviewConflict("The 30-day restore window has expired")
if row['content_sha256'] and db.execute("SELECT 1 FROM receipts WHERE content_sha256=? AND lifecycle_state<>'DELETED' AND receipt_id<>?", (row['content_sha256'], receipt_id)).fetchone():
if row['content_sha256'] and row['processing_status'] != 'FAILED' and db.execute("SELECT 1 FROM receipts WHERE content_sha256=? AND lifecycle_state<>'DELETED' AND processing_status<>'FAILED' AND receipt_id<>?", (row['content_sha256'], receipt_id)).fetchone():
raise ReviewConflict("An identical retained receipt exists. Open that record instead")
else:
if row['lifecycle_state'] != 'ACTIVE':
Expand Down
8 changes: 8 additions & 0 deletions app/reprocessing.py
Original file line number Diff line number Diff line change
Expand Up @@ -94,6 +94,14 @@ def finish(store, rid, request, data=None, error=None):
row = db.execute('SELECT * FROM receipts WHERE receipt_id=?', (rid,)).fetchone()
result = json.loads(attempt['result_json'])
changed = row is None or row['lifecycle_state'] != 'ACTIVE' or row['lifecycle_version'] != request.expected_lifecycle_version or record_version(db, rid) != request.expected_record_version
if not changed and data is not None and row['processing_status'] == 'FAILED' and row['content_sha256']:
changed = db.execute(
"SELECT 1 FROM receipts WHERE receipt_id<>? AND content_sha256=? "
"AND lifecycle_state<>'DELETED' AND processing_status<>'FAILED'",
(rid, row['content_sha256']),
).fetchone() is not None
if changed:
error = 'An identical active receipt exists. Review the newer upload instead.'
result.update(status='SUPERSEDED' if changed else 'FAILED' if error else 'SUCCEEDED',
extracted_data=data, error=error, finished_at=datetime.now(timezone.utc).isoformat())
db.execute('UPDATE receipt_reprocessing SET status=?, result_json=? WHERE request_id=?', (result['status'], json.dumps(result), result['request_id']))
Expand Down
2 changes: 1 addition & 1 deletion docs/docker.md
Original file line number Diff line number Diff line change
Expand Up @@ -14,7 +14,7 @@ remain attached.
Paddle's memory needs must be measured on the target machine; a passing health
check does not prove that OCR models fit in RAM.

The current application migrates supported SQLite databases through schema v9
The current application migrates supported SQLite databases through schema v10
on first use. Create the database-and-evidence backup below before replacing
an older image; an older app image cannot read the upgraded schema.

Expand Down
2 changes: 1 addition & 1 deletion docs/pdf.md
Original file line number Diff line number Diff line change
Expand Up @@ -38,7 +38,7 @@ invoices before upload.

- 202: processing completed; `ocr_engine` begins with `pdf:` and identifies native
text, OCR, or a per-page mixture.
- 409: exact original bytes were already uploaded; use `existing_receipt_id`.
- 409: exact original bytes belong to another active or voided, non-failed receipt; use `existing_receipt_id`. A failed attempt can be retried without deletion.
- 415: media type or `%PDF-` signature did not match.
- 422: encrypted, malformed, excessive-page, unsafe-dimension, or textless PDF.
- 504: PDF inspection/rendering or OCR exceeded its timeout.
Expand Down
2 changes: 1 addition & 1 deletion docs/receipt-lifecycle.md
Original file line number Diff line number Diff line change
Expand Up @@ -20,7 +20,7 @@ The authenticated workspace supports **Move to deleted receipts**, **Restore rec

Deletion hides records from ordinary history, review queues, dashboard counts/totals and exports. Deleted receipts appear only in their dedicated UI tab or an explicit `state=DELETED` list query. Voided receipts remain searchable in history but are excluded from dashboards and exports (including explicit ID exports). Previously downloaded spreadsheets do not change.

Identical file uploads may be repeated after soft deletion. Restoration is blocked when the hash matches any non-deleted record, including a voided one. Voiding does not release the file hash. Potential-duplicate searches exclude deleted and voided records.
Identical file uploads may be repeated after soft deletion or a failed processing attempt. The failed record remains in history; a successful retry gets a new receipt ID. Restoration of an accepted receipt is blocked when the hash matches another active or voided, non-failed record. Voiding does not release the file hash. Potential-duplicate searches exclude deleted and voided records.

## Retention and operations

Expand Down
2 changes: 1 addition & 1 deletion docs/reprocessing.md
Original file line number Diff line number Diff line change
Expand Up @@ -6,7 +6,7 @@ checks arithmetic. None silently approves or posts an expense.

After deletion, **Undo deletion** restores the record with one click, retaining a restore audit event under the same self-reported name. It uses the normal restore rules: stale changes, expiry and an identical retained receipt prevent restoration. An uncertain response retries the same request UUID. The banner lasts while the receipt detail remains mounted; after leaving, restore from Deleted receipts within 30 days. Voiding cannot be undone through this action.

**Reprocess receipt** runs extraction against saved OCR text. It does not rerun OCR, classification, upload, or payment processing. The confirmation explains the possible AI cost. A name and reason are required. Active pending, failed, approved, auto-filed and amended records with saved OCR are eligible; processing, deleted, voided and rejected records are not. If original OCR was unavailable, upload a new readable file instead (or delete a failed duplicate first).
**Reprocess receipt** runs extraction against saved OCR text. It does not rerun OCR, classification, upload, or payment processing. The confirmation explains the possible AI cost. A name and reason are required. Active pending, failed, approved, auto-filed and amended records with saved OCR are eligible; processing, deleted, voided and rejected records are not. To retry the full OCR, extraction and classification workflow after a failed upload, select **Try again** on the Upload page; it keeps the selected file and purpose. If you already left that page, select the same file again. The retry creates a new receipt ID and keeps the failed attempt visible. Identical active files remain blocked.

Each attempt has a durable UUID, start time, model identifier, self-reported reviewer, reason, status and result. Identical retries return the saved result/status without repeating the paid request. Only one extraction can run per receipt. Calls time out after 180 seconds. An interrupted process retains its running marker until its ten-minute lease expires; a new attempt may then begin. The same interrupted UUID is never automatically rerun. Errors shown to users exclude upstream diagnostics.

Expand Down
4 changes: 2 additions & 2 deletions docs/reviews.md
Original file line number Diff line number Diff line change
Expand Up @@ -125,7 +125,7 @@ Uploading a new test image may consume gateway credits; reviewing existing recor
does not. Model routing can vary, so an ambiguous image is not guaranteed to queue.

The review and amendment features first arrived with schema v3; the current
application migrates supported databases transactionally through schema v9.
application migrates supported databases transactionally through schema v10.
Previously accepted invalid requests remain visible in review history;
use an audited amendment to correct an approved record. Keep invalid test records out of reports.

Expand All @@ -137,7 +137,7 @@ Review never modifies vendor rules.

## Migration, verification and deployment

Supported older schemas upgrade transactionally through version 9 on first
Supported older schemas upgrade transactionally through version 10 on first
database access (including container startup). The historical v3 step added
duplicate metadata, amendments and audit protection triggers; later versions
added lifecycle, statement, FX and monthly review data. Back up the database
Expand Down
2 changes: 1 addition & 1 deletion docs/user-guide.md
Original file line number Diff line number Diff line change
Expand Up @@ -20,7 +20,7 @@ The sign-in image is a current capture from the deployed page. The dashboard, hi

*Upload a receipt*

Upload a JPEG, PNG or bounded PDF through the web workspace. The selected file appears beside the form, and its preview remains visible after processing. Select **Open processed receipt** to inspect extracted fields and any review reasons, or **Upload another** to clear the form for the next file. This local preview is a convenience; the saved receipt detail shows the retained original evidence. Telegram receipts reach the same authenticated backend through the OpenClaw relay. If an API restart interrupts an upload, refresh Receipt history after the service returns: the interrupted record becomes **Failed**. Open it to reprocess from saved OCR text (when available), or move it to Deleted receipts and upload the file again. An active upload cannot be deleted.
Upload a JPEG, PNG or bounded PDF through the web workspace. The selected file appears beside the form, and its preview remains visible after processing. Select **Open processed receipt** to inspect extracted fields and any review reasons, or **Upload another** to clear the form for the next file. This local preview is a convenience; the saved receipt detail shows the retained original evidence. Telegram receipts reach the same authenticated backend through the OpenClaw relay. If processing fails, the Upload page keeps the selected file and business purpose: check the saved result, then select **Try again**. If an API restart interrupts an upload, refresh Receipt history after the service returns; the interrupted record becomes **Failed**, and you can select the same file on the Upload page again. Failed attempts remain visible for inspection and do not block retrying the same file. You can also reprocess saved OCR as a review draft. An active upload cannot be deleted.

![Pending reviews](assets/screenshots/2026-09-23/03-pending-reviews.jpg)

Expand Down
33 changes: 28 additions & 5 deletions frontend/src/App.tsx
Original file line number Diff line number Diff line change
Expand Up @@ -961,7 +961,8 @@ function UploadForm({
[otherPurpose, setOtherPurpose] = useState(""),
[busy, setBusy] = useState(false),
[error, setError] = useState(""),
[failedId, setFailedId] = useState<string | null>(null);
[failedId, setFailedId] = useState<string | null>(null),
[retryableFailedId, setRetryableFailedId] = useState<string | null>(null);
const active = useRef(true),
submitting = useRef(false);
useEffect(() => {
Expand Down Expand Up @@ -1000,19 +1001,38 @@ function UploadForm({
body.set("receipt", file);
body.set("business_purpose", purpose);
try {
if (retryableFailedId) {
const previous = await request<{ processing_status: string }>(
`/receipts/${retryableFailedId}`, token,
);
if (previous.processing_status !== "FAILED") {
setError("Your previous upload is still processing or has completed. Open its saved record before trying again.");
return;
}
}
const result = await request<{ receipt_id: string }>(
"/receipts/upload",
token,
{ method: "POST", body, timeoutMs: 330000 },
);
if (active.current) setUploadedId(result.receipt_id);
if (active.current) {
setUploadedId(result.receipt_id);
setRetryableFailedId(null);
}
} catch (err) {
if (!active.current) return;
setError(
message(err) +
" Check history before uploading again; processing may already have started.",
(err instanceof ApiError && err.status === 409
? " Open the existing record to see why this upload was rejected."
: err instanceof ApiError && err.receiptId
? " Try again will check the saved status before processing this same file."
: " Check history before trying again; processing may already have started."),
);
if (err instanceof ApiError) setFailedId(err.receiptId);
if (err instanceof ApiError) {
setFailedId(err.receiptId);
setRetryableFailedId(err.status === 409 ? null : err.receiptId);
}
} finally {
submitting.current = false;
if (active.current) setBusy(false);
Expand All @@ -1036,6 +1056,7 @@ function UploadForm({
onChange={(e) => {
const chosen = e.target.files?.[0] || null;
setFile(chosen);
setRetryableFailedId(null);
setPreviewUrl(
chosen && ["image/jpeg", "image/png", "application/pdf"].includes(chosen.type)
? URL.createObjectURL(chosen)
Expand Down Expand Up @@ -1127,6 +1148,8 @@ function UploadForm({
setOtherPurpose("");
setUploadedId(null);
setError("");
setFailedId(null);
setRetryableFailedId(null);
}}
>
Upload another
Expand All @@ -1145,7 +1168,7 @@ function UploadForm({
}
>
{busy ? <LoaderCircle className="animate-spin" /> : <Upload />}
{busy ? "Processing receipt…" : "Upload and process"}
{busy ? "Processing receipt…" : retryableFailedId ? "Try again" : "Upload and process"}
</Button>
)}
<p className="muted">
Expand Down
22 changes: 21 additions & 1 deletion tests/test_database.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,7 +3,7 @@

import pytest

from app.database import DatabaseError, ReceiptStore
from app.database import DatabaseError, DuplicateReceiptError, ReceiptStore


def test_concurrent_initialization_and_writes(tmp_path):
Expand Down Expand Up @@ -72,3 +72,23 @@ def test_failure_retains_ocr(tmp_path):
assert result["processing_status"] == "FAILED"
assert result["classification"] is None
assert store.list(None, "FAILED", 20, 0)["total"] == 1


def test_v9_migration_releases_failed_hash_but_keeps_active_hash_unique(tmp_path):
path = tmp_path / "expenses.db"
store = ReceiptStore(path)
store.start("failed", "image/jpeg", 10, "private/path", None, "same-file")
store.fail("failed", "Original OCR failed")
with sqlite3.connect(path) as db:
db.execute("DROP INDEX exact_receipt_content")
db.execute("CREATE UNIQUE INDEX exact_receipt_content ON receipts(content_sha256) "
"WHERE content_sha256 IS NOT NULL AND lifecycle_state<>'DELETED'")
db.execute("PRAGMA user_version=9")

store.start("retry", "image/jpeg", 10, "private/path", None, "same-file")
with store.connect() as db:
assert db.execute("PRAGMA user_version").fetchone()[0] == 10
assert db.execute("SELECT processing_status FROM receipts WHERE receipt_id='failed'").fetchone()[0] == "FAILED"
with pytest.raises(DuplicateReceiptError) as duplicate:
store.start("concurrent", "image/jpeg", 10, "private/path", None, "same-file")
assert duplicate.value.receipt_id == "retry"
38 changes: 38 additions & 0 deletions tests/test_duplicates_amendments.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,8 @@
import pytest

from app.database import DatabaseError, ReceiptStore
from app.extraction import ExtractionTimeoutError
from app.ocr import OCRTimeoutError
from test_api import TEST_KEY, configured_client


Expand Down Expand Up @@ -52,6 +54,42 @@ def test_exact_duplicate_stops_before_ocr_and_returns_existing_id(monkeypatch, t
assert len(list(tmp_path.glob("*.jpg"))) == 1


def test_failed_receipt_allows_same_file_upload_without_deleting(monkeypatch, tmp_path):
client, ocr, extractor, _ = configured_client(monkeypatch, tmp_path)
extractor.error = ExtractionTimeoutError("private gateway detail")
failed = upload(client)
assert failed.status_code == 504
failed_id = failed.headers["X-Receipt-ID"]
extractor.error = None

retried = upload(client)
assert retried.status_code == 202, retried.text
new_id = retried.json()["receipt_id"]
assert new_id != failed_id
assert len(ocr.paths) == 2
assert client.get(f"/receipts/{failed_id}", headers=HEADERS).json()["processing_status"] == "FAILED"
assert client.get(f"/receipts/{new_id}", headers=HEADERS).json()["processing_status"] == "COMPLETED"
assert upload(client).status_code == 409


def test_failed_ocr_without_retained_file_allows_same_upload(monkeypatch, tmp_path):
client, ocr, _, _ = configured_client(monkeypatch, tmp_path)
original_extract = ocr.extract

def fail_ocr(path):
raise OCRTimeoutError("OCR service timed out")

ocr.extract = fail_ocr
failed = upload(client)
assert failed.status_code == 504
failed_id = failed.headers["X-Receipt-ID"]
ocr.extract = original_extract
retried = upload(client)
assert retried.status_code == 202, retried.text
assert retried.json()["receipt_id"] != failed_id
assert client.get(f"/receipts/{failed_id}", headers=HEADERS).json()["processing_status"] == "FAILED"


def test_different_file_is_not_blocked_by_vendor_and_amount(monkeypatch, tmp_path):
client, *_ = configured_client(monkeypatch, tmp_path)
assert upload(client, b"\xff\xd8\xffone").status_code == 202
Expand Down
Loading
Loading