diff --git a/README.md b/README.md index 2086142..56d2df4 100644 --- a/README.md +++ b/README.md @@ -32,6 +32,14 @@ It supports: ## πŸ— Architecture overview +### Global system architecture + +![Global system architecture](docs/images/summarization-tool-global-architecture.png) + +*Architecture overview. The platform connects document ingestion, the React user interface, FastAPI orchestration, parser outputs, LLM workflows, evaluation, collaboration, persistence, authentication, and observability.* + +Detailed backend diagrams: [Backend visual workflow map](docs/backend/README.md). + The application is organized around a small set of core services: | Service | Port | Purpose | @@ -244,9 +252,14 @@ SummarizationTool-dev/ ## πŸ“š Additional documentation +- [Documentation index](docs/INDEX.md) β€” **start here** β€” full navigation map for all backend, frontend, and deployment docs +- [Frontend technical design docs](docs/frontend/README.md) β€” frontend architecture, page-by-page docs, component index, hooks, and TypeScript interfaces +- [Glossary](docs/glossary.md) β€” definitions for all Azure services, tools, and project-specific terms - [Backend README](backend/README.md) β€” backend setup and processing details -- [Migration guide](docs/migration-guide.md) β€” architecture migration and platform transition notes -- [GitHub auth setup](docs/setup-github-auth.md) β€” Better Auth GitHub OAuth configuration +- [Backend technical design docs](docs/backend/README.md) β€” backend architecture, workflows, diagrams, data models, schemas, and appendices +- [Backend class reference](docs/backend/appendices/class-reference.md) β€” field-level reference for backend ORM models, schemas, dataclasses, service attributes, and provider classes +- [Migration guide](docs/superpowers/migration-guide.md) β€” architecture migration and platform transition notes +- [GitHub auth setup](docs/superpowers/setup-github-auth.md) β€” Better Auth GitHub OAuth configuration - [Dockerize & deploy to Azure](docs/superpowers/plans/dockerize-and-deploy.md) β€” deployment architecture and implementation notes --- diff --git a/auth-service/src/index.ts b/auth-service/src/index.ts index 393655f..35b9abf 100644 --- a/auth-service/src/index.ts +++ b/auth-service/src/index.ts @@ -16,14 +16,29 @@ import "dotenv/config"; import express from "express"; import cors from "cors"; import { betterAuth } from "better-auth"; +import { getMigrations } from "better-auth/db/migration"; import { toNodeHandler } from "better-auth/node"; import { Pool } from "pg"; // ---------- Shared DB Pool ---------- +function databaseRequiresSSL(url: string | undefined): boolean { + if (!url) return false; + try { + const parsed = new URL(url); + const host = parsed.hostname.toLowerCase(); + if (host === "azure.com" || host.endsWith(".azure.com")) return true; + if (parsed.searchParams.get("sslmode") === "require") return true; + } catch { + // Non-standard connection string β€” fall back to explicit sslmode check only + if (/[?&]sslmode=require(&|$)/.test(url)) return true; + } + return false; +} + const dbPool = new Pool({ connectionString: process.env.DATABASE_URL, - ssl: { rejectUnauthorized: false }, + ssl: databaseRequiresSSL(process.env.DATABASE_URL) ? { rejectUnauthorized: false } : false, }); function getAllowedEmails(): Set { @@ -155,10 +170,23 @@ app.get("/api/auth/validate", async (req, res) => { }); const PORT = process.env.PORT || 3001; -app.listen(PORT, () => { - console.log(`βœ… Better Auth sidecar running on http://localhost:${PORT}`); - console.log(` Auth endpoints: http://localhost:${PORT}/api/auth/*`); - console.log( - ` GitHub OAuth: ${process.env.GITHUB_CLIENT_ID ? "ENABLED" : "DISABLED (no credentials)"}` - ); + +async function start() { + console.log("[Migration] Running Better Auth database migrations..."); + const { runMigrations } = await getMigrations((auth as any).options); + await runMigrations(); + console.log("[Migration] Done."); + + app.listen(PORT, () => { + console.log(`βœ… Better Auth sidecar running on http://localhost:${PORT}`); + console.log(` Auth endpoints: http://localhost:${PORT}/api/auth/*`); + console.log( + ` GitHub OAuth: ${process.env.GITHUB_CLIENT_ID ? "ENABLED" : "DISABLED (no credentials)"}` + ); + }); +} + +start().catch((err) => { + console.error("[Startup] Fatal error:", err); + process.exit(1); }); diff --git a/backend/api/documents/router.py b/backend/api/documents/router.py index 2c1c5e2..08c383a 100644 --- a/backend/api/documents/router.py +++ b/backend/api/documents/router.py @@ -1351,6 +1351,81 @@ async def extract_figure_content( return await generate_figure_summary(document_id, figure_id, request, http_request) +@router.get("/{document_id}/tables/download", dependencies=[Depends(get_current_user)]) +async def download_all_tables_zip(document_id: str): + """ + Download all extracted tables as a single ZIP archive. + + Resolves the processor once, fetches all table HTML files concurrently via + asyncio.gather(), packages them into an in-memory ZIP, and returns it as + application/zip. + + IMPORTANT: must be registered before /{document_id}/tables/{table_filename} + so FastAPI does not route the literal segment "download" to get_table_html. + """ + import asyncio + import io + import zipfile + from fastapi.responses import Response as _Response + + try: + resolved_processor = await file_service.resolve_processed_processor(document_id) + if not resolved_processor: + raise HTTPException(status_code=404, detail="Processed document not found") + + metadata = await file_service.get_processed_metadata( + document_id, resolved_processor + ) + tables_count = (metadata or {}).get("tables_found", 0) + + if not tables_count: + blobs = await file_service._blob.list_blobs_with_prefix( + f"global/{document_id}/processed/{resolved_processor}/tables/", + limit=500, + ) + tables_count = len([b for b in blobs if b.lower().endswith(".html")]) + + if not tables_count: + raise HTTPException( + status_code=404, detail="No tables found for this document" + ) + + print(f"[TABLE-ZIP] Building ZIP for {document_id}: {tables_count} tables") + + async def fetch_one(i: int): + return i, await file_service.get_processing_file_bytes( + document_id, resolved_processor, f"tables/table-{i}.html" + ) + + results = await asyncio.gather( + *[fetch_one(i) for i in range(1, tables_count + 1)] + ) + + buf = io.BytesIO() + with zipfile.ZipFile(buf, mode="w", compression=zipfile.ZIP_DEFLATED) as zf: + for i, data in results: + if data: + zf.writestr(f"table-{i}.html", data) + buf.seek(0) + + print(f"[TABLE-ZIP] βœ… ZIP ready for {document_id}") + return _Response( + content=buf.read(), + media_type="application/zip", + headers={ + "Content-Disposition": f'attachment; filename="tables-{document_id[:8]}.zip"' + }, + ) + + except HTTPException: + raise + except Exception as e: + print(f"[TABLE-ZIP] Error: {str(e)}") + raise HTTPException( + status_code=500, detail=f"Error creating tables ZIP: {str(e)}" + ) + + @router.get( "/{document_id}/tables/{table_filename}", dependencies=[Depends(get_current_user)] ) diff --git a/backend/services/document/organized_file_service.py b/backend/services/document/organized_file_service.py index 2d8c258..c3fae4c 100644 --- a/backend/services/document/organized_file_service.py +++ b/backend/services/document/organized_file_service.py @@ -13,6 +13,7 @@ path, and as a read cache to avoid redundant blob downloads within one container lifetime. """ +import asyncio import json import os import hashlib @@ -187,14 +188,14 @@ async def get_processing_file_bytes( / proc_str / Path(norm_path) ) - if tmp_file.exists(): - return tmp_file.read_bytes() + if await asyncio.to_thread(tmp_file.exists): + return await asyncio.to_thread(tmp_file.read_bytes) blob_path = f"global/{file_hash}/processed/{proc_str}/{norm_path}" data = await self._blob.download_bytes(blob_path) if data: - tmp_file.parent.mkdir(parents=True, exist_ok=True) - tmp_file.write_bytes(data) + await asyncio.to_thread(tmp_file.parent.mkdir, parents=True, exist_ok=True) + await asyncio.to_thread(tmp_file.write_bytes, data) return data async def processing_file_exists( diff --git a/docs/INDEX.md b/docs/INDEX.md new file mode 100644 index 0000000..a362c83 --- /dev/null +++ b/docs/INDEX.md @@ -0,0 +1,84 @@ +# Science-GPT Documentation Index + +Start here. Every document in this repository is listed below with its audience and a short description of what it covers. + +**New to the project?** Read the [Product Overview](README.md) first, then the [Glossary](glossary.md), then the entry point for the area you're working in (backend or frontend). + +--- + +## Start here + +| Document | Audience | What it covers | +|---|---|---| +| [Product Overview](README.md) | Everyone | What the tool does, who uses it (PMRA reviewers), how it was tested, SME evaluation results | +| [Glossary](glossary.md) | Everyone | Plain-language definitions for all Azure services, tools, and project-specific terms | + +--- + +## Backend + +| Document | Audience | What it covers | +|---|---|---| +| [Backend TDD β€” Entry Point](backend/README.md) | Engineers | Master index for the full backend technical design; start here for anything backend | +| [01 β€” Architecture](backend/01-architecture.md) | Engineers | Service boundaries, package responsibilities, five major data flows, dependency direction | +| [02 β€” API Surface](backend/02-api-surface.md) | Engineers | All 14 routers documented with every endpoint, request/response shapes, auth, and exceptions | +| [03 β€” Data Models](backend/03-data-models.md) | Engineers | All 13 ORM models with field-level types, constraints, indexes, and migration notes | +| [04 β€” Schemas](backend/04-schemas.md) | Engineers | All Pydantic request/response schemas with design notes | +| [05 β€” Document Processing](backend/05-document-processing.md) | Engineers | Upload, SHA-256 deduplication, parser selection, Azure DI and Docling pipelines, blob storage, bounding boxes | +| [06 β€” LLM Layer](backend/06-llm-layer.md) | Engineers | All 7 provider clients, dispatch table, timeout budgets, structured output handling, cost tracking | +| [07 β€” Extraction Flow](backend/07-extraction-flow.md) | Engineers | Entity extraction end-to-end: concurrency model, provider dispatch, reference/bbox matching, session persistence | +| [08 β€” Evaluation Flow](backend/08-evaluation-flow.md) | Engineers | G-Eval scoring, combined JSON parsing, background job lifecycle, cancellation, cost tracking | +| [09 β€” Sessions, Sharing & Groups](backend/09-session-sharing-groups.md) | Engineers | Session lifecycle, restore-view construction, group membership rules, shared session read path | +| [10 β€” Template System](backend/10-template-system.md) | Engineers | Template CRUD, version snapshots, fork, scope change, access-control algorithms, folder operations | +| [11 β€” Auth, Security & Observability](backend/11-auth-security-observability.md) | Engineers | Better Auth session validation, auth proxy, CORS, secrets loading, structlog, Prometheus, OpenTelemetry, CostTracker | + +### Backend appendices + +| Document | What it covers | +|---|---| +| [API Endpoint Index](backend/appendices/api-endpoint-index.md) | Compact table of every route β€” method, path, purpose | +| [Class Index](backend/appendices/class-index.md) | All backend classes organised by package | +| [Class Reference](backend/appendices/class-reference.md) | Field-level reference for ORM models, Pydantic schemas, and service classes | +| [Data Flow Diagrams](backend/appendices/data-flow-diagrams.md) | 15 text-format diagrams covering every major request flow | +| [Risks, Assumptions & Testing](backend/appendices/risks-assumptions-testing.md) | Runtime/data/provider assumptions, risk table with mitigations, 13-category test strategy, 12-step smoke test | + +--- + +## Frontend + +| Document | Audience | What it covers | +|---|---|---| +| [Frontend TDD β€” Entry Point](frontend/README.md) | Engineers | Tech stack, page map, workflow diagram, architecture overview | +| [01 β€” App Shell](frontend/01-app-shell.md) | Engineers | `App.tsx` β€” `DocumentData` interface, step routing, `onComplete()` pattern, session persistence, navigation guards | +| [02 β€” Auth](frontend/02-auth.md) | Engineers | `LoginPage`, `AuthCallback`, `authUtils.ts` β€” OAuth flow, token lifecycle, `authenticatedFetch()`, visibility refresh | +| [03 β€” Upload](frontend/03-upload.md) | Engineers | Workflow step 1 β€” file upload, SHA-256 deduplication, auto-processing, parser selection | +| [04 β€” Processing](frontend/04-processing.md) | Engineers | Workflow step 2 β€” parsed content inspection, re-processing, PDF bounding box viewer | +| [05 β€” Study Config](frontend/05-study-config.md) | Engineers | Workflow step 3 β€” study type, entity editor, template loading, model selection | +| [06 β€” Extraction](frontend/06-extraction.md) | Engineers | Workflow step 4 β€” concurrent entity extraction, PDF reference highlighting, multi-model comparison, in-place editing | +| [07 β€” Evaluation](frontend/07-evaluation.md) | Engineers | Workflow step 5 β€” G-Eval metrics, background jobs, human score overrides, Excel export | +| [08 β€” Simplified Flow](frontend/08-simplified-flow.md) | Engineers | One-click pipeline, `useSimplifiedPipeline` hook, batched extraction, stage progression | +| [09 β€” Chat](frontend/09-chat.md) | Engineers | Freeform document Q&A, multi-document context, message ratings | +| [10 β€” Session History](frontend/10-session-history.md) | Engineers | Browse, restore, share, and delete sessions; shared sessions (read-only) | +| [11 β€” Templates](frontend/11-templates.md) | Engineers | Template CRUD, version history, fork, scope change, folder organisation | +| [12 β€” Groups](frontend/12-groups.md) | Engineers | Group lifecycle, member roles, add/remove members, user search | +| [13 β€” Executive Mode](frontend/13-executive-mode.md) | Engineers | Standalone summary generation without structured entity review | +| [14 β€” Batch Results](frontend/14-batch-results.md) | Engineers | Cross-file results table, fuzzy search, column visibility, Excel export | + +### Frontend appendices + +| Document | What it covers | +|---|---| +| [Component Index](frontend/appendices/component-index.md) | All shared components with props and usage | +| [Hooks & Contexts](frontend/appendices/hooks-contexts.md) | All custom hooks and `ThemeContext` with exported APIs | +| [Types & Interfaces](frontend/appendices/types-interfaces.md) | Key TypeScript interfaces: `DocumentData`, `Entity`, `Template`, `Group`, and more | + +--- + +## Deployment & operations + +| Document | Audience | What it covers | +|---|---|---| +| [GitHub Auth Setup](superpowers/setup-github-auth.md) | DevOps | 10-step guide to configuring GitHub Enterprise Cloud OAuth | +| [Migration Guide](superpowers/migration-guide.md) | Engineers | Supabase β†’ Azure Postgres + Better Auth migration history | +| [Deployment Plan](superpowers/plans/dockerize-and-deploy.md) | DevOps | Azure Container Apps architecture, CI/CD pipeline, provisioning record | +| [Logging Stack](../logging/README.md) | DevOps | LGTM stack setup: Grafana, Loki, Tempo, Prometheus; NSG firewall rules | diff --git a/docs/README.md b/docs/README.md new file mode 100644 index 0000000..057f87d --- /dev/null +++ b/docs/README.md @@ -0,0 +1,121 @@ +# Science-GPT Summarization Tool Overview + +Science-GPT is a web application that helps scientific reviewers work through large scientific documents more quickly. It can read uploaded PDF studies, extract key information, generate structured summaries, and let reviewers compare results from different AI models. + +This page is a high-level overview for readers who want to understand what the project does, who it is for, and what was learned during testing. + +## Where to go next + +- Read this page first for a plain-language project overview. +- See [`backend/`](backend/) for detailed backend architecture and implementation notes. +- See [`images/`](images/) for application screenshots and workflow diagrams. + + +## What problem does this solve? + +Scientific reviewers at PMRA review large volumes of technical documents, including toxicology studies, epidemiology studies, scientific articles, and grey literature. This work is important for pesticide and chemical safety decisions, but it is also time-consuming. + +A major part of the review process involves finding information in documents, extracting key details, organizing results, and writing summaries. Science-GPT was built to test whether AI can help with those tasks while still keeping scientific experts in control. + +The goal is not to replace reviewers. The goal is to help reviewers get to a strong first draft faster, so they can spend more time checking evidence, applying scientific judgement, and making decisions. + +## Who is this tool for? + +The main users are scientific reviewers and evaluators who review pesticide and chemical safety information. + +During Stream 2 testing, the project focused on two groups: + +- **Toxicology reviewers**, who review animal toxicity studies and summarize effects such as maternal and fetal outcomes. +- **Epidemiology reviewers**, who review human health studies and summarize study design, exposure, outcomes, measures of association, strengths, limitations, and risk-of-bias information. + +The tool may also be useful to project teams, managers, and technical partners who need to understand how AI could support scientific review workflows. + +## What does the tool do? + +At a high level, Science-GPT supports this workflow: + +1. A reviewer uploads one or more PDF studies. +2. The tool converts the PDF into structured text that AI models can use. +3. The reviewer chooses what information should be extracted. +4. One or more AI models extract the requested information. +5. The reviewer compares the outputs, checks the source evidence, and validates the result. +6. The output can be exported for reporting or further analysis. + +The application supports multiple models, including models from OpenAI, Google, Anthropic, Meta, and locally hosted open-source models. This makes it possible to compare model quality, speed, and cost on the same task. + +## How was it tested? + +Testing was led by Subject Matter Experts (SMEs). SMEs selected studies, created expected answers, reviewed AI outputs, and scored how well the models performed. + +The team tested two main use cases: + +### In vivo developmental toxicity studies + +Toxicology SMEs selected 10 published developmental toxicity studies. They manually created β€œground truth” summaries, which acted as the reference answers for scoring AI outputs. + +The AI models were tested on two levels of summary: + +- **Level 0 summaries:** simple study information such as author, title, publication date, and journal. +- **Level 1 summaries:** more detailed toxicology information such as test material, animal species, dose levels, route of administration, maternal effects, and fetal or offspring effects. + +### Epidemiology studies + +Epidemiology SMEs tested the tool on studies with different designs, including cohort, case-control, and biomonitoring studies. + +The tool was used to extract information such as participant numbers, pesticide of interest, exposure measurement, health outcomes, measures of association, study conclusions, strengths, limitations, and risk-of-bias information. + +This use case was more difficult because epidemiology studies often contain complex tables, different exposure measures, different study designs, and more variation in how results are reported. + +## How was performance measured? + +The main performance questions were: + +- **Was the answer correct?** +- **Was the answer complete?** +- **How much did it cost?** +- **How long did it take?** +- **Was the output useful to the reviewer?** + +For the toxicology testing, SMEs scored model outputs for correctness and completeness against the ground truth summaries. For epidemiology testing, SMEs also looked closely at whether the outputs were readable, concise, and useful for real review work. + +## Evaluation results + +A major part of Stream 2 was evaluation: testing the tool with SMEs, comparing model outputs, and measuring whether the results were useful for real scientific review work. + +### Toxicology results + +For the toxicology use case, SMEs tested the tool on 10 in vivo developmental toxicity studies. The models were scored against SME-created reference summaries. + +The strongest overall model was **Gemini 2.5 Pro**. It performed best across both simple study information and more detailed toxicology summary fields. + +| Model result | What it means | +| --- | --- | +| **Gemini 2.5 Pro was the top performer** | It had the best overall correctness and completeness scores for the toxicology studies. | +| **Simple fields were easier** | Most models could extract information such as author, title, publication date, and journal. | +| **Detailed scientific fields were harder** | More complex fields, such as maternal effects and fetal or offspring effects, showed bigger differences between models. | +| **Claude Sonnet 4.5 and Claude Opus 4.1 also performed strongly** | They were good options for complex summaries, but were more expensive than Gemini 2.5 Pro. | + +The testing also showed a large time-saving opportunity. SMEs spent about **300 minutes** manually creating summaries for 10 studies. Using the tool with the best-performing model and fastest document processing setup, the same set of draft summaries could be generated in slightly over **5 minutes**, before SME verification. + +### Epidemiology results + +For the epidemiology use case, the results were more mixed because epidemiology studies are less standardized and often include complex tables, different study designs, and detailed numerical results. + +The main findings were: + +| Model result | What it means | +| --- | --- | +| **Gemini 2.5 Pro and GPT-5.2 were strong general choices** | They performed well across many extraction tasks. | +| **Claude Sonnet 4.5 was best for complex table extraction** | It was especially useful when the reviewer needed numerical results organized into tables. | +| **Lower-cost models worked for simple fields** | They could handle straightforward information, but were less reliable for complex summaries and tables. | +| **The best model depended on the task** | No single model was best for every epidemiology extraction task. | + +In one applied epidemiology risk-of-bias test, the report estimated that reviewing 20 studies manually would take about **40 hours**. With AI-generated summaries plus SME verification, the work was estimated at about **11 hours total**. + +### Overall evaluation conclusion + +The evaluation showed that Science-GPT can create useful first drafts much faster than a fully manual workflow. However, the outputs still need SME review. The tool is most valuable when it helps reviewers quickly collect and organize information while leaving the final scientific judgement with the reviewer. + +The best default model from the testing was **Gemini 2.5 Pro**, especially for the toxicology use case. For epidemiology, **Gemini 2.5 Pro**, **GPT-5.2**, and **Claude Sonnet 4.5** were all useful, depending on whether the task required general extraction, lower cost, or stronger table generation. + +Overall, Stream 2 showed that Science-GPT can help scientific reviewers work faster while keeping expert judgement at the centre of the process. diff --git a/docs/backend/01-architecture.md b/docs/backend/01-architecture.md new file mode 100644 index 0000000..34799b5 --- /dev/null +++ b/docs/backend/01-architecture.md @@ -0,0 +1,261 @@ +# Backend Architecture + +> *Science-GPT's backend is a Python FastAPI service that sits between the frontend and everything else β€” the database, blob storage, document parsers, and AI providers. This document explains how those pieces connect: which packages own which responsibilities, how a request travels from the browser to the database and back, and which infrastructure assumptions the code relies on.* + +This document describes the high-level backend architecture, module boundaries, runtime setup, and major control/data flows. Lower-level class, endpoint, and schema details are split into the linked module documents. + +## 1. Application entry point + +The backend starts from `backend/main.py`. + +Key functions: + +| Function | Responsibility | +| --- | --- | +| `load_secrets_to_env()` | Loads `backend/core/secrets.toml` or nearby `core/secrets.toml` candidates and writes values into environment variables. Special-cases `Macbook.macbook_llm_base_url` into `MACBOOK_LLM_BASE_URL`. | +| `load_config()` | Loads provider-specific configuration from `backend/core/secrets.toml`, including Azure OpenAI, Azure Document Intelligence, Vertex/Gemini, Anthropic, and Google credentials. | +| `_setup_otel(app)` | Enables OpenTelemetry FastAPI instrumentation if `OTLP_ENDPOINT` is configured. | +| `lifespan(app)` | Sets the default asyncio thread pool to 64 workers so provider calls delegated through `asyncio.to_thread()` do not bottleneck on the default small pool. | +| `create_app()` | Creates the FastAPI app, installs metrics, tracing, CORS, exception handling, observability middleware, and all routers. | + +Router registration happens in `create_app()` and is intentionally ordered. The auth proxy router is mounted first so `/api/auth/*` traffic is forwarded to the Better Auth sidecar before other auth routes are considered. + +## 2. Runtime layers + +![Backend runtime architecture](images/backend-runtime-architecture.png) + +Read the diagram from top to bottom. `backend/main.py` creates the FastAPI runtime and installs cross-cutting middleware before requests reach routers. Routers stay thin: they validate HTTP inputs, call service objects, and translate service errors into API responses. The service layer owns orchestration across SQLAlchemy persistence, Azure Blob Storage, document parsers, LLM clients, evaluation jobs, sessions/groups/templates, and telemetry. External systems sit at the bottom because the backend treats them as replaceable infrastructure boundaries, not as dependencies that routers call directly. + +```text +Browser / frontend + | + | HTTP / JSON / multipart upload + v +FastAPI app (`backend/main.py`) + | + +-- core middleware/auth/config/logging + | + +-- API routers (`backend/api/*`) + | | + | v + +-- service layer (`backend/services/*`) + | | + | +-- SQLAlchemy DB service + | +-- document processors and blob storage + | +-- LLM provider clients + | +-- evaluation queue/adapters/metrics + | +-- sessions/groups/templates services + | + v +PostgreSQL + Azure Blob Storage + external model/parser providers +``` + +## 3. Package responsibilities + +### `backend/core/` + +Core cross-cutting concerns: + +- `auth.py` validates Better Auth sessions against the DB. +- `config.py` maps `secrets.toml` values into environment variables. +- `dependencies.py` re-exports auth dependencies. +- `middleware.py` configures CORS. +- `logging_config.py` configures structlog JSON logging, file logging, and optional Loki shipping. + +### `backend/api/` + +FastAPI routers. Routers translate HTTP input into Pydantic request models or primitive parameters, call services, and map exceptions to HTTP responses. + +Major router groups: + +- auth proxy and login history; +- file upload/download/listing; +- document processing, content, analysis, figures, tables; +- extraction and paragraph generation; +- evaluation and background evaluation jobs; +- sessions and shared sessions; +- groups and memberships; +- templates and folders; +- server health, config, models, telemetry, and logs; +- chat query endpoint. + +See [02-api-surface.md](02-api-surface.md) and [appendices/api-endpoint-index.md](appendices/api-endpoint-index.md). + +### `backend/models/` + +SQLAlchemy ORM models and database helpers. These classes define the physical data model used by the service layer. + +Important groups: + +- Better Auth tables: `User`, `AuthSession`, `Account`, `Verification`. +- Workflow tables: `AppSession`, `Document`, `ExtractionResult`, `EvaluationResult`. +- Collaboration: `Group`, `UserGroup`. +- Templates: `TemplateFolder`, `PromptTemplate`, `TemplateVersion`, `TemplatePermission`. +- Preferences and audit: `UserPreferences`, `LoginHistory`, `UserPromptTemplate`. +- Evaluation jobs: `EvalJobRecord`. + +See [03-data-models.md](03-data-models.md). + +### `backend/schemas/` + +Pydantic models used for API input/output and session aggregates. These are separate from ORM models so HTTP contracts can remain explicit even when the database representation changes. + +See [04-schemas.md](04-schemas.md). + +### `backend/services/` + +Domain and infrastructure services: + +| Subpackage | Responsibility | +| --- | --- | +| `database/` | SQLAlchemy persistence access layer. | +| `session/` | Session orchestration and DB-to-Pydantic conversion. | +| `document/` | Upload organization, parser orchestration, artifact access, bbox normalization. | +| `storage/` | Azure Blob Storage wrapper. | +| `llm/` | Provider-specific LLM clients and routing faΓ§ade. | +| `evaluation/` | DeepEval adapters, metric factories, result storage, background queue. | +| `groups/` | Group and membership authorization/business logic. | +| `templates/` | Prompt template CRUD, folders, permissions, versions, forks. | +| `telemetry/` | Cost and session metrics tracking. | + +## 4. Main data flows + +### 4.1 Authenticated request flow + +```text +Frontend request + -> Authorization: Bearer + -> FastAPI route with Depends(get_current_user) + -> core.auth.get_current_user() + -> query AuthSession + User from PostgreSQL + -> optional ALLOWED_EMAILS check + -> route receives current_user dict +``` + +Returned user dict contains `id`, `email`, `name`, `image`, and `is_admin`. + +### 4.2 Upload and processing flow + +```text +POST /api/upload + -> OrganizedFileService.save_uploaded_file() + -> SHA-256 file hash + -> blob: global/{hash}/original.{ext} + -> blob metadata: global/{hash}/metadata.json + -> optional Document DB row + +POST /api/documents/process/file/{file_hash} + -> OrganizedFileService cache check + -> DocumentService / processor orchestration + -> Azure Document Intelligence or Docling + -> local /tmp/summarization/{hash}/processed/{processor}/... + -> sync processed tree to blob + -> Document DB processing metadata update +``` + +See [05-document-processing.md](05-document-processing.md). + +### 4.3 Extraction flow + +```text +POST /api/extract + -> DocumentService.get_markdown_content() + -> optional figure context assembly + -> LLMService.extract_entities_from_markdown() + -> provider client call + -> normalize response/meta/cost + -> optional bbox matching against raw analysis + -> SessionService.add_extraction_result_fast() + -> extraction_results upsert +``` + +See [07-extraction-flow.md](07-extraction-flow.md). + +### 4.4 Evaluation flow + +```text +POST /api/evaluations/evaluate or /evaluate/batch + -> EvaluationService.create_evaluation_model() + -> metric factories produce GEval metrics + -> combined scoring prompt or per-metric fallback + -> cost tracking from adapter call history + -> optional JSON file result storage + +POST /api/evaluations/jobs + -> create EvalJob dataclass + -> persist EvalJobRecord for cross-worker status + -> background asyncio tasks + -> SessionService.add_evaluation_result_fast() +``` + +See [08-evaluation-flow.md](08-evaluation-flow.md). + +### 4.5 Restore-view flow + +```text +GET /api/sessions/{session_id}/restore-view + -> SessionService.get_session() + -> SessionService.build_restore_view() + -> OrganizedFileService.build_document_view() per document + -> frontend receives canonical uploadedFiles + processingResult state +``` + +See [09-session-sharing-groups.md](09-session-sharing-groups.md). + +## 5. Dependency direction + +The intended dependency direction is: + +```text +api -> services -> models/storage/provider SDKs +schemas -> api/services +core -> app/api/services as dependencies +``` + +Important exceptions: + +- `services.session` imports Pydantic schemas to build API-ready aggregate models. +- `services.telemetry.cost_tracker` writes session metrics back through the database service. +- Provider clients return dictionaries rather than shared typed result classes, so `LLMService` and API routers perform normalization. + +## 6. Infrastructure assumptions + +- PostgreSQL is available through `DATABASE_URL` or `POSTGRES_*` variables. +- Alembic migrations have been applied before serving traffic. +- Azure Blob Storage connection string is present for the organized file service in current upload/processing flows. +- Better Auth sidecar is reachable at the configured local/internal URL for `/api/auth/*` proxying. +- External provider credentials are optional per provider; unavailable providers should be reported as disabled rather than blocking the whole app. +- Production backend container is expected to run one Gunicorn worker per replica because Docling model memory and in-process concurrency controls assume a single process per container. + +## 7. Cross-cutting algorithms + +### Request observability + +`create_app()` installs an HTTP middleware that: + +1. assigns a short request ID; +2. binds the request ID into structlog context variables; +3. calls the downstream route; +4. logs method/path/status/duration at severity based on status code; +5. returns `X-Request-Id` in the response. + +### Processor selection + +`DocumentService._auto_select_processor()` currently selects Azure Document Intelligence if available; otherwise Docling. It does not yet perform content-based routing. + +### Model-provider routing + +`LLMService` switches by `model_type`. Each branch calls a provider client and records session metrics on successful responses. + +### Evaluation scoring + +`EvaluationService` prefers a combined scoring prompt for multiple metrics in one judge-model call, then falls back to per-metric concurrent evaluation if parsing fails. + +## 8. Related documents + +- [02-api-surface.md](02-api-surface.md) +- [03-data-models.md](03-data-models.md) +- [05-document-processing.md](05-document-processing.md) +- [06-llm-layer.md](06-llm-layer.md) +- [08-evaluation-flow.md](08-evaluation-flow.md) +- [11-auth-security-observability.md](11-auth-security-observability.md) diff --git a/docs/backend/02-api-surface.md b/docs/backend/02-api-surface.md new file mode 100644 index 0000000..ce51cc7 --- /dev/null +++ b/docs/backend/02-api-surface.md @@ -0,0 +1,481 @@ +# API Surface Technical Design + +> *Every action the frontend can take β€” uploading a file, running an extraction, fetching a session β€” goes through one of the 14 API routers documented here. This is the contract between the frontend and the backend: what URLs exist, what they expect, what they return, and what can go wrong. If you're building a new frontend feature or debugging an unexpected response, start here.* + +This document describes the backend FastAPI API surface by router. It focuses on interface communication: paths, methods, request/response structures, authentication, service dependencies, and important exceptions. + +For a compact route-only list, see [appendices/api-endpoint-index.md](appendices/api-endpoint-index.md). + +## 1. API architecture + +Routers are included from `backend/main.py` in this order: + +1. `api.auth.proxy.router` +2. `api.auth.router` +3. `api.files.router` +4. `api.documents.router` +5. `api.extractions.router` +6. `api.evaluations.router` +7. `api.evaluations.jobs_router` +8. `api.server.router` +9. `api.paragraphgenerator.router` +10. `api.paragraph_evaluation.router` +11. `api.sessions.router` +12. `api.groups.router` +13. `api.templates.router` +14. `api.chat.router` + +Most application endpoints require `Depends(get_current_user)`. Some file endpoints use `get_optional_user` or no explicit dependency, but still derive access through file hashes or storage lookups. + +## 2. Auth proxy and auth endpoints + +### `backend/api/auth/proxy.py` + +Router tags: `auth-proxy`. + +| Method/path | Request | Response | Auth behavior | Purpose | +| --- | --- | --- | --- | --- | +| `/{api/auth/{path:path}}` for all common HTTP methods | Raw forwarded request | Raw proxied response | Proxies Better Auth traffic; `get-session` can be email-allowlist checked | Transparent proxy from FastAPI to Better Auth sidecar. | + +Implementation details: + +- Forwards `/api/auth/*` to the auth sidecar. +- Preserves relevant headers and `Set-Cookie` behavior. +- Adds forwarded host/proto headers so Better Auth can construct public callback URLs correctly. +- Strips hop-by-hop headers. + +### `backend/api/auth/router.py` + +Router prefix: `/auth`. + +| Method/path | Request | Response | Dependencies | Purpose | +| --- | --- | --- | --- | --- | +| `GET /auth/health` | none | health JSON | `get_current_user` | Auth-protected health check. | +| `POST /auth/history` | current user + request metadata | login-history record/status | `get_current_user` | Records login audit information. | + +Service dependencies: + +- `core.auth.get_current_user` +- `SQLAlchemyDBService.record_login` + +## 3. File endpoints + +### `backend/api/files/router.py` + +Router prefix: `/api`. + +Local response models: + +- `FileUploadResponse` +- `UserFileInfo` + +| Method/path | Request | Response | Purpose | +| --- | --- | --- | --- | +| `POST /api/upload` | Multipart file upload, optional authenticated user | file hash/path metadata | Validate and store uploaded file through `OrganizedFileService`. | +| `GET /api/files/list` | optional user | list of user file metadata | List files associated with the user. | +| `GET /api/files/{file_id}` | file hash/id | file bytes response | Serve original uploaded file content. | +| `GET /api/files/{file_id}/info` | file hash/id | file metadata and processed flags | Return file metadata and parser availability state. | +| `DELETE /api/files/{file_id}` | file hash/id | status JSON | Delete behavior is stubbed/deferred in current code. | + +Important behavior: + +- Upload computes or reuses a SHA-256 file hash. +- Storage path follows `global/{file_hash}/original.{ext}` in blob storage. +- Metadata is stored at `global/{file_hash}/metadata.json`. +- Duplicate content returns the same hash and indicates deduplication. + +Service dependencies: + +- `OrganizedFileService` +- optional auth from `core.auth.get_optional_user` + +## 4. Document endpoints + +### `backend/api/documents/router.py` + +Router prefix: `/api/documents`. + +Request model: + +- `ProcessFileRequest` +- `ExtractFigureContentRequest` + +| Method/path | Request | Response | Purpose | +| --- | --- | --- | --- | +| `GET /api/documents/{document_id}/view` | optional `processor_used` query | canonical document view | Build frontend restore/view state from blob and metadata. | +| `POST /api/documents/process/file/{file_id}` | `ProcessFileRequest` | processing result + document view | Process an uploaded file, using cache when possible. | +| `GET /api/documents/{document_id}/content` | optional `processor_used` query | markdown content JSON | Return processed `document.md`. | +| `GET /api/documents/{document_id}/enhanced-content` | optional `processor_used` query | markdown with figure summaries inserted | Return enhanced markdown content. | +| `GET /api/documents/{document_id}/figures` | document id/file hash | list of figure metadata | Return figures from processor metadata. | +| `GET /api/documents/{document_id}/analysis` | optional `processor_used` query | normalized raw analysis JSON | Return processor raw analysis normalized for frontend. | +| `GET /api/documents/{document_id}/figures/{figure_filename}` | file hash + figure filename | image response | Serve figure image artifact. | +| `POST /api/documents/{document_id}/figures/{figure_id}/generate-summary` | `ExtractFigureContentRequest` | generated figure summary | Run a vision model over a figure and persist summary metadata. | +| `POST /api/documents/{document_id}/figures/{figure_id}/extract-content` | legacy alias | generated figure summary | Backward-compatible alias for figure content extraction. | +| `GET /api/documents/{document_id}/tables/{table_filename}` | file hash + table filename | HTML response | Serve table HTML artifact. | + +Important helper functions: + +- `camel_to_snake_case()` converts camelCase keys. +- `transform_keys_to_snake_case()` recursively converts nested structures. +- `_generate_figure_summary_with_retry()` handles vision-summary retries. +- `_insert_figure_summaries_inline()` injects figure summaries into markdown. + +Service dependencies: + +- `DocumentService` +- `OrganizedFileService` +- `LLMService` +- `normalize_bbox_format` +- `cost_tracker` +- SQLAlchemy DB service for document row updates + +Key exceptions/status behavior: + +- Missing processed artifacts produce 404-style HTTP exceptions. +- Processor failures surface as failed processing JSON or HTTP exceptions depending on path. +- Figure/table filenames are validated before artifact lookup to reduce path traversal risk. + +## 5. Extraction endpoint + +### `backend/api/extractions/router.py` + +Router prefix: `/api`. + +Request model: + +- `ExtractRequest` +- nested `Entity` + +| Method/path | Request | Response | Purpose | +| --- | --- | --- | --- | +| `POST /api/extract` | document conversion id, entities, model config, optional session id | extraction results per entity | Run entity extraction over markdown and persist optional session results. | + +Important behavior: + +- Loads markdown through `DocumentService`. +- Builds optional figure context from figure metadata and summaries. +- Runs one extraction task per entity concurrently. +- Uses `LLMService.extract_entities_from_markdown()` for provider routing. +- Attempts reference/bounding-box matching where raw analysis and references exist. +- Persists results with `SessionService.add_extraction_result_fast()` when `session_id` is provided. + +Service dependencies: + +- `DocumentService` +- `LLMService` +- `SessionService` +- Azure/Docling bbox matchers +- `cost_tracker` + +Provider map: + +- `azure` -> Azure OpenAI style extraction +- `gemini` -> Gemini/Vertex +- `anthropic` -> Anthropic Vertex +- `llama` / `azure-llama` +- `macbook` +- `vllm` + +## 6. Paragraph generation and paragraph evaluation + +### `backend/api/paragraphgenerator.py` + +Router prefix: `/api`. + +Request model: + +- `ParagraphGenerationRequest` + +| Method/path | Request | Response | Purpose | +| --- | --- | --- | --- | +| `POST /api/generate_paragraph` | extracted entities, model config, optional session id | generated paragraph text and metadata | Generate a scientific summary paragraph from extracted entity values. | + +Important behavior: + +- Routes to `LLMService.generate_paragraph()`. +- Persists paragraph output into session extraction results if `session_id` is provided. +- Uses provider map similar to extraction. + +### `backend/api/paragraph_evaluation.py` + +Router prefix: `/api/paragraph-evaluation`. + +Request model: + +- `ParagraphEvalGenerateRequest` + +| Method/path | Request | Response | Purpose | +| --- | --- | --- | --- | +| `POST /api/paragraph-evaluation/generate` | session/model/entity information | paragraph ground-truth/evaluation record | Build a deterministic paragraph ground truth from entity values. | + +Helper: + +- `build_paragraph_ground_truth(entities)` converts entity values into a paragraph-style expected answer. + +## 7. Evaluation endpoints + +### `backend/api/evaluations/router.py` + +Router prefix: `/api/evaluations`. + +Request models: + +- `EvaluationRequest` +- `BatchEvaluationRequest` +- `CustomMetricRequest` + +| Method/path | Request | Response | Purpose | +| --- | --- | --- | --- | +| `POST /api/evaluations/cancel` | `X-Session-Id` header | cancellation status | Mark a session evaluation as cancelled. | +| `POST /api/evaluations/evaluate` | `EvaluationRequest` | `EvaluationResponse`-like dict | Evaluate one extraction. | +| `POST /api/evaluations/evaluate/batch` | `BatchEvaluationRequest` | batch evaluation result | Evaluate multiple extraction outputs. | +| `POST /api/evaluations/evaluate/custom` | `CustomMetricRequest` | custom metric result | Evaluate using caller-provided metric steps. | +| `GET /api/evaluations/results/{evaluation_id}` | evaluation id | stored result JSON | Fetch a result from file-backed evaluation storage. | +| `GET /api/evaluations/results` | none | list of stored results | List stored evaluation outputs. | +| `GET /api/evaluations/metrics/info` | none | metric/provider metadata | Describe built-in metrics and configured providers. | + +Service dependencies: + +- `EvaluationService` +- cancellation helpers in `services.evaluation.evaluation_service` +- provider configuration from environment variables + +### `backend/api/evaluations/jobs.py` + +Router prefix: `/api/evaluations/jobs`. + +Request models: + +- `EvalTaskRequest` +- `ProviderConfigRequest` +- `SubmitJobRequest` + +| Method/path | Request | Response | Purpose | +| --- | --- | --- | --- | +| `POST /api/evaluations/jobs` | tasks + providers + session id | job status with job id | Submit background evaluation job. | +| `GET /api/evaluations/jobs/{job_id}` | job id | current job status | Poll in-memory or DB-backed job status. | +| `POST /api/evaluations/jobs/{job_id}/cancel` | job id | cancellation status | Cancel local or cross-worker job. | + +Service dependencies: + +- `services.evaluation.job_queue.create_job` +- `get_job` +- `cancel_job` +- `EvalJobRecord` persistence through DB service + +## 8. Session endpoints + +### `backend/api/sessions/router.py` + +Router prefix: `/api/sessions`. + +Request/response schemas: + +- `CreateSessionRequest` +- `UpdateSessionRequest` +- `Session` +- `SessionListResponse` +- local `ShareSessionRequest` + +| Method/path | Request | Response | Purpose | +| --- | --- | --- | --- | +| `POST /api/sessions` | `CreateSessionRequest` | `Session` | Create workflow session. | +| `GET /api/sessions` | current user | `SessionListResponse` | List user session summaries. | +| `GET /api/sessions/{session_id}` | session id | `Session` | Fetch full session owned by user, with shared fallback. | +| `GET /api/sessions/{session_id}/restore-view` | session id | restore-view dict | Build frontend restore state. | +| `PATCH /api/sessions/{session_id}` | `UpdateSessionRequest` | session dict | Update config, docs, extractions, evaluations. | +| `DELETE /api/sessions/{session_id}` | session id | deletion status | Delete owned session. | +| `POST /api/sessions/{session_id}/extractions` | `ExtractionResult` | updated session/status | Add extraction result. | +| `POST /api/sessions/{session_id}/evaluations` | `EvaluationResult` | updated session/status | Add evaluation result. | +| `GET /api/sessions/shared/list` | current user | `SessionListResponse` | List sessions shared with user groups. | +| `GET /api/sessions/shared/{session_id}` | session id | `Session` | Fetch shared session. | +| `GET /api/sessions/shared/{session_id}/restore-view` | session id | restore-view dict | Restore shared session view. | +| `POST /api/sessions/{session_id}/share` | group id | status/session | Share owned session with group. | +| `DELETE /api/sessions/{session_id}/share` | session id | status | Remove sharing from owned session. | + +Service dependencies: + +- `SessionService` +- `SQLAlchemyDBService` +- `OrganizedFileService` indirectly through restore view + +## 9. Group endpoints + +### `backend/api/groups/router.py` + +Router prefix: `/api/groups`. + +Local request/response models: + +- `CreateGroupRequest` +- `UpdateGroupRequest` +- `AddMemberRequest` +- `UpdateMemberRoleRequest` +- `GroupResponse` +- `GroupDetailResponse` +- `MemberResponse` +- `UserSearchResult` + +| Method/path | Request | Response | Purpose | +| --- | --- | --- | --- | +| `GET /api/groups` | current user | list groups | List groups for current user. | +| `POST /api/groups` | group name/description | group | Create group and owner membership. | +| `GET /api/groups/{group_id}` | group id | group detail | Fetch group with members. | +| `PUT /api/groups/{group_id}` | name/description | group | Update group metadata. | +| `DELETE /api/groups/{group_id}` | group id | 204 | Delete group. | +| `GET /api/groups/{group_id}/members` | group id | members | List group members. | +| `POST /api/groups/{group_id}/members` | user id/email + role | member | Add group member. | +| `PUT /api/groups/{group_id}/members/{user_id}` | role | member | Update member role. | +| `DELETE /api/groups/{group_id}/members/{user_id}` | member user id | 204 | Remove member. | +| `GET /api/groups/users/search` | query | users | Search users for group membership. | + +Service dependency: + +- `GroupService` + +Authorization is handled in `GroupService`, including owner/admin/member role rules and system-admin bypass where implemented. + +## 10. Template endpoints + +### `backend/api/templates/router.py` + +Router prefix: `/api/templates`. + +Local request/response models: + +- `EntityModel` +- `VariableModel` +- `CreateTemplateRequest` +- `UpdateTemplateRequest` +- `SetImmutableRequest` +- `SetPermissionRequest` +- `ForkTemplateRequest` +- `ChangeScopeRequest` +- `CreateFolderRequest` +- `RenameFolderRequest` +- `FolderResponse` +- `TemplateResponse` +- `VersionResponse` +- `PermissionResponse` + +Folder endpoints: + +| Method/path | Purpose | +| --- | --- | +| `GET /api/templates/folders` | List folders by scope/owner/parent. | +| `POST /api/templates/folders` | Create folder. | +| `PATCH /api/templates/folders/{folder_id}` | Rename folder. | +| `DELETE /api/templates/folders/{folder_id}` | Delete empty folder. | + +Template endpoints: + +| Method/path | Purpose | +| --- | --- | +| `GET /api/templates` | List accessible templates with filters. | +| `POST /api/templates` | Create template. | +| `GET /api/templates/{template_id}` | Fetch one template. | +| `PUT /api/templates/{template_id}` | Update template and create version snapshot. | +| `DELETE /api/templates/{template_id}` | Delete template. | +| `POST /api/templates/{template_id}/fork` | Copy accessible template into user scope. | +| `PUT /api/templates/{template_id}/scope` | Change template scope. | +| `PUT /api/templates/{template_id}/immutable` | Set immutability flag. | +| `GET /api/templates/{template_id}/versions` | List version history. | +| `POST /api/templates/{template_id}/revert/{version}` | Revert to a previous version. | +| `GET /api/templates/{template_id}/permissions` | List explicit permissions. | +| `POST /api/templates/{template_id}/permissions` | Upsert user permission. | +| `DELETE /api/templates/{template_id}/permissions/{user_id}` | Remove user permission. | + +Service dependencies: + +- `TemplateService` +- `FolderService` + +See [10-template-system.md](10-template-system.md). + +## 11. Server, metrics, and model endpoints + +### `backend/api/server/router.py` + +Router prefix: `/api`. + +Local request model: + +- `BatchMetricsRequest` + +| Method/path | Auth | Purpose | +| --- | --- | --- | +| `GET /api/server/health` | no schema auth | DB and service health check. | +| `POST /api/telemetry/traces` | unauthenticated internal/proxy style | Proxy browser OTLP traces to Tempo endpoint. | +| `POST /api/server/client-error` | unauthenticated report endpoint | Receive frontend client error reports. | +| `GET /api/server-config` | public | Return provider/config flags as `ServerConfig`. | +| `GET /api/models` | authenticated | Return available model catalog across providers. | +| `GET /api/server/session-metrics` | authenticated | Return session cost/latency/call totals. | +| `POST /api/server/session-metrics/load` | authenticated | Load metrics from DB into cost tracker. | +| `DELETE /api/server/session-metrics` | authenticated | Clear in-memory and DB metrics for a session. | +| `POST /api/server/batch-metrics` | authenticated | Record batch metric summary. | +| `GET /api/server/document-metrics` | authenticated | Return document-level metrics. | +| `POST /api/server/benchmark/clear` | authenticated | Clear benchmark/session cache state. | +| `GET /api/server/logs` | authenticated | Return backend logs. | + +Service dependencies: + +- `LLMService` provider clients indirectly +- `MacbookLLMClient` model discovery/health +- `cost_tracker` +- DB service +- log files under `backend/output/logs` + +## 12. Chat endpoint + +### `backend/api/chat/router.py` + +Router prefix: `/api/chat`. + +Request model: + +- `ChatQueryRequest` + +| Method/path | Request | Response | Purpose | +| --- | --- | --- | --- | +| `POST /api/chat/query` | query, optional `document_markdown`, model config | model answer JSON | General chat endpoint over optional uploaded document markdown context. | + +Important behavior: + +- Reuses LLM provider logic for a general answer rather than structured entity extraction. +- Frontend can concatenate multiple documents into the single `document_markdown` field. + +## 13. Interface conventions + +### Authentication + +Most protected routes use: + +```python +Depends(get_current_user) +``` + +The dependency returns a dict: + +```python +{ + "id": str, + "email": str, + "name": Optional[str], + "image": Optional[str], + "is_admin": bool +} +``` + +### Errors + +Routers generally use `HTTPException` for expected API errors. `main.py` also installs a global exception handler returning HTTP 500 with `{"detail": str(exc)}` for unhandled errors. + +### Response shape style + +The backend uses a mixture of: + +- Pydantic response models for sessions/groups/templates; +- plain dictionaries for processing, extraction, evaluation, server metrics, and model catalog endpoints; +- binary `Response` objects for files, figures, tables, and telemetry proxying. + +This mixed style should be considered part of the current API contract when changing clients. diff --git a/docs/backend/03-data-models.md b/docs/backend/03-data-models.md new file mode 100644 index 0000000..9177eb7 --- /dev/null +++ b/docs/backend/03-data-models.md @@ -0,0 +1,457 @@ +# Data Models Technical Design + +> *Everything a reviewer does β€” uploading a file, running an extraction, saving a session, sharing with a group β€” gets written to PostgreSQL through one of these 13 ORM models. This document describes every table: what each field stores, its type and constraints, and any gotchas between what the ORM assumes and what the database actually enforces. Read this if you're adding a new feature that needs a new table, or if you're debugging unexpected data in the database.* + +This document describes the backend physical data model implemented with SQLAlchemy models in `backend/models/` and Alembic migrations in `backend/alembic/`. + +## Visual overview + +![Backend data model relationships](images/data-model-relationships.png) + +The diagram separates the database into five operational areas. Better Auth owns login identity through `user`, `session`, `account`, and `verification`. Application workflow state starts at `app_sessions`, flows to `documents`, then to `extraction_results`, and finally to `evaluation_results`. `eval_jobs` is intentionally separate from normalized evaluation scores because it tracks background execution status and polling state. Collaboration is handled through `groups` and `user_groups`, while templates use their own scoped records, version snapshots, and optional per-user permission overrides. + +## 1. Database infrastructure + +### `backend/models/base.py` + +| Symbol | Purpose | +| --- | --- | +| `Base` | SQLAlchemy declarative base for all ORM models. | +| `_build_database_url()` | Uses `DATABASE_URL` or constructs a PostgreSQL URL from `POSTGRES_*` variables. Converts asyncpg URLs to sync SQLAlchemy URLs. | +| `get_engine()` | Lazy sync SQLAlchemy engine singleton with pool settings. | +| `get_session_factory()` | Lazy `sessionmaker` singleton. | +| `get_db_session()` | Returns one new SQLAlchemy session. | +| `db_session_scope()` | Context manager for transaction-scoped DB writes with commit/rollback. | + +The ORM is synchronous. Async FastAPI handlers call sync DB code directly or through service methods; cost-tracker DB updates are pushed through an executor where needed. + +## 2. Auth tables + +### `User` + +File: `backend/models/user.py` + +Table: `user` + +| Field | Type | Notes | +| --- | --- | --- | +| `id` | `String(36)` | Primary key from Better Auth. | +| `name` | `Text` | Required display name. | +| `email` | `Text` | Required, unique. | +| `email_verified` | `Boolean` | Column name `emailVerified`, Python default `False`. | +| `image` | `Text` | Optional avatar/image URL. | +| `created_at` | `DateTime` | Column name `createdAt`, Python default now. | +| `updated_at` | `DateTime` | Column name `updatedAt`, Python default/onupdate now. | +| `role` | `Text` | Application role, Python default `user`. | +| `is_admin` | `Boolean` | Application admin flag, Python default `False`. | + +Associations: + +- Referenced by auth/session/account tables and nearly all app-owned records. + +### `AuthSession` + +Table: `session` + +| Field | Type | Notes | +| --- | --- | --- | +| `id` | `String(36)` | Primary key. | +| `expires_at` | `DateTime` | Column name `expiresAt`; used for session expiry. | +| `token` | `Text` | Required, unique; matched by `core.auth.get_current_user`. | +| `created_at` / `updated_at` | `DateTime` | Better Auth timestamp columns. | +| `ip_address` | `Text` | Column name `ipAddress`. | +| `user_agent` | `Text` | Column name `userAgent`. | +| `user_id` | `String(36)` | Column name `userId`; logical link to `user.id`. | + +### `Account` + +Table: `account` + +Stores Better Auth account/provider linkage, including OAuth access/refresh/id tokens, token expiry fields, scopes, and optional password field. + +### `Verification` + +Table: `verification` + +Stores Better Auth verification/reset tokens: `identifier`, `value`, expiry, created/updated timestamps. + +## 3. Session and document workflow tables + +### `AppSession` + +File: `backend/models/app_session.py` + +Table: `app_sessions` + +Represents one user workflow session, not a login session. + +| Field | Type | Notes | +| --- | --- | --- | +| `id` | `UUID` | Primary key. | +| `user_id` | `String(36)` | FK to `user.id`, cascade delete. | +| `name` | `Text` | Default `Untitled Session`. | +| `status` | `Text` | Default `in_progress`. | +| `last_step` | `Text` | Default `upload`. | +| `configuration` | `JSONB` | Extraction/session config. | +| `evaluation_config` | `JSONB` | Evaluation settings. | +| `files_config` | `JSONB` | Per-file frontend/restore settings. | +| `total_cost` | `Float` | Session-level cost total. | +| `total_latency` | `Float` | Session-level latency total. | +| `total_calls` | `Integer` | Number of recorded calls. | +| `shared_with_group_id` | `UUID` | Optional FK to `groups.id`, set null on group deletion. | +| `shared_by` | `String(36)` | Optional FK to `user.id`. | +| `shared_at` | `DateTime` | Share timestamp. | +| `created_at` / `updated_at` | `DateTime` | ORM timestamps. | + +Indexes: + +- `idx_app_sessions_user_id` +- `idx_app_sessions_updated_at` +- partial `idx_app_sessions_shared_group` where `shared_with_group_id IS NOT NULL` + +### `Document` + +File: `backend/models/document.py` + +Table: `documents` + +Tracks uploaded and processed document metadata. + +| Field | Type | Notes | +| --- | --- | --- | +| `id` | `UUID` | Primary key. | +| `session_id` | `UUID` | Optional FK to `app_sessions.id`, cascade delete. | +| `user_id` | `String(36)` | FK to `user.id`, cascade delete. | +| `file_hash` | `Text` | SHA-256 content hash; indexed. | +| `filename` | `Text` | Original filename or resolved filename. | +| `file_path` | `Text` | Blob-style or local file path. | +| `study_type` | `Text` | Optional domain/study label. | +| `processor_used` | `Text` | Parser used, e.g. `azure_doc_intelligence` or `docling`. | +| `processing_status` | `Text` | Free-text status, default `pending`. | +| `processing_error` | `Text` | Error message if processing failed. | +| `extracted_text_path` | `Text` | Path to markdown or extracted content artifact. | +| `processed_at` | `DateTime` | Completed timestamp. | +| `parse_cost` | `Float` | Estimated parsing cost. | +| `page_count` | `Integer` | Parsed page count. | +| `parse_duration_seconds` | `Float` | Processing duration. | +| `figure_count` | `Integer` | Parsed figure count. | +| `table_count` | `Integer` | Parsed table count. | +| `created_at` / `updated_at` | `DateTime` | ORM timestamps. | + +Indexes: + +- `idx_documents_session_id` +- `idx_documents_user_id` +- `idx_documents_file_hash` + +### `ExtractionResult` + +File: `backend/models/extraction.py` + +Table: `extraction_results` + +Stores one extracted entity for one document/model combination. + +| Field | Type | Notes | +| --- | --- | --- | +| `id` | `UUID` | Primary key. | +| `session_id` | `UUID` | FK to `app_sessions.id`, cascade delete. | +| `document_id` | `UUID` | FK to `documents.id`, cascade delete. | +| `entity_name` | `Text` | Requested entity name. | +| `model_id` | `Text` | Provider/model identifier. | +| `extracted_text` | `Text` | Extracted answer text. | +| `bbox_references` | `JSONB` | Matched references/bounding boxes. | +| `status` | `Text` | Free-text status, default `pending`. | +| `error_message` | `Text` | Error text for failed extraction. | +| `extracted_at` | `DateTime` | Completion timestamp. | +| `prompt_tokens` | `Integer` | Provider prompt token count. | +| `completion_tokens` | `Integer` | Provider completion token count. | +| `duration_ms` | `Integer` | Extraction duration. | +| `cost` | `Float` | Estimated extraction cost. | +| `created_at` / `updated_at` | `DateTime` | ORM timestamps. | + +Constraints/indexes: + +- Unique `(document_id, entity_name, model_id)` as `uq_extraction_doc_entity_model`. +- Indexed by session, document, and entity/model. + +Upsert behavior depends on the unique constraint. + +### `EvaluationResult` + +File: `backend/models/evaluation.py` + +Table: `evaluation_results` + +Stores one evaluation score for one extraction result, metric, and judge model. + +| Field | Type | Notes | +| --- | --- | --- | +| `id` | `UUID` | Primary key. | +| `extraction_result_id` | `UUID` | FK to `extraction_results.id`, cascade delete. | +| `metric` | `Text` | Metric name, e.g. correctness, completeness, relevance, safety. | +| `score` | `Float` | Numeric score. | +| `reasoning` | `Text` | Judge explanation. | +| `judge_model` | `Text` | Model that judged the output. | +| `human_score` | `Float` | Optional human override. | +| `ground_truth` | `Text` | Expected answer. | +| `evaluation_cost` | `Float` | Estimated judge-call cost. | +| `evaluation_time` | `Float` | Evaluation duration. | +| `evaluated_at` | `DateTime` | Evaluation timestamp. | +| `created_at` / `updated_at` | `DateTime` | ORM timestamps. | + +Constraints/indexes: + +- Unique `(extraction_result_id, metric, judge_model)` as `uq_eval_extraction_metric_judge`. +- Indexed by extraction id and judge model. + +## 4. Groups and memberships + +### `Group` + +File: `backend/models/group.py` + +Table: `groups` + +| Field | Type | Notes | +| --- | --- | --- | +| `id` | `UUID` | Primary key. | +| `name` | `Text` | Group name. | +| `description` | `Text` | Optional description. | +| `created_by` | `String(36)` | FK to `user.id`. | +| `created_at` / `updated_at` | `DateTime` | ORM timestamps. | + +Indexes: + +- `idx_groups_created_by` +- `idx_groups_name` + +### `UserGroup` + +Table: `user_groups` + +Composite primary key: + +- `user_id` +- `group_id` + +Fields: + +| Field | Type | Notes | +| --- | --- | --- | +| `user_id` | `String(36)` | FK to `user.id`, cascade delete. | +| `group_id` | `UUID` | FK to `groups.id`, cascade delete. | +| `role` | `Text` | One of `viewer`, `member`, `admin`, `owner`. | +| `joined_at` | `DateTime` | Join timestamp. | + +Constraint: + +- `ck_user_groups_role` enforces allowed roles. + +## 5. Template system tables + +### `TemplateFolder` + +File: `backend/models/template.py` + +Table: `template_folders` + +Supports hierarchical folders scoped to user, group, or global template workspaces. + +| Field | Type | Notes | +| --- | --- | --- | +| `id` | `UUID` | Primary key. | +| `name` | `Text` | Folder name. | +| `scope` | `Text` | `user`, `group`, or `global`. | +| `owner_user_id` | `String(36)` | User owner for user scope. | +| `owner_group_id` | `UUID` | Group owner for group scope. | +| `parent_id` | `UUID` | Self-FK for hierarchy. | +| `created_by` | `String(36)` | Creator user id. | +| `created_at` / `updated_at` | `DateTime` | ORM timestamps. | + +Constraint: + +- `ck_template_folders_scope` enforces allowed scope. + +### `PromptTemplate` + +Table: `prompt_templates` + +Stores reusable prompt templates. + +| Field | Type | Notes | +| --- | --- | --- | +| `id` | `UUID` | Primary key. | +| `name` | `Text` | Template name. | +| `description` | `Text` | Optional description. | +| `study_type` | `Text` | Optional domain/study type. | +| `scope` | `Text` | `user`, `group`, or `global`. | +| `owner_user_id` | `String(36)` | User owner for user scope. | +| `owner_group_id` | `UUID` | Group owner for group scope. | +| `system_prompt` | `Text` | Optional shared system prompt. | +| `entities` | `JSONB` | Entity definitions. | +| `summary_prompt` | `Text` | Optional paragraph/summary prompt. | +| `variables` | `JSONB` | Template variable definitions. | +| `is_immutable` | `Boolean` | Blocks edits when true. | +| `tags` | `ARRAY(Text)` | Search/filter tags. | +| `is_default` | `Boolean` | Default template flag. | +| `version` | `Integer` | Current version number. | +| `folder_id` | `UUID` | Folder association; model does not declare FK. | +| `created_by` | `String(36)` | Creator. | +| `created_at` / `updated_at` | `DateTime` | ORM timestamps. | + +Constraint: + +- `ck_templates_scope` enforces allowed scope. + +### `TemplateVersion` + +Table: `template_versions` + +Stores snapshots before template updates. + +| Field | Type | Notes | +| --- | --- | --- | +| `id` | `UUID` | Primary key. | +| `template_id` | `UUID` | FK to `prompt_templates.id`, cascade delete. | +| `version` | `Integer` | Snapshot version. | +| `system_prompt` | `Text` | Snapshot field. | +| `entities` | `JSONB` | Snapshot field. | +| `summary_prompt` | `Text` | Snapshot field. | +| `variables` | `JSONB` | Snapshot field. | +| `changed_by` | `String(36)` | User who caused snapshot. | +| `change_summary` | `Text` | Optional summary. | +| `created_at` | `DateTime` | Snapshot timestamp. | + +Constraint: + +- Unique `(template_id, version)` as `uq_template_version`. + +### `TemplatePermission` + +Table: `template_permissions` + +Per-user template permission override. + +| Field | Type | Notes | +| --- | --- | --- | +| `id` | `UUID` | Primary key. | +| `template_id` | `UUID` | FK to `prompt_templates.id`, cascade delete. | +| `user_id` | `String(36)` | FK to `user.id`, cascade delete. | +| `can_read` | `Boolean` | Read override. | +| `can_write` | `Boolean` | Write override. | +| `granted_by` | `String(36)` | User who granted permission. | +| `created_at` | `DateTime` | Grant timestamp. | + +Constraint: + +- Unique `(template_id, user_id)` as `uq_template_permission`. + +## 6. Preferences, login history, and legacy prompt templates + +### `UserPreferences` + +Table: `user_preferences` + +One row per user. Stores default models, default temperature, and arbitrary settings JSON. + +### `LoginHistory` + +Table: `login_history` + +Audit trail with `user_id`, `ip_address`, `user_agent`, and `login_at`. + +### `UserPromptTemplate` + +Table: `user_prompt_templates` + +Legacy user-scoped prompt templates. Unique `(user_id, name, entity_name)`. + +## 7. Evaluation jobs + +### `EvalJobRecord` + +File: `backend/models/eval_job.py` + +Table: `eval_jobs` + +Persists background evaluation job state so polling can work across workers/replicas. + +| Field | Type | Notes | +| --- | --- | --- | +| `job_id` | `Text` | Primary key. | +| `session_id` | `Text` | Associated session id. | +| `user_id` | `Text` | Owner/requester user id. | +| `status` | `Text` | pending/running/completed/cancelled/failed style status. | +| `progress` | `Integer` | Completed task count. | +| `total` | `Integer` | Total task count. | +| `results` | `JSONB` | Serialized task results. | +| `errors` | `JSONB` | Serialized errors. | +| `error` | `Text` | Top-level failure message. | +| `created_at` | `DateTime(timezone=True)` | Creation timestamp. | +| `completed_at` | `DateTime(timezone=True)` | Completion timestamp. | + +Indexes: + +- `idx_eval_jobs_status` +- `idx_eval_jobs_user_id` + +## 8. Alembic migrations + +### Initial schema + +`backend/alembic/versions/03069a8f5e8c_initial_schema.py` creates: + +- Better Auth tables: `account`, `session`, `user`, `verification`. +- Collaboration and preferences: `groups`, `user_groups`, `user_preferences`, `login_history`, `user_prompt_templates`. +- Workflow tables: `app_sessions`, `documents`, `extraction_results`, `evaluation_results`. +- Template tables: `template_folders`, `prompt_templates`, `template_versions`, `template_permissions`. + +### Eval jobs migration + +`backend/alembic/versions/b5f8e2a1c9d3_add_eval_jobs_table.py` creates `eval_jobs` with server defaults for `status`, `progress`, and `total`. + +## 9. Important model-vs-migration notes + +Some defaults are Python-side ORM defaults, not database server defaults. Direct SQL inserts may behave differently from ORM inserts unless callers provide values explicitly. + +Examples: + +- `User.email_verified` has a Python default but no server default in the initial migration. +- `AppSession.status`, `last_step`, config JSON fields, metrics, and timestamps rely on ORM/application defaults. +- `Document.processing_status` and `ExtractionResult.status` are free text and have no DB check constraints. +- `PromptTemplate.entities` is non-null in the model but nullable in the initial migration. +- `PromptTemplate.folder_id` is used as a logical association but no FK is declared in the model. +- `EvalJobRecord.results` and `errors` have Python defaults but no JSON server defaults in the migration. + +These are not necessarily bugs, but they are important implementation constraints for tests and direct DB scripts. + +## 10. Relationship summary + +```text +user + -> session / account / verification Better Auth + -> app_sessions workflow ownership + -> documents uploaded/processed docs + -> groups.created_by group creator + -> user_groups group membership + -> prompt_templates / template_folders template ownership/creation + -> template_permissions explicit permissions + -> user_preferences / login_history settings/audit + +app_sessions + -> documents + -> extraction_results + +extraction_results + -> evaluation_results + +groups + -> user_groups + -> app_sessions.shared_with_group_id + -> group-scoped templates/folders + +prompt_templates + -> template_versions + -> template_permissions +``` diff --git a/docs/backend/04-schemas.md b/docs/backend/04-schemas.md new file mode 100644 index 0000000..da2fbe5 --- /dev/null +++ b/docs/backend/04-schemas.md @@ -0,0 +1,411 @@ +# Pydantic Schemas Technical Design + +> *Pydantic schemas are the typed contracts that sit at the boundary between the frontend and the backend β€” they define exactly what JSON the API accepts and what it returns. This document lists every schema class used for request bodies and response payloads, with notes on design decisions like why certain fields use free strings instead of enums. If the frontend is sending data that the backend rejects, or you're not sure what shape a response will have, start here.* + +This document describes request and response schemas in `backend/schemas/` and local router-level schemas in `backend/api/*`. These schemas define the API-facing data structures separate from SQLAlchemy ORM models. + +## 1. Shared enum schemas + +### `ProcessorType` + +File: `backend/schemas/enums.py` + +String enum values: + +| Name | Value | Meaning | +| --- | --- | --- | +| `AUTO` | `auto` | Let backend choose parser. | +| `DOCLING` | `docling` | Use Docling parser. | +| `AZURE_DOC_INTELLIGENCE` | `azure_doc_intelligence` | Use Azure Document Intelligence parser. | + +Used by document-processing requests. + +## 2. Document schemas + +File: `backend/schemas/documents.py` + +### `ProcessFileRequest` + +Request body for document processing. + +| Field | Type | Default | Notes | +| --- | --- | --- | --- | +| `processor` | `Optional[ProcessorType]` | `auto` | Parser selection. | +| `extract_figures` | `bool` | `True` | Whether figure extraction should run. | +| `batch_number` | `Optional[int]` | `None` | Frontend grouping/benchmark metadata. Description says 1-99, but no numeric bounds are enforced. | + +### `ExtractFigureContentRequest` + +Request body for figure content extraction or summary generation. + +| Field | Type | Default | Notes | +| --- | --- | --- | --- | +| `model_type` | `str` | `gemini` | Free string, not enum-constrained. | +| `model_id` | `Optional[str]` | `None` | Specific model identifier. | +| `extraction_prompt` | `str` | long OCR/scientific prompt | Prompt sent to vision model. | +| `max_tokens` | `int` | `2048` | No explicit bounds. | +| `temperature` | `float` | `0.0` | No explicit bounds. | +| `system_message` | `Optional[str]` | `None` | Optional system prompt. | + +### `FigureExtractionResult` + +Nested result attached to figure metadata. + +| Field | Type | +| --- | --- | +| `content` | `str` | +| `model_used` | `str` | +| `timestamp` | `str` | +| `duration` | `float` | + +### `FigureMetadata` + +Figure metadata shape returned to API callers. + +| Field | Type | Notes | +| --- | --- | --- | +| `id` | `str` | Figure identifier. | +| `page` | `Optional[int]` | Page number. | +| `caption` | `Optional[str]` | Figure caption. | +| `image_path` | `Optional[str]` | Relative artifact path. | +| `bounding_regions` | `Optional[list]` | Unconstrained list shape. | +| `extracted_content` | `Optional[FigureExtractionResult]` | Nested extracted summary/content. | + +## 3. Extraction schemas + +File: `backend/schemas/extractions.py` + +### `Entity` + +One entity extraction instruction. + +| Field | Type | Notes | +| --- | --- | --- | +| `name` | `str` | Entity label. | +| `prompt` | `str` | Entity-specific extraction prompt. | +| `extracted` | `Optional[str]` | Optional existing value. | +| `system_prompt` | `Optional[str]` | Optional per-entity system prompt. | + +### `ExtractRequest` + +Request body for `POST /api/extract`. + +| Field | Type | Notes | +| --- | --- | --- | +| `conversion_id` | `str` | File hash/conversion id. | +| `session_id` | `Optional[str]` | Session for persistence. | +| `deployment` | `Optional[str]` | Azure-style deployment. | +| `entities` | `List[Entity]` | Entity extraction instructions. | +| `api_version` | `Optional[str]` | Provider API version. | +| `azure_endpoint` | `Optional[str]` | Optional direct Azure endpoint. | +| `azure_api_key` | `Optional[str]` | Optional direct Azure API key. | +| `gemini_api_key` | `Optional[str]` | Optional Gemini API key. | +| `gemini_project_id` | `Optional[str]` | Optional GCP project. | +| `gemini_location` | `Optional[str]` | Optional Vertex location. | +| `max_tokens` | `int` | Default `8024`. | +| `temperature` | `float` | Default `0.0`. | +| `model_type` | `Optional[str]` | Default `azure`; free string. | +| `model_id` | `Optional[str]` | Provider model id. | +| `processor_used` | `Optional[str]` | Preferred parser output subtree. | + +Validation is intentionally light. Provider and model compatibility is handled in service/router code. + +## 4. Evaluation schemas + +File: `backend/schemas/evaluations.py` + +### `EvaluationRequest` + +Request for one extraction evaluation. + +Core fields: + +| Field | Type | Default / constraint | +| --- | --- | --- | +| `entity_name` | `str` | required | +| `extraction_prompt` | `str` | required | +| `actual_output` | `str` | required | +| `expected_output` | `Optional[str]` | `None` | +| `metrics` | `Optional[List[str]]` | `['all']`; values not enum-constrained | +| `provider` | `str` | `azure_openai` | +| `threshold` | `float` | default `0.5`, constrained `0.0 <= x <= 1.0` | +| `strict_mode` | `bool` | `False` | +| `custom_evaluation_steps` | `Optional[Dict[str, List[str]]]` | `None` | + +Provider-specific fields: + +- Azure: `azure_deployment`, `azure_endpoint`, `azure_api_key`, `azure_model_name`. +- Vertex: `vertex_model_name`, `vertex_project`, `vertex_location`. +- Anthropic/other: `model_name`. + +### `SingleExtractionEval` + +Nested item for batch evaluation: + +- `entity_name` +- `extraction_prompt` +- `actual_output` +- `expected_output` + +### `BatchEvaluationRequest` + +Batch request with: + +- `extractions: List[SingleExtractionEval]` +- metrics/custom steps/provider/threshold/strict mode +- same Azure/Vertex/model provider fields as single evaluation + +### `CustomMetricRequest` + +Request for one custom metric evaluation: + +| Field | Type | Notes | +| --- | --- | --- | +| `metric_name` | `str` | Custom metric label. | +| `evaluation_steps` | `List[str]` | Required but no min length. | +| `entity_name` | `str` | Entity label. | +| `extraction_prompt` | `str` | Original extraction prompt. | +| `actual_output` | `str` | Output being evaluated. | +| `expected_output` | `Optional[str]` | Ground truth. | +| provider fields | mixed | Same pattern as other evaluation schemas. | + +### Evaluation response models + +`MetricResult`: + +- `metric_name` +- `score` +- `threshold` +- `success` +- `reason` + +`EvaluationResponse`: + +- evaluation metadata, test case, metric results, aggregate score, pass/fail, status, optional error. + +`BatchEvaluationResponse`: + +- batch id, timing, counts, average score, pass/fail, provider, and raw result dictionaries. + +## 5. Server config schema + +File: `backend/schemas/server.py` + +### `ServerConfig` + +Returned by `/api/server-config`. + +| Field | Type | Meaning | +| --- | --- | --- | +| `is_azure_openai_configured` | `bool` | Azure OpenAI credentials/models available. | +| `is_gemini_configured` | `bool` | Gemini/Vertex configured. | +| `is_azure_document_intelligence_configured` | `bool` | Azure Document Intelligence available. | +| `is_llama_configured` | `bool` | Llama MaaS configured. | +| `is_macbook_configured` | `bool` | Macbook base URL configured. | +| `is_macbook_healthy` | `bool` | Macbook endpoint reachable. | + +## 6. Session schemas + +File: `backend/schemas/sessions.py` + +These schemas are the main aggregate response contract for workflow restore and session history. + +### `SessionEntity` + +| Field | Type | +| --- | --- | +| `name` | `str` | +| `prompt` | `str` | +| `system_prompt` | `Optional[str]` | + +### `SessionConfiguration` + +| Field | Type | Default | +| --- | --- | --- | +| `study_type` | `Optional[str]` | `None` | +| `selected_models` | `List[str]` | empty list | +| `entities` | `List[SessionEntity]` | empty list | +| `summary_prompt` | `Optional[str]` | `None` | +| `paragraph_system_prompt` | `Optional[str]` | `None` | +| `temperature` | `float` | `0.0` | +| `model_temperatures` | `Optional[Dict[str, float]]` | empty dict | +| `files_config` | `Optional[Dict[str, Any]]` | empty dict | +| `evaluation_config` | `Optional[Dict[str, Any]]` | empty dict | + +### `SessionDocument` + +Document summary in a session response: + +- `id` +- `file_hash` +- `filename` +- `processor_used` +- `parse_cost` +- `page_count` +- `parse_duration_seconds` +- `figure_count` +- `table_count` + +### `ExtractionResult` + +API/session extraction result, distinct from ORM model with the same class name. + +| Field | Type | Notes | +| --- | --- | --- | +| `entity_name` | `str` | Entity label. | +| `model_id` | `str` | Provider/model id. | +| `document_id` | `Optional[str]` | Associated document id. | +| `extracted_text` | `Optional[str]` | Answer text. | +| `references` | `Optional[List[Dict[str, Any]]]` | Bounding-box/reference data. | +| `status` | `Literal['pending','completed','error']` | Strictly validated. | +| `error_message` | `Optional[str]` | Error text. | +| `extracted_at` | `Optional[datetime]` | Completion timestamp. | +| `file_hash` | `Optional[str]` | File hash for matching. | +| token/cost fields | optional ints/floats | Usage metrics. | + +### `SessionMetrics` + +- `total_cost: float = 0.0` +- `total_latency: float = 0.0` +- `total_calls: int = 0` + +### `EvaluationScore` + +One metric score: + +- `metric` +- `score` +- `reasoning` +- `judge_model` +- `human_score` +- `evaluation_cost` +- `evaluation_time` + +### `EvaluationResult` + +Session-level grouped evaluation result: + +- document/file/entity/model identity fields; +- `ground_truth`; +- `scores: List[EvaluationScore]`; +- optional aggregate human/cost/time fields. + +### `Session` + +Full session aggregate: + +| Field | Type | Notes | +| --- | --- | --- | +| `session_id` | `str` | Default UUID string. | +| `user_id` | `str` | Owner. | +| `name` | `str` | Default `Untitled Session`. | +| `status` | `Literal['in_progress','completed']` | Strictly validated. | +| `last_step` | `Optional[str]` | UI workflow step. | +| `evaluation_config` | `Optional[Dict[str, Any]]` | Session-level eval config. | +| `files_config` | `Optional[Dict[str, Any]]` | Per-file config. | +| `created_at` / `updated_at` | `datetime` | Timestamps. | +| `configuration` | `SessionConfiguration` | Main workflow configuration. | +| `documents` | `List[SessionDocument]` | Documents in session. | +| `extraction_results` | `List[ExtractionResult]` | Extraction outputs. | +| `evaluation_results` | `List[EvaluationResult]` | Grouped evaluation outputs. | +| `session_metrics` | `Optional[SessionMetrics]` | Optional aggregate metrics. | + +### `CreateSessionRequest` + +Fields: + +- `user_id` +- optional name/last_step/config/evaluation_config/files_config/documents + +### `UpdateSessionRequest` + +All fields optional. Supports updating: + +- user/name/status/last_step; +- configuration/evaluation_config/files_config; +- documents/extraction_results/evaluation_results. + +### `SessionSummary` and `SessionListResponse` + +`SessionSummary` is list-card style metadata: + +- session id, name, status, timestamps, last step; +- study type, document count/names, extraction/evaluation counts; +- shared session display fields. + +`SessionListResponse` wraps: + +- `sessions: List[SessionSummary]` +- `total: int` + +## 7. Router-local schemas + +Some routers define local Pydantic models rather than central schemas. + +### Files router + +- `FileUploadResponse` +- `UserFileInfo` + +These cover upload metadata and user-file list entries. + +### Groups router + +- `CreateGroupRequest` +- `UpdateGroupRequest` +- `AddMemberRequest` +- `UpdateMemberRoleRequest` +- `GroupResponse` +- `MemberResponse` +- `GroupDetailResponse` +- `UserSearchResult` + +These mirror group service dictionaries and profile-enriched membership data. + +### Templates router + +- `EntityModel` +- `VariableModel` +- `CreateTemplateRequest` +- `UpdateTemplateRequest` +- `SetImmutableRequest` +- `SetPermissionRequest` +- `ForkTemplateRequest` +- `ChangeScopeRequest` +- `CreateFolderRequest` +- `RenameFolderRequest` +- `FolderResponse` +- `TemplateResponse` +- `VersionResponse` +- `PermissionResponse` + +These define the template workspace API contract. + +### Evaluation jobs router + +- `EvalTaskRequest` +- `ProviderConfigRequest` +- `SubmitJobRequest` + +These are converted into `EvalTask`, `ProviderConfig`, and `EvalJob` dataclasses in the job queue. + +### Chat router + +- `ChatQueryRequest` + +Fields include query text, optional document markdown, and model configuration. + +### Paragraph routers + +- `ParagraphGenerationRequest` +- `ParagraphEvalGenerateRequest` + +These support generated scientific summary paragraphs and paragraph-specific evaluation records. + +## 8. Schema design notes + +- Several fields are free strings instead of enums (`model_type`, `provider`, `metrics`, status fields in ORM models). Validation happens later in services. +- Session API schemas use `Literal` for `Session.status` and API extraction status, making the API stricter than some database fields. +- Many nested JSON structures intentionally use `Dict[str, Any]` or `List[Dict[str, Any]]` because provider outputs and bbox references vary by parser/provider. +- Mutable defaults use `default_factory` in session schemas where needed. +- There are naming collisions between ORM classes and Pydantic classes, especially `ExtractionResult` and `EvaluationResult`. Always qualify by package in technical discussions. diff --git a/docs/backend/05-document-processing.md b/docs/backend/05-document-processing.md new file mode 100644 index 0000000..1975372 --- /dev/null +++ b/docs/backend/05-document-processing.md @@ -0,0 +1,403 @@ +# Document Processing Technical Design + +> *When a reviewer uploads a PDF, the backend needs to convert it into structured text, figures, and tables before an AI model can work with it. This document traces that entire journey β€” from the moment the file arrives, through SHA-256 deduplication and blob storage, into either the Azure Document Intelligence or Docling parser, and finally into the "document view" that the frontend uses to display results and restore sessions. It also covers bounding boxes: how the coordinates of text passages are preserved so the PDF viewer can highlight exactly where an extracted answer came from.* + +This document describes upload, storage, parsing, artifact access, bounding-box normalization, figure/table handling, and document-view construction. + +## 1. Scope + +In scope: + +- file upload and SHA-256 deduplication; +- Azure Blob Storage paths and local cache paths; +- processor selection between Azure Document Intelligence and Docling; +- processor output artifacts; +- document viewer/restore state; +- raw analysis normalization; +- figure/table serving and figure-summary generation. + +Out of scope: + +- frontend rendering of the document viewer; +- cloud provisioning for blob storage; +- internals of the remote Docling service outside this repository. + +## Visual workflow + +![Document upload and processing workflow](images/document-processing-workflow.png) + +The workflow is content-addressed. `POST /api/upload` validates the PDF, computes a SHA-256 hash, and writes only one global original artifact per hash. Processing then uses that hash as the stable document id. A strict cache hit requires `processed/{processor}/document.md`; metadata-only or partial artifact trees are readable for inspection but do not satisfy the processing cache gate. On a cache miss, the service hydrates the original file into `/tmp/summarization/{hash}`, writes parser outputs locally, syncs the full processor tree to Azure Blob Storage, updates DB metadata when a session document exists, and returns a canonical `document_view` object for restore/viewer workflows. + +## 2. Main classes and files + +| Class/function | File | Responsibility | +| --- | --- | --- | +| `DocumentService` | `backend/services/document/document_service.py` | High-level parser faΓ§ade and artifact readers. | +| `OrganizedFileService` | `backend/services/document/organized_file_service.py` | Hash-based upload, blob paths, processed artifact access, document-view builder. | +| `OrganizedDocumentProcessor` | `backend/services/document/organized_processor.py` | Coordinates processing into organized `/tmp` and blob paths. | +| `BlobStorageClient` | `backend/services/storage/blob_storage.py` | Async Azure Blob Storage wrapper. | +| `AzureDocIntelligenceService` | `backend/services/document/processors/azure_doc_intelligence/azure_doc_intelligence_service.py` | Azure parser implementation. | +| `DoclingService` | `backend/services/document/processors/docling/docling_service.py` | Local Docling parser implementation with process pool and VRAM guard. | +| `DoclingRemoteClient` | `backend/services/document/processors/docling/docling_remote_client.py` | Remote Docling service client. | +| `VRAMGuard` | `backend/services/document/processors/docling/vram_guard.py` | GPU memory-aware concurrency guard for local Docling. | +| `normalize_bbox_format()` | `backend/services/document/bbox_normalizer.py` | Normalizes Azure/Docling raw analysis to a common shape. | +| bbox matchers | `backend/services/document/processors/*/bounding_box_matcher.py` | Match extraction references to document bounding boxes. | + +## 3. Upload and storage model + +### 3.1 Upload entry point + +`POST /api/upload` calls `OrganizedFileService.save_uploaded_file()`. + +### 3.2 Hashing and deduplication + +`OrganizedFileService.compute_file_hash(content)` computes a SHA-256 hex digest. That hash is the stable identifier used across upload, processing, retrieval, and cache lookup. + +Upload blob layout: + +```text +global/{sha256}/original.{ext} +global/{sha256}/metadata.json +``` + +Upload metadata includes: + +- `file_hash` +- `original_filename` +- `file_size` +- `mime_type` +- `created_at` +- `extension` + +If the original blob already exists, upload returns `is_new=False` and `deduplicated=True` instead of rewriting the file. + +### 3.3 Optional DB registration + +When a user id is available, `save_uploaded_file()` attempts to register a `Document` row through the DB service. DB registration failures are warning-logged and do not fail blob upload. + +## 4. Processed artifact model + +### 4.1 Local scratch path + +Processors write to local scratch space first: + +```text +/tmp/summarization/{sha256}/processed/{processor}/ +``` + +This path is returned by `OrganizedFileService.get_processing_output_path(file_hash, processor)`. + +### 4.2 Blob processed path + +After processing completes, the local tree is synced to blob: + +```text +global/{sha256}/processed/{processor}/document.md +global/{sha256}/processed/{processor}/metadata.json +global/{sha256}/processed/{processor}/raw_analysis.json +global/{sha256}/processed/{processor}/figures/{filename} +global/{sha256}/processed/{processor}/tables/{filename} +``` + +`OrganizedFileService.sync_processing_output_to_blob()` uploads the directory recursively and preserves relative paths. + +### 4.3 Read/cache behavior + +`OrganizedFileService.get_processing_file_bytes(file_hash, processor, relative_path)` reads from local `/tmp` first. If missing, it downloads from blob, writes the bytes into the matching local cache path, and returns them. + +This supports cross-replica blob persistence with per-container local cache warming. + +## 5. Processor selection + +`DocumentService.convert_document_to_markdown()` accepts a `ProcessorType`: + +- `auto` +- `azure_doc_intelligence` +- `docling` + +Current auto-selection algorithm: + +```text +if Azure Document Intelligence is available: + choose azure_doc_intelligence +else: + choose docling +``` + +If Azure is explicitly requested but unavailable, the service falls back to Docling and annotates the result with: + +- `processor_used = "docling"` +- `processor_fallback = True` +- `fallback_reason = "Azure Document Intelligence not available"` + +## 6. Azure Document Intelligence processing + +Class: `AzureDocIntelligenceService` + +### 6.1 Availability + +Azure is available only when: + +- the Azure Document Intelligence SDK imports successfully; +- `AZURE_DOC_INTELLIGENCE_ENDPOINT` is configured; +- `AZURE_DOC_INTELLIGENCE_KEY` is configured. + +### 6.2 Conversion algorithm + +For each conversion: + +1. Create a conversion id and output directory. +2. Call Azure `begin_analyze_document()` with markdown output and optional figures. +3. Wait for `poller.result()`. +4. Save the complete result dictionary as `raw_analysis.json`. +5. Save markdown content as `document.md`. +6. Extract HTML table blocks from markdown into `tables/table-{n}.html`. +7. Download figure images when result id and figure ids are available. +8. Save summary `metadata.json`. + +### 6.3 Azure output files + +```text +document.md +raw_analysis.json +metadata.json +conversion.log +figures/{figure_id}.png +tables/table-{idx}.html +``` + +Metadata includes: + +- conversion id; +- source/source type; +- processor/model id; +- status; +- conversion/log/raw/markdown paths; +- start/end time; +- conversion time and parse duration; +- content length; +- page count; +- table count; +- key-value pair count; +- figure count and figure metadata. + +Figure metadata includes id, page, caption, spans, bounding regions, and optional image path. + +## 7. Docling processing + +There are two Docling-related implementation paths: + +- `DoclingRemoteClient` for the remote Docling service used by `DocumentService` and `OrganizedDocumentProcessor`. +- `DoclingService` for local processing with multiprocessing and VRAM-aware concurrency. + +### 7.1 Local Docling worker algorithm + +`_docling_worker_process()` performs the conversion in a subprocess: + +1. Initialize or reuse a process-local `DocumentConverter`. +2. Convert the source PDF. +3. Extract picture items and save PNGs into `figures/`. +4. Export markdown to `document.md`. +5. Export table HTML to `tables/table-{idx}.html`. +6. Build a Docling-style `raw_analysis.json` containing pages, paragraphs, tables, figures, and document structure. +7. Optionally write a debug `docling_document.json`. +8. Return success metadata including parse duration, markdown, page count, image info, and peak VRAM. + +### 7.2 Docling output files + +```text +document.md +raw_analysis.json +metadata.json +conversion.log +figures/picture-{n}.png +tables/table-{idx}.html +docling_document.json # debug output when available +``` + +### 7.3 VRAM guard + +`VRAMGuard` protects local Docling conversion concurrency. + +Key behavior: + +- Computes max workers from total GPU VRAM, safety margin, and observed per-worker memory. +- Uses cold-start limits before enough jobs have completed. +- Provides async `acquire_slot()` context manager. +- Tracks active and queued workers. +- Updates per-worker estimate from recent peak VRAM observations. +- Persists state in `.vram_guard_state.json` when possible. +- On OOM, increases estimated per-worker memory and shrinks allowed concurrency. + +## 8. Organized processing flow + +`OrganizedDocumentProcessor.process_document()` provides a hash-oriented processing path: + +```text +input file path + -> read bytes and compute hash + -> OrganizedFileService.save_uploaded_file() + -> if already processed and not force_reprocess: + return cached document.md + metadata + else: + output_path = /tmp/summarization/{hash}/processed/{processor} + original_path = get_original_file_path(hash) + process with Azure or Docling + sync output_path to blob + return result +``` + +Cache hit requires `OrganizedFileService.is_file_processed()`, which strictly checks for `document.md`. + +## 9. Artifact resolution + +`OrganizedFileService.resolve_processed_processor(file_hash, preferred_processor)` chooses the processor subtree to read. + +Candidate order: + +1. preferred processor if given; +2. `azure_doc_intelligence`; +3. `docling`. + +A processor counts as resolved if any of the following exists: + +- `metadata.json` +- `document.md` +- `raw_analysis.json` +- any blob under the processed subtree prefix + +This is intentionally more permissive than `is_file_processed()` so partially available artifact trees can still be inspected. + +## 10. Document view contract + +`OrganizedFileService.build_document_view()` assembles canonical frontend state. + +Top-level fields include: + +- `fileName` +- `fileId` +- `status` +- `selectedParser` +- `processorUsed` + +Nested `processingResult` fields include both camelCase and snake_case names for compatibility: + +- `conversionId` +- `fileHash` +- `processorUsed` +- `markdownPath` +- `parseCost` / `parse_cost` +- `parseDuration` / `parse_duration_seconds` +- `pageCount` / `page_count` +- `figures` +- `figuresCount` +- `tablesCount` +- `artifactAvailability` + +Artifact availability flags: + +- `original` +- `markdown` +- `analysis` +- `figures` +- `tables` + +If figure/table counts are missing from metadata, the service enumerates blob prefixes under `figures/` and `tables/`. + +## 11. Raw analysis normalization + +`normalize_bbox_format(analysis_result)` detects processor type and returns a normalized shape. + +### 11.1 Processor detection + +Azure indicators: + +- `apiVersion` +- `modelId` +- camelCase `boundingRegions` +- page `unit` + +Docling indicators: + +- `api_version` +- `document_structure` +- snake_case `bounding_regions` + +### 11.2 Normalized output shape + +Common top-level fields: + +- `processor` +- `api_version` +- `model_id` +- `pages` +- `paragraphs` +- `tables` +- `figures` + +Azure page coordinates in inches are converted to points. Docling output is already closer to the normalized snake_case shape and is mostly copied with field normalization. + +## 12. Bounding-box and reference matching + +Azure matcher: + +- normalizes text; +- searches paragraphs first; +- falls back to page lines; +- extracts figure references with regexes such as Figure/Fig/FIG; +- maps figure references to figure metadata and bounding regions; +- returns best matches plus paragraph/line match candidates. + +Docling matcher: + +- searches paragraphs; +- falls back to page-level match; +- can extract polygon coordinates from 8-value arrays; +- returns simpler match structures than Azure. + +These matchers are used by extraction routes to attach visual grounding to entity answers. + +## 13. Figure and table handling + +### 13.1 Figure image serving + +`GET /api/documents/{document_id}/figures/{figure_filename}` reads figure bytes from processed artifacts and returns an image response. + +The router validates filenames before reading artifacts to avoid arbitrary path access. + +### 13.2 Figure summary generation + +`POST /api/documents/{document_id}/figures/{figure_id}/generate-summary`: + +1. resolves figure metadata and image path; +2. writes image bytes to a temporary file if required by provider client; +3. calls `LLMService.extract_content_from_image()`; +4. stores generated content back in metadata; +5. returns the updated figure summary. + +### 13.3 Enhanced markdown + +`GET /api/documents/{document_id}/enhanced-content` inserts available figure summaries into markdown near their figure references. + +### 13.4 Table serving + +`GET /api/documents/{document_id}/tables/{table_filename}` returns saved table HTML from the processed artifact tree. + +## 14. Failure modes + +| Failure | Current behavior | +| --- | --- | +| Missing blob connection string | `OrganizedFileService` construction fails. | +| Duplicate upload | Returns existing hash and dedupe flags. | +| Missing processed document | Content endpoints return not-found style errors. | +| Azure unavailable | Auto mode uses Docling; explicit Azure falls back with flags. | +| Figure image missing | Figure endpoint returns not found; summary generation fails for that figure. | +| Malformed raw analysis | Some readers assume valid JSON; callers may receive errors. | +| Docling OOM | Worker returns structured error, VRAM guard reports OOM and shrinks concurrency estimate. | + +## 15. Related docs + +- [02-api-surface.md](02-api-surface.md) +- [03-data-models.md](03-data-models.md) +- [07-extraction-flow.md](07-extraction-flow.md) +- [appendices/data-flow-diagrams.md](appendices/data-flow-diagrams.md) diff --git a/docs/backend/06-llm-layer.md b/docs/backend/06-llm-layer.md new file mode 100644 index 0000000..f68b0d1 --- /dev/null +++ b/docs/backend/06-llm-layer.md @@ -0,0 +1,357 @@ +# LLM Provider Layer Technical Design + +> *Science-GPT is not locked to a single AI provider. This document describes how the backend routes extraction and evaluation requests across seven different providers β€” Azure OpenAI, Gemini, Anthropic, Llama, Macbook (local), and vLLM β€” through a common interface. It covers how each provider client works, how timeouts and retries are handled, how structured outputs are extracted from different response formats, and how every call's cost and token usage is recorded.* + +This document describes the provider-routing layer implemented under `backend/services/llm/`. It covers classes, request/response contracts, timeout/retry behavior, concurrency controls, and cost-tracking integration. + +## 1. Scope + +In scope: + +- `LLMService` provider dispatch; +- provider client responsibilities; +- extraction, image extraction, and paragraph generation methods; +- return dictionary conventions; +- timeout/retry behavior; +- Macbook queue serialization; +- cost/session metric recording. + +Out of scope: + +- frontend model picker UI; +- provider account provisioning; +- exact model availability at runtime. + +## Visual workflow + +![LLM provider routing workflow](images/llm-provider-routing.png) + +All text and vision model use goes through `LLMService`. The routers provide the operation-specific context, while `LLMService` selects the provider client from `model_type`, applies timeout logging, normalizes the provider dictionary, and records session metrics on successful responses. Provider clients own SDK or REST details, including retries, model-name translation, structured output support, local Macbook serialization, and OpenAI-compatible vLLM calls. Downstream code should rely on the common result keys rather than provider-specific raw payloads whenever possible. + +## 2. Main classes + +| Class | File | Responsibility | +| --- | --- | --- | +| `LLMService` | `backend/services/llm/llm_service.py` | High-level router across provider clients. | +| `AzureLLMClient` | `backend/services/llm/azure.py` | Azure OpenAI / Azure AI Foundry REST+SDK calls. | +| `GeminiLLMClient` | `backend/services/llm/gemini.py` | Vertex AI Gemini text and vision calls. | +| `AnthropicLLMClient` | `backend/services/llm/anthropic.py` | Anthropic-on-Vertex calls. | +| `LlamaLLMClient` | `backend/services/llm/llama.py` | Vertex AI MaaS Llama calls. | +| `MacbookLLMClient` | `backend/services/llm/macbook.py` | Ollama-compatible Macbook-hosted inference. | +| `MacbookRequestQueue` | `backend/services/llm/macbook_queue.py` | FIFO single-worker queue for Macbook inference. | +| `VLLMClient` | `backend/services/llm/vllm.py` | OpenAI-compatible vLLM endpoint client. | + +## 3. `LLMService` + +`LLMService` owns provider client instances and exposes three main operations: + +- `extract_entities_from_markdown(...)` +- `extract_content_from_image(...)` +- `generate_paragraph(...)` + +It also records usage/cost metrics through `_record_session_metrics()` after successful provider calls. + +### 3.1 Provider dispatch for entity extraction + +`extract_entities_from_markdown()` dispatches by `model_type`: + +| `model_type` | Client method | +| --- | --- | +| `azure` | `AzureLLMClient.extract_entities_with_azure()` | +| `gemini` | `GeminiLLMClient.extract_entities_with_gemini()` | +| `anthropic` | `AnthropicLLMClient.extract_entities_with_anthropic()` | +| `llama` | `LlamaLLMClient.extract_entities_with_llama()` | +| `azure-llama` | currently routed through Azure client path | +| `macbook` | `MacbookLLMClient.extract_entities_with_macbook()` | +| `vllm` | `VLLMClient.extract_entities_with_vllm()` | + +Provider-disabled clients return structured `success=False` responses rather than crashing the whole app. + +### 3.2 Timeout budgets + +Default wrapper timeout is 240 seconds, with provider-specific overrides: + +| Provider path | Timeout | +| --- | ---: | +| Azure/Gemini/Anthropic | 240s default wrapper | +| Llama | 300s | +| Macbook | 1900s | +| vLLM | 600s | + +Timeouts are logged to `backend/output/timeout_logs/timeout_log.txt`. + +### 3.3 Session metrics + +On successful responses, `_record_session_metrics(session_id, provider, result)` extracts: + +- provider; +- model/deployment; +- prompt tokens; +- completion tokens; +- duration. + +Then it calls `cost_tracker.record_call(...)`. + +## 4. Common response conventions + +Provider clients return dictionaries rather than a shared class. Common keys: + +| Key | Meaning | +| --- | --- | +| `success` | Boolean success flag. | +| `content` | Main text result. | +| `answer` | Structured answer text, when available. | +| `references` | Structured references, when available. | +| `raw` | Raw provider response or parsed JSON. | +| `meta` | Provider/model/timing/token metadata. | +| `error` | Error text when `success=False`. | + +Provider-specific clients may also return compatibility fields such as `extracted_text`, `generated_text`, `usage`, `model`, or strategy metadata. + +Downstream code should check `success` and then normalize provider-specific fields. + +## 5. Azure provider + +Class: `AzureLLMClient` + +### 5.1 Configuration + +Supports: + +- global Azure OpenAI env vars; +- `AZURE_OPENAI_MODELS` JSON list for per-deployment endpoints/keys/API versions; +- Azure AI Foundry serverless endpoints detected by `.services.ai.azure.com`. + +### 5.2 Entity extraction + +Primary path uses OpenAI SDK structured outputs: + +- Pydantic `MarkdownReference` +- Pydantic `ExtractionResult` +- `client.beta.chat.completions.parse(...)` +- JSON response format with answer and references + +Fallback path uses raw REST `requests.post()` if structured parsing fails. + +Retry behavior: + +- up to 3 attempts; +- retry on 429, 500, 503, 504 or matching exception text; +- exponential backoff with jitter; +- temperature-related 400 errors can retry without `temperature`. + +### 5.3 Paragraph generation + +Uses REST chat completions. Payload includes messages, max completion tokens, `n=1`, optional temperature, and Foundry-specific model field when needed. + +### 5.4 Vision extraction + +Reads image as base64 and sends a multimodal chat payload with text and image data URL through REST using `aiohttp`. + +## 6. Gemini provider + +Class: `GeminiLLMClient` + +### 6.1 Configuration + +Requires: + +- GCP project id; +- Vertex location; +- service account credentials. + +Supports env aliases such as `GEMINI_PROJECT_ID`, `GEMINI_PROJECT`, `VERTEX_AI_PROJECT`, `GEMINI_LOCATION`, and `VERTEX_AI_LOCATION`. + +### 6.2 Core call behavior + +Builds Vertex AI publisher-model endpoint: + +- global endpoint for global-only models; +- regional endpoint otherwise. + +Payload supports: + +- `contents`; +- `generationConfig.temperature`; +- `generationConfig.maxOutputTokens`; +- optional `responseMimeType: application/json` and `responseJsonSchema`; +- optional `systemInstruction`. + +Retry behavior: + +- three attempts; +- retries 429/500/503/504; +- retries empty responses, JSON parse failures, and content extraction errors; +- returns partial content if finish reason is `MAX_TOKENS`. + +### 6.3 Supported extraction/generation models + +Allowed short names include: + +- `gemini-2.5-pro` +- `gemini-2.5-flash-lite` +- `gemini-2.5-flash` +- `gemini-3-pro-preview` + +Structured extraction uses the Gemini version of `ExtractionResult.model_json_schema()`. + +## 7. Anthropic provider + +Class: `AnthropicLLMClient` + +### 7.1 Configuration + +Uses `anthropic.AnthropicVertex`, requiring Google service account credentials. It finds credentials from `GOOGLE_APPLICATION_CREDENTIALS` or JSON files under `backend/core/`. + +### 7.2 Core call behavior + +Calls `client.messages.create(...)` with: + +- `model` +- `max_tokens` +- `messages` +- optional `system` +- optional `temperature` + +For structured extraction, the code uses prompt-enforced JSON because Vertex-hosted Anthropic structured output beta is not used here. + +### 7.3 Entity extraction + +Default model: + +- `claude-sonnet-4-5@20250929` + +Structured-output prompt requests: + +- `answer: string` +- `references: [{ text: string }]` + +## 8. Llama provider + +Class: `LlamaLLMClient` + +### 8.1 Configuration + +Uses Vertex AI MaaS OpenAI-compatible endpoint. Requires project, location/region, and service account credentials. + +### 8.2 Region routing + +- Llama 4 models route to `us-east5`. +- Llama 3.x models route to `us-central1`. + +### 8.3 Extraction algorithm + +Primary strategy: + +1. Optimize/truncate long prompts and markdown. +2. Request JSON object output. +3. Validate response with Pydantic `ExtractionResult`. +4. Return answer/references and `strategy='primary_optimized'`. + +Fallback strategy: + +1. Minimal system prompt. +2. Shortened context. +3. Low token budget. +4. Return `strategy='fallback_minimal'`. + +Parsing error handling writes diagnostic JSON logs under `backend/logs/llama_errors/` and attempts to salvage embedded JSON fragments. + +## 9. Macbook provider + +Class: `MacbookLLMClient` + +### 9.1 Configuration + +Uses: + +- `MACBOOK_LLM_BASE_URL` +- optional retry/backoff/timeout env vars; +- `backend/config/macbook_model_policy.json` for allow/deny model filtering. + +The endpoint is Ollama-style: + +- `GET /api/tags` +- `POST /api/generate` + +### 9.2 Queueing + +Macbook calls are serialized through `MacbookRequestQueue`. + +Queue behavior: + +- one background worker; +- FIFO `asyncio.Queue`; +- caller awaits a future; +- failed worker is restarted on next enqueue; +- stats expose total enqueued, processed, pending, and worker state. + +This prevents concurrent requests from overwhelming a local Macbook-hosted model runtime. + +### 9.3 Response cleanup + +`_sanitize_content()` strips reasoning/thinking tags and preserves user-facing content. + +## 10. vLLM provider + +Class: `VLLMClient` + +### 10.1 Configuration + +Uses: + +- `VLLM_BASE_URL` +- optional `VLLM_API_KEY` +- optional `VLLM_MODELS` + +If static models are configured, `fetch_available_models()` uses them. Otherwise it queries `GET /models`. + +### 10.2 Core call + +Calls OpenAI-compatible: + +```text +POST {base_url}/chat/completions +``` + +Payload: + +- `model` +- `messages` +- `max_tokens` +- `temperature` + +Default timeout is 600 seconds. + +If model id starts with `vllm-`, the prefix is stripped before sending to the backend. + +## 11. Cost tracking integration + +`LLMService` records successful calls only. + +Cost tracker uses: + +- provider; +- normalized model key; +- prompt/completion token counts; +- duration; +- optional document/page metadata for document parser costs. + +See [11-auth-security-observability.md](11-auth-security-observability.md) for telemetry details. + +## 12. Error behavior + +| Provider | Error style | +| --- | --- | +| Azure | Returns `success=False` with error/raw details; retries transient failures. | +| Gemini | Returns structured error or fallback partial content on token truncation. | +| Anthropic | Returns error dict for API/JSON parsing failures. | +| Llama | Uses primary/fallback strategies and parsing diagnostics. | +| Macbook | Retries 5xx/HTML/bad-gateway/request exceptions until attempt/total cap. | +| vLLM | Returns error for timeout, non-200, or missing choices. | + +## 13. Related docs + +- [07-extraction-flow.md](07-extraction-flow.md) +- [08-evaluation-flow.md](08-evaluation-flow.md) +- [11-auth-security-observability.md](11-auth-security-observability.md) diff --git a/docs/backend/07-extraction-flow.md b/docs/backend/07-extraction-flow.md new file mode 100644 index 0000000..77164c0 --- /dev/null +++ b/docs/backend/07-extraction-flow.md @@ -0,0 +1,296 @@ +# Entity Extraction Flow Technical Design + +> *Entity extraction is the core operation of Science-GPT: given a document and a list of fields to extract, ask an LLM to find each value and point to where it found it in the document. This document explains how that works end-to-end β€” how the markdown is loaded, how up to 48 LLM calls run concurrently through a semaphore, how the answers are matched back to bounding boxes on the PDF pages, and how results are persisted to the reviewer's session so nothing is lost if the browser closes.* + +This document describes how the backend extracts structured entity answers from processed documents, attaches reference/bounding-box data, records cost, and persists results into sessions. + +## 1. Scope + +In scope: + +- `POST /api/extract` interface behavior; +- entity request types; +- markdown and figure-context construction; +- LLM provider routing; +- provider response normalization; +- reference and bounding-box matching; +- persistence into `extraction_results`; +- timeout logging and cost/session metrics. + +Out of scope: + +- frontend prompt-template editing; +- model-provider provisioning; +- evaluation of extracted answers, which is covered in [08-evaluation-flow.md](08-evaluation-flow.md). + +## Visual workflow + +![Entity extraction and grounding workflow](images/extraction-grounding-workflow.png) + +The extraction path has two distinct technical phases. First, the router builds an enhanced markdown context from processed document content and available figure summaries, then fans out one provider call per entity. Cloud models run concurrently behind a semaphore; Macbook-backed models are submitted sequentially because the local runtime is already serialized by a FIFO queue. Second, structured references from the provider are matched back to parser raw analysis. Azure matching can use paragraph, line, and figure metadata; Docling matching uses paragraph/page structures and polygons. Persistence upserts by `(document_id, entity_name, model_id)` so reruns replace the same entity/model result instead of duplicating it. + +## 2. Main files and classes + +| Component | File | Responsibility | +| --- | --- | --- | +| extraction router | `backend/api/extractions/router.py` | HTTP endpoint, per-entity task orchestration, persistence. | +| `ExtractRequest` / `Entity` | `backend/schemas/extractions.py` | API request schemas. | +| `DocumentService` | `backend/services/document/document_service.py` | Markdown/raw-analysis/figure retrieval. | +| `LLMService` | `backend/services/llm/llm_service.py` | Provider dispatch. | +| bbox matchers | `backend/services/document/processors/*/bounding_box_matcher.py` | Reference-to-bbox matching. | +| `SessionService` | `backend/services/session/session_service.py` | Persist extraction results. | +| `ExtractionResult` ORM | `backend/models/extraction.py` | Physical extraction row. | +| `ExtractionResult` schema | `backend/schemas/sessions.py` | Session API extraction result. | + +## 3. API contract + +Endpoint: + +```text +POST /api/extract +``` + +Request body: `ExtractRequest` + +Important fields: + +- `conversion_id`: file hash / document conversion id; +- `session_id`: optional session id for persistence; +- `entities`: list of `Entity` objects; +- `model_type`: provider route, e.g. `azure`, `gemini`, `anthropic`, `llama`, `macbook`, `vllm`; +- `model_id`, `deployment`, `api_version`, provider-specific credentials/config; +- `max_tokens`, `temperature`; +- `processor_used`: optional preferred processor subtree. + +Each `Entity` contains: + +- `name` +- `prompt` +- optional `extracted` +- optional `system_prompt` + +## 4. High-level sequence + +```text +POST /api/extract + -> validate auth dependency + -> load markdown for conversion_id + -> load figure metadata/context when available + -> for each entity: + call LLMService.extract_entities_from_markdown() + normalize provider result + load raw analysis when needed + match references to bounding boxes + build entity result dict + -> persist results to session when session_id provided + -> return extraction result payload +``` + +## 5. Markdown loading + +The router uses `DocumentService.get_markdown_content(conversion_id, processor_used)`. + +Resolution behavior: + +1. Resolve processor via `OrganizedFileService.resolve_processed_processor()`. +2. Try `document.md` in the processed artifact tree. +3. If missing, optionally fall back to `raw_analysis.json` content field. + +If markdown cannot be loaded, extraction fails with a not-found/error response. + +## 6. Figure context construction + +The extraction router can build extra figure context with `_build_figures_context(figures, conversion_id)`. + +Inputs: + +- figure metadata from `DocumentService.get_figures_for_conversion()`; +- figure summaries stored in metadata, when previously generated. + +Purpose: + +- Give text-only extraction models more context about figures and visual content. +- Preserve figure ids/captions/summaries in the prompt context. + +The figure context is appended to or included with document markdown before provider calls, depending on route logic. + +## 7. Per-entity extraction task + +The router creates one async task per requested entity. + +For each entity: + +1. Combine document markdown, figure context, and the entity prompt. +2. Use the entity-level `system_prompt` if present. +3. Call `LLMService.extract_entities_from_markdown()` with provider configuration. +4. Interpret result success/failure. +5. Extract answer text from provider-specific fields. +6. Extract token usage, duration, and raw references from `meta` / provider response. +7. Match references to bounding boxes if possible. +8. Return an entity result object. + +## 8. Provider response handling + +Provider result dictionaries vary. Common extraction answer candidates include: + +- `answer` +- `content` +- `extracted_text` +- provider-specific generated fields + +Common reference candidates include: + +- `references` +- structured JSON references inside `raw` + +Common metadata fields include: + +- `meta.model` +- `meta.deployment` +- `meta.prompt_tokens` +- `meta.completion_tokens` +- `meta.duration` + +The extraction flow treats provider dictionaries as semi-structured and maps them into the stable session extraction schema. + +## 9. Bounding-box matching + +### 9.1 Raw analysis loading + +When references exist or bbox matching is requested, the router reads raw analysis through: + +```python +DocumentService.get_raw_analysis_result(conversion_id, processor_used) +``` + +Then the raw analysis is normalized with `normalize_bbox_format()` or routed to processor-specific matchers. + +### 9.2 Azure matching + +Azure matcher behavior: + +1. Normalize text. +2. Search paragraphs for exact/substring/similarity matches. +3. Fall back to page lines when paragraph match is weak or absent. +4. Extract figure references such as `Figure 1`, `Fig. 2`, etc. +5. Match figure ids to Azure figure metadata. +6. Return paragraph matches, line matches, best match, and figure-enriched matches. + +Returned data can include: + +- matched text; +- similarity score; +- page number; +- polygon/bounding regions; +- paragraph content; +- line content; +- figure id/caption/reference. + +### 9.3 Docling matching + +Docling matcher behavior: + +1. Normalize text. +2. Search paragraphs. +3. Fall back to page-level matching. +4. Extract and validate 8-number polygons where available. + +Docling matching is simpler than Azure matching because Docling raw analysis does not always include line-level structures equivalent to Azure. + +## 10. Persistence + +If `session_id` is provided, each successful or failed entity result can be persisted through: + +```python +SessionService.add_extraction_result_fast(session_id, user_id, result) +``` + +`SessionService` resolves the target document by: + +1. explicit `document_id`, if present; +2. `file_hash`, if present; +3. cached session documents; +4. DB lookup. + +Important multi-document guard: + +- If a session has multiple documents and an extraction result lacks `file_hash`/document identity, the service refuses to guess and returns `False`. + +DB upsert target: + +```text +(document_id, entity_name, model_id) +``` + +Constraint name: + +```text +uq_extraction_doc_entity_model +``` + +## 11. Extraction DB shape + +Persisted `ExtractionResult` ORM fields: + +- `session_id` +- `document_id` +- `entity_name` +- `model_id` +- `extracted_text` +- `bbox_references` +- `status` +- `error_message` +- `extracted_at` +- `prompt_tokens` +- `completion_tokens` +- `duration_ms` +- `cost` + +The session API later converts this into `schemas.sessions.ExtractionResult` with `references` instead of `bbox_references`. + +## 12. Cost and timing + +Provider clients include token and duration metadata when available. `LLMService` records session metrics on successful provider responses. The extraction persistence layer also stores per-extraction token, duration, and cost fields when provided. + +If stored cost is missing or zero, `SessionService._db_to_session()` can recompute estimated cost from token counts and model id using `cost_tracker` and backfill the DB. + +## 13. Timeout and error logging + +`LLMService` wraps provider calls with timeout logging. The extraction router also has a `log_timeout_event()` helper to write extraction timeout details. + +Common error outputs: + +- provider disabled; +- provider API failure; +- timeout; +- markdown/document not found; +- bbox analysis unavailable; +- DB persistence failure. + +Provider/API failures are generally returned as failed entity results rather than aborting the entire batch when possible. + +## 14. Algorithms to preserve + +### 14.1 Per-entity concurrency + +Entities are extracted concurrently so one request can produce multiple entity results faster. Persistence also uses async task-style behavior for entity result writes. + +### 14.2 Figure-context augmentation + +Figure metadata and summaries are converted into text context. This gives text LLMs access to visual analysis without requiring every extraction call to invoke a vision model. + +### 14.3 Reference grounding + +Structured provider references are mapped to parser raw analysis. The design decouples extraction answer generation from visual grounding: providers produce textual references, then backend matchers map those references to bounding boxes. + +### 14.4 Multi-document safety + +`SessionService` avoids cross-document contamination by requiring file/document identity when a session contains multiple documents. + +## 15. Related docs + +- [02-api-surface.md](02-api-surface.md) +- [04-schemas.md](04-schemas.md) +- [05-document-processing.md](05-document-processing.md) +- [06-llm-layer.md](06-llm-layer.md) +- [09-session-sharing-groups.md](09-session-sharing-groups.md) diff --git a/docs/backend/08-evaluation-flow.md b/docs/backend/08-evaluation-flow.md new file mode 100644 index 0000000..bfc54b4 --- /dev/null +++ b/docs/backend/08-evaluation-flow.md @@ -0,0 +1,425 @@ +# Evaluation Flow Technical Design + +> *After entities are extracted, reviewers can ask a second LLM to judge how good the answers are β€” scoring them for correctness, completeness, relevance, and safety. This document explains how that works: the G-Eval scoring algorithm, how the backend tries to parse scores from the judge's response (with several fallback strategies for malformed output), how long-running evaluation jobs are managed in the background so the frontend can poll for progress, and how individual evaluations are cancelled without losing work already done.* + +This document describes LLM-as-a-judge evaluation, DeepEval metric construction, batch evaluation, background job execution, cancellation, persistence, and cost tracking. + +## 1. Scope + +In scope: + +- single extraction evaluation; +- batch evaluation; +- custom metric evaluation; +- built-in metric factories; +- provider adapters for Azure, Vertex/Gemini, and Anthropic Vertex; +- background evaluation jobs; +- cancellation; +- result storage; +- session evaluation persistence; +- judge-call cost tracking. + +Out of scope: + +- frontend evaluation table UI; +- external DeepEval library implementation; +- provider account setup. + +## Visual workflow + +![Evaluation workflow](images/evaluation-workflow.png) + +Evaluation has a synchronous path and a background-job path, but both converge on `EvaluationService.evaluate_extraction()`. The service creates a judge adapter, selects built-in or custom metrics, prefers a combined JSON scoring prompt to reduce judge-call cost, and falls back to per-metric GEval scoring if parsing fails. Background jobs flatten `tasks x providers`, then use both global and per-job semaphores so multiple users can make progress without one job taking every judge slot. Job status is kept in memory for active work and synced to `eval_jobs` so polling can recover across workers. + +## 2. Main classes and files + +| Component | File | Responsibility | +| --- | --- | --- | +| `EvaluationService` | `backend/services/evaluation/evaluation_service.py` | Evaluation orchestration, metric creation, combined scoring, batch handling. | +| `EvaluationResultStorage` | `backend/services/evaluation/storage/result_storage.py` | JSON file storage for evaluation outputs. | +| metric factories | `backend/services/evaluation/metrics/*.py` | Create built-in/custom GEval metrics. | +| adapters | `backend/services/evaluation/adapters/*.py` | Wrap provider models for DeepEval. | +| job queue | `backend/services/evaluation/job_queue.py` | Async background evaluation job management. | +| API router | `backend/api/evaluations/router.py` | Synchronous evaluation endpoints. | +| jobs router | `backend/api/evaluations/jobs.py` | Background job submit/poll/cancel endpoints. | +| `EvalJobRecord` | `backend/models/eval_job.py` | Cross-worker job status persistence. | + +## 3. API endpoints + +Synchronous evaluation endpoints: + +```text +POST /api/evaluations/evaluate +POST /api/evaluations/evaluate/batch +POST /api/evaluations/evaluate/custom +POST /api/evaluations/cancel +GET /api/evaluations/results/{evaluation_id} +GET /api/evaluations/results +GET /api/evaluations/metrics/info +``` + +Background job endpoints: + +```text +POST /api/evaluations/jobs +GET /api/evaluations/jobs/{job_id} +POST /api/evaluations/jobs/{job_id}/cancel +``` + +## 4. Evaluation model creation + +`EvaluationService.create_evaluation_model()` supports: + +| Provider id | Adapter | +| --- | --- | +| `azure_openai` | `AzureOpenAIDeepEvalModel` | +| `vertex_ai` | `VertexAIDeepEvalModel` | +| `anthropic` | `AnthropicVertexDeepEvalModel` | + +Default models: + +- Vertex/Gemini: `EVAL_DEFAULT_GEMINI_MODEL` or `gemini-2.5-flash`. +- Anthropic: `EVAL_DEFAULT_ANTHROPIC_MODEL` or `claude-sonnet-4-5@20250929`. + +Unsupported providers raise `ValueError`. + +## 5. Built-in metric factories + +Metric factory map: + +| Metric key | Factory | Evaluation focus | +| --- | --- | --- | +| `correctness` | `CorrectnessMetricFactory` | Factual accuracy against ground truth. | +| `completeness` | `CompletenessMetricFactory` | Coverage of expected key information. | +| `relevance` | `RelevanceMetricFactory` | Focus on the requested entity/task. | +| `safety` | `SafetyMetricFactory` | PII, toxicity, bias, unsupported/harmful claims. | + +All built-in metrics are GEval metrics with async mode enabled. + +Correctness and completeness require expected output. If no expected output is provided, the service skips those metrics. + +## 6. Custom metrics + +`CustomMetricFactory.create()` accepts: + +- metric name; +- evaluation steps; +- evaluation model; +- evaluation params; +- threshold; +- strict mode. + +Default evaluation params are input, actual output, and expected output. + +## 7. Combined scoring algorithm + +`EvaluationService._evaluate_combined()` is the preferred scoring path for multiple metrics. + +Purpose: + +- score all metrics in one judge-model call; +- reduce repeated prompt context; +- lower latency/cost compared with one call per metric. + +Algorithm: + +1. Build a prompt containing extraction task, actual output, optional expected output, and one criteria block per metric. +2. Ask the judge model for a strict JSON object with metric entries. +3. Parse output using multiple recovery strategies: + - direct `json.loads()`; + - strip Markdown code fences; + - extract outermost JSON block with regex; + - salvage per-metric entries with regex. +4. Clamp scores to `[0, 1]`. +5. Set metric score/reason/success fields. +6. Return list of metric result dictionaries. + +If combined scoring fails, `evaluate_extraction()` falls back to per-metric evaluation with `asyncio.gather()`. + +## 8. Single extraction evaluation + +`EvaluationService.evaluate_extraction()` inputs: + +- `entity_name` +- `extraction_prompt` +- `actual_output` +- optional `expected_output` +- optional `metrics` +- `provider` +- `threshold` +- `strict_mode` +- optional `custom_evaluation_steps` +- optional `session_id` +- provider-specific model kwargs + +Flow: + +```text +create evaluation id + -> create evaluation model + -> decide metric set + -> skip metrics requiring missing expected output + -> build DeepEval LLMTestCase + -> try combined scoring + -> fallback to per-metric concurrent scoring if needed + -> collect adapter call history + -> estimate and record cost + -> compute aggregate score and all_passed + -> return result dict +``` + +Success result includes: + +- `evaluation_id` +- `entity_name` +- `provider` +- `model` +- `timestamp` +- `evaluation_time` +- `evaluation_cost` +- `metrics` +- `aggregate_score` +- `all_passed` +- `threshold` +- `strict_mode` +- `status='success'` + +Error result includes: + +- evaluation id; +- entity/provider/timestamp; +- `status='error'`; +- error text. + +## 9. Batch evaluation + +`EvaluationService.evaluate_multiple_extractions()` accepts a list of extraction dictionaries and runs them in chunks. + +Behavior: + +1. Clear stale cancellation flag for the session. +2. Default metrics to all four built-ins if omitted. +3. Process extraction mini-batches with `asyncio.gather()`. +4. Check session cancellation between batches. +5. Fill remaining results with `status='cancelled'` if cancelled. +6. Compute batch summary statistics. +7. Save batch result to `EvaluationResultStorage`. + +Batch result includes: + +- `batch_id` +- `timestamp` +- `batch_time` +- `total_evaluations` +- `successful_evaluations` +- `failed_evaluations` +- `avg_aggregate_score` +- `all_passed` +- `threshold` +- `provider` +- `results` + +## 10. Cancellation + +`EvaluationService` uses a process-local `CANCELLED_SESSIONS` set. + +Helpers: + +- `cancel_session(session_id)` +- `clear_cancelled_session(session_id)` +- `is_session_cancelled(session_id)` + +`POST /api/evaluations/cancel` reads `X-Session-Id` and marks the session as cancelled. + +Limitations: + +- This cancellation set is process-local. +- Background job cancellation also uses `EvalJobRecord` DB state for cross-worker cancellation. + +## 11. Provider adapters + +### 11.1 Azure adapter + +`AzureOpenAIDeepEvalModel` wraps `AzureChatOpenAI`. + +Configuration order: + +1. `backend/core/secrets.toml` +2. constructor args +3. environment variables + +Concurrency: + +- module-level semaphore of 35. + +Important behavior: + +- Extracts JSON from prose/code fences. +- Records call history with duration and token usage. +- Supports both sync and async `generate` methods required by DeepEval. + +### 11.2 Vertex adapter + +`VertexAIDeepEvalModel` wraps `ChatVertexAI`. + +Behavior: + +- Requires GCP project. +- Sets safety settings to `BLOCK_NONE` for major harm categories. +- Uses module-level semaphore of 25. +- Retries rate-limit/server errors with exponential backoff and jitter. +- Records call history. + +### 11.3 Anthropic Vertex adapter + +`AnthropicVertexDeepEvalModel` wraps Anthropic Vertex clients. + +Behavior: + +- Defaults to Claude Sonnet model. +- Uses module-level semaphore of 8. +- Sends user prompt through Anthropic messages API. +- Extracts usage objects with attribute access. +- Records call history. + +## 12. Background evaluation jobs + +File: `backend/services/evaluation/job_queue.py` + +### 12.1 Dataclasses + +| Dataclass/class | Purpose | +| --- | --- | +| `EvalTask` | One entity output to evaluate. | +| `ProviderConfig` | Judge provider/model configuration. | +| `TaskResult` | One task/provider evaluation result. | +| `EvalJob` | Job state and serialization. | +| `_JobStatusProxy` | DB-backed read-only job status for non-local jobs. | + +### 12.2 Job lifecycle + +```text +create_job() + -> EvalJob(job_id, tasks, providers, session/user id) + -> submit_job() + -> store in _JOBS + -> create EvalJobRecord asynchronously + -> start _process_job(job) background task + +_process_job() + -> mark running and sync DB + -> flatten tasks x providers + -> compute per-job concurrency + -> run work items with global and per-job semaphores + -> mark completed/cancelled/failed + -> final DB sync +``` + +### 12.3 Concurrency + +- Global LLM concurrency cap: `GLOBAL_LLM_CONCURRENCY = 30`. +- `_LLM_SEMAPHORE` protects total concurrent judge calls. +- Per-job concurrency is computed as: + +```text +max(1, GLOBAL_LLM_CONCURRENCY // active_running_jobs) +``` + +Each individual evaluation is wrapped in `asyncio.wait_for(..., timeout=60.0)`. + +### 12.4 Job persistence + +`EvalJobRecord` stores: + +- job id; +- session id; +- user id; +- status; +- progress/total; +- results/errors; +- top-level error; +- created/completed timestamps. + +`get_job(job_id)` checks in-memory `_JOBS` first, then falls back to DB status through `_JobStatusProxy`. + +### 12.5 Job cancellation + +`cancel_job(job_id)`: + +- if job is local, sets `cancelled=True` and cancels live asyncio tasks; +- if job is not local, marks the DB record cancelled. + +## 13. Session persistence of evaluation results + +During background jobs, successful task results are converted into `schemas.sessions.EvaluationScore` and `schemas.sessions.EvaluationResult`, then persisted through: + +```python +SessionService.add_evaluation_result_fast(...) +``` + +`SessionService` matches the evaluation to an extraction by: + +- entity name; +- model id; +- document id or file hash when available; +- paragraph-summary fallback for `__paragraph_summary__`. + +Evaluation DB upsert target: + +```text +(extraction_result_id, metric, judge_model) +``` + +Constraint name: + +```text +uq_eval_extraction_metric_judge +``` + +## 14. Result storage + +`EvaluationResultStorage` saves JSON files under: + +```text +backend/output/evaluations/{evaluation_id}.json +``` + +Methods: + +- `save(evaluation_id, result)` +- `get(evaluation_id)` +- `list_all()` sorted by timestamp descending +- `delete(evaluation_id)` +- `get_storage_path()` + +This storage is separate from normalized DB persistence of per-extraction evaluation scores. + +## 15. Cost tracking + +Evaluation service records judge-call costs from adapter call history. + +For each adapter call: + +1. estimate cost with `cost_tracker.estimate_call_cost()`; +2. record call with `cost_tracker.record_call()` when session id exists; +3. sum costs into `evaluation_cost`. + +The returned evaluation result includes total evaluation cost and evaluation time. + +## 16. Error classification + +The job queue includes helpers for classifying errors: + +- `_is_timeout_error()` +- `_is_rate_limit_error()` +- `_is_non_retryable_error()` + +Retry delay constants are defined for normal and rate-limit failures, but current `MAX_ATTEMPTS` is 1 for single job tasks. + +## 17. Related docs + +- [02-api-surface.md](02-api-surface.md) +- [04-schemas.md](04-schemas.md) +- [06-llm-layer.md](06-llm-layer.md) +- [09-session-sharing-groups.md](09-session-sharing-groups.md) +- [appendices/risks-assumptions-testing.md](appendices/risks-assumptions-testing.md) diff --git a/docs/backend/09-session-sharing-groups.md b/docs/backend/09-session-sharing-groups.md new file mode 100644 index 0000000..488a6e6 --- /dev/null +++ b/docs/backend/09-session-sharing-groups.md @@ -0,0 +1,421 @@ +# Sessions, Sharing, and Groups Technical Design + +> *A session is a saved record of a reviewer's work β€” their uploaded files, extracted entities, and evaluation scores. This document covers the full session lifecycle: how sessions are created, updated, and restored; how the "restore view" reconstructs a complete frontend state from the database; and how sessions are shared with colleagues through groups. It also covers group membership rules β€” who can add members, who can change roles, and what a shared session recipient can and cannot do.* + +This document describes workflow sessions, DB-to-API conversion, restore-view generation, shared sessions, group membership, and authorization rules. + +## 1. Scope + +In scope: + +- `SessionService` session lifecycle; +- `SQLAlchemyDBService` session/document/extraction/evaluation persistence operations; +- session schema conversion; +- restore-view construction; +- group CRUD and membership; +- session sharing with groups. + +Out of scope: + +- frontend session-history UI; +- template-specific group sharing, covered in [10-template-system.md](10-template-system.md). + +## Visual workflow + +![Session sharing and template workflow](images/session-sharing-template-workflow.png) + +Read the top half of the diagram for sessions and groups. Session APIs call `SessionService`, which converts DB rows into the Pydantic session aggregate and builds restore payloads by asking `OrganizedFileService` for each document view. Sharing does not copy session data. It sets share metadata on the owned `app_sessions` row and allows reads only when the requesting user has a `user_groups` membership for the target group. Group role checks protect group administration, while shared-session viewing only requires membership. + +## 2. Main classes and files + +| Component | File | Responsibility | +| --- | --- | --- | +| `SessionService` | `backend/services/session/session_service.py` | High-level workflow/session orchestration. | +| `SQLAlchemyDBService` | `backend/services/database/sqlalchemy_db_service.py` | Persistence boundary for sessions, docs, extractions, evaluations, metrics, sharing. | +| `GroupService` | `backend/services/groups/group_service.py` | Group and membership business logic. | +| session schemas | `backend/schemas/sessions.py` | API-facing session aggregate models. | +| session router | `backend/api/sessions/router.py` | HTTP session endpoints. | +| group router | `backend/api/groups/router.py` | HTTP group endpoints. | +| `AppSession` ORM | `backend/models/app_session.py` | Workflow session table. | +| `Group`, `UserGroup` ORM | `backend/models/group.py` | Collaboration tables. | + +## 3. Session lifecycle + +### 3.1 Create session + +Endpoint: + +```text +POST /api/sessions +``` + +`SessionService.create_session()` flow: + +1. Convert optional `SessionConfiguration` into a dict. +2. Create `AppSession` through `SQLAlchemyDBService.create_session()`. +3. For each optional `SessionDocument`, create a `Document` DB row. +4. If all requested document inserts fail, delete the orphaned session and raise an error. +5. Return a Pydantic `Session` aggregate. + +Partial document insert success is allowed. Failed document rows are logged and skipped. + +### 3.2 Get session + +Endpoint: + +```text +GET /api/sessions/{session_id} +``` + +`SQLAlchemyDBService.get_session()` loads: + +- the `AppSession` row for the requesting user; +- associated `Document` rows; +- associated `ExtractionResult` rows; +- associated `EvaluationResult` rows. + +Then `SessionService._db_to_session()` converts raw DB dictionaries into Pydantic models. + +If the session is not owned by the user, the service returns `None`. The router can also attempt shared-session fallback depending on endpoint path. + +### 3.3 List sessions + +Endpoint: + +```text +GET /api/sessions +``` + +`SQLAlchemyDBService.list_sessions()` returns session summary rows sorted by `updated_at` descending and enriches each with: + +- `document_count` +- `document_names` +- `extraction_count` +- `study_type` from JSON configuration + +`SessionService.list_sessions()` converts these into `SessionSummary` objects. + +### 3.4 Update session + +Endpoint: + +```text +PATCH /api/sessions/{session_id} +``` + +`SessionService.update_session()` supports three categories: + +1. Basic fields: `name`, `status`, `last_step`, `configuration`. +2. Config merges: `evaluation_config`, `files_config`. +3. Heavy updates: documents, extraction results, evaluation results. + +Important merge behavior: + +- `evaluation_config` is merged into the existing config dict. +- `files_config` is deep-merged per file id so existing per-file config is preserved. + +Heavy update behavior: + +- New documents are inserted only if their `file_hash` is not already present. +- Extraction results are matched to documents by `file_hash` or document id. +- Evaluation results are matched to extraction results by entity/model/document identity. + +If the update is config-only, the service returns a lightweight session object without reloading all child rows. If heavy updates occurred, it returns the full session. + +### 3.5 Delete session + +Endpoint: + +```text +DELETE /api/sessions/{session_id} +``` + +Deletes only when both session id and user id match. + +## 4. DB-to-session conversion + +`SessionService._db_to_session()` is the central conversion function. + +### 4.1 Document conversion + +For each DB document: + +- if parse cost is missing/zero, estimate it from processor/page count/duration; +- backfill the DB when a cost can be computed; +- build `SessionDocument` with id, file hash, filename, processor, cost, page/figure/table counts. + +### 4.2 Extraction conversion + +For each DB extraction: + +- if extraction cost is missing/zero, estimate it from token counts and model id; +- infer provider from model id; +- backfill cost where possible; +- map `bbox_references` to API field `references`; +- build `schemas.sessions.ExtractionResult`. + +### 4.3 Evaluation conversion + +DB evaluation rows are grouped by: + +```text +(document_id, entity_name, model_id) +``` + +Each group becomes one `schemas.sessions.EvaluationResult` with a list of `EvaluationScore` entries. + +This preserves per-document evaluation granularity in multi-document sessions. + +## 5. Extraction persistence rules + +`SessionService.add_extraction_result_fast()` resolves the target document before writing. + +Resolution order: + +1. Existing cached session documents. +2. DB document lookup. +3. Match by `file_hash` when provided. +4. If only one document exists, use it. +5. If multiple documents exist and no file identity is provided, return `False` instead of guessing. + +This guard prevents cross-document contamination. + +Persistence uses `SQLAlchemyDBService.upsert_extraction_result()` and the database unique constraint: + +```text +(document_id, entity_name, model_id) +``` + +## 6. Evaluation persistence rules + +`SessionService.add_evaluation_result_fast()`: + +1. loads extraction results for the session; +2. resolves target document from `document_id` or `file_hash`; +3. matches extraction by entity name, model id, and optional document id; +4. supports a special `__paragraph_summary__` fallback; +5. upserts each metric score through DB service. + +Special human-score behavior: + +- A human-score update can apply to all existing metrics for a judge model. +- If no score list exists but a human score is provided, the service updates existing evaluations or creates a `human_evaluation` placeholder. + +## 7. Restore-view construction + +Endpoint: + +```text +GET /api/sessions/{session_id}/restore-view +GET /api/sessions/shared/{session_id}/restore-view +``` + +`SessionService.build_restore_view()`: + +1. merges `session.configuration.files_config` with top-level `session.files_config`; +2. resolves each document's processor; +3. calls `OrganizedFileService.build_document_view()` for each document; +4. returns a canonical frontend payload. + +Returned top-level shape: + +```json +{ + "fileId": "primary-file-hash", + "conversionId": "primary-file-hash", + "processorUsed": "azure_doc_intelligence", + "uploadedFiles": [] +} +``` + +The first uploaded file is treated as the primary file. + +## 8. Session sharing + +### 8.1 Share session + +Endpoint: + +```text +POST /api/sessions/{session_id}/share +``` + +`SessionService.share_session()` first verifies the requester belongs to the target group using `SQLAlchemyDBService.get_user_group_ids()`. + +Then `SQLAlchemyDBService.share_session()` updates the session fields: + +- `shared_with_group_id` +- `shared_by` +- `shared_at` + +Only the owning user can share the session. + +### 8.2 Unshare session + +Endpoint: + +```text +DELETE /api/sessions/{session_id}/share +``` + +Clears share fields on an owned session. + +### 8.3 List shared sessions + +Endpoint: + +```text +GET /api/sessions/shared/list +``` + +Flow: + +1. Find groups the user belongs to. +2. Query sessions shared with those group ids. +3. Exclude sessions owned by the same user. +4. Enrich with group display name and sharer display name. +5. Return `SessionSummary` list. + +### 8.4 Shared session read + +A user can read a shared session when: + +- `AppSession.shared_with_group_id` is set; +- the requesting user has a `UserGroup` row for that group. + +Role does not matter for session shared-view access; membership is enough. + +## 9. Group lifecycle + +### 9.1 Create group + +Endpoint: + +```text +POST /api/groups +``` + +`GroupService.create_group()`: + +1. inserts a `Group` row with `created_by=user_id`; +2. inserts a `UserGroup` row for the creator with `role='owner'`; +3. returns the group dict. + +The owner membership is created in the same transaction. + +### 9.2 Get group + +Endpoint: + +```text +GET /api/groups/{group_id} +``` + +Rules: + +- System admin can read any group. +- Non-admin users must be group members. + +Returned group detail includes: + +- group fields; +- `user_role`; +- enriched `members` list. + +### 9.3 List user groups + +Endpoint: + +```text +GET /api/groups +``` + +Returns groups where the user has a membership row. Each row includes: + +- `user_role` +- `member_count` + +### 9.4 Update/delete group + +Update requires: + +- system admin; or +- group admin/owner role. + +Delete requires: + +- system admin; or +- owner role. + +## 10. Membership rules + +### 10.1 Add member + +Endpoint: + +```text +POST /api/groups/{group_id}/members +``` + +Rules: + +- Requester must be admin/owner or system admin. +- New member role `owner` is normalized to `admin`. +- Adding an existing member delegates to role update. + +### 10.2 Update role + +Endpoint: + +```text +PUT /api/groups/{group_id}/members/{user_id} +``` + +Rules: + +- Cannot change to or from owner through this endpoint. +- Requester must be admin/owner unless system admin. +- Only owner can promote someone to admin. + +### 10.3 Remove member + +Endpoint: + +```text +DELETE /api/groups/{group_id}/members/{user_id} +``` + +Rules: + +- Self-removal is allowed unless the user is the only owner. +- System admin can remove members but not owners. +- Non-admin users cannot remove others. +- Owners cannot be removed by this method. + +## 11. Member enrichment + +`GroupService._enrich_members_with_profiles()` bulk-loads `User` rows and adds: + +- `display_name = user.name or user.email` +- `email` +- `avatar_url = user.image` + +Missing users receive `None` profile fields. + +## 12. Session metrics + +`SQLAlchemyDBService.increment_session_metrics()` atomically increments: + +- `total_cost` +- `total_latency` +- `total_calls` + +`CostTracker` calls this from its record path. `SessionService._db_to_session()` can also use stored totals when building session responses. + +## 13. Related docs + +- [03-data-models.md](03-data-models.md) +- [04-schemas.md](04-schemas.md) +- [05-document-processing.md](05-document-processing.md) +- [07-extraction-flow.md](07-extraction-flow.md) +- [08-evaluation-flow.md](08-evaluation-flow.md) diff --git a/docs/backend/10-template-system.md b/docs/backend/10-template-system.md new file mode 100644 index 0000000..372a26a --- /dev/null +++ b/docs/backend/10-template-system.md @@ -0,0 +1,415 @@ +# Template System Technical Design + +> *Templates are reusable sets of entity prompts that reviewers can save, version, and share. Instead of re-typing the same 20 extraction fields for every toxicology study, a reviewer saves them as a template once and loads it in two clicks. This document explains how templates are stored, how every edit is automatically versioned (with full revert capability), how templates are scoped to an individual, a group, or all users, and how the access-control rules determine who can read or edit a given template.* + +This document describes the prompt template workspace: templates, folders, scopes, versions, immutability, forks, and permissions. + +## 1. Scope + +In scope: + +- template CRUD; +- folder CRUD; +- user/group/global scopes; +- access-control algorithms; +- version snapshots and revert; +- fork and scope-change behavior; +- explicit per-user template permissions. + +Out of scope: + +- frontend template editor UI; +- prompt quality/content strategy. + +## Visual workflow + +![Session sharing and template workflow](images/session-sharing-template-workflow.png) + +Read the bottom half of the diagram for template behavior. Template APIs call `TemplateService` and `FolderService`, which enforce user, group, and global scopes before touching `prompt_templates` or `template_folders`. Updates snapshot the current prompt fields into `template_versions` before mutation, so revert creates a new current version rather than rolling the row back in place. Explicit `template_permissions` can grant user-level read/write access, but `is_immutable` is evaluated first and always blocks edits. + +## 2. Main classes and files + +| Component | File | Responsibility | +| --- | --- | --- | +| `TemplateService` | `backend/services/templates/template_service.py` | Template CRUD, permissions, versions, forks, scope changes. | +| `FolderService` | `backend/services/templates/folder_service.py` | Folder hierarchy and folder permission checks. | +| template router | `backend/api/templates/router.py` | HTTP API for templates/folders. | +| `PromptTemplate` | `backend/models/template.py` | Main template ORM model. | +| `TemplateVersion` | `backend/models/template.py` | Version snapshot ORM model. | +| `TemplatePermission` | `backend/models/template.py` | Per-user permission ORM model. | +| `TemplateFolder` | `backend/models/template.py` | Folder ORM model. | +| `GroupService` | `backend/services/groups/group_service.py` | Group role checks for group-scoped resources. | + +## 3. Template data model + +`PromptTemplate` stores: + +- `name` +- `description` +- `study_type` +- `scope`: `user`, `group`, or `global` +- `owner_user_id` +- `owner_group_id` +- `system_prompt` +- `entities` JSONB +- `summary_prompt` +- `variables` JSONB +- `is_immutable` +- `tags` +- `is_default` +- `version` +- `folder_id` +- `created_by` +- timestamps + +The `scope` determines default read/edit behavior. + +## 4. Folder data model + +`TemplateFolder` stores: + +- `name` +- `scope` +- `owner_user_id` +- `owner_group_id` +- `parent_id` +- `created_by` +- timestamps + +Folders can be hierarchical through `parent_id`. + +Folder deletion is intentionally non-cascading: the service refuses to delete a folder that contains templates or subfolders. + +## 5. Template creation + +Endpoint: + +```text +POST /api/templates +``` + +`TemplateService.create_template()` rules: + +- `scope` must be `user`, `group`, or `global`. +- `group` scope requires `owner_group_id`. +- For group scope, requester must have role `member`, `admin`, or `owner` in that group. +- User-scope templates set `owner_user_id=user_id`. +- Group-scope templates set `owner_group_id`. +- New templates start at `version=1`. +- `created_by` is the requesting user. + +Global scope is allowed by this service without a special admin guard in current code. + +## 6. Template read/list + +### 6.1 Get one template + +Endpoint: + +```text +GET /api/templates/{template_id} +``` + +`TemplateService.get_template()`: + +1. loads template by id; +2. checks `_can_read()`; +3. returns `None` if unreadable; +4. adds `can_edit` and `is_owner` flags to response. + +### 6.2 List templates + +Endpoint: + +```text +GET /api/templates +``` + +Supports filters: + +- `scope` +- `study_type` +- search across name/description +- tags + +List behavior: + +1. query candidate templates; +2. load current user's groups; +3. apply `_can_read()` in memory; +4. annotate `can_edit`, `is_owner`, and group name where relevant; +5. apply tag filter using any-match semantics. + +Because access checks are applied after the DB query, query result count can be larger than final response count. + +## 7. Template update and versioning + +Endpoint: + +```text +PUT /api/templates/{template_id} +``` + +`TemplateService.update_template()`: + +1. requires readable template; +2. requires `_can_edit()`; +3. snapshots current content into `TemplateVersion` before mutation; +4. updates allowed fields; +5. increments `version`; +6. updates timestamp; +7. returns updated template. + +Allowed update fields: + +- `name` +- `description` +- `study_type` +- `system_prompt` +- `entities` +- `summary_prompt` +- `variables` +- `tags` +- `is_immutable` +- `folder_id` + +If no valid update fields are provided, the current template is returned unchanged. + +## 8. Version history and revert + +Endpoints: + +```text +GET /api/templates/{template_id}/versions +POST /api/templates/{template_id}/revert/{version} +``` + +Version history: + +- requires read permission; +- returns snapshots ordered by version descending. + +Revert behavior: + +1. requires read and edit permission; +2. loads requested `TemplateVersion`; +3. calls `update_template()` with snapshot fields; +4. creates a new version snapshot as part of the update path. + +Reverting does not reuse the old version number; it creates a new current version. + +## 9. Forking + +Endpoint: + +```text +POST /api/templates/{template_id}/fork +``` + +`TemplateService.fork_template()`: + +1. loads source through `get_template()` so read permission applies; +2. creates a new user-scope copy; +3. defaults name to `Copy of {source_name}` unless provided; +4. sets `is_immutable=False`. + +Forked templates are always personal/user scoped. + +## 10. Scope changes + +Endpoint: + +```text +PUT /api/templates/{template_id}/scope +``` + +`TemplateService.change_scope()` validates the target scope and enforces rules based on old scope. + +### 10.1 Old user scope + +- Requester must own the template. +- Moving to group scope requires membership/admin/owner role in target group. + +### 10.2 Old group scope + +- Requester must be admin/owner of current group. +- Moving to another group requires membership/admin/owner in target group. + +### 10.3 Old global scope + +- Requester must be `created_by`. + +### 10.4 Ownership fields + +The service updates ownership fields according to target scope: + +- user scope: `owner_user_id=user_id`, `owner_group_id=None`; +- group scope: `owner_user_id=None`, `owner_group_id=target group`; +- global scope: both owner fields cleared. + +## 11. Immutability + +Endpoint: + +```text +PUT /api/templates/{template_id}/immutable +``` + +Rules: + +- User-scope template owner can set immutability. +- Group-scope admin/owner can set immutability. +- Global scope currently returns `None` for this operation. + +`_can_edit()` always denies edits when `is_immutable=True`, regardless of other permissions. + +## 12. Explicit permissions + +Endpoints: + +```text +GET /api/templates/{template_id}/permissions +POST /api/templates/{template_id}/permissions +DELETE /api/templates/{template_id}/permissions/{user_id} +``` + +`TemplatePermission` fields: + +- `template_id` +- `user_id` +- `can_read` +- `can_write` +- `granted_by` + +Permissions can be managed by: + +- user-scope template owner; +- group-scope admin/owner. + +Global scope is disallowed for explicit permission operations. + +Permission upsert uses unique constraint: + +```text +(template_id, user_id) +``` + +## 13. Access-control algorithms + +### 13.1 `_can_read(template, user_id)` + +Algorithm: + +1. Global scope is readable by anyone. +2. User scope is readable by owner. +3. Group scope is readable by group members. +4. Otherwise, explicit `TemplatePermission.can_read` can grant read access. + +### 13.2 `_can_edit(template, user_id)` + +Algorithm: + +1. If immutable, deny. +2. If explicit permission exists, use `can_write`. +3. User scope: owner can edit. +4. Group scope: member/admin/owner can edit by default. +5. Global scope: creator can edit. + +Important implication: + +- Explicit user permission is checked before default scope edit rules, but immutability always wins. + +### 13.3 `_is_owner(template, user_id)` + +Algorithm: + +- User scope: owner user id matches. +- Group scope: group admin/owner counts as owner. +- Global scope: creator counts as owner. + +## 14. Folder operations + +### 14.1 List folders + +Endpoint: + +```text +GET /api/templates/folders +``` + +Filters: + +- scope; +- owner user/group; +- parent id. + +Top-level folders are selected with `parent_id is None`. + +### 14.2 Create folder + +Endpoint: + +```text +POST /api/templates/folders +``` + +Rules: + +- Name must be non-empty. +- User scope is manageable by the user. +- Group scope requires admin/owner role in owning group. +- Global scope is currently allowed by `_can_manage_folder()`. +- Parent folder must have the same scope. +- Group parent folder must have the same owning group. + +### 14.3 Rename folder + +Endpoint: + +```text +PATCH /api/templates/folders/{folder_id} +``` + +Allowed if: + +- user can manage the folder scope; or +- user originally created the folder. + +### 14.4 Delete folder + +Endpoint: + +```text +DELETE /api/templates/folders/{folder_id} +``` + +Allowed if: + +- user can manage folder scope; or +- user originally created the folder. + +Deletion is refused when: + +- any template has `folder_id` equal to the folder; +- any subfolder has `parent_id` equal to the folder. + +Return shape on success: + +```json +{"deleted": "folder-id"} +``` + +## 15. Risks and implementation notes + +- Global-scope creation and global folder management are permissive in current service code. +- `PromptTemplate.folder_id` is a logical association but no ORM FK is declared. +- Group-scope edit permission allows any member/admin/owner by default, unless immutable or explicit permission logic changes. +- Tag filtering is in-memory and uses any-match semantics. +- Version snapshots are created before update, so snapshot version represents the previous state. + +## 16. Related docs + +- [03-data-models.md](03-data-models.md) +- [04-schemas.md](04-schemas.md) +- [09-session-sharing-groups.md](09-session-sharing-groups.md) +- [11-auth-security-observability.md](11-auth-security-observability.md) diff --git a/docs/backend/11-auth-security-observability.md b/docs/backend/11-auth-security-observability.md new file mode 100644 index 0000000..e347e6b --- /dev/null +++ b/docs/backend/11-auth-security-observability.md @@ -0,0 +1,390 @@ +# Auth, Security, and Observability Technical Design + +> *Three cross-cutting concerns that touch every part of the backend: who is allowed in (authentication via Better Auth and GitHub OAuth), what they're allowed to do (authorization per endpoint and resource), and what the system records about its own behaviour (structured logs, Prometheus metrics, OpenTelemetry traces, and per-session cost tracking). This document covers all three β€” including the security risks table, CORS configuration, secrets loading, and how the CostTracker updates the database without blocking the request loop.* + +This document describes authentication, authorization boundaries, security-sensitive behaviors, logging, metrics, tracing, and cost/session telemetry. + +## 1. Scope + +In scope: + +- Better Auth session validation in FastAPI; +- auth proxy behavior; +- CORS configuration; +- service-level authorization boundaries; +- file/path safety considerations; +- logging and request IDs; +- Prometheus/OpenTelemetry/Loki hooks; +- cost/session telemetry. + +Out of scope: + +- Better Auth sidecar internal TypeScript implementation; +- OAuth provider setup details; +- cloud IAM policy design. + +## Visual workflow + +![Auth, security, and observability workflow](images/auth-observability-workflow.png) + +The diagram shows the cross-cutting path that every protected request follows. FastAPI dependencies extract the Better Auth session token, prefer the `Authorization` header, join `AuthSession` to `User`, check expiry and optional email allowlist, then pass a compact user dict into the route. Authorization is deliberately service-owned: session ownership, group roles, template scope, and artifact path safety are enforced after authentication. Observability is attached at two places: request middleware logs request id, status, and duration; provider/parser paths record model, token, duration, and cost telemetry into in-memory metrics, Prometheus metrics when available, and session totals in PostgreSQL. + +## 2. Authentication model + +The backend does not validate JWTs. It validates Better Auth sessions by looking up session tokens in PostgreSQL. + +Main file: + +```text +backend/core/auth.py +``` + +### 2.1 `get_current_user(request)` + +Flow: + +```text +extract token + -> query AuthSession joined to User + -> reject missing session + -> reject expired session + -> optional ALLOWED_EMAILS allowlist check + -> return user dict +``` + +Accepted token locations: + +1. `Authorization: Bearer ` header. +2. `better-auth.session_token` cookie fallback. + +The code prefers the header because Better Auth v1.2+ hashes tokens before storing them in the DB, and the frontend may send the DB/hash token from the get-session API. + +Returned user shape: + +```json +{ + "id": "user-id", + "email": "user@example.com", + "name": "Display Name", + "image": "avatar-url", + "is_admin": false +} +``` + +Failure behavior: + +- no token: HTTP 401; +- invalid token: HTTP 401; +- expired token: HTTP 401; +- email not allowed: HTTP 403; +- unexpected auth error: HTTP 401. + +### 2.2 `get_optional_user(request)` + +Returns `None` instead of raising for auth failure. Used by endpoints that can work with optional authentication. + +## 3. Auth proxy + +File: + +```text +backend/api/auth/proxy.py +``` + +Purpose: + +```text +/api/auth/{path:path} -> Better Auth sidecar /api/auth/{path} +``` + +Important behavior: + +- Auth proxy router is registered first in `main.py`. +- Forwards methods such as GET, POST, PUT, PATCH, DELETE, OPTIONS. +- Passes through `Set-Cookie` and redirect responses. +- Adds forwarded host/proto headers. +- Strips hop-by-hop headers. +- Can enforce email allowlist behavior for session lookup. + +This lets production route all auth traffic through the same public frontend/backend origin without nginx. + +## 4. Authorization boundaries + +Authentication only identifies the user. Authorization is mostly enforced in service classes. + +| Area | Authorization owner | +| --- | --- | +| Sessions | `SessionService`, `SQLAlchemyDBService` filter by `user_id`; shared sessions check group membership. | +| Groups | `GroupService` role checks and system-admin bypass. | +| Templates | `TemplateService` scope, owner, group role, explicit permission, immutability checks. | +| Folders | `FolderService` scope/group/creator checks. | +| Files/documents | File hash access is less strongly permission-scoped; user document listing uses DB user association. | +| Evaluation jobs | Job records include user/session ids, but in-memory polling primarily uses job id. | + +## 5. CORS + +File: + +```text +backend/core/middleware.py +``` + +`setup_cors(app)` reads: + +```text +CORS_ALLOWED_ORIGINS +``` + +Behavior: + +- If unset, defaults to `*` for local development. +- If set, splits comma-separated origins. +- Allows credentials, all methods, and all headers. + +Production should set explicit allowed origins. + +## 6. Configuration and secrets loading + +### 6.1 `main.load_secrets_to_env()` + +Loads TOML secrets from candidate paths and maps sections/keys to uppercase environment variables. + +Special case: + +- `Macbook.macbook_llm_base_url` -> `MACBOOK_LLM_BASE_URL` + +### 6.2 `core.config.load_config()` + +Loads provider-specific config from `backend/core/secrets.toml`. + +Supported sections include: + +- `azure_openai` +- `azure_doc_intelligence` +- `vertex_ai` +- `anthropic` + +It also supports `azure_openai.models` as a multi-model JSON list stored in `AZURE_OPENAI_MODELS`. + +Google credentials are detected from a service-account JSON file under `backend/core/` and mapped to `GOOGLE_APPLICATION_CREDENTIALS`. + +## 7. File and path safety + +Important safety mechanisms: + +- Uploaded files are addressed by SHA-256 content hash, not arbitrary user paths. +- Blob paths are generated by service code under `global/{hash}/...`. +- Processed artifacts are addressed by known relative paths such as `document.md`, `metadata.json`, `raw_analysis.json`, `figures/*`, and `tables/*`. +- Figure/table-serving endpoints validate filenames before reading artifact bytes. +- Text persisted to PostgreSQL is sanitized by `sanitize_text()` in DB service paths to remove null bytes and unsupported control characters. + +Important caveat: + +- Several document endpoints operate by file hash/document id and should be reviewed carefully before making files publicly guessable or exposing hashes outside authenticated contexts. + +## 8. Structured logging + +File: + +```text +backend/core/logging_config.py +``` + +`setup_logging()` configures: + +- structlog JSON rendering; +- stdout handler; +- file handler at `backend/output/logs/app.log`; +- optional Loki handler when `LOKI_URL` is configured. + +Noisy libraries are reduced to warning level: + +- `httpx` +- `httpcore` +- `urllib3` +- `uvicorn.access` + +## 9. Request observability middleware + +`main.create_app()` installs an HTTP middleware that: + +1. creates a 12-character request id; +2. clears and binds structlog context variables; +3. measures duration; +4. logs at: + - error for HTTP 500+; + - warning for HTTP 400+; + - info for success; +5. attaches `X-Request-Id` response header. + +This is the primary per-request logging path. + +## 10. Global exception handling + +`main.create_app()` registers a global exception handler for `Exception`. + +Behavior: + +- logs type, method, path, and traceback; +- returns HTTP 500 with: + +```json +{"detail": ""} +``` + +Expected API errors should still use `HTTPException` in routers/services. + +## 11. Prometheus metrics + +`main.create_app()` attempts to install `prometheus_fastapi_instrumentator`. + +If available: + +```text +GET /metrics +``` + +is exposed outside the OpenAPI schema. + +`CostTracker` also optionally emits Prometheus counters/histograms when `prometheus_client` is installed: + +- token counter by provider/model/token type; +- cost counter in cents; +- duration histogram. + +## 12. OpenTelemetry tracing + +`main._setup_otel(app)` enables tracing when: + +```text +OTLP_ENDPOINT +``` + +is set. + +Behavior: + +- creates `TracerProvider` with service name `summarization-backend`; +- sends spans through OTLP HTTP exporter to `{OTLP_ENDPOINT}/v1/traces`; +- instruments FastAPI app. + +Failure to configure tracing is warning-logged and non-fatal. + +## 13. Browser trace proxy + +Endpoint: + +```text +POST /api/telemetry/traces +``` + +The server router can proxy browser OTLP trace payloads to the configured tracing backend. This keeps browser instrumentation from needing direct access to the telemetry backend. + +## 14. Cost tracking + +File: + +```text +backend/services/telemetry/cost_tracker.py +``` + +Main classes: + +- `CallMetric` +- `BatchMetric` +- `SessionMetrics` +- `CostTracker` + +### 14.1 Pricing config + +Pricing is loaded from: + +```text +backend/config/pricing.json +``` + +Optional override: + +```text +PRICING_JSON_OVERRIDE +``` + +Overrides are JSON and merged into the pricing map. + +### 14.2 Cost algorithm + +`CostTracker._compute_cost()` supports: + +- token cost per million tokens; +- token cost per thousand tokens; +- page cost for document parsing; +- compute cost per minute for local/self-hosted runtimes. + +Model/provider normalization handles Azure, Vertex/Gemini, Claude, Llama, Docling, Azure Document Intelligence, Macbook, and vLLM-style ids. + +### 14.3 Recording calls + +`record_call()` updates in-memory session aggregate: + +- `total_cost` +- `total_latency` +- `total_calls` +- list of calls + +It also: + +- emits Prometheus metrics when available; +- updates DB session metrics through `SQLAlchemyDBService.increment_session_metrics()`. + +The DB write is scheduled through an executor when an event loop is running so telemetry does not block the async request path. + +### 14.4 Batch metrics + +`record_batch()` stores `BatchMetric` under the session state. + +### 14.5 Restore and clear + +- `load_session_metrics_from_db(session_id)` reconstructs aggregate metrics from DB totals. +- `clear_session(session_id)` clears in-memory metrics and resets DB totals. + +## 15. Server metrics/config endpoints + +Router file: + +```text +backend/api/server/router.py +``` + +Important endpoints: + +- `/api/server/health` +- `/api/server-config` +- `/api/models` +- `/api/server/session-metrics` +- `/api/server/session-metrics/load` +- `/api/server/batch-metrics` +- `/api/server/document-metrics` +- `/api/server/logs` + +These endpoints expose operational health, provider availability, model catalog, session metrics, document metrics, and recent logs. + +## 16. Security risks and mitigations + +| Risk | Current mitigation | Remaining concern | +| --- | --- | --- | +| Unauthenticated API access | Most routers use `Depends(get_current_user)`. | Some utility endpoints are intentionally unauthenticated; review before public deployment. | +| Token misuse | Sessions are looked up in DB and expiry checked. | Header/cookie token behavior depends on Better Auth token hashing mode. | +| Unauthorized session reads | Session queries filter by `user_id`; shared reads require group membership. | File hash endpoints should be reviewed if hashes leak. | +| Group privilege escalation | `GroupService` protects owner/admin transitions and only-owner removal. | System-admin behavior should be audited when admin assignment changes. | +| Template unauthorized edits | `TemplateService` checks scope, owner, group role, explicit permissions, and immutability. | Global-scope operations are permissive in current code. | +| Path traversal for artifacts | Service-generated blob paths and filename validation for figure/table endpoints. | Keep all future artifact reads on service-generated relative paths. | +| PostgreSQL null-byte errors | `sanitize_text()` strips null/control chars before DB writes in key paths. | Ensure new text persistence paths use sanitizer. | +| Provider rate limits | Provider clients and evaluation adapters use semaphores/retries/timeouts. | Retry policies vary by provider; background job `MAX_ATTEMPTS` is currently 1. | + +## 17. Related docs + +- [01-architecture.md](01-architecture.md) +- [02-api-surface.md](02-api-surface.md) +- [06-llm-layer.md](06-llm-layer.md) +- [08-evaluation-flow.md](08-evaluation-flow.md) +- [appendices/risks-assumptions-testing.md](appendices/risks-assumptions-testing.md) diff --git a/docs/backend/README.md b/docs/backend/README.md new file mode 100644 index 0000000..20afbb6 --- /dev/null +++ b/docs/backend/README.md @@ -0,0 +1,179 @@ +# Backend Technical Design Document + +This documentation set is a layered Technical Design Document (TDD) for the FastAPI backend in `backend/`. It is written from the implementation and is intended to help reviewers understand how the system is implemented in code: API boundaries, classes, data models, algorithms, infrastructure assumptions, risks, and testing strategy. + +The backend is not documented as one long README because the codebase contains several distinct subsystems: authentication, document processing, LLM routing, entity extraction, evaluation, sessions, groups, templates, storage, telemetry, and deployment hooks. Start here, then follow the module links for details. + +## Visual workflow map + +The backend diagrams are stored under [`images/`](images/) and are embedded below in the same order as the detailed module documents. They are intentionally implementation-oriented: boxes map to routers, services, tables, providers, or storage paths that exist in `backend/`. + +### Backend Runtime Architecture + +![Backend runtime architecture](images/backend-runtime-architecture.png) + +Runtime layers from browser/API edge to routers, services, persistence, external providers, and observability. Details: [01-architecture.md](01-architecture.md). + +### Backend Data Model Relationships + +![Backend data model relationships](images/data-model-relationships.png) + +How Better Auth tables, workflow tables, evaluation jobs, groups, and templates relate. Details: [03-data-models.md](03-data-models.md). + +### Document Upload and Processing Workflow + +![Document upload and processing workflow](images/document-processing-workflow.png) + +Upload validation, SHA-256 deduplication, parser selection, artifact generation, blob sync, and document-view output. Details: [05-document-processing.md](05-document-processing.md). + +### LLM Provider Routing Workflow + +![LLM provider routing workflow](images/llm-provider-routing.png) + +`LLMService` dispatch, timeout handling, provider clients, response normalization, and cost tracking. Details: [06-llm-layer.md](06-llm-layer.md). + +### Entity Extraction and Grounding Workflow + +![Entity extraction and grounding workflow](images/extraction-grounding-workflow.png) + +Per-entity LLM fan-out, figure context, reference extraction, bbox matching, and extraction persistence. Details: [07-extraction-flow.md](07-extraction-flow.md). + +### Evaluation Workflow + +![Evaluation workflow](images/evaluation-workflow.png) + +Synchronous and background LLM-as-a-judge evaluation, metric factories, job concurrency, cancellation, and result storage. Details: [08-evaluation-flow.md](08-evaluation-flow.md). + +### Session Sharing and Template Workflow + +![Session sharing and template workflow](images/session-sharing-template-workflow.png) + +Session restore, group sharing, template scopes, permissions, folders, and version snapshots. Details: [09-session-sharing-groups.md](09-session-sharing-groups.md) and [10-template-system.md](10-template-system.md). + +### Auth, Security, and Observability Workflow + +![Auth, security, and observability workflow](images/auth-observability-workflow.png) + +Better Auth session lookup, auth proxy, service authorization, request logs, metrics, traces, and session telemetry. Details: [11-auth-security-observability.md](11-auth-security-observability.md). + +## 1. Introduction + +### 1.1 Problem + +The application needs a backend that can ingest user-uploaded scientific documents, parse them into machine-readable artifacts, run prompt-based entity extraction with multiple model providers, evaluate extraction quality, and persist collaborative workflow state for later restoration and sharing. + +The implementation problem is larger than a single endpoint or service because the backend must coordinate: + +- authenticated access through Better Auth sessions stored in PostgreSQL; +- file upload, deduplication, blob-backed artifact storage, and local cache hydration; +- document parsing through Azure Document Intelligence and Docling; +- model-provider routing across Azure OpenAI, Gemini/Vertex, Anthropic-on-Vertex, Llama MaaS, Macbook-hosted models, and vLLM; +- extraction and figure-reference grounding against document analysis output; +- DeepEval-based evaluation with background job execution and cancellation; +- persistent sessions, groups, template workspaces, and sharing workflows; +- cost, latency, logging, metrics, and deployment-oriented observability. + +### 1.2 Background + +The backend is a FastAPI application launched from `backend/main.py`. It exposes API routers under `backend/api/`, uses SQLAlchemy models under `backend/models/`, Pydantic request/response schemas under `backend/schemas/`, and domain services under `backend/services/`. + +The current architecture replaced an earlier Supabase-based implementation with PostgreSQL, SQLAlchemy, Alembic, and a Better Auth sidecar. The migration context is documented in `../superpowers/migration-guide.md` and the deployment architecture in `../superpowers/plans/dockerize-and-deploy.md`. + +### 1.3 Requirements + +#### 1.3.1 Functional requirements + +- Accept authenticated PDF uploads and deduplicate files by SHA-256 hash. +- Store original files and processed outputs in Azure Blob Storage-backed paths. +- Process documents with Azure Document Intelligence or Docling. +- Generate markdown, raw analysis JSON, metadata, figure images, and table HTML artifacts. +- Return canonical document views for frontend restore and viewer workflows. +- Extract custom entities from document markdown using configured LLM providers. +- Preserve extraction references and bounding boxes where provider output allows it. +- Generate paragraph summaries from extracted entities. +- Evaluate extraction outputs using built-in and custom metrics. +- Support background evaluation jobs with polling and cancellation. +- Persist sessions, documents, extractions, evaluations, metrics, templates, folders, groups, and sharing metadata. +- Allow group-based sharing of sessions and templates. +- Expose model/provider availability and server metrics endpoints. + +#### 1.3.2 Non-functional requirements + +- Authentication: protected endpoints validate Better Auth session tokens against the PostgreSQL `session` table. +- Authorization: services enforce ownership, group membership, roles, and template permissions. +- Scalability: document and evaluation jobs use concurrency controls; production scales by replicas instead of multiple Gunicorn workers per container. +- Cost visibility: LLM and document-processing costs are estimated and attached to session metrics. +- Reliability: provider clients include retry, timeout, and fallback paths where necessary. +- Observability: structured JSON logs, request IDs, Prometheus metrics, optional OpenTelemetry traces, and optional Loki shipping are supported. +- Portability: the backend uses environment variables and `secrets.toml` loading to support local and containerized environments. + +## 2. Technical design map + +| Area | Primary document | Main implementation files | +| --- | --- | --- | +| Backend architecture | [01-architecture.md](01-architecture.md) | `backend/main.py`, `backend/core/*` | +| API contracts | [02-api-surface.md](02-api-surface.md) | `backend/api/*` | +| Physical data model | [03-data-models.md](03-data-models.md) | `backend/models/*`, `backend/alembic/*` | +| Pydantic schemas | [04-schemas.md](04-schemas.md) | `backend/schemas/*` | +| Document processing | [05-document-processing.md](05-document-processing.md) | `backend/services/document/*`, `backend/services/storage/*` | +| LLM provider layer | [06-llm-layer.md](06-llm-layer.md) | `backend/services/llm/*` | +| Entity extraction | [07-extraction-flow.md](07-extraction-flow.md) | `backend/api/extractions/router.py`, provider clients, bbox matchers | +| Evaluation | [08-evaluation-flow.md](08-evaluation-flow.md) | `backend/services/evaluation/*`, `backend/api/evaluations/*` | +| Sessions, groups, sharing | [09-session-sharing-groups.md](09-session-sharing-groups.md) | `backend/services/session/*`, `backend/services/groups/*` | +| Templates and folders | [10-template-system.md](10-template-system.md) | `backend/services/templates/*`, `backend/api/templates/router.py` | +| Auth, security, observability | [11-auth-security-observability.md](11-auth-security-observability.md) | `backend/core/*`, `backend/api/auth/*`, telemetry/logging files | + +## 3. Appendices + +- [API endpoint index](appendices/api-endpoint-index.md) β€” compact endpoint list by router. +- [Class index](appendices/class-index.md) β€” backend classes, dataclasses, and schema classes by package. +- [Class reference](appendices/class-reference.md) β€” field-level reference for ORM models, Pydantic schemas, dataclasses, service attributes, and provider classes. +- [Data-flow diagrams](appendices/data-flow-diagrams.md) β€” text diagrams for upload, processing, extraction, evaluation, and restore flows. +- [Risks, assumptions, and testing](appendices/risks-assumptions-testing.md) β€” risks, assumptions, and recommended test coverage. + +## 4. High-level backend stack + +| Layer | Technology / implementation | +| --- | --- | +| API framework | FastAPI | +| Auth | Better Auth sidecar, PostgreSQL-backed session validation | +| ORM | SQLAlchemy | +| Migrations | Alembic | +| Database | PostgreSQL | +| File storage | Azure Blob Storage via `BlobStorageClient`; local `/tmp/summarization` cache for processed artifacts | +| Document parsers | Azure Document Intelligence, Docling remote/local service paths | +| LLM providers | Azure OpenAI, Vertex/Gemini, Anthropic Vertex, Llama MaaS, Macbook-hosted Ollama-compatible runtime, vLLM OpenAI-compatible endpoint | +| Evaluation | DeepEval GEval metrics plus custom metric factory | +| Metrics/cost | `CostTracker`, Prometheus metrics, session metric DB fields | +| Logging/tracing | structlog JSON logs, optional Loki, optional OpenTelemetry OTLP export | + +## 5. Design boundaries + +### In scope + +- Backend API behavior and endpoint contracts. +- Backend classes, methods, data structures, and persistence models. +- Document parsing artifact formats and storage paths. +- LLM provider routing, timeout, retry, and response normalization behavior. +- Evaluation algorithms, job queue behavior, and cost recording. +- Session restore, sharing, groups, and template authorization logic. +- Security and observability mechanisms implemented in the backend. + +### Out of scope + +- Frontend component design and UI state management, except where backend restore/API contracts require context. +- Auth sidecar internal TypeScript design, except the backend proxy and session-validation boundary. +- Cloud provisioning details beyond backend design dependencies; see `../superpowers/plans/dockerize-and-deploy.md` for deployment records. +- Business Requirements Document (BRD) details not visible in the current repository. + +## 6. How to maintain these docs + +When backend code changes, update the smallest relevant module document first, then update the index or appendix only if links, class names, endpoint lists, or cross-module flows changed. + +Good update examples: + +- Adding a new SQLAlchemy table or class field: update [03-data-models.md](03-data-models.md), [appendices/class-index.md](appendices/class-index.md), and [appendices/class-reference.md](appendices/class-reference.md). +- Adding a new route: update [02-api-surface.md](02-api-surface.md) and [appendices/api-endpoint-index.md](appendices/api-endpoint-index.md). +- Adding or changing a Pydantic schema field: update [04-schemas.md](04-schemas.md) and [appendices/class-reference.md](appendices/class-reference.md). +- Adding a new model provider: update [06-llm-layer.md](06-llm-layer.md), [11-auth-security-observability.md](11-auth-security-observability.md) if new secrets are needed, and the risk/testing appendix. +- Changing extraction result shape: update [04-schemas.md](04-schemas.md), [07-extraction-flow.md](07-extraction-flow.md), and data-flow diagrams. diff --git a/docs/backend/appendices/api-endpoint-index.md b/docs/backend/appendices/api-endpoint-index.md new file mode 100644 index 0000000..c5fe927 --- /dev/null +++ b/docs/backend/appendices/api-endpoint-index.md @@ -0,0 +1,152 @@ +# API Endpoint Index + +Compact index of backend API endpoints by router. For design details, see [../02-api-surface.md](../02-api-surface.md). + +## Auth proxy + +| Method | Path | Purpose | +| --- | --- | --- | +| all common methods | `/api/auth/{path:path}` | Proxy Better Auth sidecar endpoints. | + +## Auth + +| Method | Path | Purpose | +| --- | --- | --- | +| `GET` | `/auth/health` | Auth-protected health check. | +| `POST` | `/auth/history` | Record login history. | + +## Files + +| Method | Path | Purpose | +| --- | --- | --- | +| `POST` | `/api/upload` | Upload and deduplicate file. | +| `GET` | `/api/files/list` | List current user's files. | +| `GET` | `/api/files/{file_id}` | Download uploaded file. | +| `GET` | `/api/files/{file_id}/info` | Get file metadata and processing flags. | +| `DELETE` | `/api/files/{file_id}` | Delete file placeholder/stub behavior. | + +## Documents + +| Method | Path | Purpose | +| --- | --- | --- | +| `GET` | `/api/documents/{document_id}/view` | Return canonical document view. | +| `POST` | `/api/documents/process/file/{file_id}` | Process uploaded file. | +| `GET` | `/api/documents/{document_id}/content` | Return markdown content. | +| `GET` | `/api/documents/{document_id}/enhanced-content` | Return markdown with figure summaries. | +| `GET` | `/api/documents/{document_id}/figures` | Return figure metadata. | +| `GET` | `/api/documents/{document_id}/analysis` | Return normalized raw analysis. | +| `GET` | `/api/documents/{document_id}/figures/{figure_filename}` | Serve figure image artifact. | +| `POST` | `/api/documents/{document_id}/figures/{figure_id}/generate-summary` | Generate/persist figure summary. | +| `POST` | `/api/documents/{document_id}/figures/{figure_id}/extract-content` | Legacy alias for figure extraction. | +| `GET` | `/api/documents/{document_id}/tables/{table_filename}` | Serve table HTML artifact. | + +## Extractions + +| Method | Path | Purpose | +| --- | --- | --- | +| `POST` | `/api/extract` | Extract requested entities from a processed document. | + +## Paragraph generation and evaluation + +| Method | Path | Purpose | +| --- | --- | --- | +| `POST` | `/api/generate_paragraph` | Generate paragraph from extracted entities. | +| `POST` | `/api/paragraph-evaluation/generate` | Generate paragraph evaluation/ground-truth record. | + +## Evaluations + +| Method | Path | Purpose | +| --- | --- | --- | +| `POST` | `/api/evaluations/cancel` | Cancel session evaluation. | +| `POST` | `/api/evaluations/evaluate` | Evaluate one extraction. | +| `POST` | `/api/evaluations/evaluate/batch` | Evaluate multiple extractions. | +| `POST` | `/api/evaluations/evaluate/custom` | Evaluate with custom metric. | +| `GET` | `/api/evaluations/results/{evaluation_id}` | Fetch stored evaluation result. | +| `GET` | `/api/evaluations/results` | List stored evaluation results. | +| `GET` | `/api/evaluations/metrics/info` | Return metric/provider info. | + +## Evaluation jobs + +| Method | Path | Purpose | +| --- | --- | --- | +| `POST` | `/api/evaluations/jobs` | Submit background evaluation job. | +| `GET` | `/api/evaluations/jobs/{job_id}` | Poll job status. | +| `POST` | `/api/evaluations/jobs/{job_id}/cancel` | Cancel job. | + +## Sessions + +| Method | Path | Purpose | +| --- | --- | --- | +| `POST` | `/api/sessions` | Create session. | +| `GET` | `/api/sessions` | List user sessions. | +| `GET` | `/api/sessions/{session_id}` | Get full owned session. | +| `GET` | `/api/sessions/{session_id}/restore-view` | Build restore-view payload. | +| `PATCH` | `/api/sessions/{session_id}` | Update session. | +| `DELETE` | `/api/sessions/{session_id}` | Delete session. | +| `POST` | `/api/sessions/{session_id}/extractions` | Add extraction result. | +| `POST` | `/api/sessions/{session_id}/evaluations` | Add evaluation result. | +| `GET` | `/api/sessions/shared/list` | List shared sessions. | +| `GET` | `/api/sessions/shared/{session_id}` | Get shared session. | +| `GET` | `/api/sessions/shared/{session_id}/restore-view` | Build shared restore-view payload. | +| `POST` | `/api/sessions/{session_id}/share` | Share session with group. | +| `DELETE` | `/api/sessions/{session_id}/share` | Unshare session. | + +## Groups + +| Method | Path | Purpose | +| --- | --- | --- | +| `GET` | `/api/groups` | List groups. | +| `POST` | `/api/groups` | Create group. | +| `GET` | `/api/groups/{group_id}` | Get group detail. | +| `PUT` | `/api/groups/{group_id}` | Update group. | +| `DELETE` | `/api/groups/{group_id}` | Delete group. | +| `GET` | `/api/groups/{group_id}/members` | List members. | +| `POST` | `/api/groups/{group_id}/members` | Add member. | +| `PUT` | `/api/groups/{group_id}/members/{user_id}` | Update member role. | +| `DELETE` | `/api/groups/{group_id}/members/{user_id}` | Remove member. | +| `GET` | `/api/groups/users/search` | Search users. | + +## Templates and folders + +| Method | Path | Purpose | +| --- | --- | --- | +| `GET` | `/api/templates/folders` | List folders. | +| `POST` | `/api/templates/folders` | Create folder. | +| `PATCH` | `/api/templates/folders/{folder_id}` | Rename folder. | +| `DELETE` | `/api/templates/folders/{folder_id}` | Delete folder. | +| `GET` | `/api/templates` | List templates. | +| `POST` | `/api/templates` | Create template. | +| `GET` | `/api/templates/{template_id}` | Get template. | +| `PUT` | `/api/templates/{template_id}` | Update template. | +| `DELETE` | `/api/templates/{template_id}` | Delete template. | +| `POST` | `/api/templates/{template_id}/fork` | Fork template. | +| `PUT` | `/api/templates/{template_id}/scope` | Change template scope. | +| `PUT` | `/api/templates/{template_id}/immutable` | Set immutability. | +| `GET` | `/api/templates/{template_id}/versions` | List versions. | +| `POST` | `/api/templates/{template_id}/revert/{version}` | Revert version. | +| `GET` | `/api/templates/{template_id}/permissions` | List permissions. | +| `POST` | `/api/templates/{template_id}/permissions` | Set permission. | +| `DELETE` | `/api/templates/{template_id}/permissions/{user_id}` | Remove permission. | + +## Server and telemetry + +| Method | Path | Purpose | +| --- | --- | --- | +| `GET` | `/api/server/health` | Health check. | +| `POST` | `/api/telemetry/traces` | Proxy browser traces. | +| `POST` | `/api/server/client-error` | Record frontend error. | +| `GET` | `/api/server-config` | Provider/config flags. | +| `GET` | `/api/models` | Available model catalog. | +| `GET` | `/api/server/session-metrics` | Session metrics. | +| `POST` | `/api/server/session-metrics/load` | Load metrics from DB. | +| `DELETE` | `/api/server/session-metrics` | Clear session metrics. | +| `POST` | `/api/server/batch-metrics` | Record batch metrics. | +| `GET` | `/api/server/document-metrics` | Document metrics. | +| `POST` | `/api/server/benchmark/clear` | Clear benchmark cache. | +| `GET` | `/api/server/logs` | Fetch server logs. | + +## Chat + +| Method | Path | Purpose | +| --- | --- | --- | +| `POST` | `/api/chat/query` | General chat over optional document markdown. | diff --git a/docs/backend/appendices/class-index.md b/docs/backend/appendices/class-index.md new file mode 100644 index 0000000..c7634bb --- /dev/null +++ b/docs/backend/appendices/class-index.md @@ -0,0 +1,172 @@ +# Backend Class Index + +Compact index of backend classes, dataclasses, and schema classes. For field-level details, see the generated [class reference](class-reference.md). For detailed behavior, follow the linked module docs from [../README.md](../README.md). + +## Core app/config/auth + +| Symbol | File | Type | +| --- | --- | --- | +| `get_current_user` | `backend/core/auth.py` | FastAPI dependency function | +| `get_optional_user` | `backend/core/auth.py` | FastAPI dependency function | +| `load_config` | `backend/core/config.py` | config loader | +| `setup_cors` | `backend/core/middleware.py` | middleware installer | +| `setup_logging` | `backend/core/logging_config.py` | logging setup | + +## SQLAlchemy models + +| Class | File | Table | +| --- | --- | --- | +| `Base` | `backend/models/base.py` | declarative base | +| `User` | `backend/models/user.py` | `user` | +| `AuthSession` | `backend/models/user.py` | `session` | +| `Account` | `backend/models/user.py` | `account` | +| `Verification` | `backend/models/user.py` | `verification` | +| `AppSession` | `backend/models/app_session.py` | `app_sessions` | +| `Document` | `backend/models/document.py` | `documents` | +| `ExtractionResult` | `backend/models/extraction.py` | `extraction_results` | +| `EvaluationResult` | `backend/models/evaluation.py` | `evaluation_results` | +| `Group` | `backend/models/group.py` | `groups` | +| `UserGroup` | `backend/models/group.py` | `user_groups` | +| `UserPreferences` | `backend/models/preferences.py` | `user_preferences` | +| `LoginHistory` | `backend/models/preferences.py` | `login_history` | +| `UserPromptTemplate` | `backend/models/preferences.py` | `user_prompt_templates` | +| `TemplateFolder` | `backend/models/template.py` | `template_folders` | +| `PromptTemplate` | `backend/models/template.py` | `prompt_templates` | +| `TemplateVersion` | `backend/models/template.py` | `template_versions` | +| `TemplatePermission` | `backend/models/template.py` | `template_permissions` | +| `EvalJobRecord` | `backend/models/eval_job.py` | `eval_jobs` | + +## Central Pydantic schemas + +| Class | File | Purpose | +| --- | --- | --- | +| `ProcessorType` | `backend/schemas/enums.py` | Parser enum. | +| `ProcessFileRequest` | `backend/schemas/documents.py` | Process file request. | +| `ExtractFigureContentRequest` | `backend/schemas/documents.py` | Figure summary/extraction request. | +| `FigureExtractionResult` | `backend/schemas/documents.py` | Figure extracted content. | +| `FigureMetadata` | `backend/schemas/documents.py` | Figure API metadata. | +| `Entity` | `backend/schemas/extractions.py` | Entity extraction instruction. | +| `ExtractRequest` | `backend/schemas/extractions.py` | Entity extraction request. | +| `EvaluationRequest` | `backend/schemas/evaluations.py` | Single evaluation request. | +| `SingleExtractionEval` | `backend/schemas/evaluations.py` | Batch evaluation item. | +| `BatchEvaluationRequest` | `backend/schemas/evaluations.py` | Batch evaluation request. | +| `CustomMetricRequest` | `backend/schemas/evaluations.py` | Custom metric request. | +| `MetricResult` | `backend/schemas/evaluations.py` | Metric response item. | +| `EvaluationResponse` | `backend/schemas/evaluations.py` | Single evaluation response. | +| `BatchEvaluationResponse` | `backend/schemas/evaluations.py` | Batch evaluation response. | +| `ServerConfig` | `backend/schemas/server.py` | Server/provider config flags. | +| `SessionEntity` | `backend/schemas/sessions.py` | Session entity config. | +| `SessionConfiguration` | `backend/schemas/sessions.py` | Session workflow config. | +| `SessionDocument` | `backend/schemas/sessions.py` | Session document summary. | +| `ExtractionResult` | `backend/schemas/sessions.py` | Session extraction result schema. | +| `SessionMetrics` | `backend/schemas/sessions.py` | Session metric totals. | +| `EvaluationScore` | `backend/schemas/sessions.py` | One evaluation metric score. | +| `EvaluationResult` | `backend/schemas/sessions.py` | Grouped session evaluation result. | +| `Session` | `backend/schemas/sessions.py` | Full session aggregate. | +| `CreateSessionRequest` | `backend/schemas/sessions.py` | Create session request. | +| `UpdateSessionRequest` | `backend/schemas/sessions.py` | Update session request. | +| `SessionSummary` | `backend/schemas/sessions.py` | Session list item. | +| `SessionListResponse` | `backend/schemas/sessions.py` | Session list response. | + +## Router-local schemas + +| Class | File | +| --- | --- | +| `FileUploadResponse` | `backend/api/files/router.py` | +| `UserFileInfo` | `backend/api/files/router.py` | +| `EvalTaskRequest` | `backend/api/evaluations/jobs.py` | +| `ProviderConfigRequest` | `backend/api/evaluations/jobs.py` | +| `SubmitJobRequest` | `backend/api/evaluations/jobs.py` | +| `ShareSessionRequest` | `backend/api/sessions/router.py` | +| `CreateGroupRequest` | `backend/api/groups/router.py` | +| `UpdateGroupRequest` | `backend/api/groups/router.py` | +| `AddMemberRequest` | `backend/api/groups/router.py` | +| `UpdateMemberRoleRequest` | `backend/api/groups/router.py` | +| `GroupResponse` | `backend/api/groups/router.py` | +| `MemberResponse` | `backend/api/groups/router.py` | +| `GroupDetailResponse` | `backend/api/groups/router.py` | +| `UserSearchResult` | `backend/api/groups/router.py` | +| `EntityModel` | `backend/api/templates/router.py` | +| `VariableModel` | `backend/api/templates/router.py` | +| `CreateTemplateRequest` | `backend/api/templates/router.py` | +| `UpdateTemplateRequest` | `backend/api/templates/router.py` | +| `SetImmutableRequest` | `backend/api/templates/router.py` | +| `SetPermissionRequest` | `backend/api/templates/router.py` | +| `ForkTemplateRequest` | `backend/api/templates/router.py` | +| `ChangeScopeRequest` | `backend/api/templates/router.py` | +| `CreateFolderRequest` | `backend/api/templates/router.py` | +| `RenameFolderRequest` | `backend/api/templates/router.py` | +| `FolderResponse` | `backend/api/templates/router.py` | +| `TemplateResponse` | `backend/api/templates/router.py` | +| `VersionResponse` | `backend/api/templates/router.py` | +| `PermissionResponse` | `backend/api/templates/router.py` | +| `BatchMetricsRequest` | `backend/api/server/router.py` | +| `ChatQueryRequest` | `backend/api/chat/router.py` | +| `ParagraphGenerationRequest` | `backend/api/paragraphgenerator.py` | +| `ParagraphEvalGenerateRequest` | `backend/api/paragraph_evaluation.py` | + +## Service classes + +| Class | File | Area | +| --- | --- | --- | +| `SQLAlchemyDBService` | `backend/services/database/sqlalchemy_db_service.py` | persistence | +| `SessionService` | `backend/services/session/session_service.py` | session orchestration | +| `GroupService` | `backend/services/groups/group_service.py` | groups/memberships | +| `TemplateService` | `backend/services/templates/template_service.py` | templates | +| `FolderService` | `backend/services/templates/folder_service.py` | template folders | +| `DocumentService` | `backend/services/document/document_service.py` | document faΓ§ade | +| `OrganizedFileService` | `backend/services/document/organized_file_service.py` | blob-backed file organization | +| `OrganizedDocumentProcessor` | `backend/services/document/organized_processor.py` | organized processing orchestration | +| `FileService` | `backend/services/document/file_service.py` | legacy local file service | +| `BlobStorageClient` | `backend/services/storage/blob_storage.py` | Azure Blob wrapper | +| `AzureDocIntelligenceService` | `backend/services/document/processors/azure_doc_intelligence/azure_doc_intelligence_service.py` | Azure parser | +| `DoclingRemoteClient` | `backend/services/document/processors/docling/docling_remote_client.py` | remote Docling client | +| `DoclingService` | `backend/services/document/processors/docling/docling_service.py` | local Docling parser | +| `_VRAMPeakTracker` | `backend/services/document/processors/docling/docling_service.py` | worker VRAM polling | +| `VRAMStatus` | `backend/services/document/processors/docling/vram_guard.py` | VRAM status data | +| `VRAMGuard` | `backend/services/document/processors/docling/vram_guard.py` | VRAM concurrency guard | +| `LLMService` | `backend/services/llm/llm_service.py` | provider router | +| `AzureLLMClient` | `backend/services/llm/azure.py` | Azure provider | +| `GeminiLLMClient` | `backend/services/llm/gemini.py` | Gemini provider | +| `AnthropicLLMClient` | `backend/services/llm/anthropic.py` | Anthropic Vertex provider | +| `LlamaLLMClient` | `backend/services/llm/llama.py` | Llama MaaS provider | +| `MacbookLLMClient` | `backend/services/llm/macbook.py` | Macbook provider | +| `MacbookRequestQueue` | `backend/services/llm/macbook_queue.py` | Macbook FIFO queue | +| `VLLMClient` | `backend/services/llm/vllm.py` | vLLM provider | +| `EvaluationService` | `backend/services/evaluation/evaluation_service.py` | evaluation orchestration | +| `EvaluationResultStorage` | `backend/services/evaluation/storage/result_storage.py` | JSON result storage | +| `CorrectnessMetricFactory` | `backend/services/evaluation/metrics/correctness.py` | metric factory | +| `CompletenessMetricFactory` | `backend/services/evaluation/metrics/completeness.py` | metric factory | +| `RelevanceMetricFactory` | `backend/services/evaluation/metrics/relevance.py` | metric factory | +| `SafetyMetricFactory` | `backend/services/evaluation/metrics/safety.py` | metric factory | +| `CustomMetricFactory` | `backend/services/evaluation/metrics/custom.py` | metric factory | +| `AzureOpenAIDeepEvalModel` | `backend/services/evaluation/adapters/azure_adapter.py` | evaluation adapter | +| `VertexAIDeepEvalModel` | `backend/services/evaluation/adapters/vertex_adapter.py` | evaluation adapter | +| `AnthropicVertexDeepEvalModel` | `backend/services/evaluation/adapters/anthropic_adapter.py` | evaluation adapter | +| `CallMetric` | `backend/services/telemetry/cost_tracker.py` | telemetry dataclass | +| `BatchMetric` | `backend/services/telemetry/cost_tracker.py` | telemetry dataclass | +| `SessionMetrics` | `backend/services/telemetry/cost_tracker.py` | telemetry dataclass | +| `CostTracker` | `backend/services/telemetry/cost_tracker.py` | cost/session metrics | + +## Evaluation job dataclasses/classes + +| Class | File | Purpose | +| --- | --- | --- | +| `EvalTask` | `backend/services/evaluation/job_queue.py` | One entity output to evaluate. | +| `ProviderConfig` | `backend/services/evaluation/job_queue.py` | Judge provider config. | +| `TaskResult` | `backend/services/evaluation/job_queue.py` | One task/provider result. | +| `EvalJob` | `backend/services/evaluation/job_queue.py` | Background job runtime state. | +| `_JobStatusProxy` | `backend/services/evaluation/job_queue.py` | DB-backed job status snapshot. | + +## Provider structured-output helper schemas + +| Class | File | Purpose | +| --- | --- | --- | +| `MarkdownReference` | `backend/services/llm/azure.py` | Azure structured reference. | +| `ExtractionResult` | `backend/services/llm/azure.py` | Azure structured extraction result. | +| `MarkdownReference` | `backend/services/llm/gemini.py` | Gemini structured reference. | +| `ExtractionResult` | `backend/services/llm/gemini.py` | Gemini structured extraction result. | +| `MarkdownReference` | `backend/services/llm/llama.py` | Llama structured reference. | +| `ExtractionResult` | `backend/services/llm/llama.py` | Llama structured extraction result. | +| `MarkdownReference` | `backend/services/llm/vllm.py` | vLLM structured reference. | +| `ExtractionResult` | `backend/services/llm/vllm.py` | vLLM structured extraction result. | diff --git a/docs/backend/appendices/class-reference.md b/docs/backend/appendices/class-reference.md new file mode 100644 index 0000000..a81fa63 --- /dev/null +++ b/docs/backend/appendices/class-reference.md @@ -0,0 +1,2155 @@ +# Backend Class Reference + +This appendix is a field-oriented reference for classes defined under `backend/`. It complements the compact [class index](class-index.md) and the design documents in the parent folder. + +How to read this document: + +- **ORM models** list SQLAlchemy columns as they are defined in code. The database-focused explanation lives in [../03-data-models.md](../03-data-models.md). +- **Pydantic and API schemas** list request/response fields, types, and defaults. These are the HTTP-facing contracts. +- **Dataclasses and runtime state** list in-memory job, telemetry, and guard fields. +- **Service/provider classes** usually do not expose schema fields, so this reference lists constructor-created instance attributes and public methods. +- **Scripts and utilities** are included because they are defined under `backend/`, but most are developer tooling rather than production request-path classes. +- Fields are direct fields declared on that class. If a class inherits from another schema, inherited fields are documented on the base class entry. + +This file is generated from source-level class definitions and should be updated when fields, schemas, dataclasses, or service constructors change. + +## Summary + +| Category | Class count | +| --- | --- | +| ORM models | 18 | +| Pydantic and API schemas | 66 | +| Dataclasses and runtime state | 12 | +| Enums | 1 | +| Service and provider classes | 34 | +| Utilities | 1 | +| Scripts and developer tools | 7 | +| Other backend classes | 1 | + +## ORM models + +### `AppSession` + +**File:** `backend/models/app_session.py` +**Base classes:** `Base` +**Purpose:** An extraction workflow session. Users create sessions to track + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `id` | `UUID(as_uuid=True)` | `primary_key=True, default=uuid.uuid4` | SQLAlchemy column | +| `user_id` | `String(36)` | `ForeignKey("user.id", ondelete="CASCADE"), nullable=False` | SQLAlchemy column | +| `name` | `Text` | `default="Untitled Session"` | SQLAlchemy column | +| `status` | `Text` | `default="in_progress"` | SQLAlchemy column | +| `last_step` | `Text` | `default="upload"` | SQLAlchemy column | +| `configuration` | `JSONB` | `default=dict` | SQLAlchemy column | +| `evaluation_config` | `JSONB` | `default=dict` | SQLAlchemy column | +| `files_config` | `JSONB` | `default=dict` | SQLAlchemy column | +| `total_cost` | `Float` | `default=0.0` | SQLAlchemy column | +| `total_latency` | `Float` | `default=0.0` | SQLAlchemy column | +| `total_calls` | `Integer` | `default=0` | SQLAlchemy column | +| `shared_with_group_id` | `UUID(as_uuid=True)` | `ForeignKey("groups.id", ondelete="SET NULL"), nullable=True` | SQLAlchemy column | +| `shared_by` | `String(36)` | `ForeignKey("user.id"), nullable=True` | SQLAlchemy column | +| `shared_at` | `DateTime` | `nullable=True` | SQLAlchemy column | +| `created_at` | `DateTime` | `default=datetime.utcnow` | SQLAlchemy column | +| `updated_at` | `DateTime` | `default=datetime.utcnow, onupdate=datetime.utcnow` | SQLAlchemy column | + +### `Document` + +**File:** `backend/models/document.py` +**Base classes:** `Base` +**Purpose:** A document within an extraction session. + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `id` | `UUID(as_uuid=True)` | `primary_key=True, default=uuid.uuid4` | SQLAlchemy column | +| `session_id` | `UUID(as_uuid=True)` | `ForeignKey("app_sessions.id", ondelete="CASCADE"), nullable=True` | SQLAlchemy column | +| `user_id` | `String(36)` | `ForeignKey("user.id", ondelete="CASCADE"), nullable=False` | SQLAlchemy column | +| `file_hash` | `Text` | `nullable=False` | SQLAlchemy column | +| `filename` | `Text` | `nullable=False` | SQLAlchemy column | +| `file_path` | `Text` | `nullable=True` | SQLAlchemy column | +| `study_type` | `Text` | `nullable=True` | SQLAlchemy column | +| `processor_used` | `Text` | `nullable=True` | SQLAlchemy column | +| `processing_status` | `Text` | `default="pending"` | SQLAlchemy column | +| `processing_error` | `Text` | `nullable=True` | SQLAlchemy column | +| `extracted_text_path` | `Text` | `nullable=True` | SQLAlchemy column | +| `processed_at` | `DateTime` | `nullable=True` | SQLAlchemy column | +| `parse_cost` | `Float` | `nullable=True` | SQLAlchemy column | +| `page_count` | `Integer` | `nullable=True` | SQLAlchemy column | +| `parse_duration_seconds` | `Float` | `nullable=True` | SQLAlchemy column | +| `figure_count` | `Integer` | `nullable=True` | SQLAlchemy column | +| `table_count` | `Integer` | `nullable=True` | SQLAlchemy column | +| `created_at` | `DateTime` | `default=datetime.utcnow` | SQLAlchemy column | +| `updated_at` | `DateTime` | `default=datetime.utcnow, onupdate=datetime.utcnow` | SQLAlchemy column | + +### `EvalJobRecord` + +**File:** `backend/models/eval_job.py` +**Base classes:** `Base` +**Purpose:** Persisted snapshot of an EvalJob's status and results. + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `job_id` | `Text` | `primary_key=True` | SQLAlchemy column | +| `session_id` | `Text` | `nullable=True` | SQLAlchemy column | +| `user_id` | `Text` | `nullable=True` | SQLAlchemy column | +| `status` | `Text` | `nullable=False, default="pending"` | SQLAlchemy column | +| `progress` | `Integer` | `nullable=False, default=0` | SQLAlchemy column | +| `total` | `Integer` | `nullable=False, default=0` | SQLAlchemy column | +| `results` | `JSONB` | `nullable=True, default=list` | SQLAlchemy column | +| `errors` | `JSONB` | `nullable=True, default=list` | SQLAlchemy column | +| `error` | `Text` | `nullable=True` | SQLAlchemy column | +| `created_at` | `DateTime(timezone=True)` | `nullable=True` | SQLAlchemy column | +| `completed_at` | `DateTime(timezone=True)` | `nullable=True` | SQLAlchemy column | + +### `EvaluationResult` + +**File:** `backend/models/evaluation.py` +**Base classes:** `Base` +**Purpose:** An evaluation score for an extraction result. + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `id` | `UUID(as_uuid=True)` | `primary_key=True, default=uuid.uuid4` | SQLAlchemy column | +| `extraction_result_id` | `UUID(as_uuid=True)` | `ForeignKey("extraction_results.id", ondelete="CASCADE"), nullable=False` | SQLAlchemy column | +| `metric` | `Text` | `nullable=False` | SQLAlchemy column | +| `score` | `Float` | `nullable=True` | SQLAlchemy column | +| `reasoning` | `Text` | `nullable=True` | SQLAlchemy column | +| `judge_model` | `Text` | `nullable=True` | SQLAlchemy column | +| `human_score` | `Float` | `nullable=True` | SQLAlchemy column | +| `ground_truth` | `Text` | `nullable=True` | SQLAlchemy column | +| `evaluation_cost` | `Float` | `nullable=True` | SQLAlchemy column | +| `evaluation_time` | `Float` | `nullable=True` | SQLAlchemy column | +| `evaluated_at` | `DateTime` | `default=datetime.utcnow` | SQLAlchemy column | +| `created_at` | `DateTime` | `default=datetime.utcnow` | SQLAlchemy column | +| `updated_at` | `DateTime` | `default=datetime.utcnow, onupdate=datetime.utcnow` | SQLAlchemy column | + +### `ExtractionResult` + +**File:** `backend/models/extraction.py` +**Base classes:** `Base` +**Purpose:** An extraction result for a specific entity from a specific model. + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `id` | `UUID(as_uuid=True)` | `primary_key=True, default=uuid.uuid4` | SQLAlchemy column | +| `session_id` | `UUID(as_uuid=True)` | `ForeignKey("app_sessions.id", ondelete="CASCADE"), nullable=False` | SQLAlchemy column | +| `document_id` | `UUID(as_uuid=True)` | `ForeignKey("documents.id", ondelete="CASCADE"), nullable=False` | SQLAlchemy column | +| `entity_name` | `Text` | `nullable=False` | SQLAlchemy column | +| `model_id` | `Text` | `nullable=False` | SQLAlchemy column | +| `extracted_text` | `Text` | `nullable=True` | SQLAlchemy column | +| `bbox_references` | `JSONB` | `nullable=True` | SQLAlchemy column | +| `status` | `Text` | `default="pending"` | SQLAlchemy column | +| `error_message` | `Text` | `nullable=True` | SQLAlchemy column | +| `extracted_at` | `DateTime` | `nullable=True` | SQLAlchemy column | +| `prompt_tokens` | `Integer` | `nullable=True` | SQLAlchemy column | +| `completion_tokens` | `Integer` | `nullable=True` | SQLAlchemy column | +| `duration_ms` | `Integer` | `nullable=True` | SQLAlchemy column | +| `cost` | `Float` | `nullable=True` | SQLAlchemy column | +| `created_at` | `DateTime` | `default=datetime.utcnow` | SQLAlchemy column | +| `updated_at` | `DateTime` | `default=datetime.utcnow, onupdate=datetime.utcnow` | SQLAlchemy column | + +### `Group` + +**File:** `backend/models/group.py` +**Base classes:** `Base` +**Purpose:** A user group for sharing sessions and templates. + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `id` | `UUID(as_uuid=True)` | `primary_key=True, default=uuid.uuid4` | SQLAlchemy column | +| `name` | `Text` | `nullable=False` | SQLAlchemy column | +| `description` | `Text` | `nullable=True` | SQLAlchemy column | +| `created_by` | `String(36)` | `ForeignKey("user.id"), nullable=True` | SQLAlchemy column | +| `created_at` | `DateTime` | `default=datetime.utcnow` | SQLAlchemy column | +| `updated_at` | `DateTime` | `default=datetime.utcnow, onupdate=datetime.utcnow` | SQLAlchemy column | + +### `UserGroup` + +**File:** `backend/models/group.py` +**Base classes:** `Base` +**Purpose:** User-group membership with roles. + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `user_id` | `String(36)` | `ForeignKey("user.id", ondelete="CASCADE"), primary_key=True` | SQLAlchemy column | +| `group_id` | `UUID(as_uuid=True)` | `ForeignKey("groups.id", ondelete="CASCADE"), primary_key=True` | SQLAlchemy column | +| `role` | `Text` | `default="member"` | SQLAlchemy column | +| `joined_at` | `DateTime` | `default=datetime.utcnow` | SQLAlchemy column | + +### `UserPreferences` + +**File:** `backend/models/preferences.py` +**Base classes:** `Base` +**Purpose:** User preferences for default models, temperature, etc. + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `id` | `UUID(as_uuid=True)` | `primary_key=True, default=uuid.uuid4` | SQLAlchemy column | +| `user_id` | `String(36)` | `ForeignKey("user.id", ondelete="CASCADE"), nullable=False, unique=True` | SQLAlchemy column | +| `default_models` | `JSONB` | `default=list` | SQLAlchemy column | +| `default_temperature` | `Float` | `default=0.0` | SQLAlchemy column | +| `settings` | `JSONB` | `default=dict` | SQLAlchemy column | +| `created_at` | `DateTime` | `default=datetime.utcnow` | SQLAlchemy column | +| `updated_at` | `DateTime` | `default=datetime.utcnow, onupdate=datetime.utcnow` | SQLAlchemy column | + +### `LoginHistory` + +**File:** `backend/models/preferences.py` +**Base classes:** `Base` +**Purpose:** Login history for audit trail. + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `id` | `UUID(as_uuid=True)` | `primary_key=True, default=uuid.uuid4` | SQLAlchemy column | +| `user_id` | `String(36)` | `ForeignKey("user.id", ondelete="CASCADE"), nullable=False` | SQLAlchemy column | +| `ip_address` | `Text` | `nullable=True` | SQLAlchemy column | +| `user_agent` | `Text` | `nullable=True` | SQLAlchemy column | +| `login_at` | `DateTime` | `default=datetime.utcnow` | SQLAlchemy column | + +### `UserPromptTemplate` + +**File:** `backend/models/preferences.py` +**Base classes:** `Base` +**Purpose:** Legacy user-scoped prompt templates (simple key-value). + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `id` | `UUID(as_uuid=True)` | `primary_key=True, default=uuid.uuid4` | SQLAlchemy column | +| `user_id` | `String(36)` | `ForeignKey("user.id", ondelete="CASCADE"), nullable=False` | SQLAlchemy column | +| `name` | `Text` | `nullable=False` | SQLAlchemy column | +| `entity_name` | `Text` | `nullable=False` | SQLAlchemy column | +| `prompt_content` | `Text` | `nullable=False` | SQLAlchemy column | +| `study_type` | `Text` | `nullable=True` | SQLAlchemy column | +| `system_prompt` | `Text` | `nullable=True` | SQLAlchemy column | +| `created_at` | `DateTime` | `default=datetime.utcnow` | SQLAlchemy column | +| `updated_at` | `DateTime` | `default=datetime.utcnow, onupdate=datetime.utcnow` | SQLAlchemy column | + +### `TemplateFolder` + +**File:** `backend/models/template.py` +**Base classes:** `Base` +**Purpose:** Folder for organising prompt templates hierarchically. + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `id` | `UUID(as_uuid=True)` | `primary_key=True, default=uuid.uuid4` | SQLAlchemy column | +| `name` | `Text` | `nullable=False` | SQLAlchemy column | +| `scope` | `Text` | `nullable=False, default="user"` | SQLAlchemy column | +| `owner_user_id` | `String(36)` | `ForeignKey("user.id", ondelete="CASCADE"), nullable=True` | SQLAlchemy column | +| `owner_group_id` | `UUID(as_uuid=True)` | `ForeignKey("groups.id", ondelete="CASCADE"), nullable=True` | SQLAlchemy column | +| `parent_id` | `UUID(as_uuid=True)` | `ForeignKey("template_folders.id", ondelete="CASCADE"), nullable=True` | SQLAlchemy column | +| `created_by` | `String(36)` | `ForeignKey("user.id"), nullable=True` | SQLAlchemy column | +| `created_at` | `DateTime` | `default=datetime.utcnow` | SQLAlchemy column | +| `updated_at` | `DateTime` | `default=datetime.utcnow, onupdate=datetime.utcnow` | SQLAlchemy column | + +### `PromptTemplate` + +**File:** `backend/models/template.py` +**Base classes:** `Base` +**Purpose:** A prompt template for entity extraction. + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `id` | `UUID(as_uuid=True)` | `primary_key=True, default=uuid.uuid4` | SQLAlchemy column | +| `name` | `Text` | `nullable=False` | SQLAlchemy column | +| `description` | `Text` | `nullable=True` | SQLAlchemy column | +| `study_type` | `Text` | `nullable=True` | SQLAlchemy column | +| `scope` | `Text` | `nullable=False, default="user"` | SQLAlchemy column | +| `owner_user_id` | `String(36)` | `ForeignKey("user.id", ondelete="CASCADE"), nullable=True` | SQLAlchemy column | +| `owner_group_id` | `UUID(as_uuid=True)` | `ForeignKey("groups.id", ondelete="CASCADE"), nullable=True` | SQLAlchemy column | +| `system_prompt` | `Text` | `nullable=True` | SQLAlchemy column | +| `entities` | `JSONB` | `nullable=False, default=list` | SQLAlchemy column | +| `summary_prompt` | `Text` | `nullable=True` | SQLAlchemy column | +| `variables` | `JSONB` | `default=list` | SQLAlchemy column | +| `is_immutable` | `Boolean` | `default=False` | SQLAlchemy column | +| `tags` | `ARRAY(Text)` | `default=list` | SQLAlchemy column | +| `is_default` | `Boolean` | `default=False` | SQLAlchemy column | +| `version` | `Integer` | `default=1` | SQLAlchemy column | +| `folder_id` | `UUID(as_uuid=True)` | `nullable=True` | SQLAlchemy column | +| `created_by` | `String(36)` | `ForeignKey("user.id"), nullable=True` | SQLAlchemy column | +| `created_at` | `DateTime` | `default=datetime.utcnow` | SQLAlchemy column | +| `updated_at` | `DateTime` | `default=datetime.utcnow, onupdate=datetime.utcnow` | SQLAlchemy column | + +### `TemplateVersion` + +**File:** `backend/models/template.py` +**Base classes:** `Base` +**Purpose:** Version history for prompt templates. + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `id` | `UUID(as_uuid=True)` | `primary_key=True, default=uuid.uuid4` | SQLAlchemy column | +| `template_id` | `UUID(as_uuid=True)` | `ForeignKey("prompt_templates.id", ondelete="CASCADE"), nullable=False` | SQLAlchemy column | +| `version` | `Integer` | `nullable=False` | SQLAlchemy column | +| `system_prompt` | `Text` | `nullable=True` | SQLAlchemy column | +| `entities` | `JSONB` | `nullable=False` | SQLAlchemy column | +| `summary_prompt` | `Text` | `nullable=True` | SQLAlchemy column | +| `variables` | `JSONB` | `nullable=True` | SQLAlchemy column | +| `changed_by` | `String(36)` | `ForeignKey("user.id"), nullable=True` | SQLAlchemy column | +| `change_summary` | `Text` | `nullable=True` | SQLAlchemy column | +| `created_at` | `DateTime` | `default=datetime.utcnow` | SQLAlchemy column | + +### `TemplatePermission` + +**File:** `backend/models/template.py` +**Base classes:** `Base` +**Purpose:** Per-user permission overrides for templates. + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `id` | `UUID(as_uuid=True)` | `primary_key=True, default=uuid.uuid4` | SQLAlchemy column | +| `template_id` | `UUID(as_uuid=True)` | `ForeignKey("prompt_templates.id", ondelete="CASCADE"), nullable=False` | SQLAlchemy column | +| `user_id` | `String(36)` | `ForeignKey("user.id", ondelete="CASCADE"), nullable=False` | SQLAlchemy column | +| `can_read` | `Boolean` | `default=True` | SQLAlchemy column | +| `can_write` | `Boolean` | `default=False` | SQLAlchemy column | +| `granted_by` | `String(36)` | `ForeignKey("user.id"), nullable=True` | SQLAlchemy column | +| `created_at` | `DateTime` | `default=datetime.utcnow` | SQLAlchemy column | + +### `User` + +**File:** `backend/models/user.py` +**Base classes:** `Base` +**Purpose:** Better Auth 'user' table. + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `id` | `String(36)` | `primary_key=True` | SQLAlchemy column | +| `name` | `Text` | `nullable=False` | SQLAlchemy column | +| `email` | `Text` | `nullable=False, unique=True` | SQLAlchemy column | +| `email_verified` | `Boolean` | `default=False, name="emailVerified"` | SQLAlchemy column | +| `image` | `Text` | `nullable=True` | SQLAlchemy column | +| `created_at` | `DateTime` | `default=datetime.utcnow, name="createdAt"` | SQLAlchemy column | +| `updated_at` | `DateTime` | `default=datetime.utcnow, onupdate=datetime.utcnow, name="updatedAt"` | SQLAlchemy column | +| `role` | `Text` | `default="user"` | SQLAlchemy column | +| `is_admin` | `Boolean` | `default=False` | SQLAlchemy column | + +### `AuthSession` + +**File:** `backend/models/user.py` +**Base classes:** `Base` +**Purpose:** Better Auth 'session' table. + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `id` | `String(36)` | `primary_key=True` | SQLAlchemy column | +| `expires_at` | `DateTime` | `nullable=False, name="expiresAt"` | SQLAlchemy column | +| `token` | `Text` | `nullable=False, unique=True` | SQLAlchemy column | +| `created_at` | `DateTime` | `default=datetime.utcnow, name="createdAt"` | SQLAlchemy column | +| `updated_at` | `DateTime` | `default=datetime.utcnow, onupdate=datetime.utcnow, name="updatedAt"` | SQLAlchemy column | +| `ip_address` | `Text` | `nullable=True, name="ipAddress"` | SQLAlchemy column | +| `user_agent` | `Text` | `nullable=True, name="userAgent"` | SQLAlchemy column | +| `user_id` | `String(36)` | `nullable=False, name="userId"` | SQLAlchemy column | + +### `Account` + +**File:** `backend/models/user.py` +**Base classes:** `Base` +**Purpose:** Better Auth 'account' table. + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `id` | `String(36)` | `primary_key=True` | SQLAlchemy column | +| `account_id` | `Text` | `nullable=False, name="accountId"` | SQLAlchemy column | +| `provider_id` | `Text` | `nullable=False, name="providerId"` | SQLAlchemy column | +| `user_id` | `String(36)` | `nullable=False, name="userId"` | SQLAlchemy column | +| `access_token` | `Text` | `nullable=True, name="accessToken"` | SQLAlchemy column | +| `refresh_token` | `Text` | `nullable=True, name="refreshToken"` | SQLAlchemy column | +| `id_token` | `Text` | `nullable=True, name="idToken"` | SQLAlchemy column | +| `access_token_expires_at` | `DateTime` | `nullable=True, name="accessTokenExpiresAt"` | SQLAlchemy column | +| `refresh_token_expires_at` | `DateTime` | `nullable=True, name="refreshTokenExpiresAt"` | SQLAlchemy column | +| `scope` | `Text` | `nullable=True` | SQLAlchemy column | +| `password` | `Text` | `nullable=True` | SQLAlchemy column | +| `created_at` | `DateTime` | `default=datetime.utcnow, name="createdAt"` | SQLAlchemy column | +| `updated_at` | `DateTime` | `default=datetime.utcnow, onupdate=datetime.utcnow, name="updatedAt"` | SQLAlchemy column | + +### `Verification` + +**File:** `backend/models/user.py` +**Base classes:** `Base` +**Purpose:** Better Auth 'verification' table. + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `id` | `String(36)` | `primary_key=True` | SQLAlchemy column | +| `identifier` | `Text` | `nullable=False` | SQLAlchemy column | +| `value` | `Text` | `nullable=False` | SQLAlchemy column | +| `expires_at` | `DateTime` | `nullable=False, name="expiresAt"` | SQLAlchemy column | +| `created_at` | `DateTime` | `default=datetime.utcnow, name="createdAt"` | SQLAlchemy column | +| `updated_at` | `DateTime` | `default=datetime.utcnow, onupdate=datetime.utcnow, name="updatedAt"` | SQLAlchemy column | + +## Pydantic and API schemas + +### `ChatQueryRequest` + +**File:** `backend/api/chat/router.py` +**Base classes:** `BaseModel` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `query` | `str` | | declared field | +| `document_markdown` | `Optional[str]` | `None` | declared field | +| `model_type` | `str` | | declared field | +| `model_id` | `Optional[str]` | `None` | declared field | +| `deployment` | `Optional[str]` | `None` | declared field | +| `api_version` | `Optional[str]` | `None` | declared field | + +### `EvalTaskRequest` + +**File:** `backend/api/evaluations/jobs.py` +**Base classes:** `BaseModel` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `entity_name` | `str` | | declared field | +| `source_model` | `str` | | declared field | +| `actual_output` | `str` | | declared field | +| `extraction_prompt` | `str` | | declared field | +| `expected_output` | `Optional[str]` | `None` | declared field | +| `file_hash` | `Optional[str]` | `None` | declared field | +| `file_id` | `Optional[str]` | `None` | declared field | + +### `ProviderConfigRequest` + +**File:** `backend/api/evaluations/jobs.py` +**Base classes:** `BaseModel` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `provider_id` | `str` | `Field(..., description="e.g. 'azure-gpt4o'")` | Pydantic Field | +| `provider` | `str` | `Field(..., description="'azure_openai' \| 'vertex_ai' \| 'anthropic'")` | Pydantic Field | +| `model_name` | `Optional[str]` | `None` | declared field | +| `deployment` | `Optional[str]` | `None` | declared field | +| `endpoint` | `Optional[str]` | `None` | declared field | +| `api_key` | `Optional[str]` | `None` | declared field | + +### `SubmitJobRequest` + +**File:** `backend/api/evaluations/jobs.py` +**Base classes:** `BaseModel` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `session_id` | `str` | | declared field | +| `tasks` | `List[EvalTaskRequest]` | | declared field | +| `providers` | `List[ProviderConfigRequest]` | | declared field | +| `metrics` | `List[str]` | `Field( default=["correctness", "completeness", "relevance", "safety"] )` | Pydantic Field | +| `custom_evaluation_steps` | `Optional[Dict[str, List[str]]]` | `None` | declared field | +| `threshold` | `float` | `Field(default=0.7, ge=0.0, le=1.0)` | Pydantic Field | + +### `FileUploadResponse` + +**File:** `backend/api/files/router.py` +**Base classes:** `BaseModel` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `success` | `bool` | | declared field | +| `file_hash` | `str` | | declared field | +| `original_filename` | `str` | | declared field | +| `file_size` | `int` | | declared field | +| `is_new` | `bool` | | declared field | +| `deduplicated` | `bool` | | declared field | +| `processed` | `dict` | `{}` | declared field | + +### `UserFileInfo` + +**File:** `backend/api/files/router.py` +**Base classes:** `BaseModel` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `file_hash` | `str` | | declared field | +| `original_filename` | `str` | | declared field | +| `file_size` | `int` | | declared field | +| `mime_type` | `str` | | declared field | +| `created_at` | `str` | | declared field | +| `processed` | `dict` | `{}` | declared field | + +### `CreateGroupRequest` + +**File:** `backend/api/groups/router.py` +**Base classes:** `BaseModel` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `name` | `str` | | declared field | +| `description` | `Optional[str]` | `None` | declared field | + +### `UpdateGroupRequest` + +**File:** `backend/api/groups/router.py` +**Base classes:** `BaseModel` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `name` | `Optional[str]` | `None` | declared field | +| `description` | `Optional[str]` | `None` | declared field | + +### `AddMemberRequest` + +**File:** `backend/api/groups/router.py` +**Base classes:** `BaseModel` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `user_id` | `str` | | declared field | +| `role` | `str` | `"member"` | declared field | + +### `UpdateMemberRoleRequest` + +**File:** `backend/api/groups/router.py` +**Base classes:** `BaseModel` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `role` | `str` | | declared field | + +### `GroupResponse` + +**File:** `backend/api/groups/router.py` +**Base classes:** `BaseModel` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `id` | `str` | | declared field | +| `name` | `str` | | declared field | +| `description` | `Optional[str]` | | declared field | +| `created_by` | `Optional[str]` | | declared field | +| `created_at` | `str` | | declared field | +| `updated_at` | `str` | | declared field | +| `user_role` | `Optional[str]` | `None` | declared field | +| `member_count` | `Optional[int]` | `None` | declared field | + +### `MemberResponse` + +**File:** `backend/api/groups/router.py` +**Base classes:** `BaseModel` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `user_id` | `str` | | declared field | +| `role` | `str` | | declared field | +| `joined_at` | `str` | | declared field | +| `display_name` | `Optional[str]` | `None` | declared field | +| `email` | `Optional[str]` | `None` | declared field | +| `avatar_url` | `Optional[str]` | `None` | declared field | + +### `GroupDetailResponse` + +**File:** `backend/api/groups/router.py` +**Base classes:** `GroupResponse` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `members` | `List[MemberResponse]` | `[]` | declared field | + +### `UserSearchResult` + +**File:** `backend/api/groups/router.py` +**Base classes:** `BaseModel` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `user_id` | `str` | | declared field | +| `display_name` | `Optional[str]` | `None` | declared field | +| `email` | `Optional[str]` | `None` | declared field | +| `avatar_url` | `Optional[str]` | `None` | declared field | + +### `ParagraphEvalGenerateRequest` + +**File:** `backend/api/paragraph_evaluation.py` +**Base classes:** `BaseModel` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `session_id` | `str` | | declared field | +| `file_hash` | `str` | | declared field | +| `user_id` | `Optional[str]` | `None` | declared field | +| `entity_order` | `Optional[List[str]]` | `None` | declared field | + +### `ParagraphGenerationRequest` + +**File:** `backend/api/paragraphgenerator.py` +**Base classes:** `BaseModel` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `entities` | `List[Dict]` | | declared field | +| `summary_prompt` | `str` | | declared field | +| `session_id` | `Optional[str]` | `None` | declared field | +| `file_hash` | `Optional[str]` | `None` | declared field | +| `system_prompt` | `Optional[str]` | `None` | declared field | +| `model_type` | `Optional[str]` | `"azure"` | declared field | +| `model_id` | `Optional[str]` | `None` | declared field | +| `deployment` | `Optional[str]` | `None` | declared field | +| `api_version` | `Optional[str]` | `None` | declared field | +| `azure_endpoint` | `Optional[str]` | `None` | declared field | +| `azure_api_key` | `Optional[str]` | `None` | declared field | +| `gemini_api_key` | `Optional[str]` | `None` | declared field | +| `gemini_project_id` | `Optional[str]` | `None` | declared field | +| `gemini_location` | `Optional[str]` | `None` | declared field | +| `max_tokens` | `int` | `8048` | declared field | +| `temperature` | `Optional[float]` | `None` | declared field | + +### `BatchMetricsRequest` + +**File:** `backend/api/server/router.py` +**Base classes:** `BaseModel` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `session_id` | `str` | | declared field | +| `batch_number` | `int` | | declared field | +| `batch_latency` | `float` | | declared field | +| `document_count` | `int` | | declared field | + +### `ShareSessionRequest` + +**File:** `backend/api/sessions/router.py` +**Base classes:** `BaseModel` +**Purpose:** Request to share a session with a group + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `group_id` | `str` | | declared field | + +### `EntityModel` + +**File:** `backend/api/templates/router.py` +**Base classes:** `BaseModel` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `name` | `str` | | declared field | +| `prompt` | `str` | | declared field | + +### `VariableModel` + +**File:** `backend/api/templates/router.py` +**Base classes:** `BaseModel` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `name` | `str` | | declared field | +| `description` | `Optional[str]` | `None` | declared field | +| `default` | `Optional[str]` | `None` | declared field | + +### `CreateTemplateRequest` + +**File:** `backend/api/templates/router.py` +**Base classes:** `BaseModel` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `name` | `str` | | declared field | +| `entities` | `List[EntityModel]` | | declared field | +| `scope` | `str` | `"user"` | declared field | +| `owner_group_id` | `Optional[str]` | `None` | declared field | +| `description` | `Optional[str]` | `None` | declared field | +| `study_type` | `Optional[str]` | `None` | declared field | +| `system_prompt` | `Optional[str]` | `None` | declared field | +| `summary_prompt` | `Optional[str]` | `None` | declared field | +| `variables` | `Optional[List[VariableModel]]` | `None` | declared field | +| `tags` | `Optional[List[str]]` | `None` | declared field | +| `is_immutable` | `bool` | `False` | declared field | +| `folder_id` | `Optional[str]` | `None` | declared field | + +### `UpdateTemplateRequest` + +**File:** `backend/api/templates/router.py` +**Base classes:** `BaseModel` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `name` | `Optional[str]` | `None` | declared field | +| `description` | `Optional[str]` | `None` | declared field | +| `study_type` | `Optional[str]` | `None` | declared field | +| `system_prompt` | `Optional[str]` | `None` | declared field | +| `entities` | `Optional[List[EntityModel]]` | `None` | declared field | +| `summary_prompt` | `Optional[str]` | `None` | declared field | +| `variables` | `Optional[List[VariableModel]]` | `None` | declared field | +| `tags` | `Optional[List[str]]` | `None` | declared field | +| `is_immutable` | `Optional[bool]` | `None` | declared field | +| `change_summary` | `Optional[str]` | `None` | declared field | +| `folder_id` | `Optional[str]` | `None` | declared field | + +### `SetImmutableRequest` + +**File:** `backend/api/templates/router.py` +**Base classes:** `BaseModel` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `is_immutable` | `bool` | | declared field | + +### `SetPermissionRequest` + +**File:** `backend/api/templates/router.py` +**Base classes:** `BaseModel` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `user_id` | `str` | | declared field | +| `can_read` | `bool` | `True` | declared field | +| `can_write` | `bool` | `False` | declared field | + +### `ForkTemplateRequest` + +**File:** `backend/api/templates/router.py` +**Base classes:** `BaseModel` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `new_name` | `Optional[str]` | `None` | declared field | + +### `ChangeScopeRequest` + +**File:** `backend/api/templates/router.py` +**Base classes:** `BaseModel` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `new_scope` | `str` | | declared field | +| `owner_group_id` | `Optional[str]` | `None` | declared field | + +### `CreateFolderRequest` + +**File:** `backend/api/templates/router.py` +**Base classes:** `BaseModel` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `name` | `str` | | declared field | +| `scope` | `str` | | declared field | +| `parent_id` | `Optional[str]` | `None` | declared field | +| `owner_group_id` | `Optional[str]` | `None` | declared field | + +### `RenameFolderRequest` + +**File:** `backend/api/templates/router.py` +**Base classes:** `BaseModel` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `name` | `str` | | declared field | + +### `FolderResponse` + +**File:** `backend/api/templates/router.py` +**Base classes:** `BaseModel` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `id` | `str` | | declared field | +| `name` | `str` | | declared field | +| `scope` | `str` | | declared field | +| `owner_user_id` | `Optional[str]` | | declared field | +| `owner_group_id` | `Optional[str]` | | declared field | +| `parent_id` | `Optional[str]` | | declared field | +| `created_by` | `Optional[str]` | | declared field | +| `created_at` | `str` | | declared field | + +### `TemplateResponse` + +**File:** `backend/api/templates/router.py` +**Base classes:** `BaseModel` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `id` | `str` | | declared field | +| `name` | `str` | | declared field | +| `description` | `Optional[str]` | | declared field | +| `study_type` | `Optional[str]` | | declared field | +| `scope` | `str` | | declared field | +| `owner_user_id` | `Optional[str]` | | declared field | +| `owner_group_id` | `Optional[str]` | | declared field | +| `system_prompt` | `Optional[str]` | | declared field | +| `entities` | `List[Any]` | | declared field | +| `summary_prompt` | `Optional[str]` | | declared field | +| `variables` | `Optional[List[Any]]` | | declared field | +| `tags` | `Optional[List[str]]` | | declared field | +| `is_immutable` | `bool` | | declared field | +| `version` | `int` | | declared field | +| `created_by` | `Optional[str]` | | declared field | +| `created_at` | `str` | | declared field | +| `updated_at` | `str` | | declared field | +| `can_edit` | `Optional[bool]` | `None` | declared field | +| `is_owner` | `Optional[bool]` | `None` | declared field | +| `group_name` | `Optional[str]` | `None` | declared field | +| `folder_id` | `Optional[str]` | `None` | declared field | + +### `VersionResponse` + +**File:** `backend/api/templates/router.py` +**Base classes:** `BaseModel` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `id` | `str` | | declared field | +| `template_id` | `str` | | declared field | +| `version` | `int` | | declared field | +| `system_prompt` | `Optional[str]` | | declared field | +| `entities` | `List[Any]` | | declared field | +| `summary_prompt` | `Optional[str]` | | declared field | +| `variables` | `Optional[List[Any]]` | | declared field | +| `changed_by` | `Optional[str]` | | declared field | +| `change_summary` | `Optional[str]` | | declared field | +| `created_at` | `str` | | declared field | + +### `PermissionResponse` + +**File:** `backend/api/templates/router.py` +**Base classes:** `BaseModel` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `id` | `str` | | declared field | +| `template_id` | `str` | | declared field | +| `user_id` | `str` | | declared field | +| `can_read` | `bool` | | declared field | +| `can_write` | `bool` | | declared field | +| `granted_by` | `Optional[str]` | | declared field | +| `created_at` | `str` | | declared field | + +### `ProcessFileRequest` + +**File:** `backend/schemas/documents.py` +**Base classes:** `BaseModel` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `processor` | `Optional[ProcessorType]` | `ProcessorType.AUTO` | declared field | +| `extract_figures` | `bool` | `Field( default=True, description="Extract figures/charts from document (Azure Document Intelligence only)", )` | Pydantic Field | +| `batch_number` | `Optional[int]` | `Field( default=None, description="Logical batch identifier (1–99) assigned by the frontend for grouped uploads", )` | Pydantic Field | + +### `ExtractFigureContentRequest` + +**File:** `backend/schemas/documents.py` +**Base classes:** `BaseModel` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `model_type` | `str` | `Field( default="gemini", description="LLM model type for OCR extraction (gemini, azure)", )` | Pydantic Field | +| `model_id` | `Optional[str]` | `Field( default=None, description="Specific model ID to use" )` | Pydantic Field | +| `extraction_prompt` | `str` | `Field( default="Extract all textual content, data points, axis labels, legends, and any other readable information from this scientific figure or chart. Include numerical values...` | Pydantic Field | +| `max_tokens` | `int` | `Field(default=2048, description="Maximum tokens in the response")` | Pydantic Field | +| `temperature` | `float` | `Field( default=0.0, description="Sampling temperature for extraction" )` | Pydantic Field | +| `system_message` | `Optional[str]` | `Field( default=None, description="Custom system message for the model" )` | Pydantic Field | + +### `FigureExtractionResult` + +**File:** `backend/schemas/documents.py` +**Base classes:** `BaseModel` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `content` | `str` | `Field(description="Extracted textual content from the figure")` | Pydantic Field | +| `model_used` | `str` | `Field(description="Model that was used for extraction")` | Pydantic Field | +| `timestamp` | `str` | `Field(description="ISO timestamp of extraction")` | Pydantic Field | +| `duration` | `float` | `Field(description="Processing time in seconds")` | Pydantic Field | + +### `FigureMetadata` + +**File:** `backend/schemas/documents.py` +**Base classes:** `BaseModel` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `id` | `str` | `Field(description="Figure identifier")` | Pydantic Field | +| `page` | `Optional[int]` | `Field( default=None, description="Page number where figure appears" )` | Pydantic Field | +| `caption` | `Optional[str]` | `Field( default=None, description="Figure caption if available" )` | Pydantic Field | +| `image_path` | `Optional[str]` | `Field( default=None, description="Path to figure image file" )` | Pydantic Field | +| `bounding_regions` | `Optional[list]` | `Field( default=None, description="Figure bounding regions" )` | Pydantic Field | +| `extracted_content` | `Optional[FigureExtractionResult]` | `Field( default=None, description="OCR extraction results if available" )` | Pydantic Field | + +### `EvaluationRequest` + +**File:** `backend/schemas/evaluations.py` +**Base classes:** `BaseModel` +**Purpose:** Request schema for evaluating a single entity extraction + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `entity_name` | `str` | `Field(..., description="Name of the entity being extracted")` | Pydantic Field | +| `extraction_prompt` | `str` | `Field(..., description="Prompt used for extraction")` | Pydantic Field | +| `actual_output` | `str` | `Field(..., description="The actual extracted output")` | Pydantic Field | +| `expected_output` | `Optional[str]` | `Field( None, description="Expected/ground truth output (required for correctness/completeness)", )` | Pydantic Field | +| `metrics` | `Optional[List[str]]` | `Field( default=["all"], description="List of metrics to use: 'correctness', 'completeness', 'relevance', 'safety', or 'all'", )` | Pydantic Field | +| `provider` | `str` | `Field( default="azure_openai", description="LLM provider for evaluation: 'azure_openai' or 'vertex_ai'", )` | Pydantic Field | +| `threshold` | `float` | `Field( default=0.5, ge=0.0, le=1.0, description="Score threshold for passing" )` | Pydantic Field | +| `strict_mode` | `bool` | `Field( default=False, description="If True, only perfect scores pass" )` | Pydantic Field | +| `custom_evaluation_steps` | `Optional[Dict[str, List[str]]]` | `Field( None, description="Custom evaluation steps for each metric (e.g., {'correctness': ['step1', 'step2']})", )` | Pydantic Field | +| `azure_deployment` | `Optional[str]` | `Field( None, description="Azure OpenAI deployment name" )` | Pydantic Field | +| `azure_endpoint` | `Optional[str]` | `Field(None, description="Azure OpenAI endpoint")` | Pydantic Field | +| `azure_api_key` | `Optional[str]` | `Field(None, description="Azure OpenAI API key")` | Pydantic Field | +| `azure_model_name` | `Optional[str]` | `Field(None, description="Azure OpenAI model name")` | Pydantic Field | +| `vertex_model_name` | `Optional[str]` | `Field( default="gemini-2.5-flash", description="Vertex AI model name" )` | Pydantic Field | +| `vertex_project` | `Optional[str]` | `Field(None, description="GCP project ID")` | Pydantic Field | +| `vertex_location` | `Optional[str]` | `Field( default="us-central1", description="GCP location" )` | Pydantic Field | +| `model_name` | `Optional[str]` | `Field( None, description="Model name for Anthropic providers" )` | Pydantic Field | + +### `SingleExtractionEval` + +**File:** `backend/schemas/evaluations.py` +**Base classes:** `BaseModel` +**Purpose:** Schema for a single extraction in batch evaluation + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `entity_name` | `str` | | declared field | +| `extraction_prompt` | `str` | | declared field | +| `actual_output` | `str` | | declared field | +| `expected_output` | `Optional[str]` | `None` | declared field | + +### `BatchEvaluationRequest` + +**File:** `backend/schemas/evaluations.py` +**Base classes:** `BaseModel` +**Purpose:** Request schema for batch evaluation of multiple extractions + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `extractions` | `List[SingleExtractionEval]` | `Field( ..., description="List of extractions to evaluate" )` | Pydantic Field | +| `metrics` | `Optional[List[str]]` | `Field( default=["all"], description="List of metrics to use" )` | Pydantic Field | +| `custom_evaluation_steps` | `Optional[Dict[str, List[str]]]` | `Field( None, description="Custom evaluation steps for each metric (e.g., {'correctness': ['step1', 'step2']})", )` | Pydantic Field | +| `provider` | `str` | `Field( default="azure_openai", description="LLM provider for evaluation" )` | Pydantic Field | +| `threshold` | `float` | `Field(default=0.5, ge=0.0, le=1.0)` | Pydantic Field | +| `strict_mode` | `bool` | `Field(default=False)` | Pydantic Field | +| `azure_deployment` | `Optional[str]` | `None` | declared field | +| `azure_endpoint` | `Optional[str]` | `None` | declared field | +| `azure_api_key` | `Optional[str]` | `None` | declared field | +| `azure_model_name` | `Optional[str]` | `None` | declared field | +| `vertex_model_name` | `Optional[str]` | `"gemini-2.5-flash"` | declared field | +| `vertex_project` | `Optional[str]` | `None` | declared field | +| `vertex_location` | `Optional[str]` | `"us-central1"` | declared field | +| `model_name` | `Optional[str]` | `None` | declared field | + +### `CustomMetricRequest` + +**File:** `backend/schemas/evaluations.py` +**Base classes:** `BaseModel` +**Purpose:** Request schema for creating and running a custom G-Eval metric + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `metric_name` | `str` | `Field(..., description="Name of the custom metric")` | Pydantic Field | +| `evaluation_steps` | `List[str]` | `Field( ..., description="List of evaluation steps for the metric" )` | Pydantic Field | +| `entity_name` | `str` | `Field(..., description="Name of the entity being extracted")` | Pydantic Field | +| `extraction_prompt` | `str` | `Field(..., description="Prompt used for extraction")` | Pydantic Field | +| `actual_output` | `str` | `Field(..., description="The actual extracted output")` | Pydantic Field | +| `expected_output` | `Optional[str]` | `None` | declared field | +| `provider` | `str` | `Field(default="azure_openai")` | Pydantic Field | +| `threshold` | `float` | `Field(default=0.5, ge=0.0, le=1.0)` | Pydantic Field | +| `strict_mode` | `bool` | `Field(default=False)` | Pydantic Field | +| `azure_deployment` | `Optional[str]` | `None` | declared field | +| `azure_endpoint` | `Optional[str]` | `None` | declared field | +| `azure_api_key` | `Optional[str]` | `None` | declared field | +| `azure_model_name` | `Optional[str]` | `None` | declared field | +| `vertex_model_name` | `Optional[str]` | `"gemini-2.5-flash"` | declared field | +| `vertex_project` | `Optional[str]` | `None` | declared field | +| `vertex_location` | `Optional[str]` | `"us-central1"` | declared field | + +### `MetricResult` + +**File:** `backend/schemas/evaluations.py` +**Base classes:** `BaseModel` +**Purpose:** Schema for a single metric evaluation result + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `metric_name` | `str` | | declared field | +| `score` | `float` | | declared field | +| `threshold` | `float` | | declared field | +| `success` | `bool` | | declared field | +| `reason` | `str` | | declared field | + +### `EvaluationResponse` + +**File:** `backend/schemas/evaluations.py` +**Base classes:** `BaseModel` +**Purpose:** Response schema for evaluation results + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `evaluation_id` | `str` | | declared field | +| `entity_name` | `str` | | declared field | +| `provider` | `str` | | declared field | +| `model` | `str` | | declared field | +| `timestamp` | `str` | | declared field | +| `evaluation_time` | `float` | | declared field | +| `test_case` | `Dict[str, Any]` | | declared field | +| `metrics` | `List[MetricResult]` | | declared field | +| `aggregate_score` | `float` | | declared field | +| `all_passed` | `bool` | | declared field | +| `threshold` | `float` | | declared field | +| `strict_mode` | `bool` | | declared field | +| `status` | `str` | | declared field | +| `error` | `Optional[str]` | `None` | declared field | + +### `BatchEvaluationResponse` + +**File:** `backend/schemas/evaluations.py` +**Base classes:** `BaseModel` +**Purpose:** Response schema for batch evaluation results + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `batch_id` | `str` | | declared field | +| `timestamp` | `str` | | declared field | +| `batch_time` | `float` | | declared field | +| `total_evaluations` | `int` | | declared field | +| `successful_evaluations` | `int` | | declared field | +| `failed_evaluations` | `int` | | declared field | +| `avg_aggregate_score` | `float` | | declared field | +| `all_passed` | `bool` | | declared field | +| `threshold` | `float` | | declared field | +| `provider` | `str` | | declared field | +| `results` | `List[Dict[str, Any]]` | | declared field | + +### `Entity` + +**File:** `backend/schemas/extractions.py` +**Base classes:** `BaseModel` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `name` | `str` | | declared field | +| `prompt` | `str` | | declared field | +| `extracted` | `Optional[str]` | `None` | declared field | +| `system_prompt` | `Optional[str]` | `None` | declared field | + +### `ExtractRequest` + +**File:** `backend/schemas/extractions.py` +**Base classes:** `BaseModel` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `conversion_id` | `str` | | declared field | +| `session_id` | `Optional[str]` | `None` | declared field | +| `deployment` | `Optional[str]` | `None` | declared field | +| `entities` | `List[Entity]` | | declared field | +| `api_version` | `Optional[str]` | `None` | declared field | +| `azure_endpoint` | `Optional[str]` | `None` | declared field | +| `azure_api_key` | `Optional[str]` | `None` | declared field | +| `gemini_api_key` | `Optional[str]` | `None` | declared field | +| `gemini_project_id` | `Optional[str]` | `None` | declared field | +| `gemini_location` | `Optional[str]` | `None` | declared field | +| `max_tokens` | `int` | `8024` | declared field | +| `temperature` | `float` | `0.0` | declared field | +| `model_type` | `Optional[str]` | `"azure"` | declared field | +| `model_id` | `Optional[str]` | `None` | declared field | +| `processor_used` | `Optional[str]` | `None` | declared field | + +### `ServerConfig` + +**File:** `backend/schemas/server.py` +**Base classes:** `BaseModel` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `is_azure_openai_configured` | `bool` | | declared field | +| `is_gemini_configured` | `bool` | `False` | declared field | +| `is_azure_document_intelligence_configured` | `bool` | `False` | declared field | +| `is_llama_configured` | `bool` | `False` | declared field | +| `is_macbook_configured` | `bool` | `False` | declared field | +| `is_macbook_healthy` | `bool` | `False` | declared field | + +### `SessionEntity` + +**File:** `backend/schemas/sessions.py` +**Base classes:** `BaseModel` +**Purpose:** Entity configuration within a session + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `name` | `str` | | declared field | +| `prompt` | `str` | | declared field | +| `system_prompt` | `Optional[str]` | `None` | declared field | + +### `SessionConfiguration` + +**File:** `backend/schemas/sessions.py` +**Base classes:** `BaseModel` +**Purpose:** Configuration snapshot for a session + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `study_type` | `Optional[str]` | `None` | declared field | +| `selected_models` | `List[str]` | `Field(default_factory=list)` | Pydantic Field | +| `entities` | `List[SessionEntity]` | `Field(default_factory=list)` | Pydantic Field | +| `summary_prompt` | `Optional[str]` | `None` | declared field | +| `paragraph_system_prompt` | `Optional[str]` | `None` | declared field | +| `temperature` | `float` | `0.0` | declared field | +| `model_temperatures` | `Optional[Dict[str, float]]` | `Field( default_factory=dict )` | Pydantic Field | +| `files_config` | `Optional[Dict[str, Any]]` | `Field( default_factory=dict )` | Pydantic Field | +| `evaluation_config` | `Optional[Dict[str, Any]]` | `Field( default_factory=dict )` | Pydantic Field | + +### `SessionDocument` + +**File:** `backend/schemas/sessions.py` +**Base classes:** `BaseModel` +**Purpose:** Document reference within a session + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `file_hash` | `str` | | declared field | +| `filename` | `str` | | declared field | +| `id` | `Optional[str]` | `None` | declared field | +| `processor_used` | `Optional[str]` | `None` | declared field | +| `parse_cost` | `Optional[float]` | `None` | declared field | +| `page_count` | `Optional[int]` | `None` | declared field | +| `parse_duration_seconds` | `Optional[float]` | `None` | declared field | +| `figure_count` | `Optional[int]` | `None` | declared field | +| `table_count` | `Optional[int]` | `None` | declared field | + +### `ExtractionResult` + +**File:** `backend/schemas/sessions.py` +**Base classes:** `BaseModel` +**Purpose:** Extraction result for a single entity + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `entity_name` | `str` | | declared field | +| `model_id` | `str` | | declared field | +| `document_id` | `Optional[str]` | `None` | declared field | +| `extracted_text` | `Optional[str]` | `None` | declared field | +| `references` | `Optional[List[Dict[str, Any]]]` | `None` | declared field | +| `status` | `Literal["pending", "completed", "error"]` | `"pending"` | declared field | +| `error_message` | `Optional[str]` | `None` | declared field | +| `extracted_at` | `Optional[datetime]` | `None` | declared field | +| `file_hash` | `Optional[str]` | `None` | declared field | +| `prompt_tokens` | `Optional[int]` | `None` | declared field | +| `completion_tokens` | `Optional[int]` | `None` | declared field | +| `duration_ms` | `Optional[int]` | `None` | declared field | +| `cost` | `Optional[float]` | `None` | declared field | + +### `SessionMetrics` + +**File:** `backend/schemas/sessions.py` +**Base classes:** `BaseModel` +**Purpose:** Aggregated session metrics (stored in sessions table) + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `total_cost` | `float` | `0.0` | declared field | +| `total_latency` | `float` | `0.0` | declared field | +| `total_calls` | `int` | `0` | declared field | + +### `EvaluationScore` + +**File:** `backend/schemas/sessions.py` +**Base classes:** `BaseModel` +**Purpose:** Evaluation score from a judge + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `metric` | `str` | | declared field | +| `score` | `Optional[float]` | `None` | declared field | +| `reasoning` | `Optional[str]` | `None` | declared field | +| `judge_model` | `Optional[str]` | `None` | declared field | +| `human_score` | `Optional[float]` | `None` | declared field | +| `evaluation_cost` | `Optional[float]` | `None` | declared field | +| `evaluation_time` | `Optional[float]` | `None` | declared field | + +### `EvaluationResult` + +**File:** `backend/schemas/sessions.py` +**Base classes:** `BaseModel` +**Purpose:** Evaluation result for an extraction + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `document_id` | `Optional[str]` | `None` | declared field | +| `file_hash` | `Optional[str]` | `None` | declared field | +| `entity_name` | `str` | | declared field | +| `model_id` | `str` | | declared field | +| `ground_truth` | `Optional[str]` | `None` | declared field | +| `scores` | `List[EvaluationScore]` | `Field(default_factory=list)` | Pydantic Field | +| `human_score` | `Optional[float]` | `None` | declared field | +| `evaluated_at` | `Optional[datetime]` | `None` | declared field | +| `evaluation_cost` | `Optional[float]` | `None` | declared field | +| `evaluation_time` | `Optional[float]` | `None` | declared field | + +### `Session` + +**File:** `backend/schemas/sessions.py` +**Base classes:** `BaseModel` +**Purpose:** Full session model with configuration, results, and evaluations + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `session_id` | `str` | `Field(default_factory=lambda: str(uuid.uuid4()))` | Pydantic Field | +| `user_id` | `str` | | declared field | +| `name` | `str` | `"Untitled Session"` | declared field | +| `status` | `Literal["in_progress", "completed"]` | `"in_progress"` | declared field | +| `last_step` | `Optional[str]` | `"upload"` | declared field | +| `evaluation_config` | `Optional[Dict[str, Any]]` | `Field(default_factory=dict)` | Pydantic Field | +| `files_config` | `Optional[Dict[str, Any]]` | `Field(default_factory=dict)` | Pydantic Field | +| `created_at` | `datetime` | `Field(default_factory=datetime.utcnow)` | Pydantic Field | +| `updated_at` | `datetime` | `Field(default_factory=datetime.utcnow)` | Pydantic Field | +| `configuration` | `SessionConfiguration` | `Field(default_factory=SessionConfiguration)` | Pydantic Field | +| `documents` | `List[SessionDocument]` | `Field(default_factory=list)` | Pydantic Field | +| `extraction_results` | `List[ExtractionResult]` | `Field(default_factory=list)` | Pydantic Field | +| `evaluation_results` | `List[EvaluationResult]` | `Field(default_factory=list)` | Pydantic Field | +| `session_metrics` | `Optional[SessionMetrics]` | `None` | declared field | + +### `CreateSessionRequest` + +**File:** `backend/schemas/sessions.py` +**Base classes:** `BaseModel` +**Purpose:** Request to create a new session + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `user_id` | `str` | | declared field | +| `name` | `Optional[str]` | `"Untitled Session"` | declared field | +| `last_step` | `Optional[str]` | `"upload"` | declared field | +| `configuration` | `Optional[SessionConfiguration]` | `None` | declared field | +| `evaluation_config` | `Optional[Dict[str, Any]]` | `None` | declared field | +| `files_config` | `Optional[Dict[str, Any]]` | `None` | declared field | +| `documents` | `Optional[List[SessionDocument]]` | `None` | declared field | + +### `UpdateSessionRequest` + +**File:** `backend/schemas/sessions.py` +**Base classes:** `BaseModel` +**Purpose:** Request to update an existing session + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `user_id` | `Optional[str]` | `None` | declared field | +| `name` | `Optional[str]` | `None` | declared field | +| `status` | `Optional[Literal["in_progress", "completed"]]` | `None` | declared field | +| `last_step` | `Optional[str]` | `None` | declared field | +| `configuration` | `Optional[SessionConfiguration]` | `None` | declared field | +| `evaluation_config` | `Optional[Dict[str, Any]]` | `None` | declared field | +| `files_config` | `Optional[Dict[str, Any]]` | `None` | declared field | +| `documents` | `Optional[List[SessionDocument]]` | `None` | declared field | +| `extraction_results` | `Optional[List[ExtractionResult]]` | `None` | declared field | +| `evaluation_results` | `Optional[List[EvaluationResult]]` | `None` | declared field | + +### `SessionSummary` + +**File:** `backend/schemas/sessions.py` +**Base classes:** `BaseModel` +**Purpose:** Lightweight session summary for list views + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `session_id` | `str` | | declared field | +| `name` | `str` | | declared field | +| `status` | `Literal["in_progress", "completed"]` | | declared field | +| `created_at` | `datetime` | | declared field | +| `updated_at` | `datetime` | | declared field | +| `last_step` | `Optional[str]` | `None` | declared field | +| `study_type` | `Optional[str]` | `None` | declared field | +| `document_count` | `int` | | declared field | +| `document_names` | `List[str]` | `Field(default_factory=list)` | Pydantic Field | +| `extraction_count` | `int` | | declared field | +| `evaluation_count` | `int` | | declared field | +| `shared_by_name` | `Optional[str]` | `None` | declared field | +| `shared_group_name` | `Optional[str]` | `None` | declared field | +| `shared_at` | `Optional[datetime]` | `None` | declared field | +| `owner_user_id` | `Optional[str]` | `None` | declared field | + +### `SessionListResponse` + +**File:** `backend/schemas/sessions.py` +**Base classes:** `BaseModel` +**Purpose:** Response for listing sessions + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `sessions` | `List[SessionSummary]` | | declared field | +| `total` | `int` | | declared field | + +### `MarkdownReference` + +**File:** `backend/services/llm/azure.py` +**Base classes:** `BaseModel` +**Purpose:** A reference to a specific section of the markdown that was used + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `text` | `str` | `Field( description="The exact text excerpt from the markdown that was referenced" )` | Pydantic Field | + +### `ExtractionResult` + +**File:** `backend/services/llm/azure.py` +**Base classes:** `BaseModel` +**Purpose:** Structured result containing both the extracted answer and its references + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `answer` | `str` | `Field( description="The extracted information or answer based on the prompt" )` | Pydantic Field | +| `references` | `List[MarkdownReference]` | `Field( description="List of specific text excerpts from the markdown that were used to generate this answer" )` | Pydantic Field | + +### `MarkdownReference` + +**File:** `backend/services/llm/gemini.py` +**Base classes:** `BaseModel` +**Purpose:** A reference to a specific section of the markdown that was used + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `text` | `str` | `Field( description="The exact text excerpt from the markdown that was referenced" )` | Pydantic Field | + +### `ExtractionResult` + +**File:** `backend/services/llm/gemini.py` +**Base classes:** `BaseModel` +**Purpose:** Structured result containing both the extracted answer and its references + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `answer` | `str` | `Field( description="The extracted information or answer based on the prompt" )` | Pydantic Field | +| `references` | `List[MarkdownReference]` | `Field( description="List of specific text excerpts from the markdown that were used to generate this answer" )` | Pydantic Field | + +### `MarkdownReference` + +**File:** `backend/services/llm/llama.py` +**Base classes:** `BaseModel` +**Purpose:** A reference to a specific section of the markdown that was used + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `text` | `str` | `Field( description="The exact text excerpt from the markdown that was referenced" )` | Pydantic Field | + +### `ExtractionResult` + +**File:** `backend/services/llm/llama.py` +**Base classes:** `BaseModel` +**Purpose:** Structured result containing both the extracted answer and its references + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `answer` | `Union[str, Dict[str, Any], List[Any]]` | `Field( description="The extracted information or answer based on the prompt (string, structured data, or list)" )` | Pydantic Field | +| `references` | `List[MarkdownReference]` | `Field( description="List of specific text excerpts from the markdown that were used to generate this answer" )` | Pydantic Field | + +### `MarkdownReference` + +**File:** `backend/services/llm/vllm.py` +**Base classes:** `BaseModel` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `text` | `str` | `Field(description="Exact text excerpt from the markdown")` | Pydantic Field | + +### `ExtractionResult` + +**File:** `backend/services/llm/vllm.py` +**Base classes:** `BaseModel` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `answer` | `str` | `Field(description="Extracted information or answer")` | Pydantic Field | +| `references` | `List[MarkdownReference]` | `Field( description="Text excerpts used to generate the answer" )` | Pydantic Field | + +## Dataclasses and runtime state + +### `Config` + +**File:** `backend/scripts/evaluate_macbook_models.py` +**Decorators:** `dataclass` +**Purpose:** Configuration settings for the evaluation script. + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `base_url` | `str` | `"http://macbook1.sciencegpt.ca"` | declared field | +| `api_endpoint` | `str` | `"/api/generate"` | declared field | +| `delay_between_requests` | `int` | `60` | declared field | +| `max_tokens` | `int` | `4096` | declared field | +| `temperature` | `float` | `0.0` | declared field | +| `request_timeout` | `int` | `600` | declared field | +| `project_root` | `Path` | `Path(__file__).resolve().parents[2]` | declared field | +| `model_list_file` | `Path` | `field(default_factory=lambda: Path("macbookmodelnames.csv"))` | dataclass field | +| `prompt_file` | `Path` | `field(default_factory=lambda: Path("prompt.md"))` | dataclass field | +| `test_document_file` | `Path` | `field( default_factory=lambda: Path( "64596011f75ffd2916b1ce50131f3d7cb36c10141e914e435fd5dc0e007b2b52_base.md" ) )` | dataclass field | +| `output_file` | `Path` | `field( default_factory=lambda: Path("model_evaluation_results.xlsx") )` | dataclass field | + +**Public methods:** `from_env()` + +### `TestResult` + +**File:** `backend/scripts/evaluate_macbook_models.py` +**Decorators:** `dataclass` +**Purpose:** Result of testing a single model. + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `model_name` | `str` | | declared field | +| `entity_responses` | `Dict[str, str]` | | declared field | +| `ground_truth` | `Dict[str, str]` | | declared field | +| `score` | `float` | | declared field | +| `total_latency_seconds` | `float` | | declared field | +| `error` | `Optional[str]` | `None` | declared field | + +**Public methods:** `to_dict()` + +### `EntityPrompt` + +**File:** `backend/scripts/evaluate_macbook_models.py` +**Decorators:** `dataclass` +**Purpose:** Represents a single entity extraction prompt. + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `entity_name` | `str` | | declared field | +| `prompt_with_examples` | `str` | | declared field | + +### `VRAMStatus` + +**File:** `backend/services/document/processors/docling/vram_guard.py` +**Decorators:** `dataclass` +**Purpose:** Snapshot of current GPU VRAM and worker state. + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `total_mb` | `float` | | declared field | +| `used_mb` | `float` | | declared field | +| `free_mb` | `float` | | declared field | +| `active_workers` | `int` | | declared field | +| `max_workers` | `int` | | declared field | +| `can_accept_worker` | `bool` | | declared field | +| `estimated_per_worker_mb` | `float` | | declared field | +| `safety_margin_mb` | `float` | | declared field | +| `jobs_completed` | `int` | `0` | declared field | +| `jobs_queued` | `int` | `0` | declared field | +| `is_cold_start` | `bool` | `False` | declared field | + +**Public methods:** `utilization_pct()` + +### `EvalTask` + +**File:** `backend/services/evaluation/job_queue.py` +**Decorators:** `dataclass` +**Purpose:** One atomic unit of work: evaluate a single entity extraction. + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `entity_name` | `str` | | declared field | +| `source_model` | `str` | | declared field | +| `actual_output` | `str` | | declared field | +| `extraction_prompt` | `str` | | declared field | +| `expected_output` | `Optional[str]` | `None` | declared field | +| `file_hash` | `Optional[str]` | `None` | declared field | +| `file_id` | `Optional[str]` | `None` | declared field | + +### `ProviderConfig` + +**File:** `backend/services/evaluation/job_queue.py` +**Decorators:** `dataclass` +**Purpose:** Configuration for a single judge LLM. + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `provider_id` | `str` | | declared field | +| `provider` | `str` | | declared field | +| `model_name` | `Optional[str]` | `None` | declared field | +| `deployment` | `Optional[str]` | `None` | declared field | +| `endpoint` | `Optional[str]` | `None` | declared field | +| `api_key` | `Optional[str]` | `None` | declared field | + +### `TaskResult` + +**File:** `backend/services/evaluation/job_queue.py` +**Decorators:** `dataclass` +**Purpose:** Result of evaluating one task with one provider. + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `entity_name` | `str` | | declared field | +| `source_model` | `str` | | declared field | +| `file_id` | `Optional[str]` | | declared field | +| `file_hash` | `Optional[str]` | | declared field | +| `provider_id` | `str` | | declared field | +| `provider` | `str` | | declared field | +| `model` | `str` | | declared field | +| `aggregate_score` | `float` | | declared field | +| `all_passed` | `bool` | | declared field | +| `evaluation_time` | `float` | | declared field | +| `evaluation_cost` | `float` | | declared field | +| `metrics` | `List[Dict[str, Any]]` | | declared field | +| `ground_truth` | `Optional[str]` | `None` | declared field | + +### `EvalJob` + +**File:** `backend/services/evaluation/job_queue.py` +**Decorators:** `dataclass` +**Purpose:** A batch evaluation job submitted by a user. + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `job_id` | `str` | | declared field | +| `session_id` | `str` | | declared field | +| `user_id` | `str` | | declared field | +| `tasks` | `List[EvalTask]` | | declared field | +| `providers` | `List[ProviderConfig]` | | declared field | +| `metrics` | `List[str]` | | declared field | +| `custom_evaluation_steps` | `Dict[str, List[str]]` | | declared field | +| `threshold` | `float` | `0.7` | declared field | +| `status` | `str` | `"pending"` | declared field | +| `progress` | `int` | `0` | declared field | +| `total` | `int` | `0` | declared field | +| `results` | `List[TaskResult]` | `field(default_factory=list)` | dataclass field | +| `errors` | `List[Dict[str, str]]` | `field( default_factory=list )` | dataclass field | +| `cancelled` | `bool` | `False` | declared field | +| `created_at` | `datetime` | `field(default_factory=lambda: datetime.now(timezone.utc))` | dataclass field | +| `completed_at` | `Optional[datetime]` | `None` | declared field | +| `error` | `Optional[str]` | `None` | declared field | +| `_asyncio_tasks` | `List[Any]` | `field(default_factory=list, repr=False)` | dataclass field | + +**Public methods:** `to_status_dict()` + +### `_JobStatusProxy` + +**File:** `backend/services/evaluation/job_queue.py` +**Decorators:** `dataclass` +**Purpose:** Read-only snapshot of a job loaded from DB (cross-worker lookup). + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `_data` | `Dict[str, Any]` | | declared field | + +**Public methods:** `to_status_dict()` + +### `CallMetric` + +**File:** `backend/services/telemetry/cost_tracker.py` +**Decorators:** `dataclass` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `provider` | `str` | | declared field | +| `model` | `str` | | declared field | +| `prompt_tokens` | `int` | | declared field | +| `completion_tokens` | `int` | | declared field | +| `duration` | `float` | | declared field | +| `cost` | `float` | | declared field | +| `timestamp` | `str` | | declared field | +| `document_name` | `Optional[str]` | `None` | declared field | +| `page_count` | `int` | `0` | declared field | +| `figure_count` | `int` | `0` | declared field | +| `table_count` | `int` | `0` | declared field | +| `batch_number` | `Optional[int]` | `None` | declared field | + +### `BatchMetric` + +**File:** `backend/services/telemetry/cost_tracker.py` +**Decorators:** `dataclass` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `batch_number` | `int` | | declared field | +| `batch_latency` | `float` | | declared field | +| `document_count` | `int` | | declared field | + +### `SessionMetrics` + +**File:** `backend/services/telemetry/cost_tracker.py` +**Decorators:** `dataclass` + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `session_id` | `str` | | declared field | +| `total_cost` | `float` | `0.0` | declared field | +| `total_latency` | `float` | `0.0` | declared field | +| `total_calls` | `int` | `0` | declared field | +| `calls` | `List[CallMetric]` | `field(default_factory=list)` | dataclass field | +| `batches` | `Dict[int, BatchMetric]` | `field(default_factory=dict)` | dataclass field | + +## Enums + +### `ProcessorType` + +**File:** `backend/schemas/enums.py` +**Base classes:** `str, Enum` +**Purpose:** Document processor types + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `AUTO` | | `"auto"` | declared field | +| `DOCLING` | | `"docling"` | declared field | +| `AZURE_DOC_INTELLIGENCE` | | `"azure_doc_intelligence"` | declared field | + +## Service and provider classes + +### `SQLAlchemyDBService` + +**File:** `backend/services/database/sqlalchemy_db_service.py` +**Purpose:** Database service using SQLAlchemy / Azure Postgres directly. + +_No class-level data fields or constructor instance attributes are defined._ + +**Public methods:** `create_session()`, `get_session()`, `list_sessions()`, `get_session_basic()`, `update_session()`, `delete_session()`, `get_session_for_shared_view()`, `record_login()`, `create_document()`, `get_document()`, `get_documents_by_session()`, `get_parse_cost_by_file_hash()`, `list_user_documents()`, `update_document()`, `update_document_processing()`, `upsert_extraction_result()`, `update_extraction_cost()`, `get_extraction_results_by_session()`, `get_extraction_results_by_document()`, `upsert_evaluation_result()`, `get_evaluation_results_by_extraction()`, `get_or_create_preferences()`, `update_preferences()`, `save_prompt_template()`, `get_prompt_templates()`, `delete_prompt_template()`, `share_session()`, `unshare_session()`, `list_shared_sessions()`, `get_user_group_ids()`, `get_group_name()`, `get_user_display_name()`, `increment_session_metrics()`, `get_session_metrics()`, `reset_session_metrics()`, `create_eval_job_record()`, `upsert_eval_job_status()`, `get_eval_job_status()`, `mark_eval_job_cancelled()` + +### `DocumentService` + +**File:** `backend/services/document/document_service.py` +**Purpose:** Main service for document processing with multiple processor support + +| Instance attribute | Initialized from | +| --- | --- | +| `available_processors` | `self._check_processor_availability()` | +| `azure_doc_intelligence_service` | `AzureDocIntelligenceService()` | +| `docling_service` | `DoclingRemoteClient()` | +| `file_service` | `get_organized_file_service()` | + +**Public methods:** `async convert_document_to_markdown()`, `async get_processor_capabilities()`, `async get_conversion_by_id()`, `async get_markdown_content()`, `async resolve_processor_used()`, `async get_figures_for_conversion()`, `async get_raw_analysis_result()`, `async get_processing_file_bytes()` + +### `FileService` + +**File:** `backend/services/document/file_service.py` +**Purpose:** Service for handling file upload, storage, and management operations + +| Instance attribute | Initialized from | +| --- | --- | +| `metadata_dir` | `self.upload_dir / "metadata"` | +| `upload_dir` | `Path(upload_dir)` | + +**Public methods:** `async save_uploaded_file()`, `async get_file_by_hash()`, `async get_file_info()`, `async get_file_by_id()`, `async delete_file()`, `async get_file_content()`, `async list_files()`, `async _save_metadata()`, `async _load_metadata()` + +### `OrganizedFileService` + +**File:** `backend/services/document/organized_file_service.py` +**Purpose:** Service for file upload, storage, and retrieval using Azure Blob Storage. + +| Instance attribute | Initialized from | +| --- | --- | +| `_blob` | `BlobStorageClient(conn_str, container)` | +| `_db` | `None` | + +**Public methods:** `db()`, `compute_file_hash()`, `async save_uploaded_file()`, `get_processing_output_path()`, `async sync_processing_output_to_blob()`, `async get_processed_metadata()`, `async update_processed_metadata()`, `async get_processing_file_bytes()`, `async processing_file_exists()`, `async resolve_processed_processor()`, `async build_document_view()`, `async is_file_processed()`, `async get_processed_content()`, `async get_file_content()`, `async get_file_metadata()`, `async get_original_file_path()`, `async list_user_files()` + +### `OrganizedDocumentProcessor` + +**File:** `backend/services/document/organized_processor.py` +**Purpose:** Document processor that uses the organized file structure. + +| Instance attribute | Initialized from | +| --- | --- | +| `azure_service` | `AzureDocIntelligenceService()` | +| `docling_service` | `DoclingRemoteClient(base_url=docling_url)` | +| `file_service` | `get_organized_file_service()` | + +**Public methods:** `async process_document()`, `async _process_with_azure()`, `async _process_with_docling()`, `async _load_metadata()`, `async get_processed_markdown()`, `async is_processed()` + +### `AzureDocIntelligenceService` + +**File:** `backend/services/document/processors/azure_doc_intelligence/azure_doc_intelligence_service.py` +**Purpose:** Service for processing documents using Azure Document Intelligence + +| Instance attribute | Initialized from | +| --- | --- | +| `base_path` | `Path(__file__).parent.parent.parent.parent.parent` | +| `client` | `self._init_client()` | +| `output_base_dir` | `self.base_path / "output" / "azure_doc_intelligence"` | + +**Public methods:** `async convert_document_to_markdown()`, `async _log()`, `async get_conversion_by_id()`, `async get_markdown_content()`, `is_available()`, `async get_figures_for_conversion()`, `async get_raw_analysis_result()` + +### `DoclingRemoteClient` + +**File:** `backend/services/document/processors/docling/docling_remote_client.py` +**Purpose:** Drop-in replacement for DoclingService.convert_document_to_markdown(). + +| Instance attribute | Initialized from | +| --- | --- | +| `base_url` | `(base_url or os.environ.get("DOCLING_SERVICE_URL", "")).rstrip( "/" )` | +| `poll_interval` | `poll_interval` | +| `timeout` | `timeout` | + +**Public methods:** `async convert_document_to_markdown()`, `async _call_sync_convert()`, `async _download_artifact_bundle()`, `async convert_async()`, `async check_health()` + +### `_VRAMPeakTracker` + +**File:** `backend/services/document/processors/docling/docling_service.py` +**Purpose:** Poll nvidia-smi in a background thread to capture the true peak VRAM. + +| Instance attribute | Initialized from | +| --- | --- | +| `_peak` | `-1.0` | +| `_poll_sec` | `poll_sec` | +| `_stop` | `_threading.Event()` | +| `_thread` | `None` | + +**Public methods:** `start()`, `stop()` + +### `DoclingService` + +**File:** `backend/services/document/processors/docling/docling_service.py` +**Purpose:** Service for handling document ingestion and conversion using Docling + +| Instance attribute | Initialized from | +| --- | --- | +| `_pool_size` | `0` | +| `_process_pool` | `None` | +| `_vram_guard` | `None` | +| `base_path` | `Path(__file__).resolve().parents[4]` | +| `image_resolution_scale` | `image_resolution_scale` | +| `output_base_dir` | `Path(markdown_dir)` | +| `output_base_dir` | `self.base_path / "output" / "docling"` | + +**Public methods:** `process_pool()`, `vram_guard()`, `max_workers()`, `async convert_document_to_markdown()`, `async start_conversion()`, `async get_conversion_by_id()`, `async get_markdown_content()`, `async list_conversions()`, `async delete_conversion()`, `async get_figures_for_conversion()`, `async get_raw_analysis_result()`, `async _save_conversion_metadata()`, `async _load_conversion_metadata()` + +### `VRAMGuard` + +**File:** `backend/services/document/processors/docling/vram_guard.py` +**Purpose:** VRAM-aware concurrency controller for GPU-accelerated document processing. + +| Instance attribute | Initialized from | +| --- | --- | +| `_active_workers` | `0` | +| `_cold_start_max` | `cold_start_workers if cold_start_workers is not None else int( os.environ.get( "VRAM_COLD_START_WORKERS", str(_COLD_START_WORKERS_DEFAULT) ) )` | +| `_cold_start_min_jobs` | `cold_start_min_jobs if cold_start_min_jobs is not None else _COLD_START_MIN_JOBS_DEFAULT` | +| `_is_cold_start` | `not loaded_state` | +| `_jobs_completed` | `0` | +| `_last_smi_free` | `self.vram_total_mb` | +| `_last_smi_time` | `0.0` | +| `_last_smi_used` | `0.0` | +| `_lock` | `asyncio.Lock()` | +| `_max_workers_cap` | `max_workers_cap if max_workers_cap is not None else ( int(os.environ["VRAM_MAX_WORKERS"]) if "VRAM_MAX_WORKERS" in os.environ else None )` | +| `_observation_window` | `observation_window if observation_window is not None else int( os.environ.get( "VRAM_OBSERVATION_WINDOW", str(_OBSERVATION_WINDOW_DEFAULT) ) )` | +| `_peak_observations` | `loaded_state.get("observations", []) if loaded_state else []` | +| `_per_worker_mb` | `loaded_state["per_worker_mb"]` | +| `_per_worker_mb` | `per_worker_init_mb if per_worker_init_mb is not None else float(os.environ.get("VRAM_PER_WORKER_INIT_MB", "2800"))` | +| `_persistence_path` | `Path(persistence_path) if persistence_path else _DEFAULT_STATE_PATH` | +| `_pool_needs_resize` | `False` | +| `_queued_workers` | `0` | +| `_slot_available` | `asyncio.Event()` | +| `_smi_lock` | `threading.Lock()` | +| `_usable_vram_mb` | `self.vram_total_mb - self.safety_margin_mb` | +| `check_interval_sec` | `check_interval_sec if check_interval_sec is not None else float(os.environ.get("VRAM_CHECK_INTERVAL_SEC", "1.0"))` | +| `safety_margin_mb` | `safety_margin_mb if safety_margin_mb is not None else float(os.environ.get("VRAM_SAFETY_MARGIN_MB", "1536"))` | +| `vram_total_mb` | `vram_total_mb or _get_gpu_vram_total_mb() or 16384.0` | + +**Public methods:** `async acquire_slot()`, `report_worker_result()`, `report_oom()`, `get_status()`, `max_workers()`, `active_workers()`, `per_worker_mb()`, `is_cuda_oom()` + +### `AnthropicVertexDeepEvalModel` + +**File:** `backend/services/evaluation/adapters/anthropic_adapter.py` +**Base classes:** `DeepEvalBaseLLM` +**Purpose:** Custom DeepEval model adapter for Anthropic via Vertex AI + +| Instance attribute | Initialized from | +| --- | --- | +| `_model_name` | `model_name` | +| `async_client` | `AsyncAnthropicVertex( region=self.location, project_id=self.project )` | +| `call_history` | `[]` | +| `client` | `AnthropicVertex(region=self.location, project_id=self.project)` | +| `location` | `location` | +| `max_tokens` | `max_tokens` | +| `project` | `project or "hcsx-scigpt2-innocentrhino-acm"` | +| `temperature` | `temperature` | + +**Public methods:** `load_model()`, `generate()`, `async a_generate()`, `get_model_name()` + +### `AzureOpenAIDeepEvalModel` + +**File:** `backend/services/evaluation/adapters/azure_adapter.py` +**Base classes:** `DeepEvalBaseLLM` +**Purpose:** Custom DeepEval model adapter for Azure OpenAI using LangChain + +| Instance attribute | Initialized from | +| --- | --- | +| `_model_name` | `model_name or (secrets_config.get("model_name") if secrets_config else None) or os.getenv("AZURE_OPENAI_MODEL_NAME", "gpt-5-mini")` | +| `api_key` | `api_key or (secrets_config.get("api_key") if secrets_config else None) or os.getenv("AZURE_OPENAI_KEY")` | +| `api_version` | `api_version or (secrets_config.get("api_version") if secrets_config else None) or os.getenv("AZURE_OPENAI_API_VERSION", "2024-12-01-preview")` | +| `call_history` | `[]` | +| `deployment` | `deployment or (secrets_config.get("deployment") if secrets_config else None) or os.getenv("AZURE_OPENAI_DEPLOYMENT") or os.getenv("AZURE_OPENAI_MODEL_NAME")` | +| `endpoint` | `endpoint or (secrets_config.get("endpoint") if secrets_config else None) or os.getenv("AZURE_OPENAI_ENDPOINT")` | +| `max_tokens` | `max_tokens` | +| `model` | `AzureChatOpenAI(**model_kwargs)` | +| `temperature` | `temperature` | + +**Public methods:** `load_model()`, `generate()`, `async a_generate()`, `get_model_name()` + +### `VertexAIDeepEvalModel` + +**File:** `backend/services/evaluation/adapters/vertex_adapter.py` +**Base classes:** `DeepEvalBaseLLM` +**Purpose:** Custom DeepEval model adapter for Vertex AI using LangChain + +| Instance attribute | Initialized from | +| --- | --- | +| `_model_name` | `model_name` | +| `call_history` | `[]` | +| `location` | `location or os.getenv("GEMINI_LOCATION", "us-central1")` | +| `model` | `ChatVertexAI(**model_kwargs)` | +| `project` | `project or os.getenv("GEMINI_PROJECT")` | +| `temperature` | `temperature` | + +**Public methods:** `load_model()`, `generate()`, `async a_generate()`, `get_model_name()` + +### `EvaluationService` + +**File:** `backend/services/evaluation/evaluation_service.py` +**Purpose:** Main evaluation service orchestrator + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `METRIC_FACTORIES` | | `{ "correctness": CorrectnessMetricFactory, "completeness": CompletenessMetricFactory, "relevance": RelevanceMetricFactory, "safety": SafetyMetricFactory, }` | declared field | + +**Public methods:** `create_evaluation_model()`, `create_metric()`, `create_custom_metric()`, `async _evaluate_combined()`, `async evaluate_extraction()`, `async evaluate_multiple_extractions()`, `async get_evaluation_result()`, `async list_evaluations()` + +### `CompletenessMetricFactory` + +**File:** `backend/services/evaluation/metrics/completeness.py` +**Purpose:** Factory for creating Completeness evaluation metrics + +_No class-level data fields or constructor instance attributes are defined._ + +**Public methods:** `create()`, `get_required_params()`, `get_description()` + +### `CorrectnessMetricFactory` + +**File:** `backend/services/evaluation/metrics/correctness.py` +**Purpose:** Factory for creating Correctness evaluation metrics + +_No class-level data fields or constructor instance attributes are defined._ + +**Public methods:** `create()`, `get_required_params()`, `get_description()` + +### `CustomMetricFactory` + +**File:** `backend/services/evaluation/metrics/custom.py` +**Purpose:** Factory for creating custom G-Eval metrics + +_No class-level data fields or constructor instance attributes are defined._ + +**Public methods:** `create()`, `get_description()` + +### `RelevanceMetricFactory` + +**File:** `backend/services/evaluation/metrics/relevance.py` +**Purpose:** Factory for creating Relevance evaluation metrics + +_No class-level data fields or constructor instance attributes are defined._ + +**Public methods:** `create()`, `get_required_params()`, `get_description()` + +### `SafetyMetricFactory` + +**File:** `backend/services/evaluation/metrics/safety.py` +**Purpose:** Factory for creating Safety evaluation metrics + +_No class-level data fields or constructor instance attributes are defined._ + +**Public methods:** `create()`, `get_required_params()`, `get_description()` + +### `EvaluationResultStorage` + +**File:** `backend/services/evaluation/storage/result_storage.py` +**Purpose:** Handles storage and retrieval of evaluation results + +| Instance attribute | Initialized from | +| --- | --- | +| `output_dir` | `Path(output_dir) if output_dir else base_path / "output" / "evaluations"` | + +**Public methods:** `async save()`, `async get()`, `async list_all()`, `async delete()`, `get_storage_path()` + +### `GroupService` + +**File:** `backend/services/groups/group_service.py` +**Purpose:** Service for managing groups and memberships + +_No class-level data fields or constructor instance attributes are defined._ + +**Public methods:** `create_group()`, `get_group()`, `list_user_groups()`, `update_group()`, `delete_group()`, `get_group_members()`, `add_member()`, `update_member_role()`, `remove_member()` + +### `AnthropicLLMClient` + +**File:** `backend/services/llm/anthropic.py` + +| Instance attribute | Initialized from | +| --- | --- | +| `disabled` | `not self.service_account_path` | +| `location` | `os.environ.get("ANTHROPIC_LOCATION", "global")` | +| `project_id` | `os.environ.get("ANTHROPIC_PROJECT_ID") or "hcsx-scigpt2-innocentrhino-acm"` | +| `service_account_path` | `self._find_service_account_file()` | + +**Public methods:** `async _call_anthropic_api()`, `async extract_entities_with_anthropic()`, `async generate_paragraph_with_anthropic()` + +### `AzureLLMClient` + +**File:** `backend/services/llm/azure.py` + +| Instance attribute | Initialized from | +| --- | --- | +| `api_key` | `os.environ.get("AZURE_OPENAI_KEY")` | +| `api_version` | `os.environ.get( "AZURE_OPENAI_API_VERSION", "2024-08-01-preview" )` | +| `default_deployment` | `os.environ.get("AZURE_OPENAI_DEPLOYMENT")` | +| `default_model_name` | `os.environ.get("AZURE_OPENAI_MODEL_NAME")` | +| `disabled` | `not has_global_creds and not has_configured_models` | +| `endpoint` | `os.environ.get("AZURE_OPENAI_ENDPOINT")` | + +**Public methods:** `async generate_paragraph_with_azure()`, `async extract_entities_with_azure()`, `async extract_content_from_image()` + +### `GeminiLLMClient` + +**File:** `backend/services/llm/gemini.py` + +| Instance attribute | Initialized from | +| --- | --- | +| `disabled` | `not self.project_id or not self.location or not self.service_account_path` | +| `location` | `os.environ.get("GEMINI_LOCATION") or os.environ.get("VERTEX_AI_LOCATION") or "us-central1"` | +| `project_id` | `os.environ.get("GEMINI_PROJECT_ID") or os.environ.get("GEMINI_PROJECT") or os.environ.get("VERTEX_AI_PROJECT")` | +| `service_account_path` | `self._find_service_account_file()` | + +**Public methods:** `async _call_gemini_api()`, `async extract_entities_with_gemini()`, `async generate_paragraph_with_gemini()`, `async extract_content_from_image()` + +### `LlamaLLMClient` + +**File:** `backend/services/llm/llama.py` + +| Instance attribute | Initialized from | +| --- | --- | +| `disabled` | `not self.project_id or not self.location or not self.region or not self.service_account_path` | +| `location` | `os.environ.get("LLAMA_LOCATION", "us-east5")` | +| `project_id` | `os.environ.get("LLAMA_PROJECT_ID") or os.environ.get( "GEMINI_PROJECT_ID" )` | +| `region` | `os.environ.get("LLAMA_REGION", "us-east5")` | +| `service_account_path` | `self._find_service_account_file()` | + +**Public methods:** `async _call_llama_api()`, `async extract_entities_with_llama()`, `async _try_llama_extraction_strategy()`, `async _try_llama_fallback_strategy()`, `async _handle_llama_parsing_error()`, `async generate_paragraph_with_llama()`, `async warm_up()` + +### `LLMService` + +**File:** `backend/services/llm/llm_service.py` +**Purpose:** LLM Service for entity extraction and paragraph generation. + +| Instance attribute | Initialized from | +| --- | --- | +| `anthropic_client` | `AnthropicLLMClient()` | +| `azure_client` | `AzureLLMClient()` | +| `gemini_client` | `GeminiLLMClient()` | +| `llama_client` | `LlamaLLMClient()` | +| `macbook_client` | `MacbookLLMClient()` | +| `timeout_log_dir` | `Path(__file__).resolve().parents[2] / "output" / "timeout_logs"` | +| `timeout_log_file` | `self.timeout_log_dir / "timeout_log.txt"` | +| `vllm_client` | `VLLMClient()` | + +**Public methods:** `async _call_with_timeout_logging()`, `async extract_entities_from_markdown()`, `async extract_content_from_image()`, `async generate_paragraph()` + +### `MacbookLLMClient` + +**File:** `backend/services/llm/macbook.py` + +| Instance attribute | Initialized from | +| --- | --- | +| `_fail_count` | `0` | +| `_tags_cache` | `[]` | +| `_tags_cache_ts` | `0.0` | +| `_tags_cache_ttl_seconds` | `120` | +| `base_url` | `(self.base_url or "").rstrip("/")` | +| `base_url` | `os.environ.get("MACBOOK_LLM_BASE_URL", "").rstrip("/")` | +| `base_url` | `self._load_base_url_from_secrets()` | +| `disable_reasoning` | `disable_reasoning_env not in [ "false", "0", "no", "off", ]` | +| `disabled` | `not bool(self.base_url)` | +| `initial_backoff` | `float(os.environ.get("MACBOOK_INITIAL_BACKOFF", 1.0))` | +| `max_attempts` | `int(os.environ.get("MACBOOK_MAX_ATTEMPTS", 2))` | +| `max_backoff` | `float(os.environ.get("MACBOOK_MAX_BACKOFF", 8.0))` | +| `per_attempt_timeout` | `max( 1800.0, float(os.environ.get("MACBOOK_PER_ATTEMPT_TIMEOUT", 1800.0)) )` | +| `total_retry_cap` | `max( 1800.0, float(os.environ.get("MACBOOK_TOTAL_RETRY_CAP", 1800.0)) )` | + +**Public methods:** `async fetch_available_models()`, `async check_health()`, `async _call_macbook_api()`, `async extract_entities_with_macbook()`, `async generate_paragraph_with_macbook()` + +### `MacbookRequestQueue` + +**File:** `backend/services/llm/macbook_queue.py` +**Purpose:** FIFO queue with a single worker for serializing Macbook LLM requests. + +| Instance attribute | Initialized from | +| --- | --- | +| `_queue` | `asyncio.Queue()` | +| `_started` | `False` | +| `_total_enqueued` | `0` | +| `_total_processed` | `0` | +| `_worker_task` | `None` | + +**Public methods:** `async _worker()`, `async enqueue()`, `pending_count()`, `stats()` + +### `VLLMClient` + +**File:** `backend/services/llm/vllm.py` +**Purpose:** OpenAI-compatible client for VLLM inference servers. + +| Instance attribute | Initialized from | +| --- | --- | +| `_static_models` | `[]` | +| `_static_models` | `json.loads(models_json)` | +| `api_key` | `os.environ.get("VLLM_API_KEY", "EMPTY")` | +| `base_url` | `os.environ.get("VLLM_BASE_URL", "").rstrip("/")` | +| `disabled` | `not bool(self.base_url)` | + +**Public methods:** `async fetch_available_models()`, `async check_health()`, `async _call_vllm_api()`, `async extract_entities_with_vllm()`, `async generate_paragraph_with_vllm()` + +### `SessionService` + +**File:** `backend/services/session/session_service.py` +**Purpose:** Service for managing user sessions with database storage + +| Instance attribute | Initialized from | +| --- | --- | +| `_doc_cache` | `{}` | +| `_file_service` | `None` | +| `db` | `get_db_service()` | + +**Public methods:** `file_service()`, `create_session()`, `get_session()`, `list_sessions()`, `update_session()`, `delete_session()`, `add_extraction_result()`, `add_extraction_result_fast()`, `add_evaluation_result()`, `add_evaluation_result_fast()`, `clear_cache()`, `share_session()`, `unshare_session()`, `list_shared_sessions()`, `get_session_for_shared_view()`, `async build_restore_view()` + +### `BlobStorageClient` + +**File:** `backend/services/storage/blob_storage.py` +**Purpose:** Async Azure Blob Storage client. + +| Instance attribute | Initialized from | +| --- | --- | +| `_container` | `container_name` | +| `_service` | `BlobServiceClient.from_connection_string(connection_string)` | + +**Public methods:** `async _ensure_container()`, `async upload_bytes()`, `async download_bytes()`, `async exists()`, `async upload_directory()`, `async list_blobs_with_prefix()`, `from_env()` + +### `CostTracker` + +**File:** `backend/services/telemetry/cost_tracker.py` + +| Instance attribute | Initialized from | +| --- | --- | +| `_db_service` | `None` | +| `_pricing` | `self._load_pricing()` | +| `_sessions` | `{}` | + +**Public methods:** `estimate_call_cost()`, `record_call()`, `record_batch()`, `get_session_metrics()`, `load_session_metrics_from_db()`, `clear_session()` + +### `FolderService` + +**File:** `backend/services/templates/folder_service.py` +**Purpose:** Service for managing template folders. + +| Instance attribute | Initialized from | +| --- | --- | +| `group_service` | `get_group_service()` | + +**Public methods:** `list_folders()`, `create_folder()`, `rename_folder()`, `delete_folder()` + +### `TemplateService` + +**File:** `backend/services/templates/template_service.py` +**Purpose:** Service for managing prompt templates + +| Instance attribute | Initialized from | +| --- | --- | +| `group_service` | `get_group_service()` | + +**Public methods:** `create_template()`, `get_template()`, `list_templates()`, `update_template()`, `delete_template()`, `get_version_history()`, `revert_to_version()`, `fork_template()`, `change_scope()`, `set_immutable()`, `set_permission()`, `get_permissions()`, `remove_permission()` + +## Utilities + +### `PDFBBoxVisualizer` + +**File:** `backend/utils/pdf_bbox_visualizer.py` +**Purpose:** Visualizes bounding boxes from Azure Document Intelligence on PDFs + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `COLORS` | | `{ "word": (0.2, 0.6, 1.0), # Light blue "line": (0.0, 0.5, 0.8), # Blue "paragraph": (0.5, 0.0, 0.5), # Purple "table": (1.0, 0.5, 0.0), # Orange "table_cell": (1.0, 0.7, 0.3), ...` | declared field | + +**Public methods:** `visualize_words()`, `visualize_lines()`, `visualize_paragraphs()`, `visualize_tables()`, `visualize_figures()`, `visualize_selection_marks()`, `visualize_all()`, `save()`, `close()` + +## Scripts and developer tools + +### `DocumentAnalyzer` + +**File:** `backend/scripts/analyze_and_visualize_pdf.py` +**Purpose:** Analyzes documents using Azure Document Intelligence + +| Instance attribute | Initialized from | +| --- | --- | +| `client` | `DocumentIntelligenceClient( endpoint=self.endpoint, credential=AzureKeyCredential(self.key) )` | +| `endpoint` | `os.getenv("AZURE_DOC_INTELLIGENCE_ENDPOINT")` | +| `key` | `os.getenv("AZURE_DOC_INTELLIGENCE_KEY")` | + +**Public methods:** `analyze_pdf()` + +### `ModelListParser` + +**File:** `backend/scripts/evaluate_macbook_models.py` +**Purpose:** Parse model names from the CSV file. + +| Field | Type | Default / definition | Kind | +| --- | --- | --- | --- | +| `EXPECTED_MODELS` | | `[ # 3B - 4B "llama3.2:3b-instruct-fp16", "llama3.2:3b-instruct-q4_K_M", "MedAIBase/MedGemma1.5:4b", "phi4-mini:3.8b", "phi3.5:3.8b", "nemotron-mini:4b-instruct-q4_K_M", "nemotro...` | declared field | + +**Public methods:** `parse_model_list()` + +### `PromptLoader` + +**File:** `backend/scripts/evaluate_macbook_models.py` +**Purpose:** Load prompt template and test document. + +| Instance attribute | Initialized from | +| --- | --- | +| `config` | `config` | + +**Public methods:** `load()` + +### `MacbookClient` + +**File:** `backend/scripts/evaluate_macbook_models.py` +**Purpose:** Client for interacting with MacBook LLM API. + +| Instance attribute | Initialized from | +| --- | --- | +| `base_url` | `config.base_url.rstrip("/")` | +| `config` | `config` | +| `endpoint` | `config.api_endpoint` | +| `session` | `requests.Session()` | + +**Public methods:** `generate()`, `close()` + +### `ResponseEvaluator` + +**File:** `backend/scripts/evaluate_macbook_models.py` +**Purpose:** Evaluate model responses against ground truth. + +| Instance attribute | Initialized from | +| --- | --- | +| `ground_truth` | `ground_truth` | +| `weights` | `{ "study_author(s)": 0.12, "author_affiliations": 0.12, "study_title": 0.12, "publication_date": 0.10, "test_material": 0.12, "vehicle_or_solvent_used": 0.12, "dose_le...` | + +**Public methods:** `compute_score()` + +### `ExcelReporter` + +**File:** `backend/scripts/evaluate_macbook_models.py` +**Purpose:** Generate Excel report from test results. + +| Instance attribute | Initialized from | +| --- | --- | +| `config` | `config` | + +**Public methods:** `generate()` + +### `ModelTestRunner` + +**File:** `backend/scripts/evaluate_macbook_models.py` +**Purpose:** Orchestrates the model testing process. + +| Instance attribute | Initialized from | +| --- | --- | +| `client` | `MacbookClient(config)` | +| `config` | `config` | +| `reporter` | `ExcelReporter(config)` | + +**Public methods:** `run()` + +## Other backend classes + +### `Base` + +**File:** `backend/models/base.py` +**Base classes:** `DeclarativeBase` +**Purpose:** Base class for all SQLAlchemy models + +_No class-level data fields or constructor instance attributes are defined._ + diff --git a/docs/backend/appendices/data-flow-diagrams.md b/docs/backend/appendices/data-flow-diagrams.md new file mode 100644 index 0000000..ac1ac54 --- /dev/null +++ b/docs/backend/appendices/data-flow-diagrams.md @@ -0,0 +1,345 @@ +# Backend Data-Flow Diagrams + +Text diagrams for major backend flows. See the module docs for field-level details. + +For visual architecture diagrams, start with the [visual workflow map](../README.md#visual-workflow-map). The SVG diagrams under [`../images/`](../images/) are the primary visual companion to this appendix; the text flows below remain useful for exact request-by-request sequencing. + +## 1. Authenticated API request + +```text +Frontend + | + | Authorization: Bearer + v +FastAPI route + | + | Depends(get_current_user) + v +core.auth.get_current_user() + | + | SELECT AuthSession JOIN User WHERE token = ... + v +PostgreSQL + | + | session row + user row + v +expiry check + optional ALLOWED_EMAILS check + | + v +route receives current_user dict +``` + +## 2. Better Auth proxy + +```text +Browser + | + | /api/auth/sign-in/*, /api/auth/callback/*, etc. + v +FastAPI auth proxy router + | + | forwards request to localhost/auth sidecar + v +Better Auth sidecar + | + | reads/writes Better Auth DB tables + v +PostgreSQL + | + | Set-Cookie / redirect / JSON response + v +Browser +``` + +## 3. File upload + +```text +POST /api/upload multipart file + | + v +files router + | + v +OrganizedFileService.save_uploaded_file() + | + +-- compute SHA-256 hash + +-- infer extension and mime type + +-- check blob exists: global/{hash}/original.{ext} + | + +-- if missing: + | upload original bytes + | upload global/{hash}/metadata.json + | + +-- if user_id: + | create Document DB row best-effort + v +response: file_hash, blob path, dedupe flags, filename, size +``` + +## 4. Document processing + +```text +POST /api/documents/process/file/{file_hash} + | + v +documents router + | + +-- check existing processed document.md + | global/{hash}/processed/{processor}/document.md + | + +-- if cache hit: + | build document view from metadata/artifacts + | + +-- if cache miss: + get original file from blob to /tmp/summarization/{hash}/original.{ext} + choose processor + write local artifacts to /tmp/summarization/{hash}/processed/{processor}/ + sync local artifact tree to blob + update Document DB processing metadata + build document view +``` + +## 5. Azure Document Intelligence processing + +```text +local original file or URL + | + v +AzureDocIntelligenceService.convert_document_to_markdown() + | + +-- begin_analyze_document(..., output markdown + figures) + +-- wait for poller.result() + +-- save raw_analysis.json + +-- save document.md + +-- regex-extract HTML tables to tables/table-N.html + +-- download figure PNGs to figures/{figure_id}.png + +-- save metadata.json + v +organized processor syncs output tree to blob +``` + +## 6. Docling processing + +```text +local original file + | + v +DoclingService / DoclingRemoteClient + | + +-- acquire VRAM slot if local Docling path + +-- worker process converts PDF + +-- save document.md + +-- save raw_analysis.json + +-- save figures/picture-N.png + +-- save tables/table-N.html + +-- save metadata.json + +-- report peak VRAM / OOM to VRAMGuard + v +organized processor syncs output tree to blob +``` + +## 7. Document content/read path + +```text +GET /api/documents/{file_hash}/content + | + v +DocumentService.get_markdown_content() + | + +-- resolve processor: + | preferred -> azure_doc_intelligence -> docling + | + +-- get_processed_content(file_hash, processor) + | /tmp cache first + | blob fallback + | + +-- optional raw_analysis.content fallback + v +return markdown_content +``` + +## 8. Entity extraction + +```text +POST /api/extract + | + v +load markdown + optional figure context + | + v +for each Entity concurrently: + | + +-- LLMService.extract_entities_from_markdown() + | provider dispatch by model_type + | provider call + timeout logging + | record cost/session metrics on success + | + +-- normalize answer/references/meta + | + +-- if references and raw analysis available: + | match references to bounding boxes + | + +-- build extraction result + | + +-- if session_id: + SessionService.add_extraction_result_fast() + DB upsert by (document_id, entity_name, model_id) + v +return all entity results +``` + +## 9. Figure summary generation + +```text +POST /api/documents/{file_hash}/figures/{figure_id}/generate-summary + | + v +resolve processor + metadata + | + v +find figure metadata and image path + | + v +read figure image bytes from /tmp cache or blob + | + v +LLMService.extract_content_from_image() + | + v +update metadata.json with extracted_content / summary + | + v +return figure summary result +``` + +## 10. Single evaluation + +```text +POST /api/evaluations/evaluate + | + v +EvaluationService.evaluate_extraction() + | + +-- create evaluation model adapter + +-- select metrics + +-- skip correctness/completeness if no expected_output + +-- create LLMTestCase + +-- try combined JSON scoring prompt + | parse direct JSON / fenced JSON / extracted JSON / salvaged metric entries + +-- fallback to per-metric async GEval scoring if needed + +-- collect call history + +-- estimate and record cost + +-- compute aggregate score/all_passed + v +return evaluation result +``` + +## 11. Background evaluation job + +```text +POST /api/evaluations/jobs + | + v +create EvalJob(tasks, providers, session_id, user_id) + | + v +submit_job() + | + +-- store in in-memory _JOBS + +-- create EvalJobRecord asynchronously + +-- start _process_job background task + | + v +_process_job() + | + +-- status=running, sync DB + +-- flatten tasks x providers + +-- compute per-job concurrency + +-- run _run_single_eval under per-job and global semaphores + +-- persist session evaluation result + +-- status=completed/cancelled/failed, sync DB + +GET /api/evaluations/jobs/{job_id} + | + +-- check _JOBS + +-- else load EvalJobRecord and return _JobStatusProxy snapshot +``` + +## 12. Session restore view + +```text +GET /api/sessions/{session_id}/restore-view + | + v +SessionService.get_session() + | + +-- DB session + documents + extractions + evaluations + +-- convert to Pydantic Session + | + v +SessionService.build_restore_view() + | + +-- merge files_config sources + +-- for each document: + OrganizedFileService.build_document_view() + check artifact availability + enumerate figure/table artifacts if needed + v +return primary file ids + uploadedFiles restore payload +``` + +## 13. Shared session read + +```text +GET /api/sessions/shared/{session_id} + | + v +SQLAlchemyDBService.get_session_for_shared_view() + | + +-- load AppSession + +-- require shared_with_group_id is not null + +-- require UserGroup row for requesting user/group + +-- load docs/extractions/evaluations + v +SessionService._db_to_session() + | + v +return shared Session +``` + +## 14. Template update/versioning + +```text +PUT /api/templates/{template_id} + | + v +TemplateService.update_template() + | + +-- load template + +-- _can_read() + +-- _can_edit() + +-- insert TemplateVersion snapshot of current content + +-- update allowed fields + +-- increment template.version + +-- update timestamp + v +return updated template +``` + +## 15. Cost recording + +```text +provider call succeeds + | + v +LLMService._record_session_metrics() + | + v +cost_tracker.record_call(session_id, provider, model, tokens, duration) + | + +-- update in-memory SessionMetrics + +-- emit Prometheus metrics if available + +-- DB increment_session_metrics through executor when event loop exists + v +session totals visible through server/session-metrics endpoints +``` diff --git a/docs/backend/appendices/risks-assumptions-testing.md b/docs/backend/appendices/risks-assumptions-testing.md new file mode 100644 index 0000000..7f4151b --- /dev/null +++ b/docs/backend/appendices/risks-assumptions-testing.md @@ -0,0 +1,228 @@ +# Risks, Assumptions, and Testing Strategy + +This appendix captures implementation assumptions, technical risks, and recommended tests for the backend TDD. + +## 1. Assumptions + +### 1.1 Runtime assumptions + +- PostgreSQL is available through `DATABASE_URL` or `POSTGRES_*` environment variables. +- Alembic migrations have been applied before the backend serves production traffic. +- Better Auth sidecar is running and shares the same PostgreSQL database. +- In production, `/api/auth/*` requests reach FastAPI first and are proxied to the auth sidecar. +- Azure Blob Storage connection string is present when using the organized file service. +- External model providers may be partially configured; unavailable providers should not prevent app startup. +- Production backend uses one Gunicorn worker per container replica to avoid duplicating heavyweight parser/model state. + +### 1.2 Data assumptions + +- File hash is the stable identity for uploaded file content. +- Processed artifact trees are organized by file hash and processor name. +- `document.md` is the strict signal that a file is fully processed. +- Partial artifact trees may still be useful for analysis/debug endpoints. +- Session extraction/evaluation persistence depends on matching document id or file hash, especially for multi-document sessions. +- Prompt template scopes are one of `user`, `group`, or `global`. + +### 1.3 Provider assumptions + +- Provider result dictionaries include enough metadata for token/cost tracking when calls succeed. +- Structured extraction providers may return `answer` and `references`, but downstream code must tolerate provider-specific shapes. +- Evaluation adapters can expose call history with token usage for cost estimation. +- Provider timeouts/retries vary by provider and are part of current behavior. + +## 2. Technical risks + +| Risk | Likelihood | Impact | Mitigation / management | +| --- | --- | --- | --- | +| ORM defaults differ from DB server defaults | Medium | Medium | Prefer ORM writes; add tests for direct migration schema; document model/migration mismatches. | +| File hash leakage allows artifact probing | Low/Medium | High | Keep file endpoints authenticated where possible; validate access model before public sharing. | +| Global template operations are too permissive | Medium | Medium | Review global-scope creation/folder management before enabling broad users. | +| Provider response shape drift | High | Medium | Normalize provider outputs in one place; add tests with recorded sample responses. | +| Evaluation combined JSON parsing fails | Medium | Medium | Keep per-metric fallback path; test malformed/fenced/partial JSON outputs. | +| Docling OOM or GPU pressure | Medium | High | VRAMGuard, worker peak reporting, OOM estimate bumping, one worker per container process. | +| Background jobs across replicas lose in-memory state | Medium | Medium | Persist `EvalJobRecord`; polling falls back to DB snapshot. | +| Cancellation is process-local for synchronous batch evaluation | Medium | Medium | Use job-based evaluation for cross-worker cancellation; document process-local limitation. | +| Blob/local cache inconsistency | Medium | Medium | Read from `/tmp` first but blob fallback; sync output tree only after successful processing. | +| Text with null bytes breaks PostgreSQL writes | Medium | Low/Medium | Use `sanitize_text()` in DB write paths. | +| CORS default is permissive | Medium | High in production | Set explicit `CORS_ALLOWED_ORIGINS` in production. | +| Telemetry write blocks request loop | Low | Medium | CostTracker schedules DB metrics updates through executor. | + +## 3. Recommended test coverage + +## 3.1 Auth and security tests + +- `get_current_user` accepts valid Authorization bearer token. +- `get_current_user` rejects missing token, invalid token, expired token. +- `ALLOWED_EMAILS` denies non-allowlisted email. +- Auth proxy forwards headers and preserves `Set-Cookie` behavior. +- CORS uses `*` when unset and comma-separated origins when configured. +- File/figure/table endpoints reject unsafe filenames. +- `sanitize_text()` removes null bytes/control chars and preserves tab/newline/carriage return. + +## 3.2 Database/model tests + +- Alembic migrations create all expected tables, indexes, and constraints. +- ORM insert defaults populate expected fields for `AppSession`, `Document`, `ExtractionResult`, and `EvalJobRecord`. +- Unique upsert constraints work for: + - extraction `(document_id, entity_name, model_id)`; + - evaluation `(extraction_result_id, metric, judge_model)`; + - template permission `(template_id, user_id)`; + - template version `(template_id, version)`. +- `PromptTemplate.entities` model/migration nullability mismatch is either fixed or documented by tests. +- Direct SQL insert behavior is known for fields without server defaults. + +## 3.3 File and document processing tests + +- Upload same bytes twice returns same hash and dedupe flags. +- Upload metadata is written to expected blob path. +- DB document registration failure does not fail upload. +- `resolve_processed_processor()` respects preferred processor and fallback order. +- `is_file_processed()` requires `document.md`. +- `get_processing_file_bytes()` reads local cache before blob and caches blob downloads. +- `build_document_view()` returns stable top-level and `processingResult` fields. +- Artifact availability flags reflect blob state. +- Figure/table fallback enumeration works when metadata counts are missing. + +## 3.4 Azure parser tests + +- Azure unavailable returns false availability. +- File source and URL source build different Azure analyze requests. +- Successful conversion writes `document.md`, `raw_analysis.json`, `metadata.json`, figures, and tables. +- Table extraction from markdown HTML is correct. +- Missing Azure figure result id does not fail the whole conversion. +- Error path returns structured failure with conversion id. + +## 3.5 Docling parser and VRAM tests + +- Worker returns structured success with markdown, raw analysis, image info, page count, peak VRAM. +- Worker returns structured failure on exception. +- Markdown table replacement preserves intended table order. +- `VRAMGuard.acquire_slot()` handles admission, queue count, release, and timeout. +- `VRAMGuard.report_worker_result()` updates per-worker estimate and max workers. +- `VRAMGuard.report_oom()` bumps estimate and can shrink max workers. +- Persisted VRAM state loads only when version and age are valid. + +## 3.6 Bounding-box tests + +- Azure normalization converts inch page dimensions and polygons to points. +- Azure paragraph match works for exact and fuzzy references. +- Azure line fallback works when paragraph match is absent. +- Azure figure-reference extraction detects Figure/Fig variants. +- Docling normalization returns expected page/paragraph/table/figure shape. +- Docling polygon extraction handles valid and malformed polygons. +- Unknown processor raw analysis passes through unchanged. + +## 3.7 LLM provider tests + +Use mocked provider responses, not live API calls, for unit tests. + +- `LLMService` dispatches correctly by `model_type`. +- Disabled provider returns `success=False` without crashing. +- Timeout wrapper logs and re-raises timeout. +- Azure structured extraction success maps answer/references/meta. +- Azure fallback path handles structured-output failure. +- Gemini structured output parses JSON and records retry metadata. +- Anthropic JSON-prompt mode parses answer/references. +- Llama primary and fallback strategies return expected strategy metadata. +- Macbook queue serializes concurrent requests. +- vLLM strips `vllm-` prefix before sending model id. +- Successful provider calls record session metrics. + +## 3.8 Extraction flow tests + +- Missing markdown returns API error. +- One request with multiple entities runs all entity tasks and returns per-entity results. +- Entity-level system prompt is passed to provider call. +- Figure context includes generated figure summaries when present. +- Provider references are matched to bbox data when raw analysis exists. +- Extraction persistence upserts existing result instead of duplicating. +- Multi-document session without file hash refuses to guess target document. +- Failed entity extraction can coexist with successful entities in one response. + +## 3.9 Evaluation tests + +- Metric factory map creates correctness/completeness/relevance/safety metrics. +- Correctness/completeness are skipped without expected output. +- Combined scoring parses direct JSON, fenced JSON, extracted JSON block, and partial salvaged metric entries. +- Combined scoring clamps scores to `[0,1]`. +- Per-metric fallback runs when combined parse fails. +- Batch evaluation honors cancellation between chunks. +- Evaluation cost is computed from adapter call history. +- Result storage saves, reads, lists, and deletes JSON files. + +## 3.10 Evaluation job tests + +- `create_job()` builds expected total from tasks x providers. +- `submit_job()` stores in memory and creates DB record. +- `get_job()` returns memory job first, DB proxy second. +- `cancel_job()` cancels local task handles when local. +- `cancel_job()` marks DB cancelled when non-local. +- `_run_single_eval()` persists session evaluation result on success. +- Per-job concurrency changes when multiple jobs run. +- Completed jobs are cleaned after TTL. + +## 3.11 Session tests + +- Create session with no docs, one doc, partial doc failures, and all doc failures. +- Config-only update returns lightweight session. +- `evaluation_config` and `files_config` merge instead of replace nested values unexpectedly. +- Extraction result matching uses file hash/document id correctly. +- Evaluation result matching preserves document-specific results. +- Human-score update applies to intended judge/model metrics. +- Restore-view returns expected primary file and uploaded files. +- Shared session requires group membership. + +## 3.12 Group tests + +- Group creation creates owner membership. +- Non-member cannot read group detail. +- System admin can read/update/delete according to service logic. +- Admin/owner can add members. +- Adding owner role normalizes to admin for new members. +- Cannot change to/from owner through role update endpoint. +- Only owner can promote member to admin. +- Only-owner self-removal is blocked. +- Membership responses are enriched with user profile data. + +## 3.13 Template/folder tests + +- Create template validates scope. +- Group-scope create requires group id and membership. +- Get/list templates enforce `_can_read()`. +- `_can_edit()` denies immutable templates. +- Update creates `TemplateVersion` snapshot before mutation. +- Revert creates a new version through update path. +- Fork creates user-scope mutable copy. +- Scope transitions enforce old/new scope permission rules. +- Explicit permission upsert uses unique constraint. +- Folder create validates parent scope/group. +- Folder delete refuses non-empty folders. + +## 4. Manual smoke tests + +For integrated backend verification: + +1. Start Postgres, auth sidecar, backend, and frontend. +2. Log in through Better Auth. +3. Upload a PDF. +4. Process with Azure Document Intelligence. +5. Fetch markdown/content/analysis/figures. +6. Extract at least two entities with one model. +7. Save session and reload restore view. +8. Run evaluation for one extraction. +9. Submit background evaluation job and poll until completed. +10. Create group, share session, verify another group member can open shared restore view. +11. Create template, update it, verify version history, fork it, and change scope. +12. Check `/api/server/session-metrics` and `/api/server/logs`. + +## 5. Documentation maintenance checklist + +When backend changes: + +- New route: update `../02-api-surface.md` and `api-endpoint-index.md`. +- New ORM model/migration: update `../03-data-models.md` and `class-index.md`. +- New schema: update `../04-schemas.md` and `class-index.md`. +- New parser/artifact: update `../05-document-processing.md` and data-flow diagrams. +- New model provider: update `../06-llm-layer.md`, security/config docs, and tests. +- New evaluation metric/provider: update `../08-evaluation-flow.md`. +- New auth/permission behavior: update `../11-auth-security-observability.md` and risk table. diff --git a/docs/backend/images/auth-observability-workflow.png b/docs/backend/images/auth-observability-workflow.png new file mode 100644 index 0000000..3223942 Binary files /dev/null and b/docs/backend/images/auth-observability-workflow.png differ diff --git a/docs/backend/images/auth-observability-workflow.svg b/docs/backend/images/auth-observability-workflow.svg new file mode 100644 index 0000000..314a594 --- /dev/null +++ b/docs/backend/images/auth-observability-workflow.svg @@ -0,0 +1,110 @@ + + Auth, security, and observability workflow + Cross-cutting workflow for Better Auth session validation, auth proxying, service-level authorization, structured logging, request IDs, metrics, tracing, and cost tracking. + + + + + + + + + + + + Auth, Security, and Observability + Authentication is DB-backed Better Auth session lookup; observability is attached at request middleware and provider-call boundaries. + + + AUTHENTICATED API REQUEST + + Frontend request + Bearer token preferred + + get_current_user() + session token dependency + + PostgreSQL lookup + AuthSession joined to User + + Access checks + expiry and ALLOWED_EMAILS + + + + + + AUTH PROXY AND SERVICE AUTHORIZATION + + /api/auth/{path} + registered before generic auth routes + + Better Auth sidecar + Set-Cookie, redirects, JSON + + Service-level authorization + sessions, groups, templates + + Path safety + hash paths and filename validation + + + + + + REQUEST OBSERVABILITY + + HTTP middleware + bind 12-char request id + + Structured logs + status, method, path, duration + + Prometheus + /metrics and provider metrics + + OpenTelemetry + optional OTLP traces + + + + + + + COST AND SESSION TELEMETRY + + Provider call succeeds + LLM or parser metadata + + CostTracker + model pricing and duration + + Session totals + total_cost, latency, calls + + Server endpoints + metrics, logs, model catalog + + + + + Non-fatal integrations: Prometheus instrumentation, Loki handler, and OTLP tracing log warnings if unavailable so the API can still serve traffic. + diff --git a/docs/backend/images/backend-runtime-architecture.png b/docs/backend/images/backend-runtime-architecture.png new file mode 100644 index 0000000..4220962 Binary files /dev/null and b/docs/backend/images/backend-runtime-architecture.png differ diff --git a/docs/backend/images/backend-runtime-architecture.svg b/docs/backend/images/backend-runtime-architecture.svg new file mode 100644 index 0000000..f40ecfe --- /dev/null +++ b/docs/backend/images/backend-runtime-architecture.svg @@ -0,0 +1,145 @@ + + Backend runtime architecture + Layered architecture of the Summarization Tool backend from browser requests through FastAPI routers, services, persistence, external providers, and observability. + + + + + + + + + + + + Backend Runtime Architecture + FastAPI coordinates authenticated workflows, hash-addressed document artifacts, provider calls, persistence, and telemetry. + + + ENTRY AND AUTH EDGE + + Frontend browser + + FastAPI app + backend/main.py create_app() + + Core boundary + auth, CORS, config, logging + + Better Auth sidecar + proxied /api/auth/* traffic + + + + + + API ROUTERS + + files + + documents + + extractions + + evaluations + + sessions + + groups + + templates + + chat + + server metrics + + paragraph APIs + + + + SERVICE LAYER + + Document services + organized files, processors + bbox normalization + + LLMService + provider dispatch + timeouts, metrics + + Evaluation service + DeepEval adapters + job queue and scoring + + Workflow services + sessions, groups + templates and folders + + Telemetry + CostTracker + session metrics + + + + + + + + PERSISTENCE, PROVIDERS, AND OPERATIONS + + PostgreSQL + ORM models and Alembic + + Azure Blob Storage + global/{hash}/ artifacts + + Document parsers + Azure DI and Docling + + Model providers + Azure, Vertex, vLLM, local + + Observability + Prometheus, OTLP, Loki + + + + + + + + + + + + Runtime app + + Routers + + Services + + Persistence + + External providers + diff --git a/docs/backend/images/data-model-relationships.png b/docs/backend/images/data-model-relationships.png new file mode 100644 index 0000000..fc6b175 Binary files /dev/null and b/docs/backend/images/data-model-relationships.png differ diff --git a/docs/backend/images/data-model-relationships.svg b/docs/backend/images/data-model-relationships.svg new file mode 100644 index 0000000..c80943e --- /dev/null +++ b/docs/backend/images/data-model-relationships.svg @@ -0,0 +1,132 @@ + + Backend data model relationships + Entity relationship style overview showing Better Auth tables, workflow session tables, evaluation jobs, groups, and template tables in the backend database. + + + + + + + + + + + + Backend Data Model Relationships + The database separates login/auth tables from application workflow state, collaboration, template assets, and background job status. + + + BETTER AUTH + + user + profile and admin flag + + session + token and expiry + + account + + verification + + + + WORKFLOW SESSION STATE + + app_sessions + workflow owner, config, metrics + + documents + file hash and parser metadata + + extraction_results + entity, model, answer, refs + + + + + EVALUATION + + evaluation_results + metric scores per extraction + + eval_jobs + background job status + + + + + COLLABORATION + + groups + team workspace + + user_groups + viewer/member/admin/owner + + + + + + TEMPLATE WORKSPACE + + template_folders + scope and hierarchy + + prompt_templates + entities, prompts, scope + + template_versions + snapshots before update + + template_permissions + per-user overrides + + + + + + + USER SUPPORT TABLES + + user_preferences + default model settings + + login_history + audit trail + + legacy user_prompt_templates + + + + + + + + auth identity + + workflow records + + evaluation records + + group access + + template assets + diff --git a/docs/backend/images/document-processing-workflow.png b/docs/backend/images/document-processing-workflow.png new file mode 100644 index 0000000..6dd50a2 Binary files /dev/null and b/docs/backend/images/document-processing-workflow.png differ diff --git a/docs/backend/images/document-processing-workflow.svg b/docs/backend/images/document-processing-workflow.svg new file mode 100644 index 0000000..a46c802 --- /dev/null +++ b/docs/backend/images/document-processing-workflow.svg @@ -0,0 +1,129 @@ + + Document upload and processing workflow + Workflow from PDF upload through validation, SHA-256 deduplication, blob storage, processor selection, artifact generation, synchronization, database updates, and document view creation. + + + + + + + + + + + + Document Upload and Processing Workflow + Files are content-addressed by SHA-256; parser outputs are persisted to blob storage and cached under /tmp for the current container. + + + UPLOAD AND DEDUPLICATION + + PDF upload + POST /api/upload + multipart file + + Validate file + PDF extension and magic + 20 MB body limit + + Compute hash + SHA-256 digest + stable file id + + Azure Blob + global/{hash}/original.pdf + global/{hash}/metadata.json + dedupe if already present + + Document row + best-effort DB register + when user is known + + + + + + + PROCESSING REQUEST + + Process endpoint + POST /api/documents/process + file/{file_hash} + + Strict cache check + processed/{processor}/document.md + hit returns cached view + + Cache hit path + read metadata and markdown + rebuild document_view + + Cache miss path + download original to /tmp + prepare output directory + + Processor selection + auto prefers Azure DI when available + otherwise falls back to Docling + + + + + + + PARSER EXECUTION AND ARTIFACT TREE + + Azure Document Intelligence + markdown, raw analysis + figures and tables + + Docling + remote client or local worker + VRAM guard on local path + + Local output in /tmp + document.md + raw_analysis.json + metadata.json + figures/* and tables/* + + Persist processed tree + sync to blob prefix + global/{hash}/processed/{processor} + warm /tmp cache on reads + cross-replica persistence + + + + + + + BACKEND OUTPUT CONTRACT + + Document DB metadata update + + canonical document_view payload + + restore-ready uploadedFiles state + + + diff --git a/docs/backend/images/evaluation-workflow.png b/docs/backend/images/evaluation-workflow.png new file mode 100644 index 0000000..be280c8 Binary files /dev/null and b/docs/backend/images/evaluation-workflow.png differ diff --git a/docs/backend/images/evaluation-workflow.svg b/docs/backend/images/evaluation-workflow.svg new file mode 100644 index 0000000..69af071 --- /dev/null +++ b/docs/backend/images/evaluation-workflow.svg @@ -0,0 +1,113 @@ + + Evaluation workflow + LLM-as-a-judge evaluation workflow showing synchronous evaluation, batch jobs, provider adapters, metrics, concurrency controls, persistence, cancellation, and result storage. + + + + + + + + + + + + Evaluation Workflow + Evaluation can run synchronously or as a persisted background job; both paths converge on the same judge-model service and metrics. + + + ENTRYPOINTS + + Synchronous endpoints + evaluate, batch, custom + + Background jobs + POST /api/evaluations/jobs + + Polling and cancel + GET job, POST cancel + + Frontend status + progress, errors, scores + + + SCORING SERVICE + + EvaluationService + create_evaluation_model() + evaluate_extraction() + + Metric factories + correctness, completeness + relevance, safety, custom + + Combined scoring + one JSON judge prompt + fallback to per-metric GEval + + Judge adapters + Azure, Vertex/Gemini + Anthropic Vertex + + + + + + + BACKGROUND JOB ORCHESTRATION + + EvalJob + tasks x providers flattened + + Concurrency gates + global 30 plus per-job slots + + Durable status + eval_jobs DB record + + Cancellation + cancel live tasks or mark DB + + + + + + + + RESULTS, COST, AND SESSION PERSISTENCE + + EvaluationResultStorage + JSON files for batch outputs + + SessionService + persist scores to evaluation_results + + CostTracker + judge-call cost and latency + + API response/status + aggregate score and metric reasons + + + + + diff --git a/docs/backend/images/extraction-grounding-workflow.png b/docs/backend/images/extraction-grounding-workflow.png new file mode 100644 index 0000000..7d450f0 Binary files /dev/null and b/docs/backend/images/extraction-grounding-workflow.png differ diff --git a/docs/backend/images/extraction-grounding-workflow.svg b/docs/backend/images/extraction-grounding-workflow.svg new file mode 100644 index 0000000..44e192b --- /dev/null +++ b/docs/backend/images/extraction-grounding-workflow.svg @@ -0,0 +1,111 @@ + + Entity extraction and grounding workflow + Workflow for loading processed markdown, augmenting it with figure context, running one LLM extraction per entity, normalizing results, matching references to bounding boxes, and persisting extraction rows. + + + + + + + + + + + + Entity Extraction and Reference Grounding + The backend separates answer generation from visual grounding: providers return text and references, then processor-specific matchers attach boxes. + + + REQUEST AND DOCUMENT CONTEXT + + POST /api/extract + session, file hash, entities, model + + Load markdown + raw_analysis.content or document.md + + Add figure context + captions and generated summaries + + Enhanced markdown + document plus figures section + + + + + + PER-ENTITY CONCURRENT WORK + + Entity tasks + async gather for cloud + semaphore limit 48 + + LLMService call + provider dispatch by model_type + Macbook processed sequentially + + Provider response + answer/content plus references + meta tokens and duration + + Normalize result + stable entity payload + cost estimated from meta + + + + + + + REFERENCE GROUNDING + + Load raw analysis + processor artifact from blob/cache + + Azure matcher + paragraph and line matching + figure reference expansion + bounding regions retained + + Docling matcher + paragraph and page matching + polygon extraction + normalized output shape + + Grounded references + page, polygon, matched text + + + + + + + PERSISTENCE AND RESPONSE + + SessionService.add_extraction_result_fast() + + upsert by document, entity, model + + return extracted_entities payload + + + + diff --git a/docs/backend/images/llm-provider-routing.png b/docs/backend/images/llm-provider-routing.png new file mode 100644 index 0000000..43bbc6e Binary files /dev/null and b/docs/backend/images/llm-provider-routing.png differ diff --git a/docs/backend/images/llm-provider-routing.svg b/docs/backend/images/llm-provider-routing.svg new file mode 100644 index 0000000..84d357c --- /dev/null +++ b/docs/backend/images/llm-provider-routing.svg @@ -0,0 +1,114 @@ + + LLM provider routing workflow + LLMService routes extraction, paragraph generation, and figure summary requests to provider clients, wraps calls with timeouts, normalizes responses, and records session metrics. + + + + + + + + + + + + LLM Provider Routing + A single service facade keeps router code stable while provider clients handle model-specific API details. + + + CALLERS + + entity extraction router + + paragraph generator + + figure summary generation + + chat and future workflows + + + SERVICE FACADE + + LLMService + extract_entities_from_markdown() + generate_paragraph(), image content + + Timeout wrapper + 240s default, provider overrides + timeout log file for diagnostics + + Response normalization + success, content, answer, references + meta, token usage, raw payload + + CostTracker + records provider/model + tokens, latency, cost + + + + + + + PROVIDER CLIENTS + + Azure + OpenAI SDK or REST + 3 retry attempts + + Gemini + Vertex AI endpoint + JSON schema support + + Anthropic + Claude via Vertex + prompt-enforced JSON + + Llama + Vertex MaaS + primary plus fallback + + Macbook + Ollama-compatible + FIFO single worker + + vLLM + OpenAI-compatible + 600s timeout + + + + + + + + + COMMON CONTRACT BACK TO ROUTERS + + success / error + + answer / content + + references / raw + + meta: model, tokens, duration + + diff --git a/docs/backend/images/session-sharing-template-workflow.png b/docs/backend/images/session-sharing-template-workflow.png new file mode 100644 index 0000000..515773f Binary files /dev/null and b/docs/backend/images/session-sharing-template-workflow.png differ diff --git a/docs/backend/images/session-sharing-template-workflow.svg b/docs/backend/images/session-sharing-template-workflow.svg new file mode 100644 index 0000000..eaeedfd --- /dev/null +++ b/docs/backend/images/session-sharing-template-workflow.svg @@ -0,0 +1,108 @@ + + Session sharing and template workflow + Workflow diagram for session persistence, restore-view construction, group-based sharing, and prompt template/folder authorization and versioning. + + + + + + + + + + + + Sessions, Sharing, and Templates + Workflow state, collaborative visibility, and reusable prompts are separate service boundaries that meet through user and group authorization. + + + SESSION LIFECYCLE AND RESTORE + + Session API + create, list, update, delete + + SessionService + DB-to-Pydantic aggregate + + Workflow rows + session, docs, extractions, evals + + Restore view + build_document_view per file + + + + + + GROUP SHARING + + GroupService + create group + add/update/remove members + + Role checks + owner, admin, member, viewer + system admin bypass + + Share session + set shared_with_group_id + shared_by and shared_at + + Shared read + membership is enough to view + ownership still controls edits + + + + + + + TEMPLATE AND FOLDER WORKSPACE + + Template API + CRUD, fork, scope change + permissions, versions + + TemplateService + read/edit/owner checks + immutability guard + + Versioning + snapshot before mutation + revert creates new current version + + FolderService + scope-compatible hierarchy + no delete while non-empty + + Tables: prompt_templates, template_versions, template_permissions, template_folders + + + + + + + + + + SHARED AUTHORIZATION PRINCIPLE + Sessions are owned by users and optionally shared to groups. Templates are scoped to user, group, or global workspaces; explicit permissions can override reads/writes, but immutability always blocks edits. + diff --git a/docs/frontend/01-app-shell.md b/docs/frontend/01-app-shell.md new file mode 100644 index 0000000..688a42b --- /dev/null +++ b/docs/frontend/01-app-shell.md @@ -0,0 +1,167 @@ +# App Shell + +> *`App.tsx` is the spine of the entire frontend. It owns all workflow state, controls which page is visible, persists sessions across browser reloads, and guards against accidental navigation away from in-progress work. Every page component is a child of App.tsx β€” pages read their inputs from it and report their outputs back to it via callbacks.* + +## 1. Overview + +`App.tsx` (~2100 lines) does four things: + +1. **Owns `DocumentData`** β€” the single object that accumulates everything a reviewer has done in a session: uploaded files, processing results, entity configs, extraction results, evaluation results. +2. **Controls navigation** β€” a `currentStep` string determines which page renders. There is no URL router; the browser address bar does not change between steps. +3. **Persists state** β€” saves `currentStep` and `sessionId` to `localStorage` so a reload restores where the reviewer left off. +4. **Guards navigation** β€” prevents accidental navigation away from a step with in-flight API calls or unsaved results. + +--- + +## 2. The `DocumentData` interface + +`DocumentData` is the central data structure. It is passed as a prop to every page and updated via the `onComplete()` callback pattern. + +```typescript +interface DocumentData { + // --- Upload step --- + file: File | null; // Primary uploaded file (single-file path) + fileId?: string; // SHA-256 hash of the primary file + uploadResult?: any; // Raw response from POST /api/upload + parser: string; // Selected processor: "azure" | "docling" | "auto" + uploadedFiles?: UploadedFile[]; // All uploaded files (multi-file path) + + // --- Processing step --- + extractedText: string; // Processed markdown from primary file + annotatedOutput: string; // Enhanced markdown (with figure summaries injected) + + // --- Study config step --- + studyType: string; // "toxicology" | "epidemiology" | "custom" + summaryPrompt?: string; // System prompt for paragraph summary generation + selectedModel: string; // Primary model identifier + selectedModels?: string[]; // All selected models (multi-model extraction) + temperature?: number; // Model temperature setting + entities: Entity[]; // Entity definitions + their extraction results + + // --- Extraction step (populated per entity) --- + // Entity.extracted, Entity.answer, Entity.references, + // Entity.extractionsByModel, Entity.duration, etc. + // See Entity interface in appendices/types-interfaces.md + + // --- Session --- + sessionId?: string; // ID of the saved AppSession in the DB + + // --- Config --- + filesConfig?: FilesConfig; // Per-file parser and model settings + evaluationConfig?: EvaluationConfig; // Evaluation metric and model settings +} +``` + +`UploadedFile` extends the single-file fields to support multi-document sessions. Each entry carries its own `fileId`, `extractedText`, `entities`, and so on. + +--- + +## 3. Routing and navigation + +### Step identifiers + +Navigation is driven by a `currentStep` string, not a URL path. The full set: + +| Step ID | Page rendered | Role | +|---|---|---| +| `login` | `LoginPage` | Auth entry point | +| `auth_callback` | `AuthCallback` | OAuth redirect handler | +| `upload` | `UploadPage` | Workflow step 1 | +| `processing` | `ProcessingPage` | Workflow step 2 | +| `study_selection` | `BatchStudySelectionPage` | Workflow step 3 | +| `extraction` | `EntityExtractionPage` | Workflow step 4 | +| `evaluation` | `EvaluationPage` | Workflow step 5 | +| `simplified` | `SimplifiedFlowPage` | One-click pipeline | +| `chat` | `ChatPage` | Document Q&A | +| `executive` | `ExecutiveModePage` | Executive summary | +| `history` | `SessionHistoryPage` | Session browser | +| `templates` | `TemplateWorkspacePage` | Template manager | +| `groups` | `GroupManagementPage` | Group manager | + +### Tool overlay navigation + +Tool overlays (`chat`, `executive`, `history`, `templates`, `groups`) can be opened from any workflow step. When the reviewer enters an overlay, App.tsx saves the current step in `previousWorkflowStep`. The overlay's Back button calls `setCurrentStep(previousWorkflowStep)` to return exactly where they were. + +### Completed steps + +App.tsx derives a `completedSteps` set from the contents of `DocumentData`: + +- `upload` is complete when `documentData.uploadedFiles` or `documentData.fileId` is populated. +- `processing` is complete when `documentData.extractedText` is non-empty. +- `study_selection` is complete when `documentData.entities` has at least one entry. +- `extraction` is complete when any entity has an `extracted` or `answer` value. +- `evaluation` is complete when any entity has `evaluationResults`. + +Reviewers can only navigate forward to the next step or backward to a completed step. Skipping ahead is blocked. + +--- + +## 4. The `onComplete()` callback pattern + +Pages do not write to global state directly. Instead, each page receives an `onComplete` prop: + +```typescript +// Example: UploadPage reports its results back to App.tsx +onComplete: (updates: Partial) => void +``` + +When a page finishes its work, it calls `onComplete({ uploadedFiles: [...], parser: "azure" })`. App.tsx merges this into `documentData` with a shallow merge (top-level keys only) and advances `currentStep` to the next step. + +Pages also receive: + +| Prop | Type | Purpose | +|---|---|---| +| `documentData` | `DocumentData` | Read-only access to current workflow state | +| `onComplete` | `(updates) => void` | Report results and advance to next step | +| `onInvalidateDownstream` | `(fromStep) => void` | Mark later steps as stale | +| `onInFlightChange` | `(bool) => void` | Signal that an API call is in progress (blocks navigation) | +| `onNavigate` | `(step) => void` | Navigate to a specific step (used by overlays) | + +--- + +## 5. Downstream invalidation + +When a reviewer modifies an early step β€” for example, re-uploading a file on the Upload page β€” the work done in later steps may no longer be valid. App.tsx tracks this with a `staleDownstream` flag. + +When `onInvalidateDownstream("processing")` is called: +- `extractedText`, `annotatedOutput`, entities, extraction results, and evaluation results are cleared from `documentData`. +- The user sees a visual banner: "Results invalidated β€” re-run to update." +- Navigation to downstream steps is still permitted (for inspection), but the stale indicator makes clear the results are outdated. + +--- + +## 6. Session persistence + +On every meaningful state change, App.tsx writes two values to `localStorage`: + +``` +summarization_current_step β†’ "extraction" +summarization_session_id β†’ "abc123" +``` + +On app load, if both values are present: + +1. App.tsx calls `GET /api/sessions/{sessionId}/restore-view`. +2. The response re-hydrates `documentData` with the full session state. +3. `currentStep` is restored to the saved step. +4. The reviewer continues exactly where they left off. + +If the restore call fails (session deleted, token expired), App.tsx clears localStorage and starts a fresh session at `upload`. + +--- + +## 7. Navigation guards + +Two guards prevent accidental data loss: + +**In-flight guard** (`onInFlightChange`): When a page signals that an API call is in progress, App.tsx disables the step navigation bar. Clicking another step shows a confirmation dialog: "An extraction is in progress. Leave anyway?" + +**Unsaved changes guard** (`rerunConfirm`): When a reviewer tries to navigate backward from a step with results (e.g., going back from Extraction to Study Config), App.tsx shows: "Going back will clear your extraction results. Continue?" The reviewer must confirm before the state is cleared. + +--- + +## 8. Authentication flow in App.tsx + +On mount, App.tsx calls `getSession()` from `authUtils.ts`. If no session exists, it redirects to `login`. If a session exists, it fetches user info and proceeds to restore state from localStorage or start at `upload`. + +The session check is non-blocking β€” the app renders a loading spinner while `getSession()` is in flight and never flashes unauthenticated content to the user. diff --git a/docs/frontend/02-auth.md b/docs/frontend/02-auth.md new file mode 100644 index 0000000..bf4b5ff --- /dev/null +++ b/docs/frontend/02-auth.md @@ -0,0 +1,188 @@ +# Authentication + +> *Before a reviewer can use the tool they need to prove who they are. Science-GPT uses GitHub OAuth β€” the reviewer clicks "Sign in with GitHub", gets redirected to GitHub, approves access, and is sent back to the app with a session token. This document covers the login page, the OAuth callback handler, and `authUtils.ts` β€” the utility module that attaches auth tokens to every API call the app makes.* + +## 1. Components + +| File | Purpose | +|---|---| +| `components/LoginPage.tsx` | GitHub login UI | +| `components/AuthCallback.tsx` | Handles the redirect back from GitHub | +| `utils/authUtils.ts` | Session management, token refresh, authenticated fetch | + +--- + +## 2. LoginPage + +**File:** `components/LoginPage.tsx` (~192 lines) + +### What the user sees + +A split-screen layout: + +- **Left (60%)** β€” hero section with the headline "Complexity, solved. Focus, restored.", animated `AuroraText`, and three stat callouts (10Γ— faster, 98% accuracy, 500+ documents processed). +- **Right (40%)** β€” login form with a single "Sign in with GitHub" button, a loading spinner during auth, and an error message if the flow fails. + +### What happens on click + +``` +User clicks "Sign in with GitHub" + β”‚ + β–Ό +signInWithGitHub() ← authUtils.ts + β”‚ + β–Ό +POST /api/auth/sign-in/social ← Better Auth sidecar + β”‚ + β–Ό +Redirect to github.com/login/oauth/authorize +``` + +The browser leaves the app entirely. GitHub handles the credential check, then redirects back to the app's callback URL. + +### State + +| State | Type | Purpose | +|---|---|---| +| `isLoading` | `boolean` | Shows spinner while auth is in flight | +| `error` | `string \| null` | Displays error message if GitHub rejects or network fails | + +--- + +## 3. AuthCallback + +**File:** `components/AuthCallback.tsx` (~151 lines) + +This component renders briefly after GitHub redirects the browser back to the app. It never shows meaningful UI β€” it processes the OAuth result and then navigates away. + +### What happens + +``` +Browser lands on /auth/callback?code=... + β”‚ + β–Ό +AuthCallback mounts + β”‚ + β–Ό +Better Auth sidecar exchanges code for session + β”‚ + β–Ό +Session cookie set by sidecar response + β”‚ + β–Ό +App.tsx detects valid session β†’ navigate to "upload" + β”‚ + β–Ό (on failure) +Toast notification shown β†’ redirect back to "login" +``` + +The `code` query parameter from GitHub is handled server-side by the Better Auth sidecar β€” `AuthCallback.tsx` does not parse it directly. Its job is to wait for the sidecar to finish and then let `App.tsx`'s session check take over. + +--- + +## 4. `authUtils.ts` + +**File:** `utils/authUtils.ts` + +This is the most important auth file in the frontend. Every single API call in the app goes through `authenticatedFetch()`, which lives here. + +### Key functions + +#### `getSession()` + +```typescript +async function getSession(): Promise +``` + +Fetches the current session from the Better Auth sidecar at `GET /api/auth/get-session`. Caches the result in `_cachedSession` to avoid redundant network calls on rapid re-renders. + +**Deduplication:** If `getSession()` is called concurrently (e.g., two components mount at the same time), all calls share the same in-flight promise and receive the same result. This prevents a burst of session requests on app load. + +#### `getValidToken()` + +```typescript +async function getValidToken(): Promise +``` + +Returns the current bearer token. If the token expires within 30 seconds, it refreshes the session first before returning. This ensures tokens passed to API calls are never about to expire mid-request. + +#### `authenticatedFetch()` + +```typescript +async function authenticatedFetch(url: string, options?: RequestInit): Promise +``` + +A drop-in replacement for `fetch()` that automatically: + +1. Calls `getValidToken()` to get a fresh bearer token. +2. Adds `Authorization: Bearer {token}` to the request headers. +3. Adds `X-Session-Id: {sessionId}` if a session ID is available (used for cost tracking on the backend). +4. If the response is `401`, refreshes the session and retries the request **once**. +5. If the retry also returns `401`, throws an auth error (does not loop). + +All API calls across every page component use `authenticatedFetch()` β€” not raw `fetch()`. + +#### `installVisibilityRefreshListener()` + +```typescript +function installVisibilityRefreshListener(): void +``` + +Registers a `visibilitychange` event listener on `document`. When the browser tab becomes visible again (after being hidden), this immediately calls `getSession()` to refresh the cached token. + +**Why this exists:** Chrome aggressively throttles background tabs. A reviewer who leaves the app open in a background tab for 30+ minutes may return to find their session token expired. Without this listener, the next API call would fail with 401 and show an error. With it, the token is silently refreshed the moment they switch back to the tab. + +`App.tsx` calls `installVisibilityRefreshListener()` once on mount. + +#### `signInWithGitHub()` + +Initiates the OAuth flow. Calls the Better Auth client to redirect the browser to GitHub's authorization URL. + +#### `signOut()` + +Posts to `POST /api/auth/sign-out`, which invalidates the server-side session. Clears `_cachedSession` locally, then reloads the page to land on the login screen. + +#### `getCurrentUser()` + +Returns `{ id, email, name, image }` from the cached session. Used by App.tsx to display the user's name and avatar in the navigation bar. + +--- + +## 5. Session token lifecycle + +``` +App loads + β”‚ + β–Ό +getSession() β†’ cache result in _cachedSession + β”‚ + β”œβ”€β”€ No session β†’ navigate to "login" + β”‚ + └── Valid session + β”‚ + β–Ό + App renders + β”‚ + β–Ό + User makes API call + β”‚ + β–Ό + authenticatedFetch() + β”‚ + β”œβ”€β”€ Token still valid β†’ attach header β†’ send request + β”‚ + └── Token expires within 30s β†’ getValidToken() refreshes first β†’ attach header β†’ send + β”‚ + └── Response 401 β†’ refresh session β†’ retry once + β”‚ + └── Still 401 β†’ throw AuthError β†’ user redirected to login +``` + +--- + +## 6. Security notes + +- Tokens are stored in `HttpOnly` cookies by the Better Auth sidecar β€” JavaScript cannot read them directly. The frontend only handles the bearer token returned from `getSession()`, which is a short-lived JWT. +- `authenticatedFetch()` never logs tokens. +- The visibility refresh listener only refreshes the session cache β€” it does not store credentials anywhere. +- `signOut()` always invalidates the server-side session, not just the local cache. A reviewer logging out on one device/tab invalidates the session for all. diff --git a/docs/frontend/03-upload.md b/docs/frontend/03-upload.md new file mode 100644 index 0000000..158a02d --- /dev/null +++ b/docs/frontend/03-upload.md @@ -0,0 +1,111 @@ +# Upload Page (Workflow Step 1) + +> *The first thing a reviewer does is bring their documents into the tool. The Upload page handles file selection, uploads each file to the backend, immediately triggers document processing, and lets the reviewer choose which parser to use. By the time the reviewer clicks "Next", every file has been uploaded and converted into structured text that the AI models can read.* + +**File:** `components/UploadPage.tsx` (~914 lines) + +--- + +## 1. UI sections + +| Section | Purpose | +|---|---| +| Drag-and-drop zone | Drop files or click to open file picker | +| File list | Shows each file with upload + processing status indicators | +| Parser selector (global) | Choose Azure Document Intelligence, Docling, or Auto for all files | +| Per-file parser override | Override the global parser for individual files | +| Status indicators | Pending / uploading / processing / processed / error per file | +| Error panel | Per-file error messages with retry option | + +--- + +## 2. Upload and processing flow + +Upload and processing happen automatically β€” the reviewer does not need to click a separate "Process" button. + +``` +Reviewer drops files + β”‚ + β–Ό +For each file (concurrent): + POST /api/upload (multipart) + β”‚ Returns: { file_hash, blob_path, is_duplicate, filename, size } + β”‚ + β–Ό + Store file_hash in uploadResults[filename] + β”‚ + β–Ό + POST /api/documents/process/file/{file_hash} + β”‚ Body: { processor: "azure" | "docling" | "auto" } + β”‚ Returns: document view (markdown, figures, tables, metadata) + β”‚ + β–Ό + Store result in processedFiles[filename] + β”‚ + β–Ό + Mark file as "processed" in status display +``` + +If the backend returns a cached result (file already processed with this parser), processing completes in milliseconds. The `is_duplicate` flag from the upload response is shown in the UI so the reviewer knows the file was recognised. + +--- + +## 3. State + +| State field | Type | Purpose | +|---|---|---| +| `selectedFiles` | `File[]` | Files the reviewer has selected but not yet uploaded | +| `uploadResults` | `Map` | Keyed by filename; stores file hash and blob path | +| `processedFiles` | `Map` | Keyed by filename; stores markdown and artifact metadata | +| `uploadErrors` | `Map` | Per-file upload error messages | +| `processingErrors` | `Map` | Per-file processing error messages | +| `processingFiles` | `Set` | Filenames currently being processed (for spinner display) | +| `fileParsers` | `Map` | Per-file parser override | +| `globalParser` | `string` | Default parser applied to all files | + +--- + +## 4. API calls + +| Method | Path | When | Purpose | +|---|---|---|---| +| `POST` | `/api/upload` | On file drop/select | Upload file bytes, get file hash | +| `POST` | `/api/documents/process/file/{file_hash}` | Immediately after upload | Convert file to markdown/figures/tables | + +--- + +## 5. `onComplete()` payload + +When the reviewer clicks "Next", the page calls: + +```typescript +onComplete({ + uploadedFiles: [ + { + fileId: "sha256hash", + filename: "study_001.pdf", + uploadResult: { ... }, + processingResult: { markdown, figureCount, tableCount, ... }, + parser: "azure", + }, + // ... + ], + parser: globalParser, +}) +``` + +App.tsx merges this into `documentData` and navigates to `processing`. + +--- + +## 6. Downstream invalidation + +If the reviewer adds or removes files after some have already been processed, the page calls `onInvalidateDownstream("processing")`. This clears all downstream results (processing output, entity configs, extraction results, evaluation results) from `documentData` and shows the stale warning banner. + +--- + +## 7. Error handling + +- **Upload failure:** Shows per-file error with a Retry button. The file is marked with a red error indicator. Other files continue processing normally. +- **Processing failure:** Shows the backend error message. The reviewer can change the parser and click Retry β€” the backend may succeed with a different processor. +- **Unsupported file type:** Rejected client-side before upload. Accepted types: PDF, DOCX, XLSX, PPTX. diff --git a/docs/frontend/04-processing.md b/docs/frontend/04-processing.md new file mode 100644 index 0000000..52e0b51 --- /dev/null +++ b/docs/frontend/04-processing.md @@ -0,0 +1,117 @@ +# Processing Page (Workflow Step 2) + +> *After files are uploaded, the Processing page lets reviewers inspect what the parser actually extracted β€” the structured text, any figures, and any tables. It's an inspection and verification step: reviewers can check that the document parsed correctly, view the raw analysis, and re-process with a different parser if the output looks wrong. They can also view figures and tables with their bounding-box locations highlighted on the original PDF.* + +**File:** `components/ProcessingPage.tsx` (~534 lines) + +--- + +## 1. UI sections + +| Section | Purpose | +|---|---| +| File list (sidebar) | Select which uploaded file to inspect | +| Parser selector | Change parser for the selected file; triggers re-processing | +| Re-process button | Re-run document processing with the current parser selection | +| Tab: Text | Shows the extracted markdown content | +| Tab: Figures | Gallery of extracted figures with captions and metadata | +| Tab: Tables | HTML table viewer for extracted tables | +| Tab: Raw | Syntax-highlighted raw analysis JSON from the parser | +| PDF bounding box viewer | Highlights figure/table locations on the original PDF pages | +| Status indicators | Processing / processed / error per file | + +--- + +## 2. How processing results are loaded + +When this page mounts, `documentData.uploadedFiles` already contains the processing results from the Upload step. The Processing page reads these results β€” it does not re-call the backend unless the reviewer explicitly clicks Re-process. + +``` +Page mounts + β”‚ + β–Ό +Read uploadedFiles from documentData + β”‚ + β–Ό +Populate file list + set first file as active + β”‚ + β–Ό +Display processing result for active file (text / figures / tables) +``` + +--- + +## 3. Re-processing a file + +If a reviewer switches to a different parser and clicks Re-process: + +``` +Reviewer changes parser + clicks Re-process + β”‚ + β–Ό +POST /api/documents/process/file/{file_hash} + Body: { processor: "azure" | "docling" } + β”‚ + β–Ό +Backend returns new document view (may be cached if already processed with this parser) + β”‚ + β–Ό +Update processedFiles[filename] with new result + β”‚ + β–Ό +onInvalidateDownstream("processing") β€” clears extraction + evaluation results +``` + +Re-processing with a different parser does not re-upload the file β€” the original bytes are already in blob storage under the file hash. + +--- + +## 4. The PDF bounding box viewer + +The `PDFBoundingBoxViewer` shared component (see [appendices/component-index.md](appendices/component-index.md)) renders the original PDF in-browser using `pdfjs-dist` and draws coloured overlaid boxes at the coordinates returned by the parser. + +- Azure Document Intelligence returns bounding polygons in inches; the backend normalizes these to page-relative coordinates before sending to the frontend. +- Docling returns bounding boxes in page-relative coordinates directly. +- Clicking a figure or table in the Figures or Tables tab scrolls the PDF viewer to the relevant page and highlights the bounding box. + +--- + +## 5. State + +| State field | Type | Purpose | +|---|---|---| +| `files` | `FileStatus[]` | All uploaded files with their current processing state | +| `activeFile` | `string` | Filename of the currently selected file | +| `selectedTab` | `"text" \| "figures" \| "tables" \| "raw"` | Active content tab | +| `parserOverrides` | `Map` | Tracks parser selection per file | +| `reprocessingFiles` | `Set` | Files currently being re-processed | + +--- + +## 6. API calls + +| Method | Path | When | Purpose | +|---|---|---|---| +| `POST` | `/api/documents/process/file/{file_hash}` | On Re-process click | Re-run parsing with chosen parser | + +--- + +## 7. `onComplete()` payload + +When the reviewer clicks "Next": + +```typescript +onComplete({ + uploadedFiles: updatedUploadedFiles, // with any re-processing results merged in +}) +``` + +App.tsx navigates to `study_selection`. + +--- + +## 8. Error handling + +- **Processing failure on re-process:** Shows the backend error message inline. The original result (from Upload step) is preserved β€” the reviewer can dismiss the error and continue with the original. +- **Missing artifacts:** If a figure image or table HTML is not found in blob storage, the gallery shows a placeholder with the artifact filename. This can happen if the blob sync was interrupted; re-processing recovers it. +- **Parser unavailable:** If Azure Document Intelligence credentials are not configured, the Azure option is disabled with a tooltip explaining why. diff --git a/docs/frontend/05-study-config.md b/docs/frontend/05-study-config.md new file mode 100644 index 0000000..d661c58 --- /dev/null +++ b/docs/frontend/05-study-config.md @@ -0,0 +1,116 @@ +# Study Config Page (Workflow Step 3) + +> *Before running extraction, a reviewer needs to tell the tool what to look for. The Study Config page is where reviewers pick a study type (toxicology, epidemiology, or custom), load or define a set of entities to extract, and choose which AI models to use. Think of it as building the extraction checklist. The output of this step is the list of entity definitions that the Extraction page will work through.* + +**File:** `components/BatchStudySelectionPage.tsx` (~1242 lines) + +--- + +## 1. UI sections + +| Section | Purpose | +|---|---| +| File selector | Choose which uploaded file is being configured (multi-file mode) | +| Study type picker | Select toxicology, epidemiology, or custom | +| Template picker | Load entity definitions from a saved template | +| Entity list | Shows all entities with their extraction prompts | +| Entity editor | Add, remove, or edit the name and prompt of each entity | +| Summary prompt | Optional: customise the paragraph summary system prompt | +| Paragraph system prompt | Optional: customise the paragraph generation model instructions | +| Model selector | Choose which AI models to use for extraction | +| Global vs. per-file toggle | Apply the same entity config to all files, or configure each file separately | + +--- + +## 2. Study types and built-in templates + +Three built-in study types are available, loaded from `components/TemplateLoader.tsx`: + +| Study type | Description | Typical entities | +|---|---|---| +| `toxicology` | In vivo developmental toxicity studies | Test material, species, dose levels, route, maternal effects, fetal effects | +| `epidemiology` | Human health observational studies | Participant count, pesticide of interest, exposure measurement, health outcomes, measures of association, strengths, limitations, risk of bias | +| `custom` | Reviewer-defined | Any fields the reviewer specifies | + +When the reviewer selects a study type, `loadStudyTypeTemplate()` populates the entity list with the default prompts for that study type. The reviewer can edit, add, or remove entities freely after loading. + +--- + +## 3. Loading from a saved template + +The `TemplatePicker` component (see [appendices/component-index.md](appendices/component-index.md)) shows all templates accessible to the reviewer (personal, group-shared, and global). Selecting a template replaces the current entity list with the template's entities and system prompt. + +The reviewer can then modify the loaded entities before running extraction β€” selecting a template is always a starting point, not a locked-in configuration. + +--- + +## 4. Multi-file configuration + +When multiple files are uploaded, the reviewer has two modes: + +- **Global config:** All files use the same study type, entity list, and model selection. Changes apply to all files at once. +- **Per-file config:** Each file can have a different study type and entity list. The file selector at the top switches between file-specific configurations. + +The active mode is tracked in `useGlobalConfig` (boolean state). Switching from per-file to global merges all per-file configs into a single shared config, with a confirmation dialog. + +--- + +## 5. Model selection + +The model selector shows all configured AI providers from `SettingsManager`. The reviewer can select multiple models β€” extraction will run once per model for each entity, enabling side-by-side comparison on the Extraction page. + +Provider groups shown: +- Azure OpenAI (GPT-4o, GPT-4.1, etc.) +- Google Vertex AI (Gemini 2.5 Pro, Gemini 2.0 Flash, etc.) +- Anthropic (Claude Sonnet 4.5, Claude Opus 4.1, etc.) +- Cohere, Llama, local models β€” shown only if configured + +--- + +## 6. State + +| State field | Type | Purpose | +|---|---|---| +| `fileConfigs` | `Map` | Per-file entity + study type config | +| `useGlobalConfig` | `boolean` | Whether all files share one config | +| `globalEntities` | `Entity[]` | Entity list when in global config mode | +| `studyType` | `string` | Selected study type | +| `summaryPrompt` | `string` | Optional summary system prompt | +| `selectedModels` | `string[]` | All models selected for extraction | + +--- + +## 7. API calls + +This page makes no direct API calls. Entity templates are loaded from `TemplateLoader` (built-in JSON) or from the `useTemplates` hook (saved templates). Model metadata is read from `SettingsManager` (localStorage). + +--- + +## 8. `onComplete()` payload + +```typescript +onComplete({ + studyType: "toxicology", + entities: [ + { name: "Test material", prompt: "What is the test material or compound used in this study?" }, + { name: "Species", prompt: "What animal species and strain were used?" }, + // ... + ], + summaryPrompt: "Summarise the key findings...", + selectedModels: ["azure-gpt4o", "gemini-2.5-pro"], + uploadedFiles: updatedUploadedFiles, // with per-file entity configs merged in +}) +``` + +App.tsx navigates to `extraction`. + +--- + +## 9. Validation + +Before allowing "Next", the page validates: +- At least one entity is defined. +- All entity names are non-empty. +- At least one model is selected. + +Validation errors are shown inline next to the relevant field. The "Next" button is disabled until all errors are resolved. diff --git a/docs/frontend/06-extraction.md b/docs/frontend/06-extraction.md new file mode 100644 index 0000000..c73bfaf --- /dev/null +++ b/docs/frontend/06-extraction.md @@ -0,0 +1,177 @@ +# Extraction Page (Workflow Step 4) + +> *This is where the AI models do the work. The Extraction page sends each entity prompt to the selected LLMs, shows the answers as they come back, and lets reviewers see exactly where in the document each answer came from β€” highlighted in the original PDF. Reviewers can run multiple models side by side, edit individual answers, and see token counts and cost for every extraction. This is the largest and most complex component in the frontend.* + +**File:** `components/EntityExtractionPage.tsx` (~4300 lines) + +--- + +## 1. UI sections + +| Section | Purpose | +|---|---| +| File/document selector | Switch between uploaded files in multi-file sessions | +| Entity list | All configured entities with extraction status (pending, running, complete, error) | +| Model tabs | One tab per selected model showing that model's results | +| Entity detail panel | Expanded view of one entity: extracted answer, source references, cost metrics | +| PDF viewer with highlighting | Shows the pages referenced by the extracted answer, with bounding boxes drawn | +| Full text viewer | Shows the raw markdown with the relevant passage highlighted | +| In-place editor | Edit the extracted value directly in the results panel | +| Paragraph summary section | Auto-generated paragraph summary from all extracted entities | +| Export buttons | Download results as Word document or Markdown file | +| Re-run controls | Re-run a single entity, all entities for a model, or all entities across all models | + +--- + +## 2. Extraction flow + +``` +Reviewer clicks "Run Extraction" + β”‚ + β–Ό +For each entity Γ— each selected model (concurrent): + β”‚ + POST /api/extract + β”‚ Body: { + β”‚ document_conversion_id: file_hash, + β”‚ entities: [{ name, prompt }], + β”‚ model_config: { model_type, model_id, ... }, + β”‚ session_id: sessionId, + β”‚ } + β”‚ Returns: { + β”‚ entity_name: string, + β”‚ answer: string, + β”‚ references: [{ text, page, bounding_box }], + β”‚ duration_ms: number, + β”‚ prompt_tokens: number, + β”‚ completion_tokens: number, + β”‚ cost_usd: number, + β”‚ } + β”‚ + β–Ό +Update entities[i].extractionsByModel[modelId] with result + β”‚ + β–Ό +Mark entity as complete; show answer + reference count +``` + +Each entity-model combination is a separate API call. With 20 entities and 3 models, 60 concurrent requests are made. The backend handles concurrency with a semaphore. + +--- + +## 3. PDF reference highlighting + +When an extraction result includes `references` (source passages from the document), the `EntityPDFViewerBeta` component renders the original PDF and draws coloured bounding boxes at the referenced locations. + +How it works: +1. The backend's extraction response includes `references[].bounding_box` β€” page-relative coordinates of the source passage. +2. `EntityPDFViewerBeta` uses `pdfjs-dist` to render the PDF pages as canvases. +3. Bounding boxes are drawn as coloured overlays scaled to the rendered page dimensions. +4. Clicking a reference in the entity detail panel scrolls the PDF to that page and pulses the highlight. + +If references are not available (the model did not return them, or the document was processed without a parser that supports bounding boxes), the PDF viewer shows the full document without highlights, and the text viewer highlights the passage by string search. + +--- + +## 4. Multi-model comparison + +When multiple models are selected in the Study Config step, the extraction page shows a tab for each model. Within each tab, all entities are shown with that model's answers. + +A **comparison view** can be toggled to show all models' answers for a single entity side by side. This lets reviewers quickly assess where models agree or disagree. + +The `extractionsByModel` field on each entity stores results keyed by model ID: + +```typescript +entity.extractionsByModel = { + "azure-gpt4o": { answer: "...", references: [...], duration_ms: 2100, ... }, + "gemini-2.5-pro": { answer: "...", references: [...], duration_ms: 1800, ... }, +} +``` + +--- + +## 5. In-place editing + +Reviewers can edit any extracted answer by clicking the edit icon in the entity detail panel. Edits are stored in `entity.extracted` (the reviewer-accepted value). The original model answer is preserved in `entity.extractionsByModel[modelId].answer` β€” edits never overwrite the raw model output. + +The edited value is what gets included in Word document exports and session saves. + +--- + +## 6. Paragraph summary generation + +After all entities are extracted, a "Generate Summary" button appears at the bottom of the entity list. Clicking it sends all extracted entity values to: + +``` +POST /api/generate_paragraph + Body: { + entities: [{ name, extracted_value }], + model_config: { ... }, + session_id: sessionId, + summary_prompt: "...", + } + Returns: { paragraph: string, duration_ms, tokens, cost_usd } +``` + +The generated paragraph is shown in the summary section and included in Word exports. + +--- + +## 7. Session auto-save + +Every time an extraction result arrives, the page patches the session on the backend: + +``` +POST /api/sessions/{sessionId}/extractions + Body: ExtractionResult +``` + +This means the reviewer's work is persisted to the database continuously β€” they can close the browser and restore from Session History without losing results. + +--- + +## 8. State + +| State field | Type | Purpose | +|---|---|---| +| `entities` | `Entity[]` | Full entity list with extraction results | +| `activeEntity` | `string \| null` | Currently selected entity for detail view | +| `activeModel` | `string` | Currently selected model tab | +| `extractionInFlight` | `Set` | Entity-model pairs currently running | +| `pdfDoc` | `PDFDocumentProxy \| null` | Loaded PDF for reference viewer | +| `editingEntity` | `string \| null` | Entity currently being edited in-place | +| `paragraphSummary` | `string` | Generated paragraph text | +| `paragraphInFlight` | `boolean` | Paragraph generation in progress | + +--- + +## 9. API calls + +| Method | Path | When | Purpose | +|---|---|---|---| +| `POST` | `/api/extract` | On "Run Extraction" | Extract one entity with one model | +| `POST` | `/api/generate_paragraph` | On "Generate Summary" | Generate paragraph from extracted values | +| `POST` | `/api/sessions/{id}/extractions` | After each extraction | Persist result to session | + +--- + +## 10. `onComplete()` payload + +```typescript +onComplete({ + entities: updatedEntities, // with extractionsByModel populated + uploadedFiles: updatedFiles, // with per-file entities merged in + sessionId: sessionId, +}) +``` + +App.tsx navigates to `evaluation`. + +--- + +## 11. Error handling + +- **Entity extraction failure:** The failed entity is marked with an error badge. Other entities continue. A "Retry" button on the entity re-sends that specific entity-model pair. +- **Partial completion:** The reviewer can proceed to evaluation with only some entities extracted. Unevaluated entities are skipped automatically on the evaluation page. +- **Token limit exceeded:** The backend returns a specific error code. The entity is marked with a "Context too long" error. The reviewer can shorten the entity prompt and retry. +- **In-flight navigation guard:** If extractions are running when the reviewer tries to navigate away, App.tsx shows a confirmation dialog. diff --git a/docs/frontend/07-evaluation.md b/docs/frontend/07-evaluation.md new file mode 100644 index 0000000..1086994 --- /dev/null +++ b/docs/frontend/07-evaluation.md @@ -0,0 +1,166 @@ +# Evaluation Page (Workflow Step 5) + +> *Once entities are extracted, reviewers can ask the tool to judge how good the extractions are. The Evaluation page uses G-Eval β€” a technique where a separate LLM acts as a judge, scoring each extraction against criteria like correctness and completeness. Reviewers can also provide their own expected answers as ground truth, add custom evaluation steps, and review a cost/score breakdown across all models. Results can be exported to Excel.* + +**File:** `components/EvaluationPage.tsx` (~5000 lines) + +--- + +## 1. UI sections + +| Section | Purpose | +|---|---| +| Metric selector | Choose which G-Eval metrics to run (correctness, completeness, relevance, safety, custom) | +| Judge model selector | Choose which LLM acts as the evaluator | +| Entity selector | Which entities to evaluate (all, or a subset) | +| Expected output fields | Optional: enter ground truth answers per entity for correctness/completeness scoring | +| Custom metric editor | Define custom evaluation criteria with step-by-step rubrics | +| Run / Stop buttons | Start or cancel evaluation | +| Results table | Metric scores per entity per extraction model (pass/fail + numeric score) | +| Per-model breakdown | Aggregate scores grouped by extraction model | +| Cost breakdown | Tokens and cost per evaluation call | +| Charts | Score distribution and model comparison (Recharts) | +| Excel export | Download full results as a structured Excel workbook | +| Human score override | Manually override the LLM score for any entity/metric | + +--- + +## 2. G-Eval metrics + +| Metric | Requires ground truth | What it measures | +|---|---|---| +| `correctness` | Yes | Does the extracted answer match the expected answer? | +| `completeness` | Yes | Does the extraction capture all relevant information from the expected answer? | +| `relevance` | No | Is the extracted answer relevant to the entity prompt? | +| `safety` | No | Does the extracted answer contain harmful or inappropriate content? | +| `custom` | Configurable | Reviewer-defined rubric with step-by-step evaluation instructions | + +If no expected output is provided, `correctness` and `completeness` are automatically skipped. The remaining metrics can still run. + +--- + +## 3. Evaluation flow + +``` +Reviewer configures metrics + judge model β†’ clicks "Run Evaluation" + β”‚ + β–Ό +For each entity with extraction results: + POST /api/evaluations/jobs + β”‚ Body: { + β”‚ tasks: [{ entity_name, extracted_value, expected_output, metric }], + β”‚ providers: [{ model_type, model_id }], + β”‚ session_id: sessionId, + β”‚ } + β”‚ Returns: { job_id, status: "queued" } + β”‚ + β–Ό + Poll: GET /api/evaluations/jobs/{job_id} + β”‚ Returns: { status, progress, results } + β”‚ + β–Ό + Results arrive incrementally as job progresses + β”‚ + β–Ό + POST /api/sessions/{sessionId}/evaluations ← persist each result +``` + +Evaluation is run as a background job on the backend. The frontend polls `GET /api/evaluations/jobs/{job_id}` every 2 seconds until `status` is `completed`, `cancelled`, or `failed`. Results are shown in the table as they arrive, not all at once. + +--- + +## 4. Stopping evaluation + +Clicking "Stop" calls: + +``` +POST /api/evaluations/jobs/{job_id}/cancel +``` + +The backend marks the job as cancelled. Already-completed results are preserved and shown. The reviewer can review partial results and re-run only the remaining entities. + +--- + +## 5. Custom metric editor + +Reviewers can define their own evaluation metric by providing: +- A metric name +- A series of evaluation steps (e.g. "1. Check if the answer mentions the test material. 2. Check if the dose is specified.") + +The custom steps are sent to the backend as `custom_steps` in the evaluation request. The judge LLM follows the steps and returns a score from 0 to 1 for each step, which are averaged into a final metric score. + +--- + +## 6. Human score override + +Any LLM-generated score can be overridden manually. The reviewer clicks the score cell in the results table and enters their own assessment. Overrides are stored separately from LLM scores in `entity.evaluationResults[].human_score` and are preserved through session saves. Both scores are visible in the Excel export. + +--- + +## 7. Results table structure + +The results table has one row per entity and one column group per extraction model. Each cell shows: +- The numeric score (0.00–1.00) +- A pass/fail badge (threshold configurable, default 0.5) +- The judge model used +- A detail icon that opens the judge's reasoning + +The table is sortable by entity name or any metric score. + +--- + +## 8. Excel export + +The Excel workbook has: +- **Sheet 1 β€” Results:** One row per entity, columns for each metric Γ— model combination +- **Sheet 2 β€” Extracted values:** The raw extraction answers per entity per model +- **Sheet 3 β€” Cost:** Token usage and cost per evaluation call +- **Sheet 4 β€” Config:** Which models, metrics, and judge were used + +--- + +## 9. State + +| State field | Type | Purpose | +|---|---|---| +| `evaluationConfig` | `EvaluationConfig` | Selected metrics, models, custom steps | +| `selectedEntities` | `string[]` | Entities included in the current evaluation run | +| `expectedOutputs` | `Map` | Ground truth per entity (entered by reviewer) | +| `evaluationResults` | `EvaluationResult[]` | All score results so far | +| `activeJobId` | `string \| null` | Currently running background job ID | +| `pollInterval` | `number \| null` | Interval handle for job status polling | +| `aggregateScores` | `Map` | Average score per extraction model | + +--- + +## 10. API calls + +| Method | Path | When | Purpose | +|---|---|---|---| +| `POST` | `/api/evaluations/jobs` | On "Run Evaluation" | Submit background evaluation job | +| `GET` | `/api/evaluations/jobs/{job_id}` | Every 2s while running | Poll job status and retrieve partial results | +| `POST` | `/api/evaluations/jobs/{job_id}/cancel` | On "Stop" | Cancel running job | +| `POST` | `/api/sessions/{id}/evaluations` | After each result | Persist result to session | + +--- + +## 11. `onComplete()` payload + +```typescript +onComplete({ + entities: entitiesWithEvaluationResults, + uploadedFiles: updatedFiles, + evaluationConfig: currentConfig, +}) +``` + +After evaluation, the reviewer is at the end of the workflow. The "Next" button navigates to the Batch Results page for a consolidated view. + +--- + +## 12. Error handling + +- **Job failure:** The job status shows `failed` with an error message. Already-completed evaluations within the job are preserved. +- **Judge model unavailable:** If the selected judge model is not configured, the error is shown before the job is submitted. +- **Score parse failure:** If the judge LLM returns malformed output, the backend falls back to a per-metric scoring approach. If that also fails, the metric is marked as `error` in the results table. +- **In-flight navigation guard:** If a job is running when the reviewer tries to navigate away, App.tsx shows a confirmation dialog. diff --git a/docs/frontend/08-simplified-flow.md b/docs/frontend/08-simplified-flow.md new file mode 100644 index 0000000..5ac477f --- /dev/null +++ b/docs/frontend/08-simplified-flow.md @@ -0,0 +1,111 @@ +# Simplified Flow + +> *The simplified flow is a one-click version of the full workflow for reviewers who don't need to inspect intermediate steps. The reviewer drops their files in, selects a study type, and clicks Run. The app automatically uploads, processes, extracts all entities, and generates a summary β€” showing a progress table as it goes. Results can be downloaded as a Word document when complete.* + +**Files:** `components/SimplifiedFlowPage.tsx` (~995 lines), `hooks/useSimplifiedPipeline.ts` + +--- + +## 1. UI sections + +| Section | Purpose | +|---|---| +| Study type selector | Pick toxicology, epidemiology, or custom | +| File upload area | Drag-and-drop; accepts multiple PDFs | +| Advanced options (collapsed) | Override parser, model selection, temperature | +| Progress table | One row per file: filename, current stage, per-entity extraction progress | +| Results preview | Summary text + key extracted entities per file | +| Download buttons | Download extraction report or executive summary as Word documents | +| Cancel / Restart controls | Stop the pipeline mid-run or start over | + +--- + +## 2. Pipeline stages + +Each file moves through these stages independently: + +| Stage | What happens | +|---|---| +| `queued` | File is waiting to start | +| `uploading` | `POST /api/upload` in progress | +| `processing` | `POST /api/documents/process/file/{hash}` in progress | +| `extracting` | `POST /api/extract` running for each entity (batched, 5 at a time) | +| `summarizing` | `POST /api/generate_paragraph` in progress | +| `exporting` | Building Word document in memory | +| `complete` | All done; download available | +| `error` | A stage failed; error message shown inline | + +--- + +## 3. `useSimplifiedPipeline` hook + +All the pipeline logic lives in `hooks/useSimplifiedPipeline.ts`, not in the page component. This keeps the component focused on display and makes the pipeline independently testable. + +**API:** + +```typescript +const { state, results, run, reset, downloadResults, downloadSingleResult } = + useSimplifiedPipeline(); + +// Start the pipeline +run(files, studyType, entities, summaryPrompt, options); + +// Download all results as a single Word document +downloadResults(); + +// Download results for one file +downloadSingleResult(filename); +``` + +**Batching:** + +Entity extraction runs 5 entities at a time per file. This avoids overwhelming the backend while still parallelising work. The progress bar for the `extracting` stage shows `N / total` as batches complete. + +**Error recovery:** + +If a stage fails for one file, that file is marked as `error` and the pipeline continues with other files. The reviewer can see the error message in the progress table row and use the Restart button to retry from the beginning. + +--- + +## 4. Difference from the advanced workflow + +| | Advanced workflow | Simplified flow | +|---|---|---| +| Steps | 5 separate pages | 1 page | +| Inspection | Reviewer checks each step's output | No intermediate inspection | +| Parser choice | Per-file, with preview | Global option in Advanced settings | +| Entity editing | Edit prompts before extraction | Uses default template prompts | +| Extraction results | Viewable inline with PDF | Download only | +| Evaluation | Full G-Eval scoring page | Not included | +| Target user | Reviewers who want control | Reviewers who want speed | + +--- + +## 5. State + +State is managed inside `useSimplifiedPipeline`. The page receives it as `state` and `results`: + +```typescript +interface PipelineState { + status: "idle" | "running" | "complete" | "error"; + fileStatuses: Map; +} + +interface FileResult { + filename: string; + entities: Entity[]; + paragraphSummary: string; + wordDocBytes?: Uint8Array; +} +``` + +--- + +## 6. API calls + +| Method | Path | When | Purpose | +|---|---|---|---| +| `POST` | `/api/upload` | Per file | Upload file bytes | +| `POST` | `/api/documents/process/file/{hash}` | Per file | Parse document | +| `POST` | `/api/extract` | Per entity batch | Extract entities | +| `POST` | `/api/generate_paragraph` | Per file | Generate summary | diff --git a/docs/frontend/09-chat.md b/docs/frontend/09-chat.md new file mode 100644 index 0000000..13c25f1 --- /dev/null +++ b/docs/frontend/09-chat.md @@ -0,0 +1,99 @@ +# Chat Page + +> *The Chat page is a freeform document Q&A interface β€” think of it as asking questions directly to an uploaded PDF. Unlike the main extraction workflow, there are no structured entities or rubrics. The reviewer types a question, attaches up to five documents as context, and gets a markdown-rendered answer. It's useful for quick lookups, sanity checks, and exploratory reading before setting up a formal extraction.* + +**File:** `components/ChatPage.tsx` (~949 lines) + +--- + +## 1. UI sections + +| Section | Purpose | +|---|---| +| Message history | Alternating user (right) and assistant (left) message bubbles | +| Message actions | Copy, thumbs up/down rating, regenerate last response | +| Document upload area | Drag-and-drop zone; accepts up to 5 PDFs | +| Document chips | Shows attached documents with processing status and a remove button | +| Model selector | Auto (backend picks best available) or manual model selection | +| Input box | Multi-line text input; Submit on Enter or button click | +| Markdown renderer | Assistant responses rendered with full GFM markdown (tables, code blocks, lists) | + +--- + +## 2. How documents are used as context + +Uploaded documents are converted to markdown and concatenated into the `document_markdown` field of the chat request. The backend passes this as context to the LLM before the user's question. + +``` +Reviewer uploads PDF + β”‚ + β–Ό +POST /api/upload β†’ file_hash + β”‚ + β–Ό +POST /api/documents/process/file/{hash} β†’ markdown + β”‚ + β–Ό +document_markdown stored in documentContexts[hash] + +Reviewer sends message + β”‚ + β–Ό +POST /api/chat/query + Body: { + query: "What was the NOAEL for maternal effects?", + document_markdown: "[doc1 markdown]\n\n[doc2 markdown]", + model_config: { ... }, + } + Returns: { answer: string, model_used: string, tokens, cost } +``` + +Up to 5 documents can be attached simultaneously. The combined markdown is sent in a single request. If the combined length exceeds the model's context window, the backend truncates from the oldest document first and indicates truncation in the response metadata. + +--- + +## 3. Message ratings + +Each assistant message has thumbs up / thumbs down buttons. Ratings are stored in local component state (`messageRatings`) and are not persisted to the backend or database. They serve as in-session quality tracking for the reviewer's own reference. + +--- + +## 4. State + +| State field | Type | Purpose | +|---|---|---| +| `messages` | `Message[]` | Full conversation history (user + assistant turns) | +| `attachedDocuments` | `AttachedDoc[]` | Uploaded documents with hash, filename, markdown, status | +| `selectedModel` | `string` | Model identifier or `"auto"` | +| `isLoading` | `boolean` | Request in progress (disables input) | +| `messageRatings` | `Map` | Per-message rating keyed by message index | + +--- + +## 5. API calls + +| Method | Path | When | Purpose | +|---|---|---|---| +| `POST` | `/api/upload` | On document drop | Upload PDF bytes, get hash | +| `POST` | `/api/documents/process/file/{hash}` | After upload | Convert PDF to markdown | +| `POST` | `/api/chat/query` | On message send | Get LLM answer with document context | + +--- + +## 6. Differences from the main workflow + +| | Chat page | Main workflow | +|---|---|---| +| Structure | Freeform conversation | Structured entity extraction | +| Session saving | Not saved to database | Saved as AppSession | +| References | No bounding box highlights | PDF page + location highlights | +| Multi-model | Single model per message | Multiple models in parallel | +| Export | Copy to clipboard only | Word / Excel / Markdown | + +--- + +## 7. Error handling + +- **Processing failure:** The document chip shows a red error indicator. The reviewer can remove and re-add the file. +- **Chat API failure:** An error bubble appears in the message history with the error text. The input is re-enabled so the reviewer can try again. +- **Context too long:** If the combined document markdown exceeds the model limit, the backend returns a specific error. The page shows a warning suggesting the reviewer remove one or more documents. diff --git a/docs/frontend/10-session-history.md b/docs/frontend/10-session-history.md new file mode 100644 index 0000000..d53b62c --- /dev/null +++ b/docs/frontend/10-session-history.md @@ -0,0 +1,120 @@ +# Session History + +> *Every time a reviewer completes extraction, their work is automatically saved as a session. The Session History page is where they can find past sessions, restore them exactly as they were left, share them with colleagues, and clean up old work. It also shows sessions that have been shared with the reviewer by others.* + +**File:** `components/SessionHistoryPage.tsx` (~800 lines) + +--- + +## 1. UI sections + +| Section | Purpose | +|---|---| +| Tab: My Sessions | Sessions the reviewer created | +| Tab: Shared With Me | Sessions shared by other users via a group | +| Session list | Searchable table: name, date, study type, file count, entity count | +| Session metrics | Cost, total tokens, duration per session | +| Session actions | Restore, Rename, Share, Delete | +| Share dialog | Search for a group to share with; shows current share status | +| Confirmation dialogs | Required for Delete and unshare | + +--- + +## 2. Restoring a session + +Restoring loads the full session state from the backend and navigates to the appropriate workflow step. + +``` +Reviewer clicks "Restore" + β”‚ + β–Ό +GET /api/sessions/{sessionId}/restore-view + β”‚ + Returns: { + primaryFileId: string, + uploadedFiles: [{ fileId, filename, processingResult, entities, ... }], + currentStep: "extraction" | "evaluation" | ..., + } + β”‚ + β–Ό +App.tsx re-hydrates documentData from restore view + β”‚ + β–Ό +Navigate to the step the session was last on +``` + +The restore-view endpoint reconstructs the full `documentData` shape β€” it re-checks which artifacts are available in blob storage and rebuilds the uploaded files array. This means a session can be restored even if the reviewer has cleared their browser cache. + +--- + +## 3. Sharing a session + +Sharing makes a session visible in the "Shared With Me" tab for all members of the selected group. + +``` +Reviewer clicks "Share" on a session + β”‚ + β–Ό +Share dialog opens β€” reviewer searches for a group by name + β”‚ + β–Ό +POST /api/sessions/{sessionId}/share + Body: { group_id: "..." } + β”‚ + β–Ό +Session is now visible to all group members under "Shared With Me" +``` + +A session can only be shared with one group at a time. Sharing with a different group replaces the previous share. The sharing reviewer retains ownership β€” only they can delete or unshare. + +To remove sharing: + +``` +DELETE /api/sessions/{sessionId}/share +``` + +--- + +## 4. Shared session behaviour + +When a reviewer opens a session from "Shared With Me": + +- They see the full extraction and evaluation results. +- They **cannot** re-run extraction or evaluation (read-only). +- They **can** restore the session view (load it into the main app for inspection). +- They **cannot** delete or rename the session (only the owner can). + +--- + +## 5. State + +| State field | Type | Purpose | +|---|---|---| +| `sessions` | `SessionSummary[]` | All sessions for the current tab | +| `activeTab` | `"mine" \| "shared"` | Which tab is displayed | +| `selectedSession` | `string \| null` | Session ID for the detail/action panel | +| `searchQuery` | `string` | Filter sessions by name or study type | +| `shareDialogOpen` | `boolean` | Whether the share dialog is visible | +| `pendingDelete` | `string \| null` | Session ID awaiting delete confirmation | + +--- + +## 6. API calls + +| Method | Path | When | Purpose | +|---|---|---|---| +| `GET` | `/api/sessions` | On mount | Load reviewer's own sessions | +| `GET` | `/api/sessions/shared/list` | On "Shared With Me" tab | Load sessions shared with reviewer | +| `GET` | `/api/sessions/{id}/restore-view` | On "Restore" | Rebuild documentData from session | +| `PATCH` | `/api/sessions/{id}` | On rename | Update session name | +| `POST` | `/api/sessions/{id}/share` | On share confirm | Share session with group | +| `DELETE` | `/api/sessions/{id}/share` | On unshare | Remove group sharing | +| `DELETE` | `/api/sessions/{id}` | On delete confirm | Permanently delete session | + +--- + +## 7. Error handling + +- **Restore failure:** If the session's files are no longer in blob storage (e.g. deleted from Azure), the restore view returns a partial result. The page shows a warning listing which files could not be restored and proceeds with the available files. +- **Delete failure:** Shows a toast error; session remains in the list. +- **Share failure (group not found):** The share dialog shows an inline error. diff --git a/docs/frontend/11-templates.md b/docs/frontend/11-templates.md new file mode 100644 index 0000000..6d3053b --- /dev/null +++ b/docs/frontend/11-templates.md @@ -0,0 +1,172 @@ +# Template Workspace + +> *Templates are reusable sets of entity definitions and prompts. Instead of manually re-entering the same 20 extraction fields every time, a reviewer saves them as a template once and loads it in seconds on the Study Config page. The Template Workspace is where those templates are created, edited, versioned, and shared β€” either privately, with a group, or globally across all users.* + +**Files:** `components/TemplateWorkspace/TemplateWorkspacePage.tsx`, `TemplateList.tsx`, `TemplateEditor.tsx`, `TemplateVersionHistory.tsx`, `FolderCard.tsx` + +**Hook:** `hooks/useTemplates.ts`, `hooks/useFolders.ts` + +--- + +## 1. UI sections + +| Section | Purpose | +|---|---| +| Folder sidebar | Hierarchical folder tree for organising templates | +| Template list | Filterable list of templates with scope badges (personal / group / global) | +| Search + filters | Filter by study type, scope, tags, or free-text name search | +| Template editor | Form: name, description, study type, entities, system prompt, summary prompt, tags | +| Entity list editor | Add, reorder, and edit entity name + prompt within a template | +| Version history panel | Timestamped list of previous versions with a Revert button | +| Fork button | Copy any accessible template to your personal scope | +| Scope selector | Change template visibility: personal β†’ group β†’ global | +| Immutable toggle | Lock a template against further edits (admin/owner only) | +| Permission editor | Grant explicit read/edit access to specific users | + +--- + +## 2. Template scopes + +| Scope | Who can see it | Who can edit it | +|---|---|---| +| `user` (personal) | Owner only | Owner only | +| `group` | All members of the linked group | Group admins and owner | +| `global` | All users | System admins only | + +Built-in study type templates (toxicology, epidemiology) are `global` and `immutable` β€” they can be read and forked but not edited. + +--- + +## 3. Creating and editing a template + +``` +Reviewer clicks "New Template" + β”‚ + β–Ό +TemplateEditor opens (blank form) + β”‚ + β–Ό +Reviewer fills in name, study type, entities, prompts + β”‚ + β–Ό +POST /api/templates + Body: { name, study_type, scope, entities, system_prompt, summary_prompt, tags } + Returns: Template + β”‚ + β–Ό +Template appears in list under "My Templates" +``` + +Editing an existing template: + +``` +Reviewer clicks "Edit" on a template + β”‚ + β–Ό +TemplateEditor opens with current values + β”‚ + β–Ό +Reviewer makes changes + saves + β”‚ + β–Ό +PUT /api/templates/{id} + β”‚ + β–Ό +Backend saves a TemplateVersion snapshot of the previous content + Template.version increments +``` + +Every save creates a version snapshot automatically β€” the reviewer never has to manually checkpoint. + +--- + +## 4. Version history and revert + +``` +Reviewer opens version history panel + β”‚ + β–Ό +GET /api/templates/{id}/versions + Returns: [{ version, created_at, entities, system_prompt, ... }] + β”‚ + β–Ό +Reviewer clicks "Revert to v3" + β”‚ + β–Ό +POST /api/templates/{id}/revert/3 + β”‚ + β–Ό +Backend creates a new version (v5) with v3's content + (reverting does not delete the intermediate versions) +``` + +--- + +## 5. Forking a template + +Forking creates a personal copy of any template the reviewer can see, regardless of scope. + +``` +POST /api/templates/{id}/fork + Returns: new Template with scope="user", owner=reviewer +``` + +The fork is independent β€” changes to the fork do not affect the original, and vice versa. + +--- + +## 6. Folder organisation + +Templates can be placed in folders. Folders are scoped the same way as templates (user / group / global) and can be nested. + +``` +GET /api/templates/folders?scope=user +POST /api/templates/folders β†’ create folder +PATCH /api/templates/folders/{id} β†’ rename +DELETE /api/templates/folders/{id} β†’ delete (only if empty) +``` + +Dragging a template into a folder calls `PUT /api/templates/{id}` with the updated `folder_id`. + +--- + +## 7. State (via `useTemplates` hook) + +```typescript +const { + templates, // Template[] β€” current filtered list + loading, + error, + filters, // { scope, study_type, search, tags } + setFilters, + fetchTemplates, + createTemplate, + updateTemplate, + deleteTemplate, + forkTemplate, + setImmutable, + changeScope, +} = useTemplates(); +``` + +The hook fetches templates on mount and re-fetches after any mutation. Filtering happens server-side via query parameters. + +--- + +## 8. API calls + +| Method | Path | Purpose | +|---|---|---| +| `GET` | `/api/templates` | List accessible templates | +| `POST` | `/api/templates` | Create template | +| `GET` | `/api/templates/{id}` | Fetch one template | +| `PUT` | `/api/templates/{id}` | Update template (creates version snapshot) | +| `DELETE` | `/api/templates/{id}` | Delete template | +| `POST` | `/api/templates/{id}/fork` | Fork to personal scope | +| `PUT` | `/api/templates/{id}/scope` | Change scope | +| `PUT` | `/api/templates/{id}/immutable` | Set immutability | +| `GET` | `/api/templates/{id}/versions` | Version history | +| `POST` | `/api/templates/{id}/revert/{v}` | Revert to version | +| `GET/POST` | `/api/templates/folders` | List / create folders | +| `PATCH` | `/api/templates/folders/{id}` | Rename folder | +| `DELETE` | `/api/templates/folders/{id}` | Delete folder | diff --git a/docs/frontend/12-groups.md b/docs/frontend/12-groups.md new file mode 100644 index 0000000..6bd76bf --- /dev/null +++ b/docs/frontend/12-groups.md @@ -0,0 +1,129 @@ +# Group Management + +> *Groups are how reviewers share their work with colleagues. A group is a named set of users with roles β€” once a group exists, a reviewer can share a session or a template with it and every group member gets access. The Group Management page is where groups are created, members are added, and roles are managed.* + +**Files:** `components/GroupManagement/GroupManagementPage.tsx` + +**Hook:** `hooks/useGroups.ts` + +--- + +## 1. UI sections + +| Section | Purpose | +|---|---| +| Group list | All groups the reviewer belongs to, with their role badge | +| Create group button | Opens the create dialog | +| Group detail panel | Expanded view: name, description, member list | +| Add member dialog | Search users by email; set their role before adding | +| Member list | Shows each member's name, email, role, and join date | +| Member actions | Change role, Remove member | +| Edit group dialog | Rename group or update description | +| Delete group button | Owner-only; requires confirmation | + +--- + +## 2. Roles + +| Role | Permissions | +|---|---| +| `owner` | Full control: edit group, manage all members, delete group. Only one owner per group (the creator). | +| `admin` | Add/remove members, change member roles (up to admin). Cannot delete the group or change the owner. | +| `member` | View group and shared resources. Cannot manage members. | + +Role rules enforced both in the UI (buttons are hidden or disabled based on the reviewer's role) and on the backend (`GroupService`). + +--- + +## 3. Creating a group + +``` +Reviewer clicks "Create Group" + β”‚ + β–Ό +Dialog opens β€” enter name + optional description + β”‚ + β–Ό +POST /api/groups + Body: { name, description } + Returns: Group (with reviewer added as owner automatically) + β”‚ + β–Ό +Group appears in list with "owner" badge +``` + +--- + +## 4. Adding members + +``` +Owner/admin clicks "Add Member" + β”‚ + β–Ό +Search input β€” type email or name + β”‚ + GET /api/groups/users/search?q=... + Returns: [{ id, email, name, image }] + β”‚ + β–Ό +Select user + choose role (member / admin) + β”‚ + β–Ό +POST /api/groups/{groupId}/members + Body: { user_id, role } +``` + +A user must already have an account in the system to be added. Inviting by email address only works if the user has previously logged in. + +--- + +## 5. Changing roles and removing members + +``` +PUT /api/groups/{groupId}/members/{userId} + Body: { role: "admin" | "member" } + +DELETE /api/groups/{groupId}/members/{userId} +``` + +Constraints enforced by the backend: +- The owner's role cannot be changed through this endpoint. +- The only owner cannot remove themselves (would leave the group ownerless). +- An admin cannot promote a member to owner. + +--- + +## 6. State (via `useGroups` hook) + +```typescript +const { + groups, // Group[] β€” all groups for current user + loading, + error, + fetchGroups, + createGroup, + updateGroup, + deleteGroup, + addMember, + updateMemberRole, + removeMember, + searchUsers, +} = useGroups(); +``` + +--- + +## 7. API calls + +| Method | Path | Purpose | +|---|---|---| +| `GET` | `/api/groups` | List reviewer's groups | +| `POST` | `/api/groups` | Create group | +| `GET` | `/api/groups/{id}` | Fetch group with members | +| `PUT` | `/api/groups/{id}` | Update name/description | +| `DELETE` | `/api/groups/{id}` | Delete group | +| `GET` | `/api/groups/{id}/members` | List members | +| `POST` | `/api/groups/{id}/members` | Add member | +| `PUT` | `/api/groups/{id}/members/{userId}` | Change member role | +| `DELETE` | `/api/groups/{id}/members/{userId}` | Remove member | +| `GET` | `/api/groups/users/search` | Search users by email/name | diff --git a/docs/frontend/13-executive-mode.md b/docs/frontend/13-executive-mode.md new file mode 100644 index 0000000..a9dc51a --- /dev/null +++ b/docs/frontend/13-executive-mode.md @@ -0,0 +1,70 @@ +# Executive Mode + +> *Executive Mode is a standalone summary generator that skips the structured entity extraction step entirely. A reviewer uploads a document, selects a template, and gets back a narrative paragraph summary β€” without going through the full five-step workflow. It's designed for situations where a reviewer needs a quick high-level overview rather than a field-by-field extraction.* + +**File:** `components/ExecutiveModePage.tsx` (~881 lines) + +--- + +## 1. UI sections + +| Section | Purpose | +|---|---| +| File upload area | Drag-and-drop or file picker | +| Study type selector | Toxicology, epidemiology, or custom | +| Template selector | Load entity prompts from a saved template | +| Summary prompt editor | Editable text area for the paragraph generation instructions | +| Model selector | Choose the LLM to use for summarisation | +| Advanced options (collapsed) | Parser selection, temperature, ingestion settings | +| Progress table | Per-file: upload β†’ processing β†’ extracting β†’ summarising stages | +| Results preview | Generated paragraph summary per file | +| Download button | Export summary as a Word document | + +--- + +## 2. Pipeline + +Executive Mode runs the same pipeline as the Simplified Flow but skips the evaluation stage and focuses the output on the paragraph summary rather than individual entity values. + +``` +Upload file β†’ Process (PDF to markdown) β†’ Extract entities β†’ Generate paragraph +``` + +Internally, the page uses `useSimplifiedPipeline` with a flag that skips storing per-entity results and focuses output on the `paragraphSummary` field. + +--- + +## 3. Difference from Simplified Flow + +| | Executive Mode | Simplified Flow | +|---|---|---| +| Output focus | Paragraph summary | Entity values + summary | +| Results display | Summary text only | Entity table + summary | +| Download | Summary Word document | Full extraction report | +| Entity values visible | No | Yes | +| Intended use | Quick overview | Full extraction record | + +--- + +## 4. State + +State is managed by `useSimplifiedPipeline`. The page reads `results[].paragraphSummary` and ignores `results[].entities` for display purposes. + +| State field | Type | Purpose | +|---|---|---| +| `studyType` | `string` | Selected study type | +| `summaryPrompt` | `string` | Editable paragraph generation instructions | +| `selectedModel` | `string` | Model for extraction and summarisation | +| `advancedOpen` | `boolean` | Whether the advanced options panel is expanded | + +--- + +## 5. API calls + +Same as Simplified Flow β€” see [08-simplified-flow.md](08-simplified-flow.md). The difference is only in what is displayed, not in which API calls are made. + +--- + +## 6. Error handling + +Identical to Simplified Flow. Per-file errors are shown inline in the progress table. The reviewer can retry individual files without restarting the entire run. diff --git a/docs/frontend/14-batch-results.md b/docs/frontend/14-batch-results.md new file mode 100644 index 0000000..6ccdd0a --- /dev/null +++ b/docs/frontend/14-batch-results.md @@ -0,0 +1,107 @@ +# Batch Results + +> *After running extraction and evaluation across multiple files or multiple models, the Batch Results page brings everything together in one searchable table. It's the consolidated view for comparing results, spotting outliers, and exporting the full dataset. Reviewers can filter by model or metric, sort any column, and drill into any single result for the full detail.* + +**File:** `components/BatchResultsPage.tsx` (~1356 lines) + +--- + +## 1. UI sections + +| Section | Purpose | +|---|---| +| Results table | One row per entity Γ— file combination; columns for each metric Γ— model | +| Search bar | Fuzzy full-text search across entity names, answers, and file names (Fuse.js) | +| Column visibility toggle | Show/hide individual metric or model columns | +| Sort controls | Click any column header to sort ascending/descending | +| Filter dropdown | Filter rows by model, metric pass/fail status, or study type | +| Row detail modal | Click a row to see the full extraction answer, all metric scores, and judge reasoning | +| Excel export | Download structured workbook with all results | +| Word export | Download formatted report | +| Human score column | Shows reviewer overrides alongside LLM scores | + +--- + +## 2. Table structure + +Each row represents one entity extracted from one file by one model: + +| Column | Source | +|---|---| +| File | `uploadedFile.filename` | +| Entity | `entity.name` | +| Model | `extractionsByModel` key | +| Extracted value | `entity.extractionsByModel[model].answer` | +| Correctness | `evaluationResults[].correctness` | +| Completeness | `evaluationResults[].completeness` | +| Relevance | `evaluationResults[].relevance` | +| Safety | `evaluationResults[].safety` | +| Human score | `evaluationResults[].human_score` (if set) | +| Cost | `extractionsByModel[model].cost_usd` | +| Duration | `extractionsByModel[model].duration_ms` | + +Columns for metrics that were not run are hidden by default. + +--- + +## 3. Fuzzy search + +Search is powered by [Fuse.js](https://fusejs.io/). The index is built from entity names, extracted answers, and filenames. Results are ranked by relevance score β€” an exact match scores higher than a partial match. + +The search index is rebuilt whenever `documentData.uploadedFiles` changes (i.e. when a new extraction is run). + +--- + +## 4. Row detail modal + +Clicking any row opens a modal with: + +- The full extracted answer (not truncated) +- All metric scores with the judge's reasoning text +- The source references (which pages the answer was drawn from) +- Token usage and cost for that extraction +- Human score override field (editable inline) + +Human score overrides entered in the modal are propagated back to `documentData` via `onInvalidateDownstream` (false) + a targeted update, so they persist through session saves without clearing downstream results. + +--- + +## 5. Excel export + +The exported workbook mirrors the table exactly: + +- **Sheet 1 β€” Results:** All rows visible in the current filtered/sorted view +- **Sheet 2 β€” Full results:** All rows unfiltered +- **Sheet 3 β€” Config:** Study type, models used, metrics run, judge model + +Column headers in the workbook match the table headers. Numeric scores are formatted as decimals (0.00–1.00); pass/fail is a separate boolean column. + +--- + +## 6. State + +| State field | Type | Purpose | +|---|---|---| +| `results` | `ResultRow[]` | Flattened rows derived from `documentData` | +| `fuseIndex` | `Fuse` | Search index over results | +| `searchQuery` | `string` | Active search string | +| `sortBy` | `string` | Active sort column key | +| `sortOrder` | `"asc" \| "desc"` | Sort direction | +| `columnVisibility` | `Map` | Which columns are shown | +| `activeFilter` | `FilterConfig` | Active model/metric/status filters | +| `selectedRow` | `ResultRow \| null` | Row open in detail modal | + +--- + +## 7. API calls + +This page makes no API calls on mount β€” all data is read from `documentData` passed down from `App.tsx`. The Excel and Word exports are generated entirely client-side using `exceljs` and `docx` respectively. + +The exception is human score overrides: when a reviewer saves a human score in the detail modal, the page calls: + +``` +PATCH /api/sessions/{sessionId} + Body: { entities: updatedEntities } +``` + +to persist the override to the session. diff --git a/docs/frontend/README.md b/docs/frontend/README.md new file mode 100644 index 0000000..c7c8239 --- /dev/null +++ b/docs/frontend/README.md @@ -0,0 +1,184 @@ +# Science-GPT Frontend Technical Design + +This directory documents the frontend of the Science-GPT Summarization Tool β€” a React single-page application that guides scientific reviewers through document upload, AI-powered entity extraction, and quality evaluation. + +Read this page first for orientation, then follow the numbered module docs for detail on each page or feature area. + +--- + +## Visual workflow map + +The app has two entry paths: a **step-by-step advanced workflow** (the primary path) and a **simplified one-click flow** for users who want to skip the manual steps. + +**Advanced workflow (linear):** + +``` +[Login] β†’ [Upload] β†’ [Processing] β†’ [Study Config] β†’ [Extraction] β†’ [Evaluation] + 02 03 04 05 06 07 +``` + +**Tool overlays (accessible from any step):** + +``` +[Chat] [Session History] [Templates] [Groups] [Executive Mode] [Batch Results] + 09 10 11 12 13 14 +``` + +Tool overlays remember the previous workflow step β€” clicking Back returns the reviewer to where they were. + +**Visual architecture diagrams** are in [`../images/`](../images/). + +--- + +## Tech stack + +| Layer | Technology | Version | Purpose | +|---|---|---|---| +| Framework | React | 18.2.0 | UI rendering | +| Language | TypeScript | 5.2.2 | Type safety | +| Build tool | Vite | 5.1.4 | Dev server and production builds | +| Styling | Tailwind CSS | 3.4.1 | Utility-first CSS | +| Components | shadcn/ui (Radix UI) | β€” | Base UI components (buttons, dialogs, tables, etc.) | +| Auth | Better Auth | 1.5.6 | GitHub OAuth session management | +| PDF rendering | pdfjs-dist | 5.4.394 | In-browser PDF display with bounding box overlays | +| Word export | docx + markdown-docx | 9.5.1 / 1.5.1 | Export results as Word documents | +| Excel export | exceljs + xlsx | 4.4.0 / 0.18.5 | Export results as Excel workbooks | +| Charts | Recharts | 2.12.0 | Evaluation result visualizations | +| Animation | Framer Motion | 12.23.24 | Page and component transitions | +| Markdown | React Markdown + remark-gfm | 10.1.0 | Render LLM markdown responses | +| Notifications | Sonner | 1.4.0 | Toast messages | +| Telemetry | OpenTelemetry (OTLP) | β€” | Browser-side distributed tracing | + +No external state management library (no Redux, Zustand, etc.). All state lives in `App.tsx` and local component state. See [01-app-shell.md](01-app-shell.md) for how that works. + +--- + +## Page map + +| # | Step / Page | Component file | Workflow role | +|---|---|---|---| +| β€” | App shell | `App.tsx` | Central state container and router | +| β€” | Login | `components/LoginPage.tsx` | Auth entry point | +| β€” | Auth callback | `components/AuthCallback.tsx` | OAuth redirect handler | +| 03 | Upload | `components/UploadPage.tsx` | Workflow step 1 β€” upload files, choose parser | +| 04 | Processing | `components/ProcessingPage.tsx` | Workflow step 2 β€” parse PDFs into text/figures/tables | +| 05 | Study config | `components/BatchStudySelectionPage.tsx` | Workflow step 3 β€” select study type, configure entities | +| 06 | Extraction | `components/EntityExtractionPage.tsx` | Workflow step 4 β€” run entity extraction with LLMs | +| 07 | Evaluation | `components/EvaluationPage.tsx` | Workflow step 5 β€” score extraction quality with G-Eval | +| 08 | Simplified flow | `components/SimplifiedFlowPage.tsx` | One-click end-to-end pipeline | +| 09 | Chat | `components/ChatPage.tsx` | Freeform document Q&A | +| 10 | Session history | `components/SessionHistoryPage.tsx` | Browse, restore, share previous sessions | +| 11 | Templates | `components/TemplateWorkspace/TemplateWorkspacePage.tsx` | Create and manage extraction template libraries | +| 12 | Groups | `components/GroupManagement/GroupManagementPage.tsx` | Create and manage user groups for sharing | +| 13 | Executive mode | `components/ExecutiveModePage.tsx` | Standalone executive summary generation | +| 14 | Batch results | `components/BatchResultsPage.tsx` | Tabular view of all results across files | + +--- + +## Architecture overview + +### State management + +The app uses a single top-level state object called `DocumentData` in `App.tsx`. Every page reads from it on mount and reports updates back via an `onComplete()` callback. There is no global store or context for workflow state β€” it flows through props. + +See [01-app-shell.md](01-app-shell.md) for the full `DocumentData` interface and callback pattern. + +### Navigation + +Navigation is controlled by a `currentStep` string in `App.tsx`, not a URL router. Clicking "Next" calls `setCurrentStep()` directly. The browser URL does not change between workflow steps. + +### Authentication + +Every API call uses `authenticatedFetch()` from `utils/authUtils.ts`, which automatically attaches a `Authorization: Bearer {token}` header and retries once on 401. Sessions are established via GitHub OAuth through the Better Auth sidecar. + +See [02-auth.md](02-auth.md) for the full auth flow. + +### Session persistence + +When a reviewer closes the browser mid-workflow, the app saves `currentStep` and `sessionId` to `localStorage` and auto-restores on next load by calling `/api/sessions/{id}/restore-view`. + +--- + +## Directory structure + +``` +frontend/ +β”œβ”€β”€ App.tsx Main routing + central state (~2100 lines) +β”œβ”€β”€ main.tsx React entry point +β”œβ”€β”€ index.html HTML template +β”œβ”€β”€ vite.config.ts Build config +β”œβ”€β”€ tailwind.config.js Tailwind config +β”œβ”€β”€ package.json Dependencies +β”‚ +β”œβ”€β”€ components/ All React components +β”‚ β”œβ”€β”€ LoginPage.tsx +β”‚ β”œβ”€β”€ AuthCallback.tsx +β”‚ β”œβ”€β”€ UploadPage.tsx +β”‚ β”œβ”€β”€ ProcessingPage.tsx +β”‚ β”œβ”€β”€ BatchStudySelectionPage.tsx +β”‚ β”œβ”€β”€ EntityExtractionPage.tsx (~4300 lines, largest component) +β”‚ β”œβ”€β”€ EvaluationPage.tsx (~5000 lines) +β”‚ β”œβ”€β”€ SimplifiedFlowPage.tsx +β”‚ β”œβ”€β”€ ChatPage.tsx +β”‚ β”œβ”€β”€ SessionHistoryPage.tsx +β”‚ β”œβ”€β”€ ExecutiveModePage.tsx +β”‚ β”œβ”€β”€ BatchResultsPage.tsx +β”‚ β”œβ”€β”€ TemplateWorkspace/ +β”‚ β”‚ β”œβ”€β”€ TemplateWorkspacePage.tsx +β”‚ β”‚ β”œβ”€β”€ TemplateList.tsx +β”‚ β”‚ β”œβ”€β”€ TemplateEditor.tsx +β”‚ β”‚ β”œβ”€β”€ TemplateVersionHistory.tsx +β”‚ β”‚ └── FolderCard.tsx +β”‚ β”œβ”€β”€ GroupManagement/ +β”‚ β”‚ └── GroupManagementPage.tsx +β”‚ └── ui/ shadcn/ui base components (50+) +β”‚ +β”œβ”€β”€ hooks/ Custom React hooks +β”‚ β”œβ”€β”€ useTemplates.ts +β”‚ β”œβ”€β”€ useGroups.ts +β”‚ β”œβ”€β”€ useFolders.ts +β”‚ └── useSimplifiedPipeline.ts +β”‚ +β”œβ”€β”€ contexts/ +β”‚ └── ThemeContext.tsx Light/dark theme +β”‚ +β”œβ”€β”€ utils/ +β”‚ β”œβ”€β”€ authUtils.ts Auth, token management, authenticated fetch +β”‚ β”œβ”€β”€ session.ts Session ID tracking +β”‚ β”œβ”€β”€ modelSelection.ts Model picker logic +β”‚ β”œβ”€β”€ wordExport.ts Word document generation +β”‚ β”œβ”€β”€ excelExport.ts Excel export +β”‚ └── executiveSummaryExport.ts Executive summary export +β”‚ +└── types/ + └── session.ts Shared TypeScript types +``` + +--- + +## Module docs + +| Document | What it covers | +|---|---| +| [01-app-shell.md](01-app-shell.md) | `App.tsx` β€” `DocumentData`, routing, session persistence, navigation guards | +| [02-auth.md](02-auth.md) | Login, OAuth callback, `authUtils.ts` β€” token management, authenticated fetch | +| [03-upload.md](03-upload.md) | Upload page β€” file upload, deduplication, parser selection | +| [04-processing.md](04-processing.md) | Processing page β€” document parsing, figures, tables, PDF viewer | +| [05-study-config.md](05-study-config.md) | Study config page β€” study type, entity definitions, template loading | +| [06-extraction.md](06-extraction.md) | Extraction page β€” entity extraction, multi-model comparison, PDF reference highlighting | +| [07-evaluation.md](07-evaluation.md) | Evaluation page β€” G-Eval scoring, background jobs, result visualization | +| [08-simplified-flow.md](08-simplified-flow.md) | Simplified flow β€” one-click pipeline, `useSimplifiedPipeline` hook | +| [09-chat.md](09-chat.md) | Chat page β€” document Q&A, multi-document context | +| [10-session-history.md](10-session-history.md) | Session history β€” browse, restore, share, delete sessions | +| [11-templates.md](11-templates.md) | Template workspace β€” CRUD, versioning, scopes, folders | +| [12-groups.md](12-groups.md) | Group management β€” create groups, manage membership and roles | +| [13-executive-mode.md](13-executive-mode.md) | Executive mode β€” standalone summary generation | +| [14-batch-results.md](14-batch-results.md) | Batch results β€” tabular view, search, export | + +## Appendices + +| Document | What it covers | +|---|---| +| [appendices/component-index.md](appendices/component-index.md) | All shared components with usage notes | +| [appendices/hooks-contexts.md](appendices/hooks-contexts.md) | All custom hooks and `ThemeContext` with exported APIs | +| [appendices/types-interfaces.md](appendices/types-interfaces.md) | Key TypeScript interfaces (`DocumentData`, `Template`, `Group`, etc.) | diff --git a/docs/frontend/appendices/component-index.md b/docs/frontend/appendices/component-index.md new file mode 100644 index 0000000..36cd7a9 --- /dev/null +++ b/docs/frontend/appendices/component-index.md @@ -0,0 +1,181 @@ +# Shared Component Index + +All reusable components in `frontend/components/` that are used across multiple pages. shadcn/ui base components (`components/ui/`) are not listed here β€” they are standard Radix UI wrappers and are documented at [ui.shadcn.com](https://ui.shadcn.com). + +--- + +## PDF and document viewers + +### `PDFBoundingBoxViewer` + +Renders a PDF in-browser using `pdfjs-dist` and draws coloured rectangular overlays at coordinates returned by the parser (figures, tables). + +**Props:** +- `fileHash: string` β€” identifies the file in blob storage +- `boundingBoxes: BoundingBox[]` β€” coordinates to highlight +- `activeBoxId?: string` β€” scrolls to and pulses this box on change + +**Used in:** [Processing](../04-processing.md), [Extraction](../06-extraction.md) + +--- + +### `EntityPDFViewerBeta` + +Advanced PDF viewer with entity-reference highlighting. Renders all referenced pages for the currently selected entity and draws coloured boxes at the bounding-box coordinates returned in the extraction response. + +**Props:** +- `fileHash: string` +- `references: Reference[]` β€” page + bounding box per source passage +- `activeReference?: number` β€” index of the reference to scroll to + +**Used in:** [Extraction](../06-extraction.md) + +--- + +### `FigureGallery` + +Paginated image carousel for figures extracted from a document. Shows the figure image, caption, and page number. Falls back to a placeholder if the image is unavailable in blob storage. + +**Props:** +- `fileHash: string` +- `figures: FigureMetadata[]` +- `processor: string` + +**Used in:** [Processing](../04-processing.md) + +--- + +### `TablesGallery` + +Paginated viewer for HTML tables extracted from a document. Renders each table's HTML in an isolated container to prevent style bleed. + +**Props:** +- `fileHash: string` +- `tables: TableMetadata[]` +- `processor: string` + +**Used in:** [Processing](../04-processing.md) + +--- + +## Content rendering + +### `MarkdownViewer` + +Renders markdown strings with GitHub Flavoured Markdown (GFM) extensions β€” tables, strikethrough, task lists, code blocks with syntax highlighting. + +**Props:** +- `content: string` +- `className?: string` + +**Used in:** [Chat](../09-chat.md), [Extraction](../06-extraction.md), [Evaluation](../07-evaluation.md), [Batch Results](../14-batch-results.md) + +--- + +### `RawOutputViewer` + +Syntax-highlighted JSON viewer for raw parser output. Uses `react-json-view` or similar for collapsible tree rendering. + +**Props:** +- `data: object` + +**Used in:** [Processing](../04-processing.md) + +--- + +## Auth + +### `LoginPage` + +GitHub OAuth login screen. See [02-auth.md](../02-auth.md). + +### `AuthCallback` + +OAuth redirect handler. See [02-auth.md](../02-auth.md). + +--- + +## Templates and model config + +### `TemplatePicker` + +Dropdown that lists all templates accessible to the reviewer. On selection, resolves the template to its entity list and calls an `onSelect` callback. + +**Props:** +- `onSelect: (template: Template) => void` +- `studyTypeFilter?: string` + +**Used in:** [Study Config](../05-study-config.md), [Extraction](../06-extraction.md), [Executive Mode](../13-executive-mode.md) + +--- + +### `TemplateLoader` + +Utility component (no UI) that provides built-in study type templates from local JSON files. + +**Exports:** +- `loadStudyTypeTemplate(studyType: string): Entity[]` +- `getAvailableStudyTypes(): string[]` +- `getStudyTypeDisplayName(studyType: string): string` + +**Used in:** [Study Config](../05-study-config.md), [Simplified Flow](../08-simplified-flow.md) + +--- + +### `SettingsManager` + +Singleton class (not a React component) that manages global settings in `localStorage`. Stores API keys and model configurations per provider. + +**Key methods:** +- `getModelConfigs(): ModelConfig[]` β€” returns all configured models +- `getModelConfig(modelId: string): ModelConfig | null` +- `saveModelConfig(config: ModelConfig): void` +- `clearModelConfig(modelId: string): void` + +**Used by:** All pages that call API endpoints requiring model credentials. + +--- + +### `SettingsPage` + +Modal or overlay for editing `SettingsManager` values. Presents a form for each provider's API key, deployment name, and API version. + +--- + +## Metrics and export + +### `SessionMetrics` + +Displays token usage, cost, and duration for the current session. Reads from backend session metrics endpoint and updates in real time. + +**Props:** +- `sessionId: string` + +**Used in:** [Extraction](../06-extraction.md), [Evaluation](../07-evaluation.md) + +--- + +### `ExportUtils` + +Non-rendered utility module for generating and downloading files client-side. + +**Exports:** +- `generateWordDocument(entities, summary, options): Promise` β€” builds a `.docx` file from extraction results +- `generateMarkdownDocument(entities, summary): string` β€” formats results as Markdown +- `downloadFile(blob, filename): void` β€” triggers browser download + +**Used in:** [Extraction](../06-extraction.md), [Simplified Flow](../08-simplified-flow.md), [Executive Mode](../13-executive-mode.md), [Batch Results](../14-batch-results.md) + +--- + +## Error handling + +### `ErrorBoundary` + +React error boundary wrapper. Catches render errors in child components and shows a fallback UI with a "Reload" button instead of a blank screen. + +**Props:** +- `children: ReactNode` +- `fallback?: ReactNode` + +Wraps the entire app in `main.tsx` and is also used around the PDF viewers (which can throw on malformed PDFs). diff --git a/docs/frontend/appendices/hooks-contexts.md b/docs/frontend/appendices/hooks-contexts.md new file mode 100644 index 0000000..d7d6585 --- /dev/null +++ b/docs/frontend/appendices/hooks-contexts.md @@ -0,0 +1,199 @@ +# Hooks and Contexts + +All custom React hooks and context providers in the frontend. + +--- + +## Contexts + +### `ThemeContext` + +**File:** `contexts/ThemeContext.tsx` + +Manages light/dark theme across the app. + +**Behaviour:** +- On first load, reads `prefers-color-scheme` from the OS. +- Persists the reviewer's manual preference to `localStorage` under `summarization_theme`. +- Applies the theme by toggling a `dark` class on `document.documentElement` (Tailwind dark mode convention). + +**Exported hook:** + +```typescript +const { theme, toggleTheme } = useTheme(); +// theme: "light" | "dark" +// toggleTheme: () => void +``` + +**Used in:** Navigation bar theme toggle button, present in all pages. + +--- + +## Custom hooks + +### `useTemplates` + +**File:** `hooks/useTemplates.ts` + +Manages all CRUD operations and filtering for prompt templates. + +**Returns:** + +```typescript +{ + templates: Template[]; + loading: boolean; + error: string | null; + filters: TemplateFilters; + setFilters: (filters: TemplateFilters) => void; + fetchTemplates: () => Promise; + createTemplate: (data: CreateTemplateRequest) => Promise