From cc187b132ff9452e7340431a08bbef63521a6908 Mon Sep 17 00:00:00 2001 From: eric8810 Date: Tue, 29 Sep 2026 18:19:09 +0800 Subject: [PATCH] =?UTF-8?q?feat(openvino):=20=E8=8A=AF=E5=90=AF=E7=A5=9E?= =?UTF-8?q?=E6=9E=A2=EF=BC=8CNPU=20=E5=85=88=E8=A1=8C=20=C2=B7=20Add=20the?= =?UTF-8?q?=20Intel=20NPU=20OpenVINO=20backend?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- CHANGELOG.md | 13 + CMakeLists.txt | 35 ++ bindings/node/src/addon.cpp | 4 +- docs/cli-design.md | 2 +- docs/intel-npu-acceleration.md | 335 +++++++++++++ docs/linux-device-acceleration.md | 2 +- include/light_ocr/types.hpp | 2 +- packages/light-ocr-server/src/config.js | 4 +- packages/light-ocr/README.md | 2 +- packages/light-ocr/src/document-cli.cjs | 6 +- packages/runtime/src/cli.cjs | 8 +- packages/runtime/src/index.d.ts | 2 +- src/core/engine.cpp | 188 ++++++- src/core/engine_factory.hpp | 8 + src/inference/backend.hpp | 8 + src/inference/openvino/backend.cpp | 642 ++++++++++++++++++++++++ src/inference/openvino/backend.hpp | 59 +++ src/preprocess/tensor.cpp | 19 +- src/preprocess/tensor.hpp | 6 +- tests/integration/main.cpp | 33 +- tests/unit/test_image.cpp | 34 ++ tests/unit/test_model_bundle.cpp | 118 +++++ tests/unit/test_selection.cpp | 32 ++ tools/benchmark/main.cpp | 1 + tools/common/arguments.hpp | 13 +- 25 files changed, 1527 insertions(+), 49 deletions(-) create mode 100644 docs/intel-npu-acceleration.md create mode 100644 src/inference/openvino/backend.cpp create mode 100644 src/inference/openvino/backend.hpp diff --git a/CHANGELOG.md b/CHANGELOG.md index 5a11b11..0f9088b 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,19 @@ This file records user-visible changes to `light-ocr`. Published artifact details and immutable hashes remain in [`docs/releases/`](docs/releases/). +## [Unreleased] + +### Added + +- Added a source-level Intel NPU backend (`provider: "openvino"`) for Linux x64. + Builds configured with `LIGHT_OCR_OPENVINO_SDK_DIR` load the OpenVINO C + runtime at run time, run detection and 20-bucket recognition on the NPU, and + place `openvino` first in a qualification-only Auto policy + (`openvino → webgpu → cpu`); hosts without an NPU skip it with + `adapter_unavailable`. Published npm packages do not include OpenVINO yet, so + `provider: "openvino"` returns `unsupported_capability` there. See + [Intel NPU 加速技术方案](docs/intel-npu-acceleration.md). + ## [0.5.8] - 2026-09-24 ### Added diff --git a/CMakeLists.txt b/CMakeLists.txt index 61cdca4..dfe502d 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -34,6 +34,10 @@ set(LIGHT_OCR_ORACLE_PYTHON "" CACHE FILEPATH "Pinned Python interpreter for par set(LIGHT_OCR_NODE_INCLUDE_DIR "" CACHE PATH "Directory containing node_api.h") set(LIGHT_OCR_NODE_LIBRARY "" CACHE FILEPATH "Windows Node import library") set(LIGHT_OCR_NODE_EXECUTABLE "" CACHE FILEPATH "Node.js executable used for adapter tests") +set(LIGHT_OCR_OPENVINO_SDK_DIR "" CACHE PATH + "OpenVINO SDK root with include/openvino/c; enables the Intel NPU backend on Linux x64") +set(LIGHT_OCR_OPENVINO_C_LIBRARY "" CACHE FILEPATH + "OpenVINO C runtime loaded by direct C++ callers; defaults to /lib/libopenvino_c.so") if(LIGHT_OCR_ENABLE_SANITIZERS AND LIGHT_OCR_ENABLE_THREAD_SANITIZER) message(FATAL_ERROR "Address/Undefined sanitizers and ThreadSanitizer require separate builds") @@ -138,6 +142,34 @@ if(LIGHT_OCR_HAS_WEBGPU) endif() endif() +set(LIGHT_OCR_HAS_OPENVINO OFF) +if(LIGHT_OCR_OPENVINO_SDK_DIR) + if(NOT (CMAKE_SYSTEM_NAME STREQUAL "Linux" AND + CMAKE_SYSTEM_PROCESSOR MATCHES "^(x86_64|AMD64)$")) + message(FATAL_ERROR "The OpenVINO NPU backend currently targets Linux x64 only") + endif() + set(_light_ocr_openvino_include "${LIGHT_OCR_OPENVINO_SDK_DIR}/include") + if(NOT EXISTS "${_light_ocr_openvino_include}/openvino/c/openvino.h") + message(FATAL_ERROR + "LIGHT_OCR_OPENVINO_SDK_DIR has no include/openvino/c/openvino.h: ${LIGHT_OCR_OPENVINO_SDK_DIR}") + endif() + if(NOT LIGHT_OCR_OPENVINO_C_LIBRARY) + set(LIGHT_OCR_OPENVINO_C_LIBRARY "${LIGHT_OCR_OPENVINO_SDK_DIR}/lib/libopenvino_c.so") + endif() + if(NOT EXISTS "${LIGHT_OCR_OPENVINO_C_LIBRARY}") + message(FATAL_ERROR "OpenVINO C runtime not found: ${LIGHT_OCR_OPENVINO_C_LIBRARY}") + endif() + set(LIGHT_OCR_HAS_OPENVINO ON) + # Headers only: the runtime is loaded with dlopen so the library never + # becomes a link-time dependency of the Core or the Node addon. + target_sources(light_ocr_core PRIVATE src/inference/openvino/backend.cpp) + target_include_directories(light_ocr_core SYSTEM PRIVATE "${_light_ocr_openvino_include}") + target_compile_definitions(light_ocr_core PRIVATE + LIGHT_OCR_HAS_OPENVINO=1 + LIGHT_OCR_OPENVINO_DEFAULT_LIBRARY="${LIGHT_OCR_OPENVINO_C_LIBRARY}") + target_link_libraries(light_ocr_core PRIVATE ${CMAKE_DL_LIBS}) +endif() + if(MSVC) target_compile_options(light_ocr_core PRIVATE /W4 /WX /permissive- /EHsc /fp:strict) else() @@ -210,6 +242,9 @@ if(LIGHT_OCR_BUILD_TESTS) target_compile_definitions(light_ocr_unit_tests PRIVATE LIGHT_OCR_WEBGPU_QUALIFICATION_BUILD=1) endif() + if(LIGHT_OCR_HAS_OPENVINO) + target_compile_definitions(light_ocr_unit_tests PRIVATE LIGHT_OCR_HAS_OPENVINO=1) + endif() light_ocr_stage_onnxruntime(light_ocr_unit_tests) add_test(NAME light_ocr_unit_tests COMMAND light_ocr_unit_tests) diff --git a/bindings/node/src/addon.cpp b/bindings/node/src/addon.cpp index 0005ff9..1ee9c97 100644 --- a/bindings/node/src/addon.cpp +++ b/bindings/node/src/addon.cpp @@ -499,9 +499,10 @@ ExecutionProvider parse_execution_provider(napi_env env, napi_value value) { if (provider == "cpu") return ExecutionProvider::cpu; if (provider == "apple") return ExecutionProvider::apple; if (provider == "webgpu") return ExecutionProvider::webgpu; + if (provider == "openvino") return ExecutionProvider::openvino; throw AddonFailure( "invalid_argument", - "execution.provider must be auto, cpu, apple, or webgpu"); + "execution.provider must be auto, cpu, apple, webgpu, or openvino"); } SessionFallback parse_session_fallback(napi_env env, napi_value value) { @@ -1621,6 +1622,7 @@ const char* execution_provider_string(ExecutionProvider provider) { case ExecutionProvider::cpu: return "cpu"; case ExecutionProvider::apple: return "apple"; case ExecutionProvider::webgpu: return "webgpu"; + case ExecutionProvider::openvino: return "openvino"; } return "auto"; } diff --git a/docs/cli-design.md b/docs/cli-design.md index 76f7c49..f37d297 100644 --- a/docs/cli-design.md +++ b/docs/cli-design.md @@ -114,7 +114,7 @@ light-ocr detect image.png --provider webgpu | `--format` | json \| jsonl \| text | 默认 `json` | | `--region` | `x,y,w,h` | pageSpace 轴对齐矩形,整数像素;详见 §7 | | `--no-exif` | flag | 关闭默认 EXIF orientation 修正;详见 §6 | -| `--provider` | auto \| cpu \| apple \| webgpu | 映射 `execution.provider` | +| `--provider` | auto \| cpu \| apple \| webgpu \| openvino | 映射 `execution.provider`;`openvino` 仅在包含 OpenVINO NPU runtime 的构建中可用 | | `--schema-version` | 1 | 请求精确输出 schema;不支持则稳定失败 | `--help` 第二层(高级): diff --git a/docs/intel-npu-acceleration.md b/docs/intel-npu-acceleration.md new file mode 100644 index 0000000..aaa6a62 --- /dev/null +++ b/docs/intel-npu-acceleration.md @@ -0,0 +1,335 @@ +# Intel NPU 加速技术方案 + +状态:Phase B 核心已在源码实现(qualification build);Linux x64 真机 14-fixture 验证通过;未进入 released policy,npm 包未包含 OpenVINO,不代表已经发布;目标是 Intel NPU 主机上 Auto 优先使用 NPU + +更新时间:2026-09-29 + +范围:第一目标是 Linux x64 glibc 上的 Intel Core Ultra NPU;Windows x64 作为同一后端的第二平台;Intel iGPU/dGPU 继续走已发布的 Native WebGPU 主线,不在本方案内新建 GPU 后端 + +关联文档:[Linux Device 加速技术方案 §6](linux-device-acceleration.md#6-厂商-gpunpu-路线)、[Windows Device 加速技术方案](windows-device-acceleration.md)、[Apple Device 加速技术方案](apple-device-acceleration.md)、[D111](decisions.md#d111--freeze-a-provider-neutral-execution-contract-before-enabling-accelerators)、[D112](decisions.md#d112--use-platform-aware-auto-with-creation-time-ordered-fallback) + +## 1. 结论 + +Intel NPU 需要一个**新的 Direct OpenVINO 推理后端**,但不需要新的 OCR pipeline: + +- **WebGPU 无法使用 NPU。** Native WebGPU 经 Dawn 只枚举 Vulkan/D3D12/Metal 适配器;Linux 上的 Intel NPU 由 `intel_vpu` 内核驱动(`/dev/accel/accel0`)与 Level Zero 暴露,没有 Vulkan ICD。因此 NPU 只能由厂商 runtime 驱动,项目文档中既定的厂商路径是 OpenVINO。 +- **Intel GPU 继续走 WebGPU。** Arc B390(Panther Lake Xe3)经 Mesa ANV/Vulkan 已跑通锁定 14-fixture corpus,文本与 CPU FP32 199/199 一致;它属于现有开放兼容路径,只需补做 Provider Gate,不需要新代码(§9)。 +- **后端只替换 `InferenceSession`。** 新实现位于 `src/inference/openvino/`,与 ONNX Runtime CPU/WebGPU、Direct Core ML 并列;预处理、DB postprocess、crop、CTC decode 与结果组装保持不变。 +- **NPU 需要静态 shape。** Recognition 复用 Apple 路径已锁定的 20 个宽度桶,实测质量无损;Detection **不能**通过补边转成固定 shape(§3.3),必须按实际 shape 编译,或路由到 CPU。这是本方案最主要的设计取舍(§5)。 +- **有 Intel NPU 时 Auto 优先使用 NPU。** Linux x64 与 Windows x64 的目标 Auto 顺序为 `openvino → webgpu → cpu`:主机没有可用 NPU 时 `openvino` 以 `adapter_unavailable` 跳过,行为与现状相同。用户仍可用 `provider=cpu|webgpu|openvino` 手动指定后端(§8)。按 D112,`openvino` 只有在随包交付并通过平台 Gate 后才进入 released policy;在此之前只能显式指定。 +- **自包含分发,驱动为唯一前置条件。** OpenVINO Core、NPU plugin 与 TBB(合计约 28 MB)随 platform package 交付;NPU 内核驱动、用户态驱动与 compiler-in-driver 属于系统驱动栈,不打包(§7)。 + +## 2. 目标与非目标 + +### 2.1 目标 + +1. **释放 CPU。** 交互式 OCR 在 Core Ultra 笔记本上以 NPU 承担主要推理,进程 CPU-s 显著下降。 +2. **端到端提速。** 以同一 corpus、同一 CPU 基线比较完整 OCR P50,而不是只比较单模型 microbenchmark。 +3. **质量可审计。** 与 CPU FP32 locked goldens 对比文本、置信度和框;任何差异进入 parity exceptions 并经审阅。 +4. **失败可解释。** 设备缺失、驱动过旧、编译失败都映射到 D112 封闭原因;运行期不跨 backend 重试。 + +### 2.2 非目标 + +- 不把 OpenVINO GPU plugin 作为 Intel GPU 路线;Intel GPU 由 WebGPU 覆盖。 +- 不在同一 engine 内混用 WebGPU 与 NPU(§10)。 +- 不在第一阶段做 INT8/QDQ 量化;NPU 编译器默认 FP16 推理精度。 +- 不要求用户安装 OpenVINO SDK、Python 包或编译工具链,也不在 install/postinstall/首次运行时下载任何 runtime。 +- 不因 NPU 占用率上升就宣称完整 placement;placement 以 OpenVINO `EXECUTION_DEVICES` 与编译结果为准。 + +## 3. 已确认事实与 Spike 证据 + +### 3.1 测试环境 + +| 项目 | 值 | +| --- | --- | +| CPU | Intel Core Ultra X7 358H(Panther Lake) | +| NPU | `Intel(R) AI Boost`,PCI `8086:B03E`,`DEVICE_ARCHITECTURE=5010`,FP16 25.2 TOPS / INT8 50.4 TOPS(`DEVICE_GOPS`) | +| GPU | Intel Arc B390(Xe3),Mesa ANV Vulkan | +| OS | Arch 系(Omarchy),kernel 7.2.5 | +| 驱动栈 | `intel-npu-driver 1.38.0`、`intel-npu-compiler 2026.38`(compiler in driver)、`level-zero-loader 1.32.0` | +| Runtime | 系统 `openvino 2026.4.0` + `openvino-intel-npu-plugin 2026.4.0`;light-ocr CPU 基线为 ONNX Runtime 1.22.0 | +| 模型 | `ppocrv6-small-onnx-20260714.2` 的上游 FP32 ONNX(det SHA-256 `d73e0058…`,rec `5435fd74…`) | + +Spike 以临时 `InferenceSession` 实现替换 CPU 会话,由环境变量启用;其余 pipeline 未改。CPU 基线为 `cpu_fast` profile(12 intra-op 线程),每个 fixture 2 次 warmup + 10 次测量取 P50。**该基线与 WebGPU Gate 报告的 CPU 基线配置不同,倍数不可与 [Linux 报告](linux-device-acceleration.md#11-030-真实设备结论) 直接比较。** + +### 3.2 单模型 microbenchmark + +随机输入、固定 shape、20 次 P50: + +| 模型 / shape | OpenVINO CPU | NPU | NPU 首次编译 | 与 CPU 输出最大绝对差 | +| --- | ---: | ---: | ---: | ---: | +| det 1×3×960×960 | 30.6 ms | 11.7 ms | 2.7 s | 0.0012 | +| rec 1×3×48×320 | 4.0 ms | 1.2 ms | 0.75 s | 0.042 | +| rec 1×3×48×1600 | 14.7 ms | 6.3 ms | 1.1 s | 0.039 | + +`ov::cache_dir` 命中后,det 960×960 编译从 2.55 s 降到 15 ms。 + +### 3.3 完整 OCR(14-fixture corpus) + +| 方案 | 14 个 P50 之和 | 相对 CPU | 文本一致 | 进程平均 CPU 核 | +| --- | ---: | ---: | ---: | ---: | +| CPU(ORT FP32) | 2,743 ms | 1.00× | 基线 | ≈11.8 | +| WebGPU FP32(Arc B390) | 1,622 ms | 1.69× | 199/199 | ≈0.5 | +| NPU:det 实际 shape + rec 20 桶 | 541 ms | 5.07× | 198/199 | ≈0.5 | +| det WebGPU + rec NPU(仅 Spike) | 757 ms | 3.62× | 199/199 | ≈0.4 | +| det CPU + rec NPU(由分阶段数据估算) | ≈1,035 ms | ≈2.65× | 待测 | 检测阶段回到多核 | + +分阶段看,CPU 基线中 recognition 占推理时间约 78%(1,954 ms),detection 约 22%(553 ms)。以最重的 `paddleocr-xfund-form`(113 行)为例,CPU det/rec 为 68/955 ms,NPU 为 9/146 ms。 + +关键结论: + +1. **Detection 补边破坏质量。** 输入在 normalize 后以 0 补边再裁剪输出,补到 960×960 时文本一致率降到 83/200,补到 64 的倍数时降到 172/199,补到 160 的倍数时同样出现多处框合并与文本差异;边界附近的框合并或新增,`paddleocr-boarding-pass` 甚至检出水印。Detection 必须使用实际 bounded shape。 +2. **Recognition 宽度桶无损,但不得截断输出时间步。** 右侧补零到 20 个桶后,保留完整 `[N, T, C]` 交给 CTC decode,文本 198/199,与实际 shape 持平;若按宽度比例截断时间步,行尾字符会丢失。引擎已有的 `recognition_width_buckets_` 机制(Apple 路径使用)正好满足这一点。 +3. **FP16 边界字符。** 每轮完整运行约有 1 行差异,且差异行不固定(`・`/`·`、`姓名NAME`/`姓名 NAME`),均为低置信边界字符。 +4. **编译和缓存成本高。** 每个新 shape 首次编译 0.7–2 s;默认 cache blob 每个 5–15 MB,未设上限时 63 个 shape 达 685 MB;`CacheMode::OPTIMIZE_SIZE` 下 corpus 共 29 个 blob、167 MB。 + +### 3.4 Detection shape 空间 + +`bounded` 策略只缩小超过 960 的边,两边都向上取 32 的倍数,因此 detector 输入是 32–960 × 32–960 的任意 32 倍数组合,最多 **900 种 shape**,不是只有长边 960 的 59 种。corpus 实测即出现 `192×800`、`544×896`、`416×416`、`64×384`、`960×704` 等。按实际 shape 编译意味着多数新尺寸图片都要付一次编译成本。 + +### 3.5 分发相关事实 + +| 组件 | 大小 | 归属 | +| --- | ---: | --- | +| `libopenvino.so` | 19.2 MB | 随包 | +| `libopenvino_intel_npu_plugin.so` | 8.0 MB | 随包 | +| `libopenvino_c.so` | 0.3 MB | 随包(若采用 C API,§6.1) | +| `libtbb.so.12` | 0.25 MB | 随包 | +| `libopenvino_onnx_frontend.so` | 6.4 MB | 若离线转 IR 则不需要(§7.2) | +| `libopenvino_intel_npu_compiler.so` | 107 MB | 驱动栈(compiler in driver),不随包 | +| `libze_intel_npu.so`、`libze_loader.so` | — | 驱动栈,不随包 | + +插件默认 `NPU_COMPILER_TYPE=PREFER_PLUGIN`;显式设为 `DRIVER` 时,本机仍能成功编译 recognizer,实际加载的是驱动栈中的 compiler loader。NPU plugin 的直接依赖为 `libopenvino`、`libtbb`、`libpugixml` 与 C/C++ 运行库。 + +## 4. 设备与精度矩阵 + +| 设备 | 路径 | 精度 | 状态 | +| --- | --- | --- | --- | +| Intel Core Ultra NPU(Linux x64) | Direct OpenVINO NPU | FP16(编译器默认) | 本方案第一目标 | +| Intel Core Ultra NPU(Windows x64) | Direct OpenVINO NPU | FP16 | 第二目标;Windows NPU 驱动随系统 Windows Update 分发 | +| Intel iGPU/dGPU | Native WebGPU(Vulkan/D3D12) | FP32 | 已有开放兼容路径;待 Provider Gate | +| Intel CPU | ONNX Runtime CPU | FP32 | 稳定基线,不变 | +| INT8/QDQ on NPU | — | — | 后续;需要独立校准与质量 Gate | + +## 5. 路由设计 + +### 5.1 Recognition:NPU,20 个宽度桶 + +- batch 固定为 1,高度 48,宽度向上取到 Apple 路径已锁定的 20 个桶:`320 … 3200`。 +- 超过最大桶的宽度处理与 Apple 路径一致,不引入新语义。 +- 输出保持完整时间步,由现有 CTC decode 处理补齐区域产生的 blank。 +- **内容宽度保持与未补齐张量一致。** 宽度按 32 取整或按桶补齐后,现有预处理会把文字内容缩放到 `ceil(48 × 宽高比)`,而 CPU batch-1 路径使用 `trunc(48 × 宽高比)`;每行内容宽 1 像素会系统性改变插值结果。OpenVINO 路径启用 `natural_content_width`,内容保持截断宽度、只在右侧补零;关闭时 14-fixture 文本一致为 192/199,开启后为 198/199。Apple 路径保持原有语义不变。 +- 20 个编译产物构成有界集合,可预热(§6.3)。 + +### 5.2 Detection:两个候选,需 Gate 决定 + +| 候选 | 做法 | 优点 | 代价 | +| --- | --- | --- | --- | +| **D-NPU**:NPU 按实际 shape | 首次遇到的 shape 同步编译并写入有界 cache | 全部推理在 NPU,Spike 实测 5.07× | 新尺寸图片首张多 0.7–2 s;900 种 shape 需要 LRU 与磁盘上限 | +| **D-CPU**:CPU 执行 detector | OpenVINO 后端内部以 CPU 执行 detector(ORT CPU 会话或 OpenVINO CPU plugin,FP32) | 无 detector 编译成本,检测结果与 CPU goldens 对齐 | 估算约 2.65×;检测阶段仍占多核 | + +两者都是**同一 `openvino` 候选内部的 stage 路由**,与 Apple 后端把 detector 与宽 recognition 分别路由到 ANE/MLCPU/GPU 的方式一致,不构成 D112 所禁止的混合 backend。无论选哪一种,实际设备都必须写入 `SessionExecutionInfo`(detector 的 `device`、`actualProviderChain`),不能隐藏 CPU 执行。 + +建议:Phase B 同时实现两种路由,并以 qualification-only 开关切换;由 Phase C Gate 按"冷启动 + 首张新尺寸图片延迟"与"稳态 P50 + CPU-s"两组指标选定默认。若 D-NPU 被选中,D-CPU 保留为 `cpuPartition=allow` 下 detector 的受控实现,`cpuPartition=forbid` 时只允许 D-NPU。 + +以下补边改进可作为 Phase C 的附加实验,但在 Gate 证明质量等价之前不得进入产品:以非零背景值或边缘复制补边、只补短边到少量桶、按桶重采样而不是补边。 + +### 5.3 不跨 backend 回退 + +创建成功后 backend 冻结;NPU 设备丢失、驱动重置或推理失败直接返回错误,符合 D112。D-NPU 首次编译失败属于运行期错误,不得静默改走 CPU。 + +## 6. 后端实现 + +### 6.1 代码结构与 runtime 装载 + +- `src/inference/openvino/backend.{hpp,cpp}`:`OpenVinoSession : InferenceSession`,持有每个 shape 的 `CompiledModel` 与 `InferRequest`。 +- runtime 通过 `dlopen` 从 platform package 内的固定相对路径装载,不链接系统 OpenVINO,不搜索 `LD_LIBRARY_PATH`。优先使用 OpenVINO **C API**(`libopenvino_c`),避免 C++ ABI 与 addon 的 libstdc++ 版本耦合;若 C API 缺少所需属性,再评估 C++ API 与符号版本约束。 +- 装载前校验 descriptor 中每个库的字节数与 SHA-256,与 WebGPU plugin 的做法一致。 +- 与 ONNX Runtime 同进程共存:Spike 分别在 CPU flavor(ORT 1.22)与 WebGPU flavor(ORT 1.24.4 + WebGPU plugin)的进程中加载并运行 OpenVINO,均无冲突;产品实现仍需在 Gate 中覆盖 20 次 lifecycle。 + +### 6.2 Session 配置 + +| 属性 | 值 | 说明 | +| --- | --- | --- | +| device | `NPU` | 第一阶段不暴露 `deviceId`;多 NPU 主机不在范围内 | +| `PERFORMANCE_HINT` | `LATENCY` | 交互式 profile;`throughput` 后续评估 | +| `INFERENCE_PRECISION_HINT` | `f16` | 显式固定,不依赖默认值 | +| `NPU_COMPILER_TYPE` | `DRIVER` | 不随包分发 107 MB 编译器 | +| `CACHE_DIR` | 产品缓存目录(§6.3) | | +| `CACHE_MODE` | `OPTIMIZE_SIZE` | 实测 blob 显著变小;需确认权重来源始终是随包模型 | +| `NPU_TURBO` | 默认 `NO` | 功耗优先;是否开启由 Gate 评估 | + +### 6.3 编译缓存 + +- 目录沿用 Apple 编译缓存的根目录约定,键包含:模型 SHA-256、OpenVINO 版本、`NPU_DRIVER_VERSION`、`NPU_COMPILER_VERSION`、`DEVICE_ARCHITECTURE`、shape。任一变化即视为新条目。 +- 跨进程写入使用与 Apple 缓存相同的文件锁语义;损坏条目删除后重编译,不作为 creation failure。 +- 总大小设上限(初值 256 MiB,Gate 校准),LRU 淘汰;recognition 的 20 个桶优先保留。 +- Engine 创建时只同步编译最常用的 recognition 桶与 hello canary 所需 shape,其余桶在后台线程预热;后台编译不得阻塞 `recognize()`,也不得在 engine 关闭后继续写入。 +- `SessionExecutionInfo.model_cache_status` 报告 `hit | miss | disabled`。 + +### 6.4 D112 创建原因映射 + +| 情况 | 原因 | +| --- | --- | +| `/dev/accel` 不存在、OpenVINO 未枚举到 `NPU` | `adapter_unavailable` | +| `NPU_DRIVER_VERSION` 或 compiler 版本低于 descriptor 锁定下限 | `driver_version_unsupported` | +| 编译 locked 模型/shape 失败(算子或 shape 不支持) | `model_compute_unsupported` | +| NPU plugin 明确报告设备内存不足 | `device_memory_insufficient` | +| descriptor 声明的库缺失或结构无效 | `package_corrupt` | +| 库哈希不符 | `artifact_hash_mismatch` | +| OpenVINO 版本与 plugin/descriptor 不符 | `provider_abi_mismatch` | +| 其他装载失败 | `unrecoverable_load_failed` | + +分类只来自 typed 状态与版本比较,不解析异常文本。 + +## 7. 模型产物与分发 + +### 7.1 Platform package + +- `@arcships/light-ocr` 的 Linux x64 glibc(及后续 Windows x64)platform package 增加 `openvino/` 目录:Core、NPU plugin、TBB 与许可证/SBOM;约 28 MB,未超出 Linux Gate 的 256 MiB 解包 native payload 上限,但需要 package review 接受包体增长。 +- runtime descriptor 新增 `openvino` provider 条目:库清单与哈希、OpenVINO 版本、最低 NPU driver/compiler 版本、qualification ID。 +- musl 与 arm64 不在范围内;Intel 官方 NPU 驱动只面向 glibc x64。 +- Level Zero loader 视为驱动栈组件;若 Gate 发现主流发行版默认不带,再单独决定是否随包。 + +### 7.2 模型形式 + +两种选择,Phase B 决定: + +1. 直接读取已锁定的上游 FP32 ONNX:需要随包 `libopenvino_onnx_frontend`(+6.4 MB),但不新增模型产物。 +2. 构建期离线转换为 OpenVINO IR(`.xml` + `.bin`,可 FP16 压缩):不需要 ONNX frontend,加载更快;但它是新的 immutable 派生物,需要 provenance、确定性再生与 bundle manifest schema 升级,规则同 D113 的 WebGPU FP16 派生物。 + +建议先用方式 1 完成 Gate,确认收益后再评估方式 2。 + +## 8. 公共 API 与可观测性 + +- `ExecutionProvider` 增加 `openvino`;Node 类型 `provider?: 'cpu' | 'auto' | 'apple' | 'webgpu' | 'openvino'`,与 roadmap 中预留的名称一致。 +- 显式 `provider=openvino` 只尝试该后端;在未交付的平台上返回 `unsupported_capability`。 +- `precision` 只接受 `auto` 与 `fp16`;`fp32` 返回 `invalid_argument`。 +- `cpuPartition=forbid` 只在 detector 路由为 D-NPU 时有效,否则在创建前拒绝。 +- `SessionExecutionInfo`:`runtime=OpenVINO`、`runtime_version`、`device=npu:`、`device_family=`、`precision=fp16`、`shape_policy`、`model_cache_status`。 +- Per-call recognition diagnostics 复用现有 `shape_bucket` 与 `compute_unit=npu` 字段。 +- `light-ocr doctor --json` 报告 NPU 是否存在、驱动与 compiler 版本、是否满足 descriptor 下限;不上报设备 UUID。 + +### 8.1 Auto 与手动指定 + +目标 D112 policy(新 policy ID/version,以新的 D 编号记录): + +| 平台 | 当前 | 目标 | +| --- | --- | --- | +| Linux x64 glibc | `webgpu → cpu` | `openvino → webgpu → cpu` | +| Windows x64 | `webgpu → cpu` | `openvino → webgpu → cpu` | +| 其他平台 | 不变 | 不变(不交付 `openvino`) | + +- **有 NPU 就用 NPU。** Auto 在创建时先尝试 `openvino`;没有 NPU、驱动低于下限或编译失败时,分别以 `adapter_unavailable`、`driver_version_unsupported`、`model_compute_unsupported` 跳过,继续尝试 `webgpu`,最后是 `cpu`。包损坏、哈希不符等 fatal 原因仍按 D112 立即终止。 +- **NPU 优先于独显是明确的产品取舍。** 同时具备 NPU 与独显的主机,Auto 也选择 NPU,理由是省电、释放 CPU/GPU 给前台负载;需要最高吞吐的用户手动指定 `provider=webgpu`。 +- **手动指定。** `provider=cpu|webgpu|openvino` 只尝试指定后端,失败直接返回结构化错误,不回退。Node、C++ 与 CLI(`--provider`)使用同一组取值。 +- **进入 released policy 的条件。** 按 D112,`openvino` 必须由 runtime descriptor 声明、随 platform package 交付、并通过该平台 Gate 后才进入 released policy;在此之前 released policy 保持 `webgpu → cpu`,`provider=openvino` 仅在包含该 runtime 的构建中可显式使用。 +- **失败候选的清理与共存。** `openvino` 创建失败后,必须先销毁全部 OpenVINO 状态(compiled model、infer request、Core)再尝试 `webgpu`;OpenVINO 与 ORT WebGPU 需在同一进程中共存。两者都要在 Gate 中覆盖(§11),否则按 D112 不得同列一个 Auto list。 +- **Auto 下的创建延迟。** Auto 选中 NPU 意味着默认用户也会承担 NPU 的编译成本,因此 engine 创建不得同步编译全部 shape:只同步准备 canary 所需的最少 shape,其余在后台预热(§6.3)。这也使 detector 路由(§5.2)的"首张新尺寸图片延迟"成为默认体验指标。 + +## 9. Intel GPU:WebGPU 兼容路径 + +本机实测结论: + +- 锁定 WebGPU runtime(ORT 1.24.4 + WebGPU plugin 0.1.0)经 Dawn/Vulkan 选中 Intel 适配器(vendor `0x8086`),`webgpu_allow` profile 下 14 个 fixture 文本 199/199 一致,置信度最大差 1e-4。 +- 相对 `cpu_fast` 为 1.69×;recognition 是瓶颈(xfund-form 689 ms,NPU 为 146 ms)。 + +后续动作不涉及新代码:在 Arc B390 上运行 `tools/webgpu/qualify.py` 取得完整 164 项 Gate 报告,审阅后按 Linux 文档规则补充 Intel 设备行;在此之前 Intel GPU 仍属开放兼容路径,不继承已发布的性能倍数。 + +## 10. 已评估并否决的方案 + +| 方案 | 结论 | 依据 | +| --- | --- | --- | +| WebGPU 直接使用 NPU | 不可行 | NPU 无 Vulkan/D3D12 适配器 | +| det WebGPU + rec NPU(同一 engine 混合 backend) | 不采用 | 199/199 但仅 3.62×,慢于全 NPU;需要同进程加载两套 runtime,并违反 D112"候选拥有完整 detector/recognizer 对、不暴露混合 backend"的约束 | +| 多页流水线(GPU 检测下一页、NPU 识别本页) | 暂不采用 | 每页耗时由 recognition 决定(xfund ≈157 ms),与全 NPU(≈155 ms)相当 | +| recognition 行级拆分到 GPU + NPU | 不采用 | GPU recognition 慢约 4.5×,理论收益 ≤20%,同页混合 FP16/FP32 使质量不可复现 | +| OpenVINO `HETERO` 层级切分 | 不采用 | 小模型跨设备拷贝得不偿失 | +| OpenVINO `AUTO` / GPU plugin 统一调度 | 不采用 | 绕开已验收的 WebGPU 路线,且需要额外 GPU plugin | +| ORT OpenVINO EP | 不采用 | 需要替换当前官方 ORT 构建;NPU 静态 shape 仍需自行管理 | +| detector 固定尺寸补边 | 否决 | §3.3,质量显著下降 | + +## 11. Gate + +沿用 Linux WebGPU Gate 的结构,增加 NPU 特有项: + +- **质量**:14-fixture locked corpus 与 CPU FP32 goldens 对比;文本不一致行数不超过审阅后的 parity exceptions,框最大偏差与置信度差设上限;FP16 边界差异须逐条审阅。 +- **性能**:每个 case 3 次独立 cold start × (2 warmup + 10 次测量);报告 P50、P95、各阶段时间、进程 CPU-s 与 RSS。 +- **冷启动**:`generated-hello-123` canary 在空缓存下 engine 初始化 + 首个结果 ≤30 s,热缓存 ≤3 s(D111 口径);另记录"新尺寸图片首张延迟"(D-NPU)。 +- **缓存**:总大小不超过上限;并发两进程写同一缓存无损坏;驱动版本变化后正确失效。 +- **生命周期**:20 次 engine close/recreate,retained RSS 增长绝对值 ≤128 MiB;NPU 句柄无泄漏。 +- **失败路径**:无 NPU 主机、`/dev/accel` 权限不足、驱动过旧、库缺失/哈希错误分别产生 §6.4 中的原因。 +- **Auto**:有 NPU 主机选中 `openvino`;无 NPU 主机 trace 为 `openvino skipped(adapter_unavailable) → webgpu selected`;`openvino` 失败后清理完整、`webgpu` 正常创建;手动指定 `webgpu`/`cpu` 时不加载 OpenVINO。 +- **分发**:离线复装、无网络运行、无系统 OpenVINO 时可用;有系统 OpenVINO 时仍使用随包版本。 + +## 12. 分阶段落地 + +### Phase A — 证据(已完成) + +Spike 数据见 §3;临时代码不合入主干。 + +### Phase B — 后端实现(qualification build) + +1. `OpenVinoSession`、dlopen 装载、C API 封装、哈希校验。 +2. Recognition 20 桶路由;detector D-NPU 与 D-CPU 两种路由。 +3. 有界编译缓存与跨进程锁、后台预热。 +4. `ExecutionProvider::openvino`、C++/Node 参数校验、diagnostics。 +5. CMake 选项 `LIGHT_OCR_OPENVINO_SDK_DIR` 与锁定的 OpenVINO payload(参照 `tools/webgpu/runtime-lock.json` 的模式)。 +6. 单元测试:shape 桶、cache key、原因映射;CI 在无 NPU runner 上验证 `adapter_unavailable` 路径。 + +### Phase B 实施状态(2026-09-29) + +已实现: + +- `ExecutionProvider::openvino`;C++ 校验只接受 `precision=auto|fp16`、无 `deviceId`;Node addon、TypeScript 类型、runtime CLI、document CLI 与 server `EXECUTION_MODE` 接受 `openvino`。 +- `src/inference/openvino/backend.{hpp,cpp}`:`dlopen` 装载 OpenVINO C API,按 shape 编译并在内存中 LRU 保留(detector 32、recognizer 20),编译参数为 `LATENCY`、`f16`、`NPU_COMPILER_TYPE=DRIVER`,磁盘缓存为 `OPTIMIZE_SIZE`。 +- 磁盘缓存位于 `$XDG_CACHE_HOME`(或 `~/.cache`)`/com.arcships.light-ocr/openvino-v1/`,identity 为模型 SHA-256、OpenVINO 版本、设备架构、驱动与 compiler 版本的哈希;编译在跨进程 `flock` 下进行,写入后按修改时间淘汰到 256 MiB。 +- Detector 路由:默认 D-NPU;qualification build 可用 `LIGHT_OCR_QUALIFICATION_OPENVINO_DETECTOR=cpu` 切换到 D-CPU,此时 `cpuPartition=forbid` 在创建前以 `model_compute_unsupported` 拒绝。 +- D112:构建时设置 `LIGHT_OCR_OPENVINO_SDK_DIR` 后,builtin policy 为 `builtin-openvino-v1`,顺序 `openvino → [webgpu →] cpu`,并标记 qualification-only。batch≠1、非 bounded、`maxSide>960` 或不兼容的 recognizer 形状以 `model_compute_unsupported` 跳过。 +- 工具 profile `openvino_allow` / `openvino_strict`;单元测试覆盖参数校验、policy 校验、builtin 顺序、宽度桶契约与内容宽度语义。 + +本机结果(Core Ultra X7 358H,OpenVINO 2026.4,NPU driver 1.38): + +| 项目 | 结果 | +| --- | --- | +| 14-fixture 文本一致 | 198/199(唯一差异 `姓名NAME`/`姓名 NAME`) | +| 14 个 P50 之和 | 506–573 ms(4 次运行),相对 `cpu_fast` 4.8–5.4× | +| hello canary 冷缓存 init + 首个结果 | 2.1 s + 1.3 s | +| hello canary 热缓存 init + 首个结果 | 0.12 s + 0.05 s | +| 20 次 engine lifecycle | RSS 增长 86 KiB(上限 32 MiB) | +| 无 NPU(bwrap 隐藏 `/dev/accel`)Auto | `openvino skipped(adapter_unavailable) → webgpu selected` | +| 显式 `webgpu` | 进程不加载 OpenVINO | + +尚未实现(留在 Phase B/C):后台预热其余 recognition 桶、按 descriptor 的驱动/compiler 最低版本检查(`driver_version_unsupported`)、LRU 中优先保留 recognition 桶、`model_cache_status` 的 hit/miss 区分、Node runtime descriptor 中的 OpenVINO 条目(Phase D)。 + +### Phase C — Linux x64 真机 Gate + +在至少两代 Core Ultra(Meteor Lake / Lunar Lake 或 Arrow Lake / Panther Lake)上跑 §11,选定 detector 默认路由,写入审阅报告。 + +### Phase D — 分发 + +platform package staging、SBOM、许可证、runtime descriptor、package review;`provider=openvino` 以 Preview 发布。 + +### Phase E — Windows x64 + +同一后端在 Windows NPU 驱动上复用;分发 `openvino.dll` 等,重跑 Gate。 + +### Phase F — 进入 Auto + +记录新的 D 编号,确立 Linux/Windows `openvino → webgpu → cpu` policy;平台 Gate 通过、报告与产物哈希绑定 lock 后,更新 released policy,并在 CHANGELOG 中说明 Intel NPU 主机上 Auto 的选择变化。 + +## 13. 待决问题 + +1. Detector 默认走 D-NPU 还是 D-CPU(§5.2)? +2. 模型形式:直接读 ONNX 还是离线 IR 派生物(§7.2)? +3. 约 28 MB 的 platform package 增长是否接受?是否需要拆出可选 package(需同时符合"用户只安装 `@arcships/light-ocr`"的约束)? +4. Level Zero loader 是否视为驱动前置条件? +5. OpenVINO 版本升级节奏与 NPU 驱动兼容下限如何锁定? +6. 若某代 NPU 在 Gate 中慢于同机 WebGPU,是按 device family 在 descriptor 中排除该代,还是保持"有 NPU 即优先"? + +## 14. 参考 + +- [OpenVINO NPU device](https://docs.openvino.ai/2026/openvino-workflow/running-inference/inference-devices-and-modes/npu-device.html) +- [OpenVINO model caching](https://docs.openvino.ai/2026/openvino-workflow/running-inference/optimize-inference/optimizing-latency/model-caching-overview.html) +- [Intel NPU driver for Linux](https://github.com/intel/linux-npu-driver) +- [ONNX Runtime OpenVINO EP](https://onnxruntime.ai/docs/execution-providers/OpenVINO-ExecutionProvider.html) diff --git a/docs/linux-device-acceleration.md b/docs/linux-device-acceleration.md index db0c20d..7114dfb 100644 --- a/docs/linux-device-acceleration.md +++ b/docs/linux-device-acceleration.md @@ -201,7 +201,7 @@ WebGPU 成功不自动淘汰厂商 EP;失败也不代表 Linux 无法加速。 | NVIDIA GPU | ORT CUDA EP | 当前 FP32;记录 TF32 行为 | CUDA/cuDNN/driver 矩阵、runtime 体积、stream、copy、质量;收益通过后再做 FP16 | | NVIDIA GPU 高吞吐 | ORT TensorRT EP | FP16 派生物 | detector min/opt/max profile、recognition width/batch profile、engine/context cache、首次编译;必须同时处理未支持节点 | | Intel iGPU/dGPU | OpenVINO GPU | FP16 或 accuracy profile | graph coverage、driver、model cache、dynamic shape、CPU partition、包体 | -| Intel Core Ultra NPU | OpenVINO NPU | FP16;INT8/QDQ 后续 | NPU driver、固定/bounded shape、recognition buckets、detector 路由、compiled cache;不能隐藏 CPU fallback | +| Intel Core Ultra NPU | OpenVINO NPU | FP16;INT8/QDQ 后续 | NPU driver、固定/bounded shape、recognition buckets、detector 路由、compiled cache;不能隐藏 CPU fallback ;设计与真机 Spike 见 [Intel NPU 加速技术方案](intel-npu-acceleration.md) | | AMD GPU | MIGraphX | provider 资格审查后选择 FP32/FP16 | ROCm/MIGraphX 兼容矩阵、编译/cache、算子与动态 shape;旧 ORT ROCm EP 不作为新路线 | | AMD Ryzen AI NPU | Vitis AI,若 Linux 目标与分发可行 | INT8/BF16 provider-specific | 厂商模型派生、校准、编译/context、硬件与 OS 范围;与 MIGraphX 分开决策 | | Qualcomm/Rockchip/华为等 NPU | QNN/RKNPU/CANN 等 | 厂商专用 QDQ/context/IR | 通常依赖 Linux arm64 或特定设备;当前无 Tier 1 交集,等待 D110 与真实需求 | diff --git a/include/light_ocr/types.hpp b/include/light_ocr/types.hpp index 89c8671..ad2479d 100644 --- a/include/light_ocr/types.hpp +++ b/include/light_ocr/types.hpp @@ -15,7 +15,7 @@ enum class PixelFormat { gray8, rgb8, bgr8, rgba8 }; enum class DetectionStrategy { bounded, tiled, upstream_exact }; -enum class ExecutionProvider { automatic, cpu, apple, webgpu }; +enum class ExecutionProvider { automatic, cpu, apple, webgpu, openvino }; enum class SessionFallback { error, cpu }; diff --git a/packages/light-ocr-server/src/config.js b/packages/light-ocr-server/src/config.js index 32b872f..44504b7 100644 --- a/packages/light-ocr-server/src/config.js +++ b/packages/light-ocr-server/src/config.js @@ -1,6 +1,6 @@ 'use strict'; -const EXECUTION_MODES = new Set(['auto', 'cpu', 'apple', 'webgpu']); +const EXECUTION_MODES = new Set(['auto', 'cpu', 'apple', 'webgpu', 'openvino']); function readInteger(name, value, fallback, minimum, maximum) { const raw = value ?? String(fallback); @@ -17,7 +17,7 @@ function readInteger(name, value, fallback, minimum, maximum) { function readConfig(env = process.env) { const executionMode = env.EXECUTION_MODE ?? 'cpu'; if (!EXECUTION_MODES.has(executionMode)) { - throw new Error('EXECUTION_MODE must be one of: auto, cpu, apple, webgpu'); + throw new Error('EXECUTION_MODE must be one of: auto, cpu, apple, webgpu, openvino'); } return Object.freeze({ port: readInteger('PORT', env.PORT, 3000, 1, 65535), diff --git a/packages/light-ocr/README.md b/packages/light-ocr/README.md index d93732f..d25c038 100644 --- a/packages/light-ocr/README.md +++ b/packages/light-ocr/README.md @@ -227,7 +227,7 @@ index, source metadata, dimensions, applied PDF transforms, and timing. | `light-ocr doctor --json` | Print voluntary system/provider diagnostics | Image OCR supports `--format json|jsonl|text`, `--region x,y,w,h`, -`--provider auto|cpu|apple|webgpu`, `--stdin`, and automatic EXIF correction. +`--provider auto|cpu|apple|webgpu|openvino`, `--stdin`, and automatic EXIF correction. Document OCR adds `--pages N-M`, `--dpi`, and the page/file/pixel limit flags. `detect` always emits structured JSON and does not accept `--format`. diff --git a/packages/light-ocr/src/document-cli.cjs b/packages/light-ocr/src/document-cli.cjs index 7738c10..9b88a4f 100755 --- a/packages/light-ocr/src/document-cli.cjs +++ b/packages/light-ocr/src/document-cli.cjs @@ -22,7 +22,7 @@ Options: --max-page-pixels Maximum rendered pixels per page --max-total-pixels Maximum rendered pixels for the request --max-file-bytes Maximum bytes per input - --provider + --provider --quiet Suppress progress output -h, --help Show help -v, --version Show version`; @@ -111,8 +111,8 @@ function parseArgs(argv) { } else if (arg === '--provider') { provider = takeValue(args, index, arg); index++; - if (!['auto', 'cpu', 'apple', 'webgpu'].includes(provider)) { - throw argumentError('--provider must be auto, cpu, apple, or webgpu'); + if (!['auto', 'cpu', 'apple', 'webgpu', 'openvino'].includes(provider)) { + throw argumentError('--provider must be auto, cpu, apple, webgpu, or openvino'); } } else if (arg.startsWith('-')) { throw argumentError(`unknown option: ${arg}`); diff --git a/packages/runtime/src/cli.cjs b/packages/runtime/src/cli.cjs index b807af3..bdc7f1c 100755 --- a/packages/runtime/src/cli.cjs +++ b/packages/runtime/src/cli.cjs @@ -53,7 +53,7 @@ const OCR_ERROR_EXIT = { package_load_failed: EXIT.env_package, }; -const ALLOWED_PROVIDERS = new Set(['auto', 'cpu', 'apple', 'webgpu']); +const ALLOWED_PROVIDERS = new Set(['auto', 'cpu', 'apple', 'webgpu', 'openvino']); const ALLOWED_FORMATS = new Set(['json', 'jsonl', 'text']); function die(stderr, commandName, message) { @@ -185,7 +185,7 @@ function resolveProvider(flags) { if (flags.provider === undefined) return undefined; const provider = flags.provider; if (!ALLOWED_PROVIDERS.has(provider)) { - throw { code: EXIT.invalid_argument, message: `unsupported --provider ${provider}; use auto, cpu, apple, or webgpu` }; + throw { code: EXIT.invalid_argument, message: `unsupported --provider ${provider}; use auto, cpu, apple, webgpu, or openvino` }; } return provider; } @@ -575,7 +575,7 @@ function printSubcommandHelp(stdout, subcommand, config) { stdout.write('Flags:\n'); stdout.write(' --format json|jsonl|text Output format (default: json)\n'); stdout.write(' --region x,y,w,h Restrict recognition to a pageSpace rectangle\n'); - stdout.write(' --provider auto|cpu|apple|webgpu Execution provider (default: auto)\n'); + stdout.write(' --provider auto|cpu|apple|webgpu|openvino Execution provider (default: auto)\n'); stdout.write(' --no-exif Disable EXIF orientation correction\n'); stdout.write(' --schema-version 1 Request exact output schema\n'); stdout.write(' --quiet Suppress non-error stderr\n'); @@ -589,7 +589,7 @@ function printSubcommandHelp(stdout, subcommand, config) { stdout.write('Flags:\n'); stdout.write(' --region x,y,w,h Restrict detection to a pageSpace rectangle\n'); stdout.write(' --crop Reserved; currently fails as unsupported\n'); - stdout.write(' --provider auto|cpu|apple|webgpu Execution provider (default: auto)\n'); + stdout.write(' --provider auto|cpu|apple|webgpu|openvino Execution provider (default: auto)\n'); stdout.write(' --no-exif Disable EXIF orientation correction\n'); stdout.write(' --schema-version 1 Request exact output schema\n'); stdout.write(' --quiet Suppress non-error stderr\n'); diff --git a/packages/runtime/src/index.d.ts b/packages/runtime/src/index.d.ts index aead06c..54d6d6e 100644 --- a/packages/runtime/src/index.d.ts +++ b/packages/runtime/src/index.d.ts @@ -2,7 +2,7 @@ export type PixelFormat = 'gray8' | 'rgb8' | 'bgr8' | 'rgba8'; export type DetectionStrategy = 'bounded' | 'tiled' | 'upstreamExact'; -export type ExecutionProvider = 'auto' | 'cpu' | 'apple' | 'webgpu'; +export type ExecutionProvider = 'auto' | 'cpu' | 'apple' | 'webgpu' | 'openvino'; export type SessionFallback = 'error' | 'cpu'; export type CpuPartition = 'allow' | 'forbid'; export type PerformanceHint = 'latency' | 'throughput'; diff --git a/src/core/engine.cpp b/src/core/engine.cpp index 723169d..181febb 100644 --- a/src/core/engine.cpp +++ b/src/core/engine.cpp @@ -5,6 +5,7 @@ #include #include #include +#include #include #include #include @@ -22,6 +23,7 @@ #include "inference/coreml/backend.hpp" #endif #include "inference/onnxruntime/backend.hpp" +#include "inference/openvino/backend.hpp" #include "inference/selection.hpp" #include "model/bundle_data.hpp" #include "preprocess/image.hpp" @@ -98,6 +100,13 @@ bool valid_execution_options(const ExecutionOptions& options) { (options.precision == Precision::automatic || options.precision == Precision::fp32); } + if (options.provider == ExecutionProvider::openvino) { + return !options.device_id.has_value() && + (options.cpu_partition == CpuPartition::allow || + options.cpu_partition == CpuPartition::forbid) && + (options.precision == Precision::automatic || + options.precision == Precision::fp16); + } return options.provider == ExecutionProvider::apple && !options.device_id.has_value() && (options.cpu_partition == CpuPartition::allow || @@ -112,12 +121,14 @@ const char* provider_name(ExecutionProvider provider) { case ExecutionProvider::cpu: return "cpu"; case ExecutionProvider::apple: return "apple"; case ExecutionProvider::webgpu: return "webgpu"; + case ExecutionProvider::openvino: return "openvino"; } return "auto"; } bool known_provider(const std::string& provider) { - return provider == "cpu" || provider == "apple" || provider == "webgpu"; + return provider == "cpu" || provider == "apple" || provider == "webgpu" || + provider == "openvino"; } bool policy_includes_provider(const internal::RuntimePolicy& policy, @@ -140,14 +151,33 @@ std::string policy_qualification_id(const internal::RuntimePolicy& policy, return policy.provider_qualification_ids[index]; } +bool valid_sha256_or_empty(const std::string& value) { + return value.empty() || + (value.size() == 64 && + std::all_of(value.begin(), value.end(), [](char character) { + return (character >= '0' && character <= '9') || + (character >= 'a' && character <= 'f'); + })); +} + bool valid_runtime_policy(const internal::RuntimePolicy& policy) { - const bool valid_provider_hash = policy.webgpu_provider_sha256.empty() || - (policy.webgpu_provider_sha256.size() == 64 && - std::all_of(policy.webgpu_provider_sha256.begin(), - policy.webgpu_provider_sha256.end(), [](char value) { - return (value >= '0' && value <= '9') || - (value >= 'a' && value <= 'f'); - })); + const bool valid_provider_hash = + valid_sha256_or_empty(policy.webgpu_provider_sha256); + const bool openvino_artifact_declared = + policy.openvino_runtime_bytes != 0 || + !policy.openvino_runtime_sha256.empty(); + if ((!policy.openvino_runtime_library.empty() && + !policy_includes_provider(policy, "openvino")) || + (openvino_artifact_declared && + (!policy_includes_provider(policy, "openvino") || + policy.openvino_runtime_library.empty() || + policy.openvino_runtime_bytes == 0 || + policy.openvino_runtime_sha256.empty() || + !valid_sha256_or_empty(policy.openvino_runtime_sha256))) || + (policy.openvino_detector_route != "npu" && + policy.openvino_detector_route != "cpu")) { + return false; + } if (policy.id.empty() || policy.version == 0 || policy.ordered_candidates.empty() || policy.ordered_candidates.back() != "cpu" || @@ -233,6 +263,7 @@ struct CreatedSessions { std::unique_ptr recognition; std::uint32_t recognition_width_multiple = 1; std::vector recognition_width_buckets; + bool recognition_natural_content_width = false; std::uint32_t maximum_backend_batch_size = 1; }; @@ -243,6 +274,7 @@ class EngineImpl final : public Engine { std::unique_ptr recognition, EngineInfo info, std::uint32_t recognition_width_multiple, std::vector recognition_width_buckets, + bool recognition_natural_content_width, std::uint32_t maximum_backend_batch_size) : bundle_(std::move(bundle)), detection_(std::move(detection)), @@ -250,6 +282,7 @@ class EngineImpl final : public Engine { info_(std::move(info)), recognition_width_multiple_(recognition_width_multiple), recognition_width_buckets_(std::move(recognition_width_buckets)), + recognition_natural_content_width_(recognition_natural_content_width), maximum_backend_batch_size_(maximum_backend_batch_size) {} ~EngineImpl() noexcept override { close(); } @@ -514,7 +547,7 @@ class EngineImpl final : public Engine { auto plans_result = internal::plan_recognition_batches( sorted_boxes, bundle_->geometry, bundle_->recognition, batch_size, info_.limits, recognition_width_multiple_, - recognition_width_buckets_); + recognition_width_buckets_, recognition_natural_content_width_); stage_end = Clock::now(); timing.crop_and_sort_us = elapsed_us(stage_begin, stage_end); if (!plans_result) { @@ -565,7 +598,8 @@ class EngineImpl final : public Engine { stage_begin = Clock::now(); auto batch_result = internal::make_recognition_batch( crops, plan, bundle_->recognition, recognition_limits, - recognition_width_multiple_, recognition_width_buckets_); + recognition_width_multiple_, recognition_width_buckets_, + recognition_natural_content_width_); stage_end = Clock::now(); timing.recognition_preprocess_us += elapsed_us(stage_begin, stage_end); if (!batch_result) { @@ -838,6 +872,7 @@ class EngineImpl final : public Engine { EngineInfo info_; std::uint32_t recognition_width_multiple_ = 1; std::vector recognition_width_buckets_; + bool recognition_natural_content_width_ = false; std::uint32_t maximum_backend_batch_size_ = 1; mutable std::mutex state_mutex_; std::condition_variable state_changed_; @@ -851,7 +886,22 @@ Engine::~Engine() noexcept = default; internal::RuntimePolicy internal::builtin_runtime_policy() { RuntimePolicy policy; +#if defined(LIGHT_OCR_HAS_OPENVINO) + // The OpenVINO NPU backend has not passed a platform Gate, so builds that + // include it are qualification-only regardless of the other providers. + policy.id = "builtin-openvino-v1"; #if defined(LIGHT_OCR_HAS_WEBGPU) + policy.ordered_candidates = {"openvino", "webgpu", "cpu"}; +#else + policy.ordered_candidates = {"openvino", "cpu"}; +#endif + policy.qualification_only = true; + policy.released = false; + if (const char* route = std::getenv("LIGHT_OCR_QUALIFICATION_OPENVINO_DETECTOR"); + route != nullptr && std::string(route) == "cpu") { + policy.openvino_detector_route = "cpu"; + } +#elif defined(LIGHT_OCR_HAS_WEBGPU) policy.id = "builtin-webgpu-v1"; policy.ordered_candidates = {"webgpu", "cpu"}; #if defined(LIGHT_OCR_WEBGPU_QUALIFICATION_BUILD) @@ -872,6 +922,10 @@ internal::RuntimePolicy internal::builtin_runtime_policy() { #if defined(LIGHT_OCR_HAS_WEBGPU) policy.available_providers.push_back("webgpu"); policy.provider_qualification_ids.push_back("builtin-webgpu-v1"); +#endif +#if defined(LIGHT_OCR_HAS_OPENVINO) + policy.available_providers.push_back("openvino"); + policy.provider_qualification_ids.push_back("builtin-openvino-v1"); #endif return policy; } @@ -1115,6 +1169,108 @@ Result> internal::EngineFactory::create( return internal::CandidateResult::success( std::move(created)); } + if (candidate == "openvino") { + created.provider = ExecutionProvider::openvino; + const auto& recognition = bundle.data_->recognition; + const auto& buckets = openvino_recognition_width_buckets(); + if (batch_size != 1 || + detection_strategy != DetectionStrategy::bounded || + detection_max_side > 960 || recognition.height != 48 || + recognition.minimum_tensor_width > buckets.front() || + recognition.maximum_tensor_width != buckets.back()) { + return fail( + Error{ErrorCode::unsupported_capability, + "The OpenVINO NPU provider cannot create the requested model profile", + "requires batch 1, bounded detection up to 960, and 48-pixel " + "recognition up to width 3200"}, + CreationReason::model_compute_unsupported); + } + const bool detector_on_cpu = + runtime_policy.openvino_detector_route == "cpu"; + if (detector_on_cpu && + options.execution.cpu_partition == CpuPartition::forbid) { + return fail( + Error{ErrorCode::unsupported_capability, + "The OpenVINO detector route runs on the CPU", + "cpuPartition=forbid requires the NPU detector route"}, + CreationReason::model_compute_unsupported); + } +#if defined(LIGHT_OCR_HAS_OPENVINO) + auto openvino_detection_config = detection_config; + openvino_detection_config.provider = ExecutionProvider::openvino; + openvino_detection_config.qualification_id = + policy_qualification_id(runtime_policy, candidate); + openvino_detection_config.shape_policy = "nchw-static-exact-32-960-v1"; + openvino_detection_config.openvino_runtime_library = + runtime_policy.openvino_runtime_library; + openvino_detection_config.openvino_runtime_bytes = + runtime_policy.openvino_runtime_bytes; + openvino_detection_config.openvino_runtime_sha256 = + runtime_policy.openvino_runtime_sha256; + auto openvino_recognition_config = recognition_config; + openvino_recognition_config.provider = ExecutionProvider::openvino; + openvino_recognition_config.qualification_id = + openvino_detection_config.qualification_id; + openvino_recognition_config.shape_policy = + "nchw-static-width-buckets-20-v1"; + openvino_recognition_config.openvino_runtime_library = + openvino_detection_config.openvino_runtime_library; + openvino_recognition_config.openvino_runtime_bytes = + openvino_detection_config.openvino_runtime_bytes; + openvino_recognition_config.openvino_runtime_sha256 = + openvino_detection_config.openvino_runtime_sha256; + std::optional creation_reason; + auto openvino_recognition = internal::OpenVinoSession::create( + recognition_bytes, openvino_recognition_config, + internal::ModelKind::recognition, + {1, 3, static_cast(recognition.height), + static_cast(buckets.front())}, + &creation_reason); + if (!openvino_recognition) { + return fail_from_error(openvino_recognition.error(), + creation_reason); + } + if (detector_on_cpu) { + auto cpu_detection_config = detection_config; + cpu_detection_config.provider = ExecutionProvider::cpu; + cpu_detection_config.qualification_id = + openvino_detection_config.qualification_id; + auto cpu_detection = internal::OnnxSession::create( + detection_bytes, cpu_detection_config, + internal::ModelKind::detection, 0, &creation_reason); + if (!cpu_detection) { + return fail_from_error(cpu_detection.error(), creation_reason); + } + created.detection = std::move(cpu_detection).value(); + } else { + const auto minimum_detection_side = static_cast( + bundle.data_->detection.minimum_dimension); + auto openvino_detection = internal::OpenVinoSession::create( + detection_bytes, openvino_detection_config, + internal::ModelKind::detection, + {1, 3, minimum_detection_side, minimum_detection_side}, + &creation_reason); + if (!openvino_detection) { + return fail_from_error(openvino_detection.error(), + creation_reason); + } + created.detection = std::move(openvino_detection).value(); + } + created.recognition = std::move(openvino_recognition).value(); + created.recognition_width_multiple = 32; + created.recognition_width_buckets = buckets; + created.recognition_natural_content_width = true; + created.maximum_backend_batch_size = 1; + return internal::CandidateResult::success( + std::move(created)); +#else + return fail( + Error{ErrorCode::unsupported_capability, + "The runtime descriptor and Core OpenVINO capabilities disagree", + {}}, + CreationReason::provider_abi_mismatch); +#endif + } if (candidate != "apple") { return fail(Error{ErrorCode::internal_error, "Runtime policy contains an unknown provider", {}}, @@ -1205,6 +1361,8 @@ Result> internal::EngineFactory::create( created.recognition_width_multiple; auto recognition_width_buckets = std::move(created.recognition_width_buckets); + const auto recognition_natural_content_width = + created.recognition_natural_content_width; const auto maximum_backend_batch_size = created.maximum_backend_batch_size; @@ -1219,6 +1377,8 @@ Result> internal::EngineFactory::create( info.execution_provider = detection->execution_info().runtime == "Core ML" ? "CoreML" + : selected_provider == ExecutionProvider::openvino + ? "OpenVINO" : selected_provider == ExecutionProvider::webgpu ? "WebGpuExecutionProvider" : "CPUExecutionProvider"; @@ -1238,6 +1398,12 @@ Result> internal::EngineFactory::create( runtime_policy.released && !runtime_policy.qualification_only}); } + if (policy_includes_provider(runtime_policy, "openvino")) { + info.execution.provider_capabilities.push_back(ProviderCapabilityInfo{ + "openvino", true, selected_provider == ExecutionProvider::openvino, + selected_provider == ExecutionProvider::openvino && + runtime_policy.released && !runtime_policy.qualification_only}); + } info.execution.selection_trace = std::move(selection.trace); if (policy_includes_provider(runtime_policy, "apple") && bundle.data_->apple_provider) { @@ -1272,7 +1438,7 @@ Result> internal::EngineFactory::create( std::move(runtime_bundle), std::move(detection), std::move(recognition), std::move(info), recognition_width_multiple, std::move(recognition_width_buckets), - maximum_backend_batch_size))); + recognition_natural_content_width, maximum_backend_batch_size))); } catch (const std::exception& exception) { return failure>(ErrorCode::runtime_initialization_failed, "Unexpected engine initialization failure", diff --git a/src/core/engine_factory.hpp b/src/core/engine_factory.hpp index 51ab428..e199e27 100644 --- a/src/core/engine_factory.hpp +++ b/src/core/engine_factory.hpp @@ -24,6 +24,14 @@ struct RuntimePolicy { std::string webgpu_provider_library; std::uint64_t webgpu_provider_bytes = 0; std::string webgpu_provider_sha256; + // Empty only for direct C++ callers, where the backend loads the OpenVINO C + // runtime configured at build time. + std::string openvino_runtime_library; + std::uint64_t openvino_runtime_bytes = 0; + std::string openvino_runtime_sha256; + // "npu" runs both models on the NPU; "cpu" keeps the detector on the CPU + // backend and sends only recognition to the NPU. + std::string openvino_detector_route = "npu"; std::vector ordered_candidates; std::vector available_providers; // Entries are aligned with available_providers. diff --git a/src/inference/backend.hpp b/src/inference/backend.hpp index 083c5f4..a49e48d 100644 --- a/src/inference/backend.hpp +++ b/src/inference/backend.hpp @@ -55,6 +55,14 @@ struct InferenceSessionConfig { std::uint64_t webgpu_provider_bytes = 0; std::string webgpu_provider_sha256; bool webgpu_device_validated = false; + // Empty only for direct C++ callers, where the backend loads the OpenVINO C + // runtime configured at build time. Package adapters pass the + // descriptor-verified absolute path. + std::string openvino_runtime_library; + std::uint64_t openvino_runtime_bytes = 0; + std::string openvino_runtime_sha256; + // Empty selects the per-user cache directory. + std::string openvino_cache_directory; std::optional apple_package; std::string requested_provider_override; bool session_fallback_used = false; diff --git a/src/inference/openvino/backend.cpp b/src/inference/openvino/backend.cpp new file mode 100644 index 0000000..bc59d01 --- /dev/null +++ b/src/inference/openvino/backend.cpp @@ -0,0 +1,642 @@ +#include "inference/openvino/backend.hpp" + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include + +#include + +#include "util/sha256.hpp" + +#if !defined(LIGHT_OCR_OPENVINO_DEFAULT_LIBRARY) +#error "LIGHT_OCR_OPENVINO_DEFAULT_LIBRARY must name the build-time OpenVINO C runtime" +#endif + +namespace light_ocr::internal { +namespace { + +namespace fs = std::filesystem; + +constexpr const char* kDevice = "NPU"; +constexpr std::uint64_t kCacheByteLimit = 256ULL * 1024ULL * 1024ULL; +constexpr std::size_t kMaximumDetectionShapes = 32; +constexpr std::size_t kMaximumRecognitionShapes = 20; + +class OpenVinoSetupError final : public std::runtime_error { + public: + OpenVinoSetupError(CreationReason reason, const std::string& message, + std::string detail = {}) + : std::runtime_error(message), reason_(reason), detail_(std::move(detail)) {} + + CreationReason reason() const noexcept { return reason_; } + const std::string& detail() const noexcept { return detail_; } + + private: + CreationReason reason_; + std::string detail_; +}; + +struct Api { + decltype(&::ov_core_create) core_create = nullptr; + decltype(&::ov_core_get_available_devices) core_get_available_devices = nullptr; + decltype(&::ov_available_devices_free) available_devices_free = nullptr; + decltype(&::ov_core_get_property) core_get_property = nullptr; + decltype(&::ov_core_read_model_from_memory_buffer) core_read_model_from_memory_buffer = nullptr; + decltype(&::ov_core_compile_model) core_compile_model = nullptr; + decltype(&::ov_model_free) model_free = nullptr; + decltype(&::ov_model_reshape_single_input) model_reshape_single_input = nullptr; + decltype(&::ov_partial_shape_create_static) partial_shape_create_static = nullptr; + decltype(&::ov_partial_shape_free) partial_shape_free = nullptr; + decltype(&::ov_compiled_model_create_infer_request) compiled_model_create_infer_request = nullptr; + decltype(&::ov_compiled_model_free) compiled_model_free = nullptr; + decltype(&::ov_infer_request_set_input_tensor_by_index) infer_request_set_input_tensor_by_index = nullptr; + decltype(&::ov_infer_request_infer) infer_request_infer = nullptr; + decltype(&::ov_infer_request_get_output_tensor_by_index) infer_request_get_output_tensor_by_index = nullptr; + decltype(&::ov_infer_request_free) infer_request_free = nullptr; + decltype(&::ov_tensor_create_from_host_ptr) tensor_create_from_host_ptr = nullptr; + decltype(&::ov_tensor_get_shape) tensor_get_shape = nullptr; + decltype(&::ov_tensor_data) tensor_data = nullptr; + decltype(&::ov_tensor_free) tensor_free = nullptr; + decltype(&::ov_shape_create) shape_create = nullptr; + decltype(&::ov_shape_free) shape_free = nullptr; + decltype(&::ov_free) free = nullptr; + decltype(&::ov_get_openvino_version) get_openvino_version = nullptr; + decltype(&::ov_version_free) version_free = nullptr; + decltype(&::ov_get_last_err_msg) get_last_err_msg = nullptr; +}; + +// The runtime is loaded once and never unloaded: OpenVINO plugins keep +// process-wide state that is not safe to tear down while other runtimes live. +struct Runtime { + fs::path library; + Api api; + ov_core_t* core = nullptr; + OpenVinoDeviceInfo device; + bool device_available = false; +}; + +struct RuntimeState { + std::mutex mutex; + std::unique_ptr runtime; +}; + +RuntimeState& runtime_state() { + static auto* state = new RuntimeState(); + return *state; +} + +std::string last_error(const Api& api) { + const char* message = api.get_last_err_msg != nullptr ? api.get_last_err_msg() : nullptr; + return message != nullptr ? std::string(message) : std::string(); +} + +fs::path runtime_library_path(const InferenceSessionConfig& config) { + if (config.openvino_runtime_library.empty()) { + std::error_code error; + auto library = fs::canonical(fs::u8path(LIGHT_OCR_OPENVINO_DEFAULT_LIBRARY), error); + if (error) { + throw OpenVinoSetupError(CreationReason::package_corrupt, + "The build-time OpenVINO runtime library is missing", + LIGHT_OCR_OPENVINO_DEFAULT_LIBRARY); + } + return library; + } + const auto library = fs::u8path(config.openvino_runtime_library); + if (!library.is_absolute()) { + throw OpenVinoSetupError(CreationReason::internal_assertion_failed, + "The OpenVINO runtime library path must be absolute"); + } + std::error_code error; + const auto status = fs::symlink_status(library, error); + if (error || !fs::is_regular_file(status) || fs::is_symlink(status)) { + throw OpenVinoSetupError(CreationReason::package_corrupt, + "The OpenVINO runtime library is missing or is not a regular file"); + } + if (config.openvino_runtime_bytes != 0 || !config.openvino_runtime_sha256.empty()) { + const auto actual_bytes = fs::file_size(library, error); + if (error || actual_bytes != config.openvino_runtime_bytes || + config.openvino_runtime_bytes > + static_cast(std::numeric_limits::max())) { + throw OpenVinoSetupError( + CreationReason::artifact_hash_mismatch, + "The OpenVINO runtime library byte count does not match its runtime descriptor"); + } + std::vector contents(static_cast(actual_bytes)); + std::ifstream input(library, std::ios::binary); + if (!input || + !input.read(reinterpret_cast(contents.data()), + static_cast(contents.size())) || + input.peek() != std::ifstream::traits_type::eof()) { + throw OpenVinoSetupError( + CreationReason::artifact_hash_mismatch, + "The OpenVINO runtime library changed while it was being verified"); + } + if (sha256_hex(contents.data(), contents.size()) != config.openvino_runtime_sha256) { + throw OpenVinoSetupError( + CreationReason::artifact_hash_mismatch, + "The OpenVINO runtime library hash does not match its runtime descriptor"); + } + } + return library.lexically_normal(); +} + +template +void bind(void* handle, const char* name, Function* function) { + void* symbol = ::dlsym(handle, name); + if (symbol == nullptr) { + throw OpenVinoSetupError(CreationReason::provider_abi_mismatch, + "The OpenVINO runtime does not export a required C API symbol", + name); + } + *function = reinterpret_cast(symbol); +} + +Api load_api(void* handle) { + Api api; + bind(handle, "ov_core_create", &api.core_create); + bind(handle, "ov_core_get_available_devices", &api.core_get_available_devices); + bind(handle, "ov_available_devices_free", &api.available_devices_free); + bind(handle, "ov_core_get_property", &api.core_get_property); + bind(handle, "ov_core_read_model_from_memory_buffer", &api.core_read_model_from_memory_buffer); + bind(handle, "ov_core_compile_model", &api.core_compile_model); + bind(handle, "ov_model_free", &api.model_free); + bind(handle, "ov_model_reshape_single_input", &api.model_reshape_single_input); + bind(handle, "ov_partial_shape_create_static", &api.partial_shape_create_static); + bind(handle, "ov_partial_shape_free", &api.partial_shape_free); + bind(handle, "ov_compiled_model_create_infer_request", &api.compiled_model_create_infer_request); + bind(handle, "ov_compiled_model_free", &api.compiled_model_free); + bind(handle, "ov_infer_request_set_input_tensor_by_index", + &api.infer_request_set_input_tensor_by_index); + bind(handle, "ov_infer_request_infer", &api.infer_request_infer); + bind(handle, "ov_infer_request_get_output_tensor_by_index", + &api.infer_request_get_output_tensor_by_index); + bind(handle, "ov_infer_request_free", &api.infer_request_free); + bind(handle, "ov_tensor_create_from_host_ptr", &api.tensor_create_from_host_ptr); + bind(handle, "ov_tensor_get_shape", &api.tensor_get_shape); + bind(handle, "ov_tensor_data", &api.tensor_data); + bind(handle, "ov_tensor_free", &api.tensor_free); + bind(handle, "ov_shape_create", &api.shape_create); + bind(handle, "ov_shape_free", &api.shape_free); + bind(handle, "ov_free", &api.free); + bind(handle, "ov_get_openvino_version", &api.get_openvino_version); + bind(handle, "ov_version_free", &api.version_free); + bind(handle, "ov_get_last_err_msg", &api.get_last_err_msg); + return api; +} + +std::string device_property(const Runtime& runtime, const char* key) { + char* value = nullptr; + if (runtime.api.core_get_property(runtime.core, kDevice, key, &value) != OK || + value == nullptr) { + return {}; + } + std::string result(value); + runtime.api.free(value); + return result; +} + +bool npu_listed(const Runtime& runtime) { + ov_available_devices_t devices{}; + if (runtime.api.core_get_available_devices(runtime.core, &devices) != OK) return false; + bool found = false; + for (std::size_t index = 0; index < devices.size; ++index) { + const std::string name = devices.devices[index] != nullptr ? devices.devices[index] : ""; + if (name == kDevice || name.rfind(std::string(kDevice) + ".", 0) == 0) found = true; + } + runtime.api.available_devices_free(&devices); + return found; +} + +// Caller holds runtime_state().mutex. +Runtime& load_runtime(const InferenceSessionConfig& config) { + auto& state = runtime_state(); + const auto library = runtime_library_path(config); + if (state.runtime) { + if (state.runtime->library != library) { + throw OpenVinoSetupError( + CreationReason::provider_abi_mismatch, + "A different OpenVINO runtime is already loaded in this process"); + } + return *state.runtime; + } + void* handle = ::dlopen(library.c_str(), RTLD_NOW | RTLD_LOCAL); + if (handle == nullptr) { + const char* message = ::dlerror(); + throw OpenVinoSetupError(CreationReason::unrecoverable_load_failed, + "Cannot load the OpenVINO runtime", + message != nullptr ? message : ""); + } + auto runtime = std::make_unique(); + runtime->library = library; + runtime->api = load_api(handle); + if (runtime->api.core_create(&runtime->core) != OK || runtime->core == nullptr) { + throw OpenVinoSetupError(CreationReason::unrecoverable_load_failed, + "Cannot create the OpenVINO core", last_error(runtime->api)); + } + ov_version_t version{}; + if (runtime->api.get_openvino_version(&version) == OK) { + runtime->device.runtime_version = + version.buildNumber != nullptr ? version.buildNumber : ""; + runtime->api.version_free(&version); + } + runtime->device_available = npu_listed(*runtime); + if (runtime->device_available) { + runtime->device.full_name = device_property(*runtime, "FULL_DEVICE_NAME"); + runtime->device.architecture = device_property(*runtime, "DEVICE_ARCHITECTURE"); + runtime->device.driver_version = device_property(*runtime, "NPU_DRIVER_VERSION"); + runtime->device.compiler_version = device_property(*runtime, "NPU_COMPILER_VERSION"); + } + state.runtime = std::move(runtime); + return *state.runtime; +} + +Runtime& require_npu(const InferenceSessionConfig& config) { + auto& runtime = load_runtime(config); + if (!runtime.device_available) { + throw OpenVinoSetupError(CreationReason::adapter_unavailable, + "OpenVINO reports no usable Intel NPU on this host"); + } + return runtime; +} + +fs::path default_cache_root() { + if (const char* xdg = std::getenv("XDG_CACHE_HOME"); xdg != nullptr && xdg[0] == '/') { + return fs::path(xdg) / "com.arcships.light-ocr"; + } + if (const char* home = std::getenv("HOME"); home != nullptr && home[0] == '/') { + return fs::path(home) / ".cache" / "com.arcships.light-ocr"; + } + return {}; +} + +class AdvisoryLock { + public: + explicit AdvisoryLock(const fs::path& path) { + descriptor_ = ::open(path.c_str(), O_CREAT | O_RDWR | O_CLOEXEC, 0600); + if (descriptor_ >= 0 && ::flock(descriptor_, LOCK_EX) != 0) { + ::close(descriptor_); + descriptor_ = -1; + } + } + + AdvisoryLock(const AdvisoryLock&) = delete; + AdvisoryLock& operator=(const AdvisoryLock&) = delete; + + ~AdvisoryLock() noexcept { + if (descriptor_ >= 0) { + static_cast(::flock(descriptor_, LOCK_UN)); + static_cast(::close(descriptor_)); + } + } + + bool locked() const noexcept { return descriptor_ >= 0; } + + private: + int descriptor_ = -1; +}; + +// Removes the oldest compiled blobs until the whole OpenVINO cache fits the +// byte limit. Caller holds the cache lock. +void prune_cache(const fs::path& root) noexcept { + try { + struct Entry { + fs::path path; + fs::file_time_type time; + std::uint64_t bytes; + }; + std::vector entries; + std::uint64_t total = 0; + for (const auto& item : fs::recursive_directory_iterator(root)) { + if (!item.is_regular_file() || item.path().filename() == ".lock") continue; + Entry entry{item.path(), item.last_write_time(), item.file_size()}; + total += entry.bytes; + entries.push_back(std::move(entry)); + } + if (total <= kCacheByteLimit) return; + std::sort(entries.begin(), entries.end(), + [](const Entry& left, const Entry& right) { return left.time < right.time; }); + for (const auto& entry : entries) { + if (total <= kCacheByteLimit) break; + std::error_code error; + if (fs::remove(entry.path, error)) total -= entry.bytes; + } + } catch (...) { + // Pruning is best effort; an oversized cache never fails inference. + } +} + +std::vector tensor_shape(const Api& api, const ov_tensor_t* tensor) { + ov_shape_t shape{}; + if (api.tensor_get_shape(tensor, &shape) != OK) return {}; + std::vector result(shape.dims, shape.dims + shape.rank); + api.shape_free(&shape); + return result; +} + +std::string shape_text(const std::vector& shape) { + std::string text; + for (const auto dimension : shape) { + if (!text.empty()) text += "x"; + text += std::to_string(dimension); + } + return text; +} + +} // namespace + +struct OpenVinoSession::State { + struct Compiled { + ov_compiled_model_t* model = nullptr; + ov_infer_request_t* request = nullptr; + std::uint64_t last_use = 0; + }; + + Runtime* runtime = nullptr; + ov_model_t* model = nullptr; + ModelKind kind = ModelKind::detection; + fs::path cache_directory; + fs::path cache_lock; + std::map, Compiled> compiled; + std::size_t maximum_shapes = 0; + std::uint64_t clock = 0; + bool last_compile_cached = false; + + ~State() { + for (auto& entry : compiled) release(entry.second); + if (model != nullptr) runtime->api.model_free(model); + } + + void release(Compiled& entry) const { + if (entry.request != nullptr) runtime->api.infer_request_free(entry.request); + if (entry.model != nullptr) runtime->api.compiled_model_free(entry.model); + entry = Compiled{}; + } + + // Returns the request for `shape`, compiling it on first use. Throws + // std::runtime_error with the OpenVINO message on failure. + ov_infer_request_t* request_for(const std::vector& shape) { + ++clock; + auto found = compiled.find(shape); + if (found != compiled.end()) { + found->second.last_use = clock; + return found->second.request; + } + if (compiled.size() >= maximum_shapes) { + auto oldest = std::min_element( + compiled.begin(), compiled.end(), [](const auto& left, const auto& right) { + return left.second.last_use < right.second.last_use; + }); + release(oldest->second); + compiled.erase(oldest); + } + const auto& api = runtime->api; + ov_partial_shape_t partial{}; + if (api.partial_shape_create_static(static_cast(shape.size()), + shape.data(), &partial) != OK) { + throw std::runtime_error("Cannot describe the OpenVINO input shape: " + last_error(api)); + } + const auto reshaped = api.model_reshape_single_input(model, partial); + api.partial_shape_free(&partial); + if (reshaped != OK) { + throw std::runtime_error("Cannot reshape the OpenVINO model: " + last_error(api)); + } + Compiled entry; + ov_status_e status = GENERAL_ERROR; + { + std::unique_ptr lock; + if (!cache_directory.empty()) lock = std::make_unique(cache_lock); + const bool cached = lock && lock->locked(); + const std::string cache = cached ? cache_directory.string() : std::string(); + if (cached) { + status = api.core_compile_model( + runtime->core, model, kDevice, 10, &entry.model, + "PERFORMANCE_HINT", "LATENCY", "INFERENCE_PRECISION_HINT", "f16", + "NPU_COMPILER_TYPE", "DRIVER", "CACHE_DIR", cache.c_str(), + "CACHE_MODE", "OPTIMIZE_SIZE"); + if (status == OK) prune_cache(cache_directory.parent_path()); + } else { + status = api.core_compile_model( + runtime->core, model, kDevice, 6, &entry.model, + "PERFORMANCE_HINT", "LATENCY", "INFERENCE_PRECISION_HINT", "f16", + "NPU_COMPILER_TYPE", "DRIVER"); + } + last_compile_cached = cached; + } + if (status != OK || entry.model == nullptr) { + throw std::runtime_error("OpenVINO cannot compile shape " + shape_text(shape) + + " for the NPU: " + last_error(api)); + } + if (api.compiled_model_create_infer_request(entry.model, &entry.request) != OK) { + const auto message = last_error(api); + release(entry); + throw std::runtime_error("Cannot create an OpenVINO infer request: " + message); + } + entry.last_use = clock; + return compiled.emplace(shape, entry).first->second.request; + } +}; + +OpenVinoSession::OpenVinoSession(std::unique_ptr state, + SessionExecutionInfo info) + : state_(std::move(state)), execution_info_(std::move(info)) {} + +OpenVinoSession::~OpenVinoSession() noexcept { + try { + std::lock_guard lock(runtime_state().mutex); + state_.reset(); + } catch (...) { + // Destructors release native handles only; failures cannot be reported. + } +} + +Result> OpenVinoSession::create( + const SharedBytes& model, const InferenceSessionConfig& config, ModelKind kind, + const std::vector& probe_shape, + std::optional* creation_reason) noexcept { + using CreateResult = Result>; + auto fail = [&](CreationReason reason, ErrorCode code, std::string message, + std::string detail) { + if (creation_reason != nullptr) *creation_reason = reason; + return CreateResult::failure(Error{code, std::move(message), std::move(detail)}); + }; + try { + if (!model || model->empty() || config.model_id.empty() || + config.model_sha256.size() != 64 || probe_shape.size() != 4) { + return fail(CreationReason::internal_assertion_failed, ErrorCode::internal_error, + "OpenVINO session configuration is incomplete", {}); + } + std::lock_guard lock(runtime_state().mutex); + auto& runtime = require_npu(config); + auto state = std::make_unique(); + state->runtime = &runtime; + state->kind = kind; + state->maximum_shapes = kind == ModelKind::detection ? kMaximumDetectionShapes + : kMaximumRecognitionShapes; + if (runtime.api.core_read_model_from_memory_buffer( + runtime.core, reinterpret_cast(model->data()), model->size(), + nullptr, &state->model) != OK || + state->model == nullptr) { + return fail(CreationReason::unrecoverable_load_failed, + ErrorCode::runtime_initialization_failed, + "OpenVINO cannot read the locked ONNX model", last_error(runtime.api)); + } + + const auto root = config.openvino_cache_directory.empty() + ? default_cache_root() + : fs::u8path(config.openvino_cache_directory); + if (!root.empty()) { + // Every identity that changes compiled blobs gets its own directory, so + // a driver or runtime upgrade never loads a stale blob. + const std::string identity = config.model_sha256 + "\n" + + runtime.device.runtime_version + "\n" + + runtime.device.architecture + "\n" + + runtime.device.driver_version + "\n" + + runtime.device.compiler_version; + const auto key = sha256_hex(reinterpret_cast(identity.data()), + identity.size()) + .substr(0, 32); + const auto base = root / "openvino-v1"; + std::error_code error; + fs::create_directories(base / key, error); + if (!error) { + state->cache_directory = base / key; + state->cache_lock = base / ".lock"; + } + } + + try { + state->request_for(probe_shape); + } catch (const std::exception& exception) { + return fail(CreationReason::model_compute_unsupported, + ErrorCode::unsupported_capability, + "The Intel NPU cannot compile the locked model", exception.what()); + } + + SessionExecutionInfo info; + info.requested_provider = config.requested_provider_override.empty() + ? "openvino" + : config.requested_provider_override; + info.actual_provider_chain = {"OpenVINO:NPU"}; + info.device = "npu:" + runtime.device.full_name; + info.device_family = runtime.device.architecture; + info.operating_system = "linux"; + info.precision = "fp16"; + info.shape_policy = config.shape_policy; + info.model_id = config.model_id; + info.model_sha256 = config.model_sha256; + info.runtime = "OpenVINO"; + info.runtime_version = runtime.device.runtime_version; + info.provider_version = runtime.device.driver_version; + info.model_cache_status = state->last_compile_cached ? "compiled_cache" : "disabled"; + info.qualification_id = config.qualification_id; + info.device_validated = false; + return CreateResult::success(std::unique_ptr( + new OpenVinoSession(std::move(state), std::move(info)))); + } catch (const OpenVinoSetupError& error) { + return fail(error.reason(), + error.reason() == CreationReason::adapter_unavailable + ? ErrorCode::unsupported_capability + : ErrorCode::runtime_initialization_failed, + error.what(), error.detail()); + } catch (const std::exception& exception) { + return fail(CreationReason::unrecoverable_load_failed, + ErrorCode::runtime_initialization_failed, + "Cannot create the OpenVINO session", exception.what()); + } +} + +Result OpenVinoSession::run( + const std::vector& values, const std::vector& shape) noexcept { + try { + std::size_t expected = shape.empty() ? 0 : 1; + for (const auto dimension : shape) { + if (dimension <= 0) { + return Result::failure( + Error{ErrorCode::inference_failed, "OpenVINO input shape is invalid", {}}); + } + expected *= static_cast(dimension); + } + if (shape.size() != 4 || expected != values.size()) { + return Result::failure( + Error{ErrorCode::inference_failed, + "OpenVINO input does not match its shape", {}}); + } + std::lock_guard lock(runtime_state().mutex); + const auto& api = state_->runtime->api; + ov_infer_request_t* request = state_->request_for(shape); + + ov_shape_t input_shape{}; + if (api.shape_create(static_cast(shape.size()), shape.data(), + &input_shape) != OK) { + return Result::failure( + Error{ErrorCode::inference_failed, "Cannot describe the OpenVINO input", + last_error(api)}); + } + ov_tensor_t* input = nullptr; + const auto created = api.tensor_create_from_host_ptr( + F32, input_shape, const_cast(values.data()), &input); + api.shape_free(&input_shape); + if (created != OK || input == nullptr) { + return Result::failure( + Error{ErrorCode::inference_failed, "Cannot wrap the OpenVINO input", + last_error(api)}); + } + const auto set = api.infer_request_set_input_tensor_by_index(request, 0, input); + const auto inferred = set == OK ? api.infer_request_infer(request) : set; + api.tensor_free(input); + if (inferred != OK) { + return Result::failure( + Error{ErrorCode::inference_failed, "OpenVINO NPU inference failed", + last_error(api)}); + } + + ov_tensor_t* output = nullptr; + if (api.infer_request_get_output_tensor_by_index(request, 0, &output) != OK || + output == nullptr) { + return Result::failure( + Error{ErrorCode::inference_failed, "Cannot read the OpenVINO output", + last_error(api)}); + } + auto output_shape = tensor_shape(api, output); + std::size_t size = output_shape.empty() ? 0 : 1; + for (const auto dimension : output_shape) size *= static_cast(dimension); + void* data = nullptr; + if (output_shape.empty() || api.tensor_data(output, &data) != OK || data == nullptr) { + api.tensor_free(output); + return Result::failure( + Error{ErrorCode::inference_failed, "OpenVINO output is empty", last_error(api)}); + } + // The request reuses its output buffer on the next call, so results are + // copied into storage owned by the returned tensor. + auto storage = std::make_shared>( + static_cast(data), static_cast(data) + size); + api.tensor_free(output); + const float* pointer = storage->data(); + return Result::success( + TensorOutput(std::move(storage), pointer, std::move(output_shape), size)); + } catch (const std::exception& exception) { + return Result::failure( + Error{ErrorCode::inference_failed, "OpenVINO NPU inference failed", + exception.what()}); + } catch (...) { + return Result::failure( + Error{ErrorCode::inference_failed, "OpenVINO NPU inference failed", {}}); + } +} + +} // namespace light_ocr::internal diff --git a/src/inference/openvino/backend.hpp b/src/inference/openvino/backend.hpp new file mode 100644 index 0000000..4ae5ef9 --- /dev/null +++ b/src/inference/openvino/backend.hpp @@ -0,0 +1,59 @@ +#pragma once + +#include +#include +#include +#include +#include + +#include "inference/backend.hpp" +#include "light_ocr/core.hpp" + +namespace light_ocr::internal { + +// Recognition widths are rounded up to these buckets so the NPU compiles a +// bounded set of static shapes. The list matches the Apple runtime buckets, +// which the Intel NPU spike qualified without trailing-character loss. +inline const std::vector& openvino_recognition_width_buckets() { + static const std::vector buckets = { + 320, 384, 480, 544, 576, 608, 704, 736, 832, 960, + 1056, 1184, 1248, 1376, 1600, 1984, 2240, 2560, 2880, 3200}; + return buckets; +} + +struct OpenVinoDeviceInfo { + std::string full_name; + std::string architecture; + std::string driver_version; + std::string compiler_version; + std::string runtime_version; +}; + +class OpenVinoSession final : public InferenceSession { + public: + ~OpenVinoSession() noexcept override; + + // `probe_shape` is compiled during creation so an unsupported model fails + // before the backend is selected; every other shape compiles on first use. + static Result> create( + const SharedBytes& model, const InferenceSessionConfig& config, + ModelKind kind, const std::vector& probe_shape, + std::optional* creation_reason = nullptr) noexcept; + + Result run(const std::vector& values, + const std::vector& shape) noexcept override; + + const SessionExecutionInfo& execution_info() const noexcept override { + return execution_info_; + } + + struct State; + + private: + OpenVinoSession(std::unique_ptr state, SessionExecutionInfo info); + + std::unique_ptr state_; + SessionExecutionInfo execution_info_; +}; + +} // namespace light_ocr::internal diff --git a/src/preprocess/tensor.cpp b/src/preprocess/tensor.cpp index a9ba76e..712ae4b 100644 --- a/src/preprocess/tensor.cpp +++ b/src/preprocess/tensor.cpp @@ -43,7 +43,8 @@ Result make_recognition_sample( std::size_t input_index, std::uint32_t crop_width, std::uint32_t crop_height, const RecognitionConfig& config, const ResourceLimits& limits, std::uint32_t tensor_width_multiple, - const std::vector& tensor_width_buckets) { + const std::vector& tensor_width_buckets, + bool natural_content_width) { if (crop_width == 0 || crop_height == 0) { return failure(ErrorCode::postprocess_failed, "Recognition crop is empty"); @@ -54,6 +55,9 @@ Result make_recognition_sample( static_cast(config.height) * std::max(base_ratio, ratio)); tensor_width = std::max(config.minimum_tensor_width, std::min(config.maximum_tensor_width, tensor_width)); + // Rounding and buckets only add right padding when content keeps the width + // it would have in an unpadded tensor; otherwise it grows to the ceiling. + const auto natural_width = tensor_width; if (tensor_width_multiple == 0 || tensor_width_multiple > config.maximum_tensor_width || config.maximum_tensor_width % tensor_width_multiple != 0) { @@ -91,7 +95,8 @@ Result make_recognition_sample( tensor_width = *bucket; } const auto content_width = std::min( - tensor_width, static_cast(std::ceil(config.height * ratio))); + natural_content_width ? natural_width : tensor_width, + static_cast(std::ceil(config.height * ratio))); if (tensor_width > limits.max_recognition_width || content_width == 0) { return failure(ErrorCode::resource_limit_exceeded, "Recognition tensor width exceeds limits"); @@ -240,7 +245,8 @@ Result> plan_recognition_batches( const std::vector& boxes, const GeometryConfig& geometry, const RecognitionConfig& config, std::uint32_t batch_size, const ResourceLimits& limits, std::uint32_t tensor_width_multiple, - const std::vector& tensor_width_buckets) { + const std::vector& tensor_width_buckets, + bool natural_content_width) { try { if (batch_size == 0 || batch_size > config.maximum_batch_size || batch_size > limits.max_recognition_batch_size) { @@ -259,7 +265,7 @@ Result> plan_recognition_batches( const auto shape = std::move(shape_result).value(); auto sample_result = make_recognition_sample( index, shape.output_width(), shape.output_height(), config, limits, - tensor_width_multiple, tensor_width_buckets); + tensor_width_multiple, tensor_width_buckets, natural_content_width); if (!sample_result) { return Result>::failure( sample_result.error()); @@ -294,7 +300,8 @@ Result make_recognition_batch( const std::vector& crops, const RecognitionBatchPlan& plan, const RecognitionConfig& config, const ResourceLimits& limits, std::uint32_t tensor_width_multiple, - const std::vector& tensor_width_buckets) { + const std::vector& tensor_width_buckets, + bool natural_content_width) { try { const auto count = plan.samples.size(); if (count == 0 || count != crops.size() || @@ -317,7 +324,7 @@ Result make_recognition_batch( plan.samples[index].input_index, static_cast(crop.cols), static_cast(crop.rows), config, limits, - tensor_width_multiple, tensor_width_buckets); + tensor_width_multiple, tensor_width_buckets, natural_content_width); if (!actual_result) { return Result::failure(actual_result.error()); } diff --git a/src/preprocess/tensor.hpp b/src/preprocess/tensor.hpp index 294c30f..f207bd5 100644 --- a/src/preprocess/tensor.hpp +++ b/src/preprocess/tensor.hpp @@ -47,12 +47,14 @@ Result> plan_recognition_batches( const std::vector& boxes, const GeometryConfig& geometry, const RecognitionConfig& config, std::uint32_t batch_size, const ResourceLimits& limits, std::uint32_t tensor_width_multiple = 1, - const std::vector& tensor_width_buckets = {}); + const std::vector& tensor_width_buckets = {}, + bool natural_content_width = false); Result make_recognition_batch( const std::vector& crops, const RecognitionBatchPlan& plan, const RecognitionConfig& config, const ResourceLimits& limits, std::uint32_t tensor_width_multiple = 1, - const std::vector& tensor_width_buckets = {}); + const std::vector& tensor_width_buckets = {}, + bool natural_content_width = false); } // namespace light_ocr::internal diff --git a/tests/integration/main.cpp b/tests/integration/main.cpp index 45e43d7..b17b4e8 100644 --- a/tests/integration/main.cpp +++ b/tests/integration/main.cpp @@ -114,14 +114,19 @@ int main() { return 1; } const auto builtin_policy = light_ocr::internal::builtin_runtime_policy(); - const bool webgpu_runtime = - std::find(builtin_policy.available_providers.begin(), - builtin_policy.available_providers.end(), "webgpu") != - builtin_policy.available_providers.end(); + std::vector accelerators; + for (const auto* provider : {"webgpu", "openvino"}) { + if (std::find(builtin_policy.available_providers.begin(), + builtin_policy.available_providers.end(), provider) != + builtin_policy.available_providers.end()) { + accelerators.push_back(provider); + } + } + const bool webgpu_runtime = !accelerators.empty(); light_ocr::EngineOptions integration_options; if (webgpu_runtime) { // The ordinary integration suite is hardware-independent. Direct C++ - // Auto is exercised by the real-device WebGPU qualification runner. + // Auto is exercised by the real-device qualification runners. integration_options.execution.provider = light_ocr::ExecutionProvider::cpu; } auto engine = light_ocr::Engine::create(std::move(bundle).value(), @@ -145,15 +150,17 @@ int main() { execution.provider_capabilities[0].package_included && execution.provider_capabilities[0].device_available && execution.provider_capabilities[0].device_validated; - const bool provider_capabilities_valid = + bool provider_capabilities_valid = cpu_capability_valid && - (webgpu_runtime - ? execution.provider_capabilities.size() == 2 && - execution.provider_capabilities[1].provider == "webgpu" && - execution.provider_capabilities[1].package_included && - !execution.provider_capabilities[1].device_available && - !execution.provider_capabilities[1].device_validated - : execution.provider_capabilities.size() == 1); + execution.provider_capabilities.size() == 1 + accelerators.size(); + for (std::size_t index = 0; + provider_capabilities_valid && index < accelerators.size(); ++index) { + const auto& capability = execution.provider_capabilities[index + 1]; + provider_capabilities_valid = capability.provider == accelerators[index] && + capability.package_included && + !capability.device_available && + !capability.device_validated; + } const auto expected_requested = webgpu_runtime ? light_ocr::ExecutionProvider::cpu : light_ocr::ExecutionProvider::automatic; diff --git a/tests/unit/test_image.cpp b/tests/unit/test_image.cpp index 4d963d4..f1d9d64 100644 --- a/tests/unit/test_image.cpp +++ b/tests/unit/test_image.cpp @@ -233,6 +233,40 @@ LIGHT_OCR_TEST(recognition_batches_apply_and_validate_runtime_width_buckets) { EXPECT_EQ(invalid.error().code, ErrorCode::invalid_argument); } +LIGHT_OCR_TEST(recognition_buckets_can_keep_the_unpadded_content_width) { + const auto config = recognition_config(); + internal::GeometryConfig geometry; + geometry.tall_line_ratio = 1.5f; + // 48 * 400 / 47 = 408.5: an unpadded tensor truncates to 408, while the + // padded default grows the content to the 409-pixel ceiling. + const std::vector boxes{rectangle(400, 47)}; + const std::vector buckets{320, 480, 3200}; + auto padded = internal::plan_recognition_batches( + boxes, geometry, config, 1, ResourceLimits{}, 32, buckets); + auto natural = internal::plan_recognition_batches( + boxes, geometry, config, 1, ResourceLimits{}, 32, buckets, true); + auto unpadded = internal::plan_recognition_batches( + boxes, geometry, config, 1, ResourceLimits{}); + EXPECT_TRUE(padded); + EXPECT_TRUE(natural); + EXPECT_TRUE(unpadded); + EXPECT_EQ(padded.value()[0].samples[0].tensor_width, 480u); + EXPECT_EQ(natural.value()[0].samples[0].tensor_width, 480u); + EXPECT_EQ(padded.value()[0].samples[0].content_width, 409u); + EXPECT_EQ(natural.value()[0].samples[0].content_width, + unpadded.value()[0].samples[0].content_width); + EXPECT_EQ(natural.value()[0].samples[0].content_width, 408u); + + std::vector crops{cv::Mat(47, 400, CV_8UC3, cv::Scalar(0, 0, 0))}; + auto batch = internal::make_recognition_batch( + crops, natural.value()[0], config, ResourceLimits{}, 32, buckets, true); + EXPECT_TRUE(batch); + EXPECT_EQ(batch.value().shape[3], 480); + auto mismatched = internal::make_recognition_batch( + crops, natural.value()[0], config, ResourceLimits{}, 32, buckets); + EXPECT_FALSE(mismatched); +} + LIGHT_OCR_TEST(recognition_batches_reject_invalid_batch_and_memory_limit) { std::vector crops{cv::Mat(48, 320, CV_8UC3, cv::Scalar(0, 0, 0))}; const auto config = recognition_config(); diff --git a/tests/unit/test_model_bundle.cpp b/tests/unit/test_model_bundle.cpp index ef006bd..62a1e45 100644 --- a/tests/unit/test_model_bundle.cpp +++ b/tests/unit/test_model_bundle.cpp @@ -452,6 +452,124 @@ LIGHT_OCR_TEST(webgpu_fp16_is_not_a_public_execution_profile) { EXPECT_FALSE(engine.error().creation_trace.has_value()); } +namespace { + +internal::RuntimePolicy openvino_test_policy() { + internal::RuntimePolicy policy; + policy.id = "test-openvino-v1"; + policy.version = 1; + policy.qualification_only = true; + policy.released = false; + policy.ordered_candidates = {"openvino", "cpu"}; + policy.available_providers = {"openvino", "cpu"}; + policy.provider_qualification_ids = {"test-openvino-v1", "test-cpu-v1"}; + return policy; +} + +} // namespace + +LIGHT_OCR_TEST(openvino_accepts_only_fp16_precision) { + auto bundle = ModelBundle::create(valid_bundle_files()); + EXPECT_TRUE(bundle); + EngineOptions options; + options.execution.provider = ExecutionProvider::openvino; + options.execution.precision = Precision::fp32; + auto engine = internal::EngineFactory::create( + std::move(bundle).value(), options, openvino_test_policy()); + EXPECT_FALSE(engine); + EXPECT_EQ(engine.error().code, ErrorCode::invalid_argument); + EXPECT_FALSE(engine.error().creation_trace.has_value()); +} + +LIGHT_OCR_TEST(explicit_openvino_requires_a_bundled_provider) { + auto bundle = ModelBundle::create(valid_bundle_files()); + EXPECT_TRUE(bundle); + EngineOptions options; + options.execution.provider = ExecutionProvider::openvino; + internal::RuntimePolicy policy; + policy.id = "test-cpu-v1"; + policy.version = 1; + policy.ordered_candidates = {"cpu"}; + policy.available_providers = {"cpu"}; + auto engine = internal::EngineFactory::create( + std::move(bundle).value(), options, std::move(policy)); + EXPECT_FALSE(engine); + EXPECT_EQ(engine.error().code, ErrorCode::unsupported_capability); + EXPECT_EQ(engine.error().detail, std::string("openvino")); +} + +LIGHT_OCR_TEST(openvino_rejects_batched_recognition_before_loading_the_runtime) { + auto bundle = ModelBundle::create(valid_bundle_files()); + EXPECT_TRUE(bundle); + EngineOptions options; + options.execution.provider = ExecutionProvider::openvino; + options.recognition_batch_size = 2; + auto engine = internal::EngineFactory::create( + std::move(bundle).value(), options, openvino_test_policy()); + EXPECT_FALSE(engine); + EXPECT_TRUE(engine.error().creation_trace.has_value()); + const auto& attempts = engine.error().creation_trace->attempts; + EXPECT_EQ(attempts.size(), 1u); + EXPECT_EQ(attempts[0].provider, std::string("openvino")); + EXPECT_EQ(attempts[0].status, CreationAttemptStatus::fatal); + EXPECT_EQ(*attempts[0].creation_reason, CreationReason::model_compute_unsupported); +} + +LIGHT_OCR_TEST(openvino_cpu_detector_route_rejects_strict_partition) { + auto bundle = ModelBundle::create(valid_bundle_files()); + EXPECT_TRUE(bundle); + EngineOptions options; + options.execution.provider = ExecutionProvider::openvino; + options.execution.cpu_partition = CpuPartition::forbid; + auto policy = openvino_test_policy(); + policy.openvino_detector_route = "cpu"; + auto engine = internal::EngineFactory::create( + std::move(bundle).value(), options, std::move(policy)); + EXPECT_FALSE(engine); + EXPECT_EQ(engine.error().code, ErrorCode::unsupported_capability); + EXPECT_EQ(*engine.error().creation_trace->attempts[0].creation_reason, + CreationReason::model_compute_unsupported); +} + +LIGHT_OCR_TEST(runtime_policy_rejects_incomplete_openvino_artifacts) { + const auto create = [](internal::RuntimePolicy policy) { + auto bundle = ModelBundle::create(valid_bundle_files()); + EXPECT_TRUE(bundle); + return internal::EngineFactory::create(std::move(bundle).value(), EngineOptions{}, + std::move(policy)); + }; + auto missing_hash = openvino_test_policy(); + missing_hash.openvino_runtime_library = "/opt/light-ocr/openvino/libopenvino_c.so"; + missing_hash.openvino_runtime_bytes = 1; + auto unlisted = openvino_test_policy(); + unlisted.available_providers = {"cpu"}; + unlisted.ordered_candidates = {"cpu"}; + unlisted.provider_qualification_ids = {"test-cpu-v1"}; + unlisted.openvino_runtime_library = "/opt/light-ocr/openvino/libopenvino_c.so"; + auto unknown_route = openvino_test_policy(); + unknown_route.openvino_detector_route = "gpu"; + for (auto policy : {missing_hash, unlisted, unknown_route}) { + auto engine = create(std::move(policy)); + EXPECT_FALSE(engine); + EXPECT_EQ(engine.error().code, ErrorCode::internal_error); + EXPECT_EQ(engine.error().message, std::string("The package runtime policy is invalid")); + } +} + +#if !defined(LIGHT_OCR_HAS_OPENVINO) +LIGHT_OCR_TEST(openvino_policy_without_core_support_is_an_abi_mismatch) { + auto bundle = ModelBundle::create(valid_bundle_files()); + EXPECT_TRUE(bundle); + EngineOptions options; + options.execution.provider = ExecutionProvider::openvino; + auto engine = internal::EngineFactory::create( + std::move(bundle).value(), options, openvino_test_policy()); + EXPECT_FALSE(engine); + EXPECT_EQ(*engine.error().creation_trace->attempts[0].creation_reason, + CreationReason::provider_abi_mismatch); +} +#endif + LIGHT_OCR_TEST(model_bundle_rejects_schema_1_1_without_apple_provider) { auto files = valid_bundle_files(true); for (auto& file : files) { diff --git a/tests/unit/test_selection.cpp b/tests/unit/test_selection.cpp index a91a584..b78a3d6 100644 --- a/tests/unit/test_selection.cpp +++ b/tests/unit/test_selection.cpp @@ -7,6 +7,7 @@ #include #include "core/engine_factory.hpp" +#include "inference/openvino/backend.hpp" #include "inference/selection.hpp" namespace light_ocr::test { @@ -125,10 +126,41 @@ LIGHT_OCR_TEST(auto_final_candidate_failure_is_fatal) { ErrorCode::runtime_initialization_failed); } +LIGHT_OCR_TEST(builtin_openvino_policy_prefers_the_npu_and_stays_qualification_only) { + const auto policy = internal::builtin_runtime_policy(); + const bool openvino = + std::find(policy.available_providers.begin(), policy.available_providers.end(), + "openvino") != policy.available_providers.end(); +#if defined(LIGHT_OCR_HAS_OPENVINO) + EXPECT_TRUE(openvino); + EXPECT_EQ(policy.id, std::string("builtin-openvino-v1")); + EXPECT_EQ(policy.ordered_candidates.front(), std::string("openvino")); + EXPECT_EQ(policy.ordered_candidates.back(), std::string("cpu")); + EXPECT_TRUE(policy.qualification_only); + EXPECT_TRUE(!policy.released); +#else + EXPECT_FALSE(openvino); +#endif +} + +LIGHT_OCR_TEST(openvino_recognition_buckets_match_the_locked_width_contract) { + const auto& buckets = internal::openvino_recognition_width_buckets(); + EXPECT_EQ(buckets.size(), 20u); + EXPECT_EQ(buckets.front(), 320u); + EXPECT_EQ(buckets.back(), 3200u); + EXPECT_TRUE(std::is_sorted(buckets.begin(), buckets.end())); + EXPECT_TRUE(std::adjacent_find(buckets.begin(), buckets.end()) == buckets.end()); + EXPECT_TRUE(std::all_of(buckets.begin(), buckets.end(), + [](std::uint32_t width) { return width % 32 == 0; })); +} + LIGHT_OCR_TEST(builtin_webgpu_policy_places_webgpu_before_cpu_when_bundled) { const auto policy = internal::builtin_runtime_policy(); const auto webgpu = std::find(policy.available_providers.begin(), policy.available_providers.end(), "webgpu"); +#if defined(LIGHT_OCR_HAS_OPENVINO) + return; // Covered by the OpenVINO policy test above. +#endif if (webgpu == policy.available_providers.end()) { EXPECT_EQ(policy.id, std::string("builtin-cpu-v1")); EXPECT_EQ(policy.ordered_candidates, std::vector{"cpu"}); diff --git a/tools/benchmark/main.cpp b/tools/benchmark/main.cpp index 8036654..ae41012 100644 --- a/tools/benchmark/main.cpp +++ b/tools/benchmark/main.cpp @@ -64,6 +64,7 @@ const char* provider_name(light_ocr::ExecutionProvider provider) { case light_ocr::ExecutionProvider::cpu: return "cpu"; case light_ocr::ExecutionProvider::apple: return "apple"; case light_ocr::ExecutionProvider::webgpu: return "webgpu"; + case light_ocr::ExecutionProvider::openvino: return "openvino"; } return "auto"; } diff --git a/tools/common/arguments.hpp b/tools/common/arguments.hpp index da2da3e..6782014 100644 --- a/tools/common/arguments.hpp +++ b/tools/common/arguments.hpp @@ -74,6 +74,13 @@ inline EngineOptions engine_options_for_profile(const std::string& profile) { options.execution.precision = Precision::fp16; options.detection.strategy = DetectionStrategy::bounded; options.recognition_batch_size = 1; + } else if (profile == "openvino_allow" || profile == "openvino_strict") { + options.execution.provider = ExecutionProvider::openvino; + options.execution.cpu_partition = + profile == "openvino_strict" ? CpuPartition::forbid + : CpuPartition::allow; + options.detection.strategy = DetectionStrategy::bounded; + options.recognition_batch_size = 1; } else if (profile == "webgpu_allow" || profile == "webgpu_strict") { options.execution.provider = ExecutionProvider::webgpu; options.execution.cpu_partition = @@ -132,11 +139,13 @@ inline Arguments parse_arguments(int argc, char** argv, bool benchmark) { result.profile != "apple_strict" && result.profile != "apple_cpu_fallback" && result.profile != "webgpu_allow" && - result.profile != "webgpu_strict") { + result.profile != "webgpu_strict" && + result.profile != "openvino_allow" && + result.profile != "openvino_strict") { throw std::runtime_error( "profile must be upstream_exact, cpu_fast, bounded_default, runtime_default, " "tiled_v1, apple_interactive, apple_strict, apple_cpu_fallback, " - "webgpu_allow, or webgpu_strict"); + "webgpu_allow, webgpu_strict, openvino_allow, or openvino_strict"); } if (result.diagnostics_mode != "on" && result.diagnostics_mode != "off") { throw std::runtime_error("diagnostics-mode must be on or off");