Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
19 changes: 5 additions & 14 deletions .github/workflows/ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -206,9 +206,6 @@ jobs:
path: models/qwen2.5-0.5b-instruct-q4_k_m.gguf
key: qwen2.5-0.5b-q4-k-m-74a4da8c9fdb

- name: Fetch and Verify Pinned GGUF
run: ./scripts/fetch_real_test_models.sh --gguf-only

- name: Run Real Model and Public Profile
run: |
ccache -z
Expand Down Expand Up @@ -273,20 +270,14 @@ jobs:
ccache-${{ runner.os }}-${{ github.ref }}-
ccache-${{ runner.os }}-

- name: Cache Pinned Whisper Models
- name: Cache Pinned Whisper Model
uses: actions/cache@v4
with:
path: |
models/ggml-base.bin
models/qwen2.5-0.5b-instruct-q4_k_m.gguf
models/ggml-tiny-q5_1.bin
key: whisper-models-${{ hashFiles('models/asset_manifest.json') }}
path: models/ggml-base.bin
key: whisper-e2e-${{ hashFiles('models/asset_manifest.json') }}
restore-keys: |
whisper-models-
whisper-ggml-base-60ed5bc3dd14

- name: Fetch and Verify Pinned Whisper Model
run: ./scripts/fetch_real_test_models.sh --whisper
whisper-e2e-
whisper-ggml-base-

- name: Run Real Whisper E2E and Public Profile
run: |
Expand Down
2 changes: 1 addition & 1 deletion doc/VERIFIABLE_SELECTION.md
Original file line number Diff line number Diff line change
Expand Up @@ -39,7 +39,7 @@ Studio 的“另存为可运行方案”和“运行草稿”共用配置生成

[models/asset_manifest.json](../models/asset_manifest.json) 包含 11 个权重/sidecar 条目的 SHA-256 与 8 个现有资产组合。权重、tokenizer、Kite 运行配置与视觉 projector 均纳入检查。

- 下载命令为 `./scripts/fetch_real_test_models.sh --all`、`--kite`、`--whisper` 或 `--gguf-only`,精确 URL 和 SHA 统一从清单读取。
- 下载命令为 `./scripts/fetch_real_test_models.sh --all`、`--kite`、`--whisper`、`--gguf-only` 或仅供 Whisper 真实模型测试使用的 `--whisper-e2e`,精确 URL 和 SHA 统一从清单读取。
- 清单中的 Model/Backend 配置是可选择的起点;兼容性仍由当前执行文件的 Catalog 校验。
- 清单路径相对资产目录;Pipeline 的 `models[].model_path` 相对宿主部署根;tokenizer、运行配置等 sidecar 路径相对实际模型所在目录解析。
- 变更 tokenizer/运行配置路径后不会借用旧组合的校验结论,而会变为 `unregistered`。接入新资产时,补充清单中的 `artifacts`、`selections.paths` 与完整 `files`,再实际校验。
Expand Down
6 changes: 3 additions & 3 deletions models/asset_manifest.json
Original file line number Diff line number Diff line change
Expand Up @@ -7,8 +7,7 @@
"download_groups": [
"all",
"gguf-only",
"kite",
"whisper"
"kite"
]
},
"bge_base_zh_v1.5.onnx": {
Expand Down Expand Up @@ -62,7 +61,8 @@
"url": "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-base.bin",
"download_groups": [
"all",
"whisper"
"whisper",
"whisper-e2e"
]
},
"ggml-tiny-q5_1.bin": {
Expand Down
5 changes: 3 additions & 2 deletions scripts/fetch_real_test_models.sh
Original file line number Diff line number Diff line change
Expand Up @@ -8,7 +8,7 @@ MODEL_DIR="${PROJECT_ROOT}/models"
MODE="all"

if [[ $# -gt 1 ]]; then
echo "Usage: $0 [--all | --gguf-only | --kite | --whisper]"
echo "Usage: $0 [--all | --gguf-only | --kite | --whisper | --whisper-e2e]"
exit 2
fi
if [[ $# -eq 1 ]]; then
Expand All @@ -17,8 +17,9 @@ if [[ $# -eq 1 ]]; then
--gguf-only) MODE="gguf-only" ;;
--kite) MODE="kite" ;;
--whisper) MODE="whisper" ;;
--whisper-e2e) MODE="whisper-e2e" ;;
*)
echo "Usage: $0 [--all | --gguf-only | --kite | --whisper]"
echo "Usage: $0 [--all | --gguf-only | --kite | --whisper | --whisper-e2e]"
exit 2
;;
esac
Expand Down
4 changes: 2 additions & 2 deletions scripts/generate_acceptance_evidence.sh
Original file line number Diff line number Diff line change
Expand Up @@ -53,7 +53,7 @@ import json
import os

evidence = {
"schema_version": 2,
"schema_version": 3,
"scope": os.environ["EVIDENCE_SCOPE"],
"generated_at_utc": datetime.datetime.now(
datetime.timezone.utc).isoformat(),
Expand All @@ -65,7 +65,7 @@ evidence = {
"worktree_clean": os.environ["WORKTREE_CLEAN"] == "true",
"gates": {
"canonical_run_all_tests": os.environ["CANONICAL_GATE"],
"full_address_undefined_sanitizer": os.environ["SANITIZER_GATE"],
"ci_runtime_address_undefined_sanitizer": os.environ["SANITIZER_GATE"],
"real_c_abi_and_public_profile": os.environ["REAL_GATE"],
},
}
Expand Down
2 changes: 1 addition & 1 deletion scripts/run_real_model_e2e.sh
Original file line number Diff line number Diff line change
Expand Up @@ -39,7 +39,7 @@ echo "=================================================================="

# 1. 拉取或校验固定提交的真实模型权重。
if [[ "${MODE}" == "whisper" ]]; then
"${PROJECT_ROOT}/scripts/fetch_real_test_models.sh" --whisper
"${PROJECT_ROOT}/scripts/fetch_real_test_models.sh" --whisper-e2e
elif [[ "${MODE}" == "all" ]]; then
"${PROJECT_ROOT}/scripts/fetch_real_test_models.sh" --all
else
Expand Down
19 changes: 16 additions & 3 deletions tests/contract/architecture/test_quality_gate_contract.py
Original file line number Diff line number Diff line change
Expand Up @@ -252,7 +252,8 @@ def check_real_model_contract(root, env, log):
configurations.append((configure, preset, False, True))
assert "-DENABLE_REAL_MODEL_TESTS=ON" in configure
commands = [row["command"] for row in records]
assert commands[0] == ["fetch_real_test_models.sh", "--" + mode]
fetch_mode = "--whisper-e2e" if mode == "whisper" else "--" + mode
assert commands[0] == ["fetch_real_test_models.sh", fetch_mode]
for row in records:
if row["command"][0] in ("test_real_models_e2e", "alg_demo"):
assert row["executable"] == str(build / row["command"][0]), row
Expand Down Expand Up @@ -344,16 +345,28 @@ def main():
workflow = (ROOT / ".github/workflows/ci.yml").read_text()
assert "WHISPER_GATE_RESULT: ${{ needs.whisper-asr.result }}" in workflow
assert "KITELLM_GATE_RESULT: ${{ needs.kite-llm.result }}" in workflow
assert "run: ./scripts/fetch_real_test_models.sh --gguf-only" not in workflow
assert "run: ./scripts/fetch_real_test_models.sh --whisper" not in workflow
manifest = json.loads((ROOT / "models/asset_manifest.json").read_text())
artifact_groups = {
group: {name for name, artifact in manifest["artifacts"].items()
if group in artifact.get("download_groups", [])}
for group in ("whisper", "whisper-e2e")
}
assert artifact_groups["whisper-e2e"] == {"ggml-base.bin"}
assert artifact_groups["whisper"] == {"ggml-base.bin", "ggml-tiny-q5_1.bin"}
evidence = root / "evidence.json"
for state in ("success", "failure", "skipped", "cancelled"):
evidence_env = {**os.environ, "WHISPER_GATE_RESULT": state,
"KITELLM_GATE_RESULT": "skipped"}
result = run([str(root / "scripts/generate_acceptance_evidence.sh"), str(evidence),
"success", "failure", "cancelled"], env=evidence_env)
assert result.returncode == 0, result.stdout + result.stderr
gates = json.loads(evidence.read_text())["gates"]
report = json.loads(evidence.read_text())
assert report["schema_version"] == 3
gates = report["gates"]
assert gates == {"canonical_run_all_tests": "success",
"full_address_undefined_sanitizer": "failure",
"ci_runtime_address_undefined_sanitizer": "failure",
"real_c_abi_and_public_profile": "cancelled",
"whisper_asr_backend_and_real_profile": state,
"kitellm_private_release_and_real_gguf": "skipped"}
Expand Down
12 changes: 9 additions & 3 deletions tests/e2e/real_models/test_real_models_e2e.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -23,9 +23,6 @@ class RealModelE2ETest : public ::testing::Test {
std::filesystem::path(LLM_EDGEFLOW_PROJECT_SOURCE_DIR));
model_root_ = project_root_ / "models";
model_path_ = model_root_ / "qwen2.5-0.5b-instruct-q4_k_m.gguf";
ASSERT_TRUE(std::filesystem::is_regular_file(model_path_))
<< "ENABLE_REAL_MODEL_TESTS requires pinned artifacts; run "
"./scripts/fetch_real_test_models.sh --gguf-only";
}

std::filesystem::path project_root_;
Expand All @@ -49,6 +46,9 @@ class RealModelE2ETest : public ::testing::Test {

// 1. 真实 Qwen GGUF 物理前向与自回归 Token 生成测试
TEST_F(RealModelE2ETest, RealQwenGgufTextGeneration) {
ASSERT_TRUE(std::filesystem::is_regular_file(model_path_))
<< "ENABLE_REAL_MODEL_TESTS requires pinned artifacts; run "
"./scripts/fetch_real_test_models.sh --gguf-only";
auto model = CreateModel();
ASSERT_NE(model, nullptr);

Expand Down Expand Up @@ -80,6 +80,9 @@ TEST_F(RealModelE2ETest, RealQwenGgufTextGeneration) {

// 2. 真实 Qwen 模型在 FixedBatchExecutor 定长对齐批推理压测
TEST_F(RealModelE2ETest, RealQwenBatchExecutionWithPadding) {
ASSERT_TRUE(std::filesystem::is_regular_file(model_path_))
<< "ENABLE_REAL_MODEL_TESTS requires pinned artifacts; run "
"./scripts/fetch_real_test_models.sh --gguf-only";
auto model = CreateModel();
ASSERT_NE(model, nullptr);

Expand Down Expand Up @@ -110,6 +113,9 @@ TEST_F(RealModelE2ETest, RealQwenBatchExecutionWithPadding) {

// 3. 真实模型接入 Operator 全链路端到端验证
TEST_F(RealModelE2ETest, RealModelOperatorEndToEnd) {
ASSERT_TRUE(std::filesystem::is_regular_file(model_path_))
<< "ENABLE_REAL_MODEL_TESTS requires pinned artifacts; run "
"./scripts/fetch_real_test_models.sh --gguf-only";
std::vector<std::string> sentences = {
"李雷在微软北京研发中心负责AI大模型芯片开发。"};
std::ifstream corpus(project_root_ / "data/corpus_entity_extract.txt");
Expand Down
Loading