From 5c4a8d7dcded8928a5dff28eb83b5d110eb16b63 Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Tue, 29 Sep 2026 05:03:31 +0000 Subject: [PATCH] test: prune audited duplicate e2e tests MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 中文:删除经覆盖保全审查确认的 36 个重复或自比较测试,保留兼容性与唯一行为覆盖。 --- .../tests/evals/test_run_eval_dispatch.py | 71 ---------- .../test_golden_al_distribution.py | 8 -- .../matrix/test_generate_sweep_configs.py | 23 --- .../results/agentic/test_power_adapter.py | 18 --- .../agentic/test_server_log_metrics.py | 54 -------- .../results/power/test_aggregate_power.py | 40 ------ .../results/test_collect_eval_results.py | 56 -------- .../workflows/test_find_reusable_sweep_run.py | 39 ------ .../tests/workflows/test_merge_with_reuse.py | 131 ------------------ .../workflows/test_recover_failed_ingest.py | 23 --- .../tests/workflows/test_signoff_publish.py | 9 -- .../test_validate_reusable_sweep_artifacts.py | 7 - 12 files changed, 479 deletions(-) diff --git a/inferencex-e2e/infx/tests/evals/test_run_eval_dispatch.py b/inferencex-e2e/infx/tests/evals/test_run_eval_dispatch.py index f476b30a33..76718d7cce 100644 --- a/inferencex-e2e/infx/tests/evals/test_run_eval_dispatch.py +++ b/inferencex-e2e/infx/tests/evals/test_run_eval_dispatch.py @@ -218,14 +218,6 @@ def test_env_can_force_swebench_on_fixed_seqlen(): assert "DISPATCH=swebench" in _dispatch(is_agentic="0", env_fw="swebench") -def test_env_can_force_kimi_vendor_on_agentic_eval() -> None: - assert "DISPATCH=kimi-vendor" in _dispatch( - is_agentic="1", - eval_only="true", - env_fw="kimi-vendor", - ) - - def test_kimi_vendor_skips_unused_model_context_loading() -> None: script = r""" source "$BENCHMARK_LIB" @@ -429,26 +421,6 @@ def test_agentic_eval_propagates_artifact_staging_failure() -> None: assert "eval artifact staging failed with exit code 73" in result.stderr -def test_kimi_full_suite_dispatches_to_schema_runner() -> None: - script = r""" -source "$BENCHMARK_LIB" -_run_kimi_tool_call_schema_eval() { - printf 'DISPATCH=%s ARGS=<%s>\n' "$EVAL_SUITE" "$*" -} -EVAL_SUITE=kimi_tool_call_schema_full run_kimi_vendor_eval --port 9999 -""" - result = subprocess.run( - ["bash", "-c", script], - env={**os.environ, "BENCHMARK_LIB": str(BENCHMARK_LIB)}, - text=True, - capture_output=True, - check=False, - ) - - assert result.returncode == 0, result.stderr - assert "DISPATCH=kimi_tool_call_schema_full ARGS=<--port 9999>" in result.stdout - - def test_minimax_full_suite_dispatches_to_full_runner() -> None: script = r""" source "$BENCHMARK_LIB" @@ -595,24 +567,6 @@ def test_minimax_vendor_accepts_non_m3_model() -> None: assert "DISPATCH=minimax_m3_smoke" in result.stdout -def test_minimax_vendor_accepts_case_insensitive_m3_model_name() -> None: - script = r""" -source "$BENCHMARK_LIB" -_run_minimax_m3_smoke_eval() { echo "DISPATCH=$EVAL_SUITE"; } -unset MODEL_PREFIX EVAL_SUITE EVAL_RESULT_DIR -MODEL_NAME=vendor/MINIMAX-M3-custom run_minimax_vendor_eval -""" - result = subprocess.run( - ["bash", "-c", script], - env={**os.environ, "BENCHMARK_LIB": str(BENCHMARK_LIB)}, - text=True, - capture_output=True, - check=True, - ) - - assert "DISPATCH=minimax_m3_smoke" in result.stdout - - def test_minimax_vendor_setup_failure_uses_integration_error_and_stages( tmp_path: Path, ) -> None: @@ -2524,13 +2478,6 @@ def test_env_can_force_bfcl_on_agentic_eval() -> None: assert "STAGED=summary" in output -def test_cli_can_force_bfcl_on_fixed_seqlen_eval() -> None: - output = _dispatch(is_agentic="0", cli_fw="bfcl") - - assert "DISPATCH=bfcl" in output - assert "STAGED=summary" not in output - - def test_bfcl_defaults_suite_dispatches_once_without_context_loading() -> None: script = r""" source "$BENCHMARK_LIB" @@ -2578,24 +2525,6 @@ def test_bfcl_rejects_suite_from_another_provider() -> None: assert "unsupported BFCL suite 'minimax_m3_smoke'" in result.stderr -def test_bfcl_suite_is_rejected_by_mismatched_framework() -> None: - result = _run_invalid_call( - "EVAL_CONCURRENT_REQUESTS='' " - "EVAL_SUITE=bfcl_smoke " - "run_eval --framework minimax-vendor" - ) - - assert result.returncode == 2 - assert "unsupported MiniMax Provider Verifier suite 'bfcl_smoke'" in result.stderr - - -def test_bfcl_rejects_unknown_suite() -> None: - result = _run_invalid_call("EVAL_SUITE=not_a_bfcl_suite run_bfcl_eval") - - assert result.returncode == 2 - assert "unsupported BFCL suite 'not_a_bfcl_suite'" in result.stderr - - def test_bfcl_dependency_timeout_uses_integration_error_and_stages( tmp_path: Path, ) -> None: diff --git a/inferencex-e2e/infx/tests/golden_al_distribution/test_golden_al_distribution.py b/inferencex-e2e/infx/tests/golden_al_distribution/test_golden_al_distribution.py index f19d8febf7..606e48c6d0 100644 --- a/inferencex-e2e/infx/tests/golden_al_distribution/test_golden_al_distribution.py +++ b/inferencex-e2e/infx/tests/golden_al_distribution/test_golden_al_distribution.py @@ -11,7 +11,6 @@ curve_name, golden_length, list_curves, - load_curve, ) from infx.golden_al_distribution.__main__ import main @@ -52,13 +51,6 @@ def test_curve_name_resolves_committed_curves(model, spec, expected) -> None: assert (GOLDEN_DIR / f"{expected}.yaml").is_file() -def test_golden_length_reads_committed_value() -> None: - spec = {"method": "mtp", "num_speculative_tokens": 3} - assert golden_length("qwen3.5", spec, "thinking_on") == load_curve("qwen3.5_mtp").acceptance( - "thinking_on", 3 - ) - - @pytest.mark.parametrize( ("spec", "message"), [ diff --git a/inferencex-e2e/infx/tests/matrix/test_generate_sweep_configs.py b/inferencex-e2e/infx/tests/matrix/test_generate_sweep_configs.py index 439dccb870..5140fb6c04 100644 --- a/inferencex-e2e/infx/tests/matrix/test_generate_sweep_configs.py +++ b/inferencex-e2e/infx/tests/matrix/test_generate_sweep_configs.py @@ -383,10 +383,6 @@ def full_sweep_args_multi_node(): class TestSeqLenToStr: - def test_known_sequence_lengths(self): - assert seq_len_to_str(1024, 1024) == "1k1k" - assert seq_len_to_str(8192, 1024) == "8k1k" - def test_unknown_sequence_lengths(self): assert seq_len_to_str(2048, 2048) == "2048_2048" assert seq_len_to_str(4096, 1024) == "4096_1024" @@ -2190,25 +2186,6 @@ def test_all_evals_batches_each_multinode_concurrency( assert all(entry['run-eval'] is True for entry in result) assert all(entry['eval-only'] is True for entry in result) - def test_all_evals_cannot_combine_with_no_evals(self, monkeypatch): - import sys - - from infx.matrix import generate as generate_sweep_configs - - monkeypatch.setattr(sys, 'argv', [ - 'generate_sweep_configs.py', - 'test-config', - '--config-files', 'dummy.yaml', - '--config-keys', 'dummy', - '--no-evals', - '--all-evals', - ]) - - with pytest.raises(SystemExit): - generate_sweep_configs.main() - - - @pytest.fixture def sample_mixed_config(sample_single_node_config, sample_multinode_config): """Config dict containing both single-node and multinode entries.""" diff --git a/inferencex-e2e/infx/tests/results/agentic/test_power_adapter.py b/inferencex-e2e/infx/tests/results/agentic/test_power_adapter.py index be545b66c1..30f2351480 100644 --- a/inferencex-e2e/infx/tests/results/agentic/test_power_adapter.py +++ b/inferencex-e2e/infx/tests/results/agentic/test_power_adapter.py @@ -136,24 +136,6 @@ def _write_power_csv(result_dir: Path) -> None: (result_dir / "gpu_metrics.csv").write_text("\n".join(rows) + "\n", encoding="utf-8") -def test_build_power_window_uses_profile_lifecycle_and_successful_records(tmp_path: Path): - from infx.results.agentic.power_adapter import build_power_window - - result_dir = _write_artifacts(tmp_path) - - window, reasons = build_power_window(result_dir) - - assert reasons == [] - assert window == { - "benchmark_start_time_unix": 1_700_000_001.0, - "benchmark_end_time_unix": 1_700_000_004.0, - "duration": 3.0, - "completed": 2, - "total_input_tokens": 300, - "total_output_tokens": 150, - } - - def test_build_power_window_applies_captured_offset_to_naive_aiperf_times(tmp_path: Path): from infx.results.agentic.power_adapter import build_power_window diff --git a/inferencex-e2e/infx/tests/results/agentic/test_server_log_metrics.py b/inferencex-e2e/infx/tests/results/agentic/test_server_log_metrics.py index b3b559f04c..6c35b19154 100644 --- a/inferencex-e2e/infx/tests/results/agentic/test_server_log_metrics.py +++ b/inferencex-e2e/infx/tests/results/agentic/test_server_log_metrics.py @@ -20,30 +20,6 @@ def test_kv_cache_pool_tokens_from_server_log_missing() -> None: assert SglangBackend.kv_cache_pool_tokens_from_server_log("INFO no kv cache line") is None -def test_kv_cache_pool_tokens_from_data_parallel_server_log() -> None: - log = "\n".join( - [ - "INFO (EngineCore_DP0 pid=123) GPU KV cache size: 11,577,333 tokens", - "INFO (EngineCore_DP1 pid=124) GPU KV cache size: 11,577,333 tokens", - "INFO (EngineCore_DP2 pid=125) GPU KV cache size: 11,577,333 tokens", - ] - ) - - assert VllmBackend.kv_cache_pool_tokens_from_server_log(log) == 34_731_999 - - -def test_kv_cache_pool_tokens_dedupes_engine_tags() -> None: - log = "\n".join( - [ - "INFO (EngineCore_DP0 pid=123) GPU KV cache size: 11,577,333 tokens", - "INFO (EngineCore_DP0 pid=123) GPU KV cache size: 11,577,333 tokens", - "INFO (EngineCore_DP1 pid=124) GPU KV cache size: 5,000,000 tokens", - ] - ) - - assert VllmBackend.kv_cache_pool_tokens_from_server_log(log) == 16_577_333 - - def test_kv_cache_pool_tokens_sums_bare_lines() -> None: log = "\n".join( [ @@ -55,18 +31,6 @@ def test_kv_cache_pool_tokens_sums_bare_lines() -> None: assert VllmBackend.kv_cache_pool_tokens_from_server_log(log) == 3_234_567 -def test_kv_cache_pool_tokens_from_sglang_server_log() -> None: - log = "\n".join( - [ - "[2026-06-23 01:05:00] server_args=ServerArgs(dp_size=8, tp_size=8)", - "[2026-06-23 01:10:14 DP0 TP0 EP0] max_total_num_tokens=1172224, " - "chunked_prefill_size=4096", - ] - ) - - assert SglangBackend.kv_cache_pool_tokens_from_server_log(log) == 9_377_792 - - def test_kv_cache_pool_tokens_from_sglang_per_rank_lines() -> None: log = "\n".join( [ @@ -79,24 +43,6 @@ def test_kv_cache_pool_tokens_from_sglang_per_rank_lines() -> None: assert SglangBackend.kv_cache_pool_tokens_from_server_log(log) == 2200 -def test_kv_cache_pool_tokens_sums_multiple_log_files(tmp_path: Path) -> None: - first = tmp_path / "watchtower-a.out" - second = tmp_path / "watchtower-b.out" - first.write_text( - "\n".join( - [ - "INFO (EngineCore_DP0 pid=100) GPU KV cache size: 5,000,000 tokens", - "INFO (EngineCore_DP1 pid=101) GPU KV cache size: 6,500,000 tokens", - ] - ) - ) - second.write_text( - "INFO (EngineCore_DP0 pid=200) GPU KV cache size: 7,000,000 tokens" - ) - - assert VllmBackend().gpu_kv_capacity_tokens({}, (load_server_log_head(p) for p in (first, second))) == 18_500_000 - - def test_dynamo_vllm_uses_vllm_server_log_capacity_parser(tmp_path: Path) -> None: worker_log = tmp_path / "watchtower-worker.out" worker_log.write_text( diff --git a/inferencex-e2e/infx/tests/results/power/test_aggregate_power.py b/inferencex-e2e/infx/tests/results/power/test_aggregate_power.py index 00150993f9..788dab7cc5 100644 --- a/inferencex-e2e/infx/tests/results/power/test_aggregate_power.py +++ b/inferencex-e2e/infx/tests/results/power/test_aggregate_power.py @@ -62,39 +62,6 @@ def _write_amd_csv(path: Path, samples: list[tuple[float, int, float]]) -> None: # --------------------------------------------------------------------------- # -def test_detect_columns_nvidia(): - header = ["timestamp", "index", "power.draw [W]", "utilization.gpu"] - ts, pw, gpu = _detect_columns(header) - assert ts == "timestamp" - assert pw == "power.draw [W]" - assert gpu == "index" - - -def test_detect_columns_amd(): - header = ["timestamp", "gpu", "socket_power", "temperature"] - ts, pw, gpu = _detect_columns(header) - assert ts == "timestamp" - assert pw == "socket_power" - assert gpu == "gpu" - - -def test_detect_columns_amd_watch_mode_real_header(): - # AMDSMI 26.2.0 `metric -p -c -t -u -w 1 --csv` header (order-faithful - # subset, measured on MI355X): socket_power must win even though - # power_management also matches the power pattern later in the row. - header = [ - "timestamp", "gpu", "gfx_activity", "umc_activity", "mm_activity", - "vcn_activity", "jpeg_activity", "gfx_busy_inst_xcp_0", - "jpeg_busy_xcp_0", "vcn_busy_xcp_0", "socket_power", "gfx_voltage", - "soc_voltage", "mem_voltage", "throttle_status", "power_management", - "gfx_0_clk", "mem_0_clk", "edge", "hotspot", "mem", - ] - ts, pw, gpu = _detect_columns(header) - assert ts == "timestamp" - assert pw == "socket_power" - assert gpu == "gpu" - - def test_detect_columns_excludes_power_limit(): # power.limit must NOT be picked as the power column. header = ["timestamp", "index", "power.limit [W]", "power.draw [W]"] @@ -1147,13 +1114,6 @@ def test_cross_check_accumulator_flags_disagreement_beyond_tolerance(tmp_path: P assert result["within_tolerance"] is False -def test_cross_check_accumulator_without_snapshots_returns_none(tmp_path: Path): - csv = tmp_path / "gpu_metrics.csv" - _write_flat_stream(csv, base=1_700_000_000.0, watts_by_gpu={0: 500.0}) - - assert cross_check_accumulator(csv) is None - - def test_cross_check_accumulator_reports_missing_end_snapshot(tmp_path: Path): csv = tmp_path / "gpu_metrics.csv" _write_flat_stream(csv, base=1_700_000_000.0, watts_by_gpu={0: 500.0}) diff --git a/inferencex-e2e/infx/tests/results/test_collect_eval_results.py b/inferencex-e2e/infx/tests/results/test_collect_eval_results.py index eaa5830f42..a7a5119cd7 100644 --- a/inferencex-e2e/infx/tests/results/test_collect_eval_results.py +++ b/inferencex-e2e/infx/tests/results/test_collect_eval_results.py @@ -16,8 +16,6 @@ detect_lm_eval_jsons, result_concurrency, ) -from infx.evals.kimi_vendor_eval import RESULT_FORMAT as KIMI_VENDOR_RESULT_FORMAT -from infx.evals.minimax_provider_eval import RESULT_FORMAT as MINIMAX_RESULT_FORMAT from infx.results.evals import ( build_rows, extract_metrics, select_latest_result, select_latest_results, ) @@ -273,15 +271,6 @@ def test_build_row_preserves_sequence_lengths() -> None: assert "eval_suite" not in row -def test_build_row_preserves_explicit_eval_suite() -> None: - row = build_row( - {"eval_suite": "kimi_tool_call_schema"}, - {"task": "kimi_tool_call_schema"}, - ) - - assert row["eval_suite"] == "kimi_tool_call_schema" - - def _write_lm_eval_result( path: Path, score: float, @@ -436,31 +425,6 @@ def test_collect_eval_rows_ignores_failed_batch_points( -@pytest.mark.parametrize("result_format", [KIMI_VENDOR_RESULT_FORMAT, MINIMAX_RESULT_FORMAT]) -def test_collect_eval_rows_accepts_provider_compatibility_result( - tmp_path: Path, result_format: str, -) -> None: - artifact_dir = tmp_path / "eval_minimax" - artifact_dir.mkdir() - (artifact_dir / "meta_env.json").write_text( - json.dumps({"eval_suite": "minimax_m3_smoke"}) - ) - result_path = artifact_dir / "results_minimax_vendor.json" - _write_lm_eval_result(result_path, 1.0, task="minimax_m3_smoke") - result = json.loads(result_path.read_text()) - result.pop("lm_eval_version") - result["result_format"] = result_format - result["eval_adapter"] = "minimax-provider-verifier" - result_path.write_text(json.dumps(result)) - - rows = collect_eval_rows(tmp_path) - - assert len(rows) == 1 - assert rows[0]["task"] == "minimax_m3_smoke" - assert rows[0]["score"] == 1.0 - assert rows[0]["eval_suite"] == "minimax_m3_smoke" - - def test_collect_eval_rows_retains_integration_and_sample_failures( tmp_path: Path, ) -> None: @@ -639,26 +603,6 @@ def test_collect_eval_rows_retains_missing_or_out_of_range_scores( } == {"InvalidPrimaryScore"} -def test_collect_eval_rows_falls_back_for_invalid_filename_timestamp( - tmp_path: Path, -) -> None: - artifact_dir = tmp_path / "eval_invalid_timestamp" - artifact_dir.mkdir() - (artifact_dir / "meta_env.json").write_text( - json.dumps({"eval_suite": "kimi_tool_call_schema"}) - ) - _write_lm_eval_result( - artifact_dir / "results_2026-99-99T99-99-99.json", - 1.0, - task="kimi_tool_call_schema", - ) - - rows = collect_eval_rows(tmp_path) - - assert len(rows) == 1 - assert rows[0]["score"] == 1.0 - - def test_collect_eval_rows_uses_extract_filter_as_primary_score( tmp_path: Path, ) -> None: diff --git a/inferencex-e2e/infx/tests/workflows/test_find_reusable_sweep_run.py b/inferencex-e2e/infx/tests/workflows/test_find_reusable_sweep_run.py index 1b20358e8b..416b05c902 100644 --- a/inferencex-e2e/infx/tests/workflows/test_find_reusable_sweep_run.py +++ b/inferencex-e2e/infx/tests/workflows/test_find_reusable_sweep_run.py @@ -182,19 +182,6 @@ def test_latest_successful_source_skips_ineligible_runs(monkeypatch, newer, arti assert selected["id"] == 111 -def test_artifact_names_excludes_expired_artifacts(monkeypatch) -> None: - monkeypatch.setattr( - reuse.github, - "paginate", - lambda *args, **kwargs: [ - {"name": "results_bmk", "expired": True}, - {"name": "run-stats", "expired": False}, - ], - ) - - assert reuse.artifact_names("repo", 123, "token") == {"run-stats"} - - @pytest.mark.parametrize("has_artifacts", [True, False]) def test_main_checks_source_before_skipping_pr_synchronize( monkeypatch, tmp_path, has_artifacts @@ -367,32 +354,6 @@ def test_validate_reusable_run_rejects_non_success_run_by_default(conclusion) -> raise AssertionError(f"expected an unpinned {conclusion} run to be rejected") -def test_validate_reusable_run_rejects_run_for_orphaned_commit(monkeypatch) -> None: - responses = {"/pulls/1321/commits": [[{"sha": "def456"}]]} - monkeypatch.setattr(reuse.github, "api", lambda repo, path, *args, **kwargs: responses[path]) - - try: - reuse.validate_reusable_run( - "SemiAnalysisAI/InferenceX", - "run-sweep.yml", - 1321, - { - "id": 25763404168, - "event": "pull_request", - "status": "completed", - "conclusion": "success", - "path": ".github/workflows/run-sweep.yml", - "head_sha": "abc123", - "pull_requests": [], - }, - "token", - ) - except RuntimeError as error: - assert "is not in PR #1321's commit list" in str(error) - else: - raise AssertionError("expected orphaned-commit run to be rejected") - - @pytest.mark.parametrize("labels,command", [ ([], "/reuse-sweep-run"), ([], "/use"), (["documentation"], "/use"), (["full-sweep-enabled"], "/use"), diff --git a/inferencex-e2e/infx/tests/workflows/test_merge_with_reuse.py b/inferencex-e2e/infx/tests/workflows/test_merge_with_reuse.py index 2412023955..fc6378a43d 100644 --- a/inferencex-e2e/infx/tests/workflows/test_merge_with_reuse.py +++ b/inferencex-e2e/infx/tests/workflows/test_merge_with_reuse.py @@ -22,7 +22,6 @@ _poll_pr_head, _resolve_token, _retry_delay, - die, find_eligible_run, main, merge_pr, @@ -427,35 +426,6 @@ def test_timeout(self): result = wait_for_checks(pull, sha, gh_repo, timeout=10) assert result == 1 - def test_rerun_duplicate_old_cancelled_new_success(self): - """Old cancelled + new success for the same check -> pass (not fail-fast).""" - pull = MagicMock() - gh_repo = MagicMock() - commit = MagicMock() - gh_repo.get_commit.return_value = commit - # Two runs with the same name: old cancelled, new success - commit.get_check_runs.return_value = [ - make_check_run( - name="calc-success-rate", - conclusion="cancelled", - started_at="2024-01-01T00:00:00Z", - cr_id=100, - ), - make_check_run( - name="calc-success-rate", - conclusion="success", - started_at="2024-01-01T01:00:00Z", - cr_id=200, - ), - ] - combined = MagicMock() - combined.statuses = [] - commit.get_combined_status.return_value = combined - - sha = "a" * 40 - result = wait_for_checks(pull, sha, gh_repo, timeout=10) - assert result == 0 - def test_rerun_duplicate_old_failed_new_success(self): """Old failed + new success for the same check -> pass.""" pull = MagicMock() @@ -507,23 +477,6 @@ def test_stale_keeps_waiting(self): result = wait_for_checks(pull, sha, gh_repo, timeout=10) assert result == 1 - def test_genuine_failure_fail_fast(self): - """A genuine failure triggers immediate fail-fast.""" - pull = MagicMock() - gh_repo = MagicMock() - commit = MagicMock() - gh_repo.get_commit.return_value = commit - commit.get_check_runs.return_value = [ - make_check_run(name="tests", conclusion="failure"), - ] - combined = MagicMock() - combined.statuses = [] - commit.get_combined_status.return_value = combined - - sha = "a" * 40 - result = wait_for_checks(pull, sha, gh_repo, timeout=60) - assert result == 1 - def test_no_checks_yet_keeps_waiting(self): """No check runs and no statuses -> keeps waiting until timeout.""" pull = MagicMock() @@ -598,13 +551,6 @@ def test_transient_error_retried_then_succeeds(self): class TestDeduplication: - def test_latest_check_runs_keeps_newest(self): - old = make_check_run(name="ci", started_at="2024-01-01T00:00:00Z", cr_id=1) - new = make_check_run(name="ci", started_at="2024-01-01T01:00:00Z", cr_id=2) - result = _latest_check_runs([old, new]) - assert len(result) == 1 - assert result[0].id == 2 - def test_latest_check_runs_different_names_kept(self): a = make_check_run(name="ci-a", cr_id=1) b = make_check_run(name="ci-b", cr_id=2) @@ -630,12 +576,6 @@ def test_latest_statuses_keeps_newest(self): class TestTransientErrors: - def test_github_500_is_transient(self): - from github import GithubException - - exc = GithubException(status=500, data={"message": "ISE"}, headers={}) - assert _is_transient_error(exc) is True - def test_github_429_is_transient(self): from github import GithubException @@ -691,12 +631,6 @@ def test_retry_delay_capped_at_60(self): class TestPollPrHead: - def test_returns_immediately_on_match(self): - pull = MagicMock() - pull.head.sha = "expected_sha" - result = _poll_pr_head(pull, "expected_sha", retries=3, delay=1) - assert result == "expected_sha" - def test_retries_and_succeeds(self): pull = MagicMock() # head.sha changes on successive update() calls @@ -848,54 +782,6 @@ def test_comment_posted_with_eligible_run_id(self): class TestMergePrFullFlow: """Test the full merge flow with comprehensive mocking.""" - def _run_full_flow(self, *, merge_fails=False, changelog_diff=False, sha=None): - """Run merge_pr with comprehensive mocking for the happy path.""" - sha = sha or "a" * 40 - pull = make_mock_pull(head_sha=sha) - gh = make_mock_gh(pull) - git_ops = make_mock_git_ops(sha=sha) - - if merge_fails: - git_ops.merge.return_value = 1 - - if changelog_diff: - git_ops.diff_quiet.return_value = False - else: - git_ops.diff_quiet.return_value = True - - with ( - patch( - "infx.workflows.merge_with_reuse.find_eligible_run", - return_value=42, - ), - patch( - "infx.workflows.merge_with_reuse.wait_for_check", - return_value=0, - ), - patch( - "infx.workflows.merge_with_reuse.wait_for_checks", - return_value=0, - ), - patch("infx.workflows.merge_with_reuse.canonicalize_changelog"), - patch( - "infx.workflows.merge_with_reuse.resolve_changelog_conflict", - return_value=True, - ), - ): - result = merge_pr( - 7, - repo="example/repo", - head_lag_retries=0, - head_lag_delay=0, - _git_ops=git_ops, - _gh=gh, - ) - return result, git_ops, pull - - def test_clean_merge_exits_zero(self): - result, _, _ = self._run_full_flow() - assert result == 0 - def test_head_lag_retry_succeeds(self): sha = "a" * 40 pull = make_mock_pull(head_sha=sha) @@ -1201,17 +1087,6 @@ def test_synchronize_empty_commit(self): -class TestExitCodes: - def test_main_returns_two_on_bad_usage(self): - with patch("sys.argv", ["prog"]): - assert main() == 2 - - def test_die_returns_one(self): - assert die("error") == 1 - - - - class TestTokenNeverLeaks: """Verify the token string never appears in stdout/stderr output.""" @@ -1504,12 +1379,6 @@ def test_initial_state(self): state = _MergeState() assert state.local_branch == "" - def test_local_branch_settable(self): - state = _MergeState() - state.local_branch = "pr-42-reuse-1234" - assert state.local_branch == "pr-42-reuse-1234" - - def test_canonicalize_changelog_across_layout_migration(temp_repo, tmp_path, monkeypatch): base = b"- config-keys: [historical]\n description: [original]\n pr-link: XXX\n" (tmp_path / "perf-changelog.yaml").write_bytes(base) diff --git a/inferencex-e2e/infx/tests/workflows/test_recover_failed_ingest.py b/inferencex-e2e/infx/tests/workflows/test_recover_failed_ingest.py index 2a91167cc5..c1ce331d04 100644 --- a/inferencex-e2e/infx/tests/workflows/test_recover_failed_ingest.py +++ b/inferencex-e2e/infx/tests/workflows/test_recover_failed_ingest.py @@ -124,15 +124,6 @@ def test_select_failed_job_uses_explicit_job() -> None: assert select_failed_job(jobs, 2)["id"] == 2 -def test_select_failed_job_allows_unambiguous_run_only_url() -> None: - jobs = [ - {"id": 1, "status": "completed", "conclusion": "success"}, - {"id": 2, "status": "completed", "conclusion": "failure"}, - ] - - assert select_failed_job(jobs, None)["id"] == 2 - - def test_select_failed_job_rejects_ambiguous_run_only_url() -> None: jobs = [ {"id": 1, "status": "completed", "conclusion": "failure"}, @@ -143,20 +134,6 @@ def test_select_failed_job_rejects_ambiguous_run_only_url() -> None: select_failed_job(jobs, None) -def test_audit_changelog_rejects_duplicate_yaml_keys() -> None: - raw = b"""- config-keys: - - config-a - description: - - First - description: - - Second - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/1 -""" - - with pytest.raises(ChangelogValidationError, match="duplicate key"): - audit_changelog_bytes(raw, "snapshot") - - def test_audit_changelog_reports_repairable_missing_newline() -> None: raw = block( "config-a", diff --git a/inferencex-e2e/infx/tests/workflows/test_signoff_publish.py b/inferencex-e2e/infx/tests/workflows/test_signoff_publish.py index 7f19b0f85d..50a6efef04 100644 --- a/inferencex-e2e/infx/tests/workflows/test_signoff_publish.py +++ b/inferencex-e2e/infx/tests/workflows/test_signoff_publish.py @@ -165,15 +165,6 @@ def test_identical_replay_does_not_write_again(publish): assert result["writes"] == [("POST", "repos/example/repo/issues/7/comments")] -def test_replacement_signoff_gets_a_separate_verdict(publish): - previous = publish(verdict(), signoff_key="issuecomment-41")["comments"] - result = publish(verdict(), previous) - assert result["comments"][0] == previous[0] - assert result["comments"][0]["body"].startswith(marker("issuecomment-41")) - assert result["comments"][1]["body"].startswith(marker()) - assert result["writes"] == [("POST", "repos/example/repo/issues/7/comments")] - - def test_deleted_verdict_during_update_is_recreated(publish): comments = [ {"id": 1, "user": {"login": "github-actions[bot]"}, "body": marker() + "\nOld"} diff --git a/inferencex-e2e/infx/tests/workflows/test_validate_reusable_sweep_artifacts.py b/inferencex-e2e/infx/tests/workflows/test_validate_reusable_sweep_artifacts.py index b05352160a..3baffb8d63 100644 --- a/inferencex-e2e/infx/tests/workflows/test_validate_reusable_sweep_artifacts.py +++ b/inferencex-e2e/infx/tests/workflows/test_validate_reusable_sweep_artifacts.py @@ -546,13 +546,6 @@ def test_eval_validation_separates_explicit_suite_identities( assert validate_eval_artifacts(tmp_path) == [] -def test_eval_result_key_includes_task_identity() -> None: - gsm8k = single_eval_result(32, eval_suite="tool_use") - bfcl = {**gsm8k, "task": "bfcl_smoke"} - - assert eval_result_key(gsm8k) != eval_result_key(bfcl) - - def test_eval_validation_distinguishes_sequence_lengths(tmp_path: Path) -> None: write_eval_aggregate( tmp_path,