Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
71 changes: 0 additions & 71 deletions inferencex-e2e/infx/tests/evals/test_run_eval_dispatch.py
Original file line number Diff line number Diff line change
Expand Up @@ -218,14 +218,6 @@ def test_env_can_force_swebench_on_fixed_seqlen():
assert "DISPATCH=swebench" in _dispatch(is_agentic="0", env_fw="swebench")


def test_env_can_force_kimi_vendor_on_agentic_eval() -> None:
assert "DISPATCH=kimi-vendor" in _dispatch(
is_agentic="1",
eval_only="true",
env_fw="kimi-vendor",
)


def test_kimi_vendor_skips_unused_model_context_loading() -> None:
script = r"""
source "$BENCHMARK_LIB"
Expand Down Expand Up @@ -429,26 +421,6 @@ def test_agentic_eval_propagates_artifact_staging_failure() -> None:
assert "eval artifact staging failed with exit code 73" in result.stderr


def test_kimi_full_suite_dispatches_to_schema_runner() -> None:
script = r"""
source "$BENCHMARK_LIB"
_run_kimi_tool_call_schema_eval() {
printf 'DISPATCH=%s ARGS=<%s>\n' "$EVAL_SUITE" "$*"
}
EVAL_SUITE=kimi_tool_call_schema_full run_kimi_vendor_eval --port 9999
"""
result = subprocess.run(
["bash", "-c", script],
env={**os.environ, "BENCHMARK_LIB": str(BENCHMARK_LIB)},
text=True,
capture_output=True,
check=False,
)

assert result.returncode == 0, result.stderr
assert "DISPATCH=kimi_tool_call_schema_full ARGS=<--port 9999>" in result.stdout


def test_minimax_full_suite_dispatches_to_full_runner() -> None:
script = r"""
source "$BENCHMARK_LIB"
Expand Down Expand Up @@ -595,24 +567,6 @@ def test_minimax_vendor_accepts_non_m3_model() -> None:
assert "DISPATCH=minimax_m3_smoke" in result.stdout


def test_minimax_vendor_accepts_case_insensitive_m3_model_name() -> None:
script = r"""
source "$BENCHMARK_LIB"
_run_minimax_m3_smoke_eval() { echo "DISPATCH=$EVAL_SUITE"; }
unset MODEL_PREFIX EVAL_SUITE EVAL_RESULT_DIR
MODEL_NAME=vendor/MINIMAX-M3-custom run_minimax_vendor_eval
"""
result = subprocess.run(
["bash", "-c", script],
env={**os.environ, "BENCHMARK_LIB": str(BENCHMARK_LIB)},
text=True,
capture_output=True,
check=True,
)

assert "DISPATCH=minimax_m3_smoke" in result.stdout


def test_minimax_vendor_setup_failure_uses_integration_error_and_stages(
tmp_path: Path,
) -> None:
Expand Down Expand Up @@ -2524,13 +2478,6 @@ def test_env_can_force_bfcl_on_agentic_eval() -> None:
assert "STAGED=summary" in output


def test_cli_can_force_bfcl_on_fixed_seqlen_eval() -> None:
output = _dispatch(is_agentic="0", cli_fw="bfcl")

assert "DISPATCH=bfcl" in output
assert "STAGED=summary" not in output


def test_bfcl_defaults_suite_dispatches_once_without_context_loading() -> None:
script = r"""
source "$BENCHMARK_LIB"
Expand Down Expand Up @@ -2578,24 +2525,6 @@ def test_bfcl_rejects_suite_from_another_provider() -> None:
assert "unsupported BFCL suite 'minimax_m3_smoke'" in result.stderr


def test_bfcl_suite_is_rejected_by_mismatched_framework() -> None:
result = _run_invalid_call(
"EVAL_CONCURRENT_REQUESTS='' "
"EVAL_SUITE=bfcl_smoke "
"run_eval --framework minimax-vendor"
)

assert result.returncode == 2
assert "unsupported MiniMax Provider Verifier suite 'bfcl_smoke'" in result.stderr


def test_bfcl_rejects_unknown_suite() -> None:
result = _run_invalid_call("EVAL_SUITE=not_a_bfcl_suite run_bfcl_eval")

assert result.returncode == 2
assert "unsupported BFCL suite 'not_a_bfcl_suite'" in result.stderr


def test_bfcl_dependency_timeout_uses_integration_error_and_stages(
tmp_path: Path,
) -> None:
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -11,7 +11,6 @@
curve_name,
golden_length,
list_curves,
load_curve,
)
from infx.golden_al_distribution.__main__ import main

Expand Down Expand Up @@ -52,13 +51,6 @@ def test_curve_name_resolves_committed_curves(model, spec, expected) -> None:
assert (GOLDEN_DIR / f"{expected}.yaml").is_file()


def test_golden_length_reads_committed_value() -> None:
spec = {"method": "mtp", "num_speculative_tokens": 3}
assert golden_length("qwen3.5", spec, "thinking_on") == load_curve("qwen3.5_mtp").acceptance(
"thinking_on", 3
)


@pytest.mark.parametrize(
("spec", "message"),
[
Expand Down
23 changes: 0 additions & 23 deletions inferencex-e2e/infx/tests/matrix/test_generate_sweep_configs.py
Original file line number Diff line number Diff line change
Expand Up @@ -383,10 +383,6 @@ def full_sweep_args_multi_node():

class TestSeqLenToStr:

def test_known_sequence_lengths(self):
assert seq_len_to_str(1024, 1024) == "1k1k"
assert seq_len_to_str(8192, 1024) == "8k1k"

def test_unknown_sequence_lengths(self):
assert seq_len_to_str(2048, 2048) == "2048_2048"
assert seq_len_to_str(4096, 1024) == "4096_1024"
Expand Down Expand Up @@ -2190,25 +2186,6 @@ def test_all_evals_batches_each_multinode_concurrency(
assert all(entry['run-eval'] is True for entry in result)
assert all(entry['eval-only'] is True for entry in result)

def test_all_evals_cannot_combine_with_no_evals(self, monkeypatch):
import sys

from infx.matrix import generate as generate_sweep_configs

monkeypatch.setattr(sys, 'argv', [
'generate_sweep_configs.py',
'test-config',
'--config-files', 'dummy.yaml',
'--config-keys', 'dummy',
'--no-evals',
'--all-evals',
])

with pytest.raises(SystemExit):
generate_sweep_configs.main()



@pytest.fixture
def sample_mixed_config(sample_single_node_config, sample_multinode_config):
"""Config dict containing both single-node and multinode entries."""
Expand Down
18 changes: 0 additions & 18 deletions inferencex-e2e/infx/tests/results/agentic/test_power_adapter.py
Original file line number Diff line number Diff line change
Expand Up @@ -136,24 +136,6 @@ def _write_power_csv(result_dir: Path) -> None:
(result_dir / "gpu_metrics.csv").write_text("\n".join(rows) + "\n", encoding="utf-8")


def test_build_power_window_uses_profile_lifecycle_and_successful_records(tmp_path: Path):
from infx.results.agentic.power_adapter import build_power_window

result_dir = _write_artifacts(tmp_path)

window, reasons = build_power_window(result_dir)

assert reasons == []
assert window == {
"benchmark_start_time_unix": 1_700_000_001.0,
"benchmark_end_time_unix": 1_700_000_004.0,
"duration": 3.0,
"completed": 2,
"total_input_tokens": 300,
"total_output_tokens": 150,
}


def test_build_power_window_applies_captured_offset_to_naive_aiperf_times(tmp_path: Path):
from infx.results.agentic.power_adapter import build_power_window

Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -20,30 +20,6 @@ def test_kv_cache_pool_tokens_from_server_log_missing() -> None:
assert SglangBackend.kv_cache_pool_tokens_from_server_log("INFO no kv cache line") is None


def test_kv_cache_pool_tokens_from_data_parallel_server_log() -> None:
log = "\n".join(
[
"INFO (EngineCore_DP0 pid=123) GPU KV cache size: 11,577,333 tokens",
"INFO (EngineCore_DP1 pid=124) GPU KV cache size: 11,577,333 tokens",
"INFO (EngineCore_DP2 pid=125) GPU KV cache size: 11,577,333 tokens",
]
)

assert VllmBackend.kv_cache_pool_tokens_from_server_log(log) == 34_731_999


def test_kv_cache_pool_tokens_dedupes_engine_tags() -> None:
log = "\n".join(
[
"INFO (EngineCore_DP0 pid=123) GPU KV cache size: 11,577,333 tokens",
"INFO (EngineCore_DP0 pid=123) GPU KV cache size: 11,577,333 tokens",
"INFO (EngineCore_DP1 pid=124) GPU KV cache size: 5,000,000 tokens",
]
)

assert VllmBackend.kv_cache_pool_tokens_from_server_log(log) == 16_577_333


def test_kv_cache_pool_tokens_sums_bare_lines() -> None:
log = "\n".join(
[
Expand All @@ -55,18 +31,6 @@ def test_kv_cache_pool_tokens_sums_bare_lines() -> None:
assert VllmBackend.kv_cache_pool_tokens_from_server_log(log) == 3_234_567


def test_kv_cache_pool_tokens_from_sglang_server_log() -> None:
log = "\n".join(
[
"[2026-06-23 01:05:00] server_args=ServerArgs(dp_size=8, tp_size=8)",
"[2026-06-23 01:10:14 DP0 TP0 EP0] max_total_num_tokens=1172224, "
"chunked_prefill_size=4096",
]
)

assert SglangBackend.kv_cache_pool_tokens_from_server_log(log) == 9_377_792


def test_kv_cache_pool_tokens_from_sglang_per_rank_lines() -> None:
log = "\n".join(
[
Expand All @@ -79,24 +43,6 @@ def test_kv_cache_pool_tokens_from_sglang_per_rank_lines() -> None:
assert SglangBackend.kv_cache_pool_tokens_from_server_log(log) == 2200


def test_kv_cache_pool_tokens_sums_multiple_log_files(tmp_path: Path) -> None:
first = tmp_path / "watchtower-a.out"
second = tmp_path / "watchtower-b.out"
first.write_text(
"\n".join(
[
"INFO (EngineCore_DP0 pid=100) GPU KV cache size: 5,000,000 tokens",
"INFO (EngineCore_DP1 pid=101) GPU KV cache size: 6,500,000 tokens",
]
)
)
second.write_text(
"INFO (EngineCore_DP0 pid=200) GPU KV cache size: 7,000,000 tokens"
)

assert VllmBackend().gpu_kv_capacity_tokens({}, (load_server_log_head(p) for p in (first, second))) == 18_500_000


def test_dynamo_vllm_uses_vllm_server_log_capacity_parser(tmp_path: Path) -> None:
worker_log = tmp_path / "watchtower-worker.out"
worker_log.write_text(
Expand Down
40 changes: 0 additions & 40 deletions inferencex-e2e/infx/tests/results/power/test_aggregate_power.py
Original file line number Diff line number Diff line change
Expand Up @@ -62,39 +62,6 @@ def _write_amd_csv(path: Path, samples: list[tuple[float, int, float]]) -> None:
# --------------------------------------------------------------------------- #


def test_detect_columns_nvidia():
header = ["timestamp", "index", "power.draw [W]", "utilization.gpu"]
ts, pw, gpu = _detect_columns(header)
assert ts == "timestamp"
assert pw == "power.draw [W]"
assert gpu == "index"


def test_detect_columns_amd():
header = ["timestamp", "gpu", "socket_power", "temperature"]
ts, pw, gpu = _detect_columns(header)
assert ts == "timestamp"
assert pw == "socket_power"
assert gpu == "gpu"


def test_detect_columns_amd_watch_mode_real_header():
# AMDSMI 26.2.0 `metric -p -c -t -u -w 1 --csv` header (order-faithful
# subset, measured on MI355X): socket_power must win even though
# power_management also matches the power pattern later in the row.
header = [
"timestamp", "gpu", "gfx_activity", "umc_activity", "mm_activity",
"vcn_activity", "jpeg_activity", "gfx_busy_inst_xcp_0",
"jpeg_busy_xcp_0", "vcn_busy_xcp_0", "socket_power", "gfx_voltage",
"soc_voltage", "mem_voltage", "throttle_status", "power_management",
"gfx_0_clk", "mem_0_clk", "edge", "hotspot", "mem",
]
ts, pw, gpu = _detect_columns(header)
assert ts == "timestamp"
assert pw == "socket_power"
assert gpu == "gpu"


def test_detect_columns_excludes_power_limit():
# power.limit must NOT be picked as the power column.
header = ["timestamp", "index", "power.limit [W]", "power.draw [W]"]
Expand Down Expand Up @@ -1147,13 +1114,6 @@ def test_cross_check_accumulator_flags_disagreement_beyond_tolerance(tmp_path: P
assert result["within_tolerance"] is False


def test_cross_check_accumulator_without_snapshots_returns_none(tmp_path: Path):
csv = tmp_path / "gpu_metrics.csv"
_write_flat_stream(csv, base=1_700_000_000.0, watts_by_gpu={0: 500.0})

assert cross_check_accumulator(csv) is None


def test_cross_check_accumulator_reports_missing_end_snapshot(tmp_path: Path):
csv = tmp_path / "gpu_metrics.csv"
_write_flat_stream(csv, base=1_700_000_000.0, watts_by_gpu={0: 500.0})
Expand Down
Loading
Loading