diff --git a/.github/workflows/benchmark-multinode-tmpl.yml b/.github/workflows/benchmark-multinode-tmpl.yml index 9a5c5c94b3..b55af43e5f 100644 --- a/.github/workflows/benchmark-multinode-tmpl.yml +++ b/.github/workflows/benchmark-multinode-tmpl.yml @@ -5,10 +5,6 @@ on: secrets: INFERENCEX_OFFICIAL_RO_HF_TOKEN: required: true - MODAL_TOKEN_ID: - required: false - MODAL_TOKEN_SECRET: - required: false inputs: config: description: "Benchmark configuration as JSON" @@ -52,12 +48,12 @@ on: required: false default: false eval-framework: - description: "Eval runner (lm-eval, swebench, kimi-vendor, minimax-vendor, or bfcl)" + description: "Eval runner (lm-eval, kimi-vendor, minimax-vendor, or bfcl)" type: string required: false default: "lm-eval" eval-suite: - description: "Eval suite (kimi_tool_call_schema, kimi_tool_call_schema_full, minimax_m3_smoke, minimax_m3_full, bfcl_smoke, bfcl_vllm_minimax_m3, or bfcl_vllm_kimi); empty for lm-eval and swebench" + description: "Eval suite (kimi_tool_call_schema, kimi_tool_call_schema_full, minimax_m3_smoke, minimax_m3_full, bfcl_smoke, bfcl_vllm_minimax_m3, or bfcl_vllm_kimi); empty for lm-eval" type: string required: false default: "" @@ -71,11 +67,6 @@ on: required: false type: string default: "" - swebench-gen-mode: - description: "SWE-bench generation mode (single-shot | agentic). Empty = agentic (single-shot is an explicit debugging escape hatch)." - required: false - type: string - default: "" require-power: description: "Fail fixed-sequence result processing when GPU power is invalid" type: boolean @@ -156,11 +147,6 @@ env: EVAL_SUITE: ${{ inputs.eval-suite }} EVAL_CONC: ${{ inputs.eval-conc }} EVAL_LIMIT: ${{ inputs.eval-limit }} - SWEBENCH_GEN_MODE: ${{ inputs.swebench-gen-mode || 'agentic' }} - # GPU/multi-node runners lack Docker for SWE-bench scoring. - SWEBENCH_USE_MODAL: 'true' - MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} - MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} REQUIRE_POWER: ${{ inputs.require-power && '1' || '0' }} POWER_PRODUCER_SHA: ${{ inputs.power-producer-sha }} # Agentic-coding env. Fixed-seq-len jobs leave these empty. @@ -491,7 +477,6 @@ jobs: sample*.jsonl agent_preds.json predictions.jsonl - swebench_report_*.json *_artifacts.tar.gz *.traj* if-no-files-found: ${{ inputs.eval-only && 'error' || 'ignore' }} @@ -518,7 +503,7 @@ jobs: rm -f -- ./*_results.jsonl || true rm -f -- ./*_artifacts.tar.gz || true rm -f sample*.jsonl || true - rm -f agent_preds.json predictions.jsonl swebench_report_*.json *.traj* || true + rm -f agent_preds.json predictions.jsonl *.traj* || true - name: Slurm cleanup (post-run) if: always() diff --git a/.github/workflows/benchmark-tmpl.yml b/.github/workflows/benchmark-tmpl.yml index f49bc12eaf..68fd3a6033 100644 --- a/.github/workflows/benchmark-tmpl.yml +++ b/.github/workflows/benchmark-tmpl.yml @@ -4,10 +4,6 @@ on: secrets: INFERENCEX_OFFICIAL_RO_HF_TOKEN: required: true - MODAL_TOKEN_ID: - required: false - MODAL_TOKEN_SECRET: - required: false inputs: config: description: "Benchmark configuration as JSON" @@ -46,12 +42,12 @@ on: required: false default: false eval-framework: - description: "Eval runner (lm-eval, swebench, kimi-vendor, minimax-vendor, or bfcl)" + description: "Eval runner (lm-eval, kimi-vendor, minimax-vendor, or bfcl)" type: string required: false default: "lm-eval" eval-suite: - description: "Eval suite (kimi_tool_call_schema, kimi_tool_call_schema_full, minimax_m3_smoke, minimax_m3_full, bfcl_smoke, bfcl_vllm_minimax_m3, or bfcl_vllm_kimi); empty for lm-eval and swebench" + description: "Eval suite (kimi_tool_call_schema, kimi_tool_call_schema_full, minimax_m3_smoke, minimax_m3_full, bfcl_smoke, bfcl_vllm_minimax_m3, or bfcl_vllm_kimi); empty for lm-eval" type: string required: false default: "" @@ -89,11 +85,6 @@ on: required: false type: string default: "" - swebench-gen-mode: - description: "SWE-bench generation mode (single-shot | agentic). Empty = agentic (single-shot is an explicit debugging escape hatch)." - required: false - type: string - default: "" env: PORT: '8888' INFMAX_CONTAINER_WORKSPACE: '/workspace' @@ -147,15 +138,10 @@ env: REQUIRE_POWER: ${{ inputs.require-power && '1' || '0' }} AIPERF_EXPERIMENTAL_FAST: ${{ inputs.agentx-fast && '1' || '0' }} EVAL_LIMIT: ${{ inputs.eval-limit }} - SWEBENCH_GEN_MODE: ${{ inputs.swebench-gen-mode || 'agentic' }} AIPERF_FAILED_REQUEST_THRESHOLD: '0.10' RESULT_DIR: /workspace/results PYTHONDONTWRITEBYTECODE: '1' PYTHONPYCACHEPREFIX: /tmp/inferencex-pycache - # GPU runners lack Docker for SWE-bench scoring. - SWEBENCH_USE_MODAL: 'true' - MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} - MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} permissions: contents: read @@ -471,7 +457,6 @@ jobs: sample*.jsonl agent_preds.json predictions.jsonl - swebench_report_*.json *.traj* if-no-files-found: ${{ inputs.eval-only && 'error' || 'ignore' }} @@ -494,7 +479,7 @@ jobs: rm -f -- ./*_results.jsonl || true rm -f sample*.jsonl || true rm -f -- ./*_artifacts.tar.gz || true - rm -f agent_preds.json predictions.jsonl swebench_report_*.json *.traj* || true + rm -f agent_preds.json predictions.jsonl *.traj* || true - name: Repair root-owned artifacts (post-run) if: ${{ always() && inputs.runner == 'cluster:mi355x-amds' }} diff --git a/.github/workflows/e2e-tests.yml b/.github/workflows/e2e-tests.yml index ea5d9e727b..628f61c129 100644 --- a/.github/workflows/e2e-tests.yml +++ b/.github/workflows/e2e-tests.yml @@ -52,17 +52,12 @@ on: # zizmor: ignore[concurrency-limits] type: string default: "" eval-framework: - description: "Eval runner override (auto uses model-aware selection; lm-eval, swebench, kimi-vendor, minimax-vendor, or bfcl)" + description: "Eval runner override (auto uses model-aware selection; lm-eval, kimi-vendor, minimax-vendor, or bfcl)" required: false type: string default: "auto" eval-suite: - description: "Eval suite (kimi_tool_call_schema, kimi_tool_call_schema_full, minimax_m3_smoke, minimax_m3_full, bfcl_smoke, bfcl_vllm_minimax_m3, or bfcl_vllm_kimi); empty for lm-eval and swebench" - required: false - type: string - default: "" - swebench-gen-mode: - description: "SWE-bench generation mode (single-shot | agentic). Empty = agentic (single-shot is an explicit debugging escape hatch)." + description: "Eval suite (kimi_tool_call_schema, kimi_tool_call_schema_full, minimax_m3_smoke, minimax_m3_full, bfcl_smoke, bfcl_vllm_minimax_m3, or bfcl_vllm_kimi); empty for lm-eval" required: false type: string default: "" @@ -105,10 +100,6 @@ on: # zizmor: ignore[concurrency-limits] secrets: INFERENCEX_OFFICIAL_RO_HF_TOKEN: required: true - MODAL_TOKEN_ID: - required: false - MODAL_TOKEN_SECRET: - required: false inputs: generate-cli-command: description: "Command passed to generate matrix script" @@ -154,17 +145,12 @@ on: # zizmor: ignore[concurrency-limits] type: string default: "" eval-framework: - description: "Eval runner override (auto uses model-aware selection; lm-eval, swebench, kimi-vendor, minimax-vendor, or bfcl)" + description: "Eval runner override (auto uses model-aware selection; lm-eval, kimi-vendor, minimax-vendor, or bfcl)" required: false type: string default: "auto" eval-suite: - description: "Eval suite (kimi_tool_call_schema, kimi_tool_call_schema_full, minimax_m3_smoke, minimax_m3_full, bfcl_smoke, bfcl_vllm_minimax_m3, or bfcl_vllm_kimi); empty for lm-eval and swebench" - required: false - type: string - default: "" - swebench-gen-mode: - description: "SWE-bench generation mode (single-shot | agentic). Empty = agentic (single-shot is an explicit debugging escape hatch)." + description: "Eval suite (kimi_tool_call_schema, kimi_tool_call_schema_full, minimax_m3_smoke, minimax_m3_full, bfcl_smoke, bfcl_vllm_minimax_m3, or bfcl_vllm_kimi); empty for lm-eval" required: false type: string default: "" @@ -370,8 +356,7 @@ jobs: config: ${{ fromJson(needs.get-jobs.outputs.multi-node-config) }} secrets: INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} - MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} - MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} + with: config: ${{ toJSON(matrix.config) }} klaud-run: ${{ inputs.klaud-run }} @@ -396,8 +381,7 @@ jobs: config: ${{ fromJson(needs.get-jobs.outputs.multi-node-eval-config) }} secrets: INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} - MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} - MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} + with: config: ${{ toJSON(matrix.config) }} klaud-run: ${{ inputs.klaud-run }} @@ -412,7 +396,7 @@ jobs: eval-limit: ${{ inputs.eval-limit }} eval-framework: ${{ inputs.eval-framework == 'auto' && (matrix.config['eval-framework'] || 'lm-eval') || inputs.eval-framework }} eval-suite: ${{ inputs.eval-suite != '' && inputs.eval-suite || (inputs.eval-framework == 'auto' && matrix.config['eval-suite'] || '') }} - swebench-gen-mode: ${{ inputs.swebench-gen-mode }} + ref: ${{ needs.get-jobs.outputs.benchmark-ref }} test-sweep-agentic: @@ -426,8 +410,7 @@ jobs: config: ${{ fromJson(needs.get-jobs.outputs.agentic-config) }} secrets: INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} - MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} - MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} + with: config: ${{ toJSON(matrix.config) }} klaud-run: ${{ inputs.klaud-run }} @@ -453,8 +436,7 @@ jobs: config: ${{ fromJson(needs.get-jobs.outputs.agentic-eval-config) }} secrets: INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} - MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} - MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} + with: config: ${{ toJSON(matrix.config) }} klaud-run: ${{ inputs.klaud-run }} @@ -467,7 +449,7 @@ jobs: run-eval: true eval-only: true eval-limit: ${{ inputs.eval-limit }} - swebench-gen-mode: ${{ inputs.swebench-gen-mode }} + eval-framework: ${{ inputs.eval-framework == 'auto' && (matrix.config['eval-framework'] || 'lm-eval') || inputs.eval-framework }} eval-suite: ${{ inputs.eval-suite != '' && inputs.eval-suite || (inputs.eval-framework == 'auto' && matrix.config['eval-suite'] || '') }} scenario-type: agentic-coding @@ -484,8 +466,7 @@ jobs: config: ${{ fromJson(needs.get-jobs.outputs.multi-node-agentic-config) }} secrets: INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} - MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} - MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} + with: config: ${{ toJSON(matrix.config) }} klaud-run: ${{ inputs.klaud-run }} @@ -514,8 +495,7 @@ jobs: config: ${{ fromJson(needs.get-jobs.outputs.multi-node-agentic-eval-config) }} secrets: INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} - MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} - MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} + with: config: ${{ toJSON(matrix.config) }} klaud-run: ${{ inputs.klaud-run }} @@ -530,7 +510,7 @@ jobs: eval-only: true eval-conc: ${{ matrix.config['eval-conc'] }} eval-limit: ${{ inputs.eval-limit }} - swebench-gen-mode: ${{ inputs.swebench-gen-mode }} + eval-framework: ${{ inputs.eval-framework == 'auto' && (matrix.config['eval-framework'] || 'lm-eval') || inputs.eval-framework }} eval-suite: ${{ inputs.eval-suite != '' && inputs.eval-suite || (inputs.eval-framework == 'auto' && matrix.config['eval-suite'] || '') }} scenario-type: agentic-coding @@ -547,8 +527,7 @@ jobs: config: ${{ fromJson(needs.get-jobs.outputs.single-node-config) }} secrets: INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} - MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} - MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} + with: config: ${{ toJSON(matrix.config) }} klaud-run: ${{ inputs.klaud-run }} @@ -571,8 +550,7 @@ jobs: config: ${{ fromJson(needs.get-jobs.outputs.eval-config) }} secrets: INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} - MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} - MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} + with: config: ${{ toJSON(matrix.config) }} klaud-run: ${{ inputs.klaud-run }} @@ -585,7 +563,7 @@ jobs: eval-limit: ${{ inputs.eval-limit }} eval-framework: ${{ inputs.eval-framework == 'auto' && (matrix.config['eval-framework'] || 'lm-eval') || inputs.eval-framework }} eval-suite: ${{ inputs.eval-suite != '' && inputs.eval-suite || (inputs.eval-framework == 'auto' && matrix.config['eval-suite'] || '') }} - swebench-gen-mode: ${{ inputs.swebench-gen-mode }} + ref: ${{ needs.get-jobs.outputs.benchmark-ref }} collect-results: diff --git a/.github/workflows/run-sweep.yml b/.github/workflows/run-sweep.yml index df90560057..a6b091ea4d 100644 --- a/.github/workflows/run-sweep.yml +++ b/.github/workflows/run-sweep.yml @@ -480,8 +480,6 @@ jobs: config: ${{ fromJson(needs.canary-select.outputs.canary-config) }} secrets: INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} - MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} - MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: config: ${{ toJSON(matrix.config) }} # Only same-repository PRs owned by Klaud receive background priority. @@ -508,8 +506,6 @@ jobs: config: ${{ fromJson(needs.canary-select.outputs.multi-node-canary-config) }} secrets: INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} - MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} - MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: config: ${{ toJSON(matrix.config) }} klaud-run: *klaud-run @@ -545,8 +541,6 @@ jobs: config: ${{ fromJson(needs.setup.outputs.search-space-config).multi_node['1k1k'] }} secrets: INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} - MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} - MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: &multi-node-inputs config: ${{ toJSON(matrix.config) }} klaud-run: *klaud-run @@ -578,8 +572,6 @@ jobs: config: ${{ fromJson(needs.setup.outputs.search-space-config).multi_node['8k1k'] }} secrets: INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} - MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} - MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: *multi-node-inputs sweep-single-node-1k1k: @@ -602,8 +594,6 @@ jobs: config: ${{ fromJson((needs.canary-sweep.result == 'success' && needs.canary-select.outputs.remaining-search-space-config) || needs.setup.outputs.search-space-config).single_node['1k1k'] }} secrets: INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} - MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} - MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: &single-node-inputs config: ${{ toJSON(matrix.config) }} klaud-run: *klaud-run @@ -635,8 +625,6 @@ jobs: config: ${{ fromJson((needs.canary-sweep.result == 'success' && needs.canary-select.outputs.remaining-search-space-config) || needs.setup.outputs.search-space-config).single_node['8k1k'] }} secrets: INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} - MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} - MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: *single-node-inputs sweep-agentic: @@ -659,8 +647,6 @@ jobs: config: ${{ fromJson((needs.canary-sweep.result == 'success' && needs.canary-select.outputs.remaining-search-space-config) || needs.setup.outputs.search-space-config).single_node['agentic'] }} secrets: INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} - MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} - MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: config: ${{ toJSON(matrix.config) }} klaud-run: *klaud-run @@ -694,8 +680,6 @@ jobs: config: ${{ fromJson((needs.canary-multi-node-sweep.result == 'success' && needs.canary-select.outputs.remaining-search-space-config) || needs.setup.outputs.search-space-config).multi_node['agentic'] }} secrets: INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} - MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} - MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: config: ${{ toJSON(matrix.config) }} klaud-run: *klaud-run @@ -732,8 +716,6 @@ jobs: config: ${{ fromJson(needs.setup.outputs.search-space-config).evals }} secrets: INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} - MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} - MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: config: ${{ toJSON(matrix.config) }} klaud-run: *klaud-run @@ -770,8 +752,6 @@ jobs: config: ${{ fromJson(needs.setup.outputs.search-space-config).agentic_evals }} secrets: INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} - MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} - MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: config: ${{ toJSON(matrix.config) }} klaud-run: *klaud-run @@ -807,8 +787,6 @@ jobs: config: ${{ fromJson(needs.setup.outputs.search-space-config).multinode_evals }} secrets: INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} - MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} - MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: config: ${{ toJSON(matrix.config) }} klaud-run: *klaud-run @@ -850,8 +828,6 @@ jobs: config: ${{ fromJson(needs.setup.outputs.search-space-config).multinode_agentic_evals }} secrets: INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} - MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} - MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: config: ${{ toJSON(matrix.config) }} klaud-run: *klaud-run diff --git a/benchmarks/benchmark_lib.sh b/benchmarks/benchmark_lib.sh index c76f3e653f..1f30bd6fcf 100644 --- a/benchmarks/benchmark_lib.sh +++ b/benchmarks/benchmark_lib.sh @@ -2051,8 +2051,7 @@ run_lm_eval() { local top_p=1 local concurrent_requests="${EVAL_CONCURRENT_REQUESTS:-${CONC}}" check_env_vars concurrent_requests - # SWE-bench adds a repo-local task YAML, hence --include_path. --limit is - # passed only when EVAL_LIMIT requests a smoke-test slice. + # --limit is passed only when EVAL_LIMIT requests a smoke-test slice. local eval_limit="${EVAL_LIMIT:-}" local include_path="${EVAL_INCLUDE_PATH:-}" @@ -2432,7 +2431,6 @@ stage_eval_artifacts() { "$source_dir"/*_artifacts.tar.gz "$source_dir"/sample*.jsonl "$source_dir"/agent_preds.json - "$source_dir"/swebench_report_*.json "$source_dir"/predictions.jsonl "$source_dir"/*.traj* ) @@ -2449,349 +2447,6 @@ stage_eval_artifacts() { } -_install_swebench_agent_deps() { - python3 -m pip install -q --no-cache-dir --break-system-packages \ - 'mini-swe-agent==2.4.5' 'swe-rex[modal]==1.4.0' || true - _patch_swebench_agent || \ - echo "WARN: mini-swe-agent/swe-rex patches failed; sandbox cleanup, submission fallback, or non-interactive stdin handling may be degraded" >&2 -} - -_patch_swebench_agent() { - python3 "$(_eval_patches_dir)/patch_swebench_agent.py" -} - -_install_swebench_deps() { - check_env_vars SWEBENCH_USE_MODAL - # Patch anchors depend on SWE-bench 4.1.0. - python3 -m pip install -q --no-cache-dir --break-system-packages 'swebench==4.1.0' || true - if [ "${SWEBENCH_USE_MODAL}" = "true" ]; then - python3 -m pip install -q --no-cache-dir --break-system-packages modal || true - _patch_swebench_scoring || \ - echo "WARN: scoring patches failed; eval sandboxes will reserve 4 CPUs and idle-bill to their timeout" >&2 - fi -} - -_patch_swebench_scoring() { - python3 "$(_eval_patches_dir)/patch_swebench_scoring.py" -} - -# SWE-bench requires ~/.modal.toml despite env credentials. -_ensure_modal_credentials() { - check_env_vars IS_AGENTIC SWEBENCH_USE_MODAL - # Agentic generation uses swerex_modal sandboxes even when scoring is local. - if [ "${SWEBENCH_USE_MODAL}" != "true" ] \ - && [ "${IS_AGENTIC}" != "1" ] \ - && [ "${SCENARIO_TYPE:-}" != "agentic-coding" ]; then - return 0 - fi - # CI secrets may include whitespace or quotes. - if [ -n "${MODAL_TOKEN_ID:-}" ]; then - MODAL_TOKEN_ID=$(printf %s "$MODAL_TOKEN_ID" | tr -d "[:space:]\"'") - export MODAL_TOKEN_ID - fi - if [ -n "${MODAL_TOKEN_SECRET:-}" ]; then - MODAL_TOKEN_SECRET=$(printf %s "$MODAL_TOKEN_SECRET" | tr -d "[:space:]\"'") - export MODAL_TOKEN_SECRET - fi - if [ -f "${HOME:-}/.modal.toml" ]; then return 0; fi - if [ -n "${MODAL_TOKEN_ID:-}" ] && [ -n "${MODAL_TOKEN_SECRET:-}" ]; then - # Slurm may provide an unwritable HOME. - if [ -z "${HOME:-}" ] || ! mkdir -p "$HOME" 2>/dev/null || [ ! -w "$HOME" ]; then - export HOME=/tmp/inferencex-modal-home - mkdir -p "$HOME" - echo "[swebench] HOME remapped to $HOME for Modal credentials (original path missing or not writable)" - fi - printf '[default]\ntoken_id = "%s"\ntoken_secret = "%s"\nactive = true\n' \ - "$MODAL_TOKEN_ID" "$MODAL_TOKEN_SECRET" > "$HOME/.modal.toml" - chmod 600 "$HOME/.modal.toml" - echo "[swebench] wrote ~/.modal.toml from MODAL_TOKEN_ID/MODAL_TOKEN_SECRET env" - else - echo "WARN: Modal credentials required but no ~/.modal.toml and no MODAL_TOKEN_ID/MODAL_TOKEN_SECRET env; Modal sandboxes will fail authentication" >&2 - fi -} - - -_run_swebench_agentic_generation() { - check_env_vars \ - PORT SWEBENCH_AGENT_EXIT_GRACE SWEBENCH_AGENT_STEP_LIMIT SWEBENCH_AGENT_TIMEOUT \ - SWEBENCH_EXPECTED_INSTANCES SWEBENCH_SANDBOX_SWEEP SWEBENCH_WATCHDOG_POLL - local gen_dir="$1"; shift - local port="${PORT}" - while [[ $# -gt 0 ]]; do - case $1 in - --port) port="$2"; shift 2 ;; - *) shift ;; - esac - done - - _install_swebench_agent_deps - _ensure_modal_credentials - - # minisweagent logs before its config path. - local default_cfg - default_cfg=$(python3 -c 'import minisweagent, os; print(os.path.join(os.path.dirname(minisweagent.__file__), "config/benchmarks/swebench.yaml"))' 2>/dev/null | tail -n 1) - if [ ! -f "$default_cfg" ]; then - echo "ERROR: could not locate mini-swe-agent default swebench config (got: '${default_cfg}')" >&2 - return 1 - fi - - local cfg="$gen_dir/mini_swebench_overrides.yaml" - SWEBENCH_AGENT_PORT="$port" python3 - "$default_cfg" "$cfg" <<'PYGEN' -import os, sys, yaml -default_path, out_path = sys.argv[1], sys.argv[2] -d = yaml.safe_load(open(default_path)) or {} -d.setdefault("agent", {}) -step_limit = int(os.environ.get("SWEBENCH_AGENT_STEP_LIMIT", "250")) -guidance = f""" - - -- You have a hard budget of {step_limit} commands total. Plan: reproduce -> fix -> verify -> submit, finishing the submission well before the budget runs out. A correct fix that is never submitted scores ZERO. -- BEFORE submitting you MUST run the test(s) that cover the issue and confirm your fix makes them pass. Identify the failing test from the issue/PR, run it (e.g. `python -m pytest ::` or `python tests/runtests.py """ -it = d["agent"].get("instance_template", "") -d["agent"]["instance_template"] = it.rstrip() + guidance + "\n" -d["agent"]["step_limit"] = step_limit -d["agent"]["cost_limit"] = 0.0 -env = d.get("environment") or {} -env.update({ - "environment_class": "swerex_modal", - # Modal cold starts exceed the default timeout. - "startup_timeout": float(os.environ.get("SWEBENCH_AGENT_STARTUP_TIMEOUT", "900")), - "timeout": int(os.environ.get("SWEBENCH_AGENT_CMD_TIMEOUT", "300")), - # Limit billing if cleanup misses a sandbox. - "runtime_timeout": float(os.environ.get("SWEBENCH_AGENT_RUNTIME_TIMEOUT", "3600")), -}) -agent_cpu = os.environ.get("SWEBENCH_AGENT_SANDBOX_CPU", "") -if agent_cpu: - env["modal_sandbox_kwargs"] = {"cpu": float(agent_cpu)} -d["environment"] = env -model_name = os.environ.get("MODEL_NAME") or os.environ.get("MODEL", "") -d["model"] = { - "model_name": f"openai/{model_name}", - "cost_tracking": "ignore_errors", - "model_kwargs": { - "api_base": f"http://0.0.0.0:{os.environ['SWEBENCH_AGENT_PORT']}/v1", - "api_key": "dummy", - "custom_llm_provider": "openai", - "temperature": 0.0, - }, -} -yaml.safe_dump(d, open(out_path, "w"), default_flow_style=False, sort_keys=False) -PYGEN - - case "${EVAL_LIMIT:-}" in - full|FULL|0) EVAL_LIMIT="" ;; - esac - if [ -n "${EVAL_LIMIT:-}" ] && [[ ! "$EVAL_LIMIT" =~ ^[1-9][0-9]*$ ]]; then - echo "ERROR: EVAL_LIMIT='${EVAL_LIMIT}' must be a positive integer, 'full', or 0" >&2 - return 1 - fi - local slice_args=() - if [ -n "${EVAL_LIMIT:-}" ]; then - slice_args=(--slice "0:${EVAL_LIMIT}") - fi - - export MSWEA_COST_TRACKING=ignore_errors - local expected="${EVAL_LIMIT:-${SWEBENCH_EXPECTED_INSTANCES}}" - local workers="${SWEBENCH_AGENT_WORKERS:-${CONC}}" - check_env_vars workers - echo "[swebench-agentic] mini-swe-agent: workers=${workers} step_limit=${SWEBENCH_AGENT_STEP_LIMIT} slice=${EVAL_LIMIT:-full} expected=$expected" - local agen_rc=0 - mini-extra swebench \ - -c "$cfg" \ - --subset lite --split test \ - --environment-class swerex_modal \ - "${slice_args[@]}" \ - -w "${workers}" \ - -o "$gen_dir/agent_out" & - local mini_pid=$! - # preds.json detects completion despite teardown hangs. - local preds_file="$gen_dir/agent_out/preds.json" - local deadline=$(( $(date +%s) + ${SWEBENCH_AGENT_TIMEOUT} )) - local grace_until=0 - local killed_after_complete=0 - while kill -0 "$mini_pid" 2>/dev/null; do - if [ "$(date +%s)" -ge "$deadline" ]; then - echo "ERROR: generation exceeded SWEBENCH_AGENT_TIMEOUT (${SWEBENCH_AGENT_TIMEOUT}s); killing mini-extra" >&2 - kill "$mini_pid" 2>/dev/null; sleep 5; kill -9 "$mini_pid" 2>/dev/null - agen_rc=124 - break - fi - local done_count - done_count=$(python3 -c 'import json,sys; print(len(json.load(open(sys.argv[1]))))' "$preds_file" 2>/dev/null || echo 0) - if [ "${done_count:-0}" -ge "$expected" ]; then - if [ "$grace_until" -eq 0 ]; then - grace_until=$(( $(date +%s) + ${SWEBENCH_AGENT_EXIT_GRACE} )) - echo "[swebench-agentic] all $expected predictions written; waiting ${SWEBENCH_AGENT_EXIT_GRACE}s for mini-extra to exit" - elif [ "$(date +%s)" -ge "$grace_until" ]; then - echo "WARN: mini-extra hung after completing all instances; killing (known hang-on-exit)" >&2 - kill "$mini_pid" 2>/dev/null; sleep 5; kill -9 "$mini_pid" 2>/dev/null - killed_after_complete=1 - break - fi - fi - sleep "${SWEBENCH_WATCHDOG_POLL}" - done - wait "$mini_pid" 2>/dev/null - local wait_rc=$? - if [ "$killed_after_complete" -eq 1 ]; then - agen_rc=0 - elif [ "$agen_rc" -eq 0 ] && [ "$wait_rc" -ne 0 ]; then - agen_rc=$wait_rc - fi - # Isolate sweeps to avoid killing unrelated sandboxes. - [ "${SWEBENCH_SANDBOX_SWEEP}" = "1" ] && python3 - <<'PYSWEEP' || true -try: - import os - import modal - name = os.environ.get("SWEBENCH_MODAL_APP_NAME", "infx-evals-swe") - app = modal.App.lookup(name) - n = 0 - for sb in modal.Sandbox.list(app_id=app.app_id): - try: - sb.terminate() - n += 1 - except Exception as e: - print(f"[swebench-agentic] sweep: could not terminate {sb.object_id}: {e}") - print(f"[swebench-agentic] sandbox sweep ({name}): terminated {n} lingering sandbox(es)") -except Exception as e: - print(f"[swebench-agentic] sandbox sweep skipped: {e}") -PYSWEEP - if [ "$agen_rc" -ne 0 ]; then - # Partial runs may still be scoreable. - local salvage_count - salvage_count=$(python3 -c 'import json,sys; print(len(json.load(open(sys.argv[1]))))' "$gen_dir/agent_out/preds.json" 2>/dev/null || echo 0) - if [ "${salvage_count:-0}" -gt 0 ]; then - echo "WARN: generation exited rc=$agen_rc but $salvage_count/$expected predictions exist; scoring the partial set" >&2 - else - echo "ERROR: agentic generation (mini-swe-agent) failed with $agen_rc" >&2 - return "$agen_rc" - fi - fi - if [ ! -s "$gen_dir/agent_out/preds.json" ]; then - echo "ERROR: agentic generation produced no preds.json" >&2 - return 1 - fi -} - -run_swebench_eval() { - check_env_vars \ - SWEBENCH_EVAL_TIMEOUT SWEBENCH_MAX_WORKERS SWEBENCH_SCORE_TIMEOUT SWEBENCH_SKIP_SCORE \ - SWEBENCH_USE_MODAL SWEBENCH_GEN_MODE - local out_dir="${EVAL_RESULT_DIR:-$(mktemp -d /tmp/eval_out-XXXXXX)}" - local task_name="${SWEBENCH_TASK_NAME:-swebench_lite}" - export EVAL_SUITE="${EVAL_SUITE:-$task_name}" - local gen_dir - gen_dir=$(mktemp -d /tmp/swebench_gen-XXXXXX) - - # Generation and scoring must share a dataset. - local yaml_path="${EVAL_TASKS_DIR:-infx/evals/${task_name}.yaml}" - local dataset - dataset=$(awk '/^dataset_path:[[:space:]]/{print $2; exit}' "$yaml_path" 2>/dev/null) - if [ -z "$dataset" ]; then - echo "ERROR: could not read dataset_path from ${yaml_path}" >&2 - rm -rf "$gen_dir" 2>/dev/null || true - return 1 - fi - if [ -n "${SWEBENCH_DATASET:-}" ] && [ "${SWEBENCH_DATASET}" != "$dataset" ]; then - echo "ERROR: SWEBENCH_DATASET='${SWEBENCH_DATASET}' disagrees with ${yaml_path} dataset_path='${dataset}'." >&2 - echo " Generation and scoring must use the same dataset; edit the YAML or unset SWEBENCH_DATASET." >&2 - rm -rf "$gen_dir" 2>/dev/null || true - return 1 - fi - - local gen_mode="$SWEBENCH_GEN_MODE" - local score_input=() - if [ "$gen_mode" = "agentic" ]; then - # mini-extra supports only SWE-bench Lite. - case "$dataset" in - *SWE-bench_Lite|*SWE-bench_Lite/*) ;; - *) - echo "ERROR: agentic generation only produces SWE-bench_Lite instances, but ${yaml_path} dataset_path='${dataset}' is not Lite." >&2 - echo " Use gen_mode=single-shot for other datasets, or point the YAML at SWE-bench_Lite." >&2 - rm -rf "$gen_dir" 2>/dev/null || true - return 1 - ;; - esac - _run_swebench_agentic_generation "$gen_dir" "$@" || { - local agen_rc=$? - rm -rf "$gen_dir" 2>/dev/null || true - return "$agen_rc" - } - score_input=(--predictions-file "$gen_dir/agent_out/preds.json") - mkdir -p "$out_dir" - cp -f "$gen_dir/agent_out/preds.json" "$out_dir/agent_preds.json" 2>/dev/null || true - find "$gen_dir/agent_out" -name "*.traj*" -exec cp -f {} "$out_dir/" \; 2>/dev/null || true - else - local prev_tasks_dir="${EVAL_TASKS_DIR:-}" - local prev_include_path="${EVAL_INCLUDE_PATH:-}" - export EVAL_TASKS_DIR="$task_name" - export EVAL_INCLUDE_PATH="$(dirname "$yaml_path")" - local gen_rc=0 - run_lm_eval "$@" --results-dir "$gen_dir" || gen_rc=$? - export EVAL_TASKS_DIR="$prev_tasks_dir" - export EVAL_INCLUDE_PATH="$prev_include_path" - if [ "$gen_rc" -ne 0 ]; then - echo "ERROR: swebench generation (lm-eval) failed with $gen_rc" >&2 - rm -rf "$gen_dir" 2>/dev/null || true - return "$gen_rc" - fi - - mkdir -p "$out_dir" - find "$gen_dir" -name 'samples_*.jsonl' -exec cp -f {} "$out_dir"/ \; 2>/dev/null || true - score_input=(--samples-dir "$gen_dir") - fi - export EVAL_RESULT_DIR="$out_dir" - - local lm_eval_version - lm_eval_version=$(python3 -c 'import lm_eval; print(lm_eval.__version__)' 2>/dev/null || echo unknown) - - if [ "${SWEBENCH_SKIP_SCORE}" = "true" ]; then - local skip_rc=0 - python3 -m infx.evals.swebench_score \ - "${score_input[@]}" --out-dir "$out_dir" \ - --model-name "${MODEL_NAME:-$MODEL}" --task-name "$task_name" \ - --predictions-only || skip_rc=$? - echo "SWEBENCH_SKIP_SCORE=true: staged predictions only (no resolved-rate)." >&2 - rm -rf "$gen_dir" 2>/dev/null || true - return "$skip_rc" - fi - - if [ "${INFERENCEX_SWEBENCH_RUNTIME_READY:-false}" != "true" ]; then - _install_swebench_deps - export INFERENCEX_SWEBENCH_RUNTIME_READY=true - fi - _ensure_modal_credentials - local score_rc=0 - local ns_args=() - if [ "${SWEBENCH_NAMESPACE+set}" = "set" ]; then ns_args=(--namespace "$SWEBENCH_NAMESPACE"); fi - local modal_args=() - if [ "${SWEBENCH_USE_MODAL}" = "true" ]; then modal_args=(--modal); fi - local itimeout_args=(--instance-timeout "${SWEBENCH_EVAL_TIMEOUT}") - # Avoid holding the GPU on scoring stalls. - timeout "${SWEBENCH_SCORE_TIMEOUT}" \ - python3 -m infx.evals.swebench_score \ - "${score_input[@]}" \ - --out-dir "$out_dir" \ - --model-name "${MODEL_NAME:-$MODEL}" \ - --task-name "$task_name" \ - --dataset-name "$dataset" \ - --max-workers "${SWEBENCH_MAX_WORKERS}" \ - --lm-eval-version "$lm_eval_version" \ - "${modal_args[@]}" \ - "${itimeout_args[@]}" \ - "${ns_args[@]}" \ - || score_rc=$? - rm -rf "$gen_dir" 2>/dev/null || true - if [ "$score_rc" -ne 0 ]; then - echo "ERROR: swebench scoring failed with $score_rc" >&2 - return "$score_rc" - fi -} - _wait_for_openai_chat_route() { check_env_vars EVAL_ENDPOINT_READY_TIMEOUT_SECONDS EVAL_MODEL_STABILIZATION_SECONDS PORT local port="${PORT}" @@ -3037,7 +2692,6 @@ run_eval() { local eval_rc=0 case "$framework" in lm-eval|lm_eval) run_lm_eval "${forwarded[@]}" || eval_rc=$? ;; - swebench) run_swebench_eval "${forwarded[@]}" || eval_rc=$? ;; kimi-vendor) run_kimi_vendor_eval "${forwarded[@]}" || eval_rc=$? ;; minimax-vendor) run_minimax_vendor_eval "${forwarded[@]}" || eval_rc=$? ;; bfcl) run_bfcl_eval "${forwarded[@]}" || eval_rc=$? ;; diff --git a/benchmarks/multi_node/amd_utils/job.slurm b/benchmarks/multi_node/amd_utils/job.slurm index cecc989a0f..93fe1136bd 100755 --- a/benchmarks/multi_node/amd_utils/job.slurm +++ b/benchmarks/multi_node/amd_utils/job.slurm @@ -17,7 +17,7 @@ check_env_vars \ DECODE_ENABLE_DP PREFILL_TP_SIZE DECODE_TP_SIZE DECODE_MTP_SIZE ROUTER_TYPE \ ROUTER_PORT PROXY_PING_PORT HEADNODE_PORT SERVER_PORT DRY_RUN \ BENCHMARK_LOGS_DIR KEEP_CONTAINERS RUN_EVAL EVAL_ONLY EVAL_FRAMEWORK \ - IS_MULTINODE SWEBENCH_USE_MODAL IS_AGENTIC KV_OFFLOADING PREFILL_EP \ + IS_MULTINODE IS_AGENTIC KV_OFFLOADING PREFILL_EP \ PREFILL_DP_ATTN DECODE_EP DECODE_DP_ATTN DURATION ENABLE_METRICS \ PREFILL_ROUTER_POLICY DECODE_ROUTER_POLICY DISAGG @@ -261,7 +261,6 @@ export EVAL_ONLY export EVAL_CONC="${EVAL_CONC:-}" export EVAL_FRAMEWORK export EVAL_SUITE="${EVAL_SUITE:-}" -export SWEBENCH_GEN_MODE="${SWEBENCH_GEN_MODE:-}" export FRAMEWORK="${FRAMEWORK:-}" export PRECISION="${PRECISION:-}" export MODEL_PREFIX="${MODEL_PREFIX:-}" @@ -269,9 +268,6 @@ export RUNNER_TYPE="${RUNNER_TYPE:-}" export RESULT_FILENAME="${RESULT_FILENAME:-}" export SPEC_DECODING="${SPEC_DECODING:-}" export IS_MULTINODE -export SWEBENCH_USE_MODAL -export MODAL_TOKEN_ID="${MODAL_TOKEN_ID:-}" -export MODAL_TOKEN_SECRET="${MODAL_TOKEN_SECRET:-}" export HF_TOKEN="${HF_TOKEN:-}" export SCENARIO_TYPE="${SCENARIO_TYPE:-}" export EVAL_LIMIT="${EVAL_LIMIT:-}" @@ -466,7 +462,6 @@ DOCKER_ENV_COMMON=( -e EVAL_FRAMEWORK=\$EVAL_FRAMEWORK -e EVAL_LIMIT=\$EVAL_LIMIT -e EVAL_SUITE=\$EVAL_SUITE - -e SWEBENCH_GEN_MODE=\$SWEBENCH_GEN_MODE -e FRAMEWORK=\$FRAMEWORK -e PRECISION=\$PRECISION -e MODEL_PREFIX=\$MODEL_PREFIX @@ -534,11 +529,6 @@ DOCKER_ENV_COMMON=( -e DECODE_MTP_SIZE=\$DECODE_MTP_SIZE -e IS_MULTINODE=\$IS_MULTINODE -e DRY_RUN=\${DRY_RUN} - # SWE-bench agentic eval runs inside this container and needs Modal/HF - # credentials to launch sandboxes and download datasets. - -e SWEBENCH_USE_MODAL=\${SWEBENCH_USE_MODAL} - -e MODAL_TOKEN_ID=\${MODAL_TOKEN_ID:-} - -e MODAL_TOKEN_SECRET=\${MODAL_TOKEN_SECRET:-} -e HF_TOKEN=\${HF_TOKEN:-} -e SCENARIO_TYPE=\${SCENARIO_TYPE:-} -e \"EVAL_LIMIT=\${EVAL_LIMIT:-}\" diff --git a/benchmarks/multi_node/amd_utils/submit.sh b/benchmarks/multi_node/amd_utils/submit.sh index c4a5ff8c8f..f24e59840d 100755 --- a/benchmarks/multi_node/amd_utils/submit.sh +++ b/benchmarks/multi_node/amd_utils/submit.sh @@ -54,7 +54,7 @@ check_env_vars \ PREFILL_DP_ATTN PREFILL_NUM_WORKERS PREFILL_PP_SIZE PREFILL_DCP_SIZE PREFILL_PCP_SIZE \ DECODE_EP DECODE_DP_ATTN DECODE_NUM_WORKERS DECODE_PP_SIZE DECODE_DCP_SIZE \ DECODE_PCP_SIZE DECODE_MTP_SIZE BENCH_NUM_PROMPTS_MULTIPLIER DRY_RUN RUN_EVAL \ - EVAL_ONLY EVAL_FRAMEWORK IS_MULTINODE SWEBENCH_USE_MODAL BENCHMARK_LOGS_DIR \ + EVAL_ONLY EVAL_FRAMEWORK IS_MULTINODE BENCHMARK_LOGS_DIR \ KEEP_CONTAINERS ROUTER_TYPE ROUTER_PORT PROXY_PING_PORT HEADNODE_PORT \ SERVER_PORT IS_AGENTIC KV_OFFLOADING @@ -132,7 +132,6 @@ export EVAL_ONLY export EVAL_CONC="${EVAL_CONC:-}" export EVAL_FRAMEWORK export EVAL_SUITE="${EVAL_SUITE:-}" -export SWEBENCH_GEN_MODE="${SWEBENCH_GEN_MODE:-}" export FRAMEWORK="${FRAMEWORK:-}" export PRECISION="${PRECISION:-}" export MODEL_PREFIX="${MODEL_PREFIX:-}" @@ -140,9 +139,6 @@ export RUNNER_TYPE="${RUNNER_TYPE:-}" export RESULT_FILENAME="${RESULT_FILENAME:-}" export SPEC_DECODING="${SPEC_DECODING:-}" export IS_MULTINODE -export SWEBENCH_USE_MODAL -export MODAL_TOKEN_ID="${MODAL_TOKEN_ID:-}" -export MODAL_TOKEN_SECRET="${MODAL_TOKEN_SECRET:-}" export HF_TOKEN="${HF_TOKEN:-}" export SCENARIO_TYPE="${SCENARIO_TYPE:-}" export EVAL_LIMIT="${EVAL_LIMIT:-}" diff --git a/benchmarks/multi_node/llm-d/job.slurm b/benchmarks/multi_node/llm-d/job.slurm index fe36301b6a..0195cebf5f 100644 --- a/benchmarks/multi_node/llm-d/job.slurm +++ b/benchmarks/multi_node/llm-d/job.slurm @@ -162,10 +162,6 @@ exec docker run --rm \ -e EVAL_FRAMEWORK=$EVAL_FRAMEWORK \ -e EVAL_LIMIT=$EVAL_LIMIT \ -e EVAL_SUITE=$EVAL_SUITE \ - -e SWEBENCH_GEN_MODE=$SWEBENCH_GEN_MODE \ - -e SWEBENCH_USE_MODAL=$SWEBENCH_USE_MODAL \ - -e MODAL_TOKEN_ID \ - -e MODAL_TOKEN_SECRET \ -e IS_AGENTIC=$IS_AGENTIC \ -e SCENARIO_TYPE=$SCENARIO_TYPE \ -e FRAMEWORK=$FRAMEWORK \ @@ -213,7 +209,6 @@ elif [[ "$LLMD_CONTAINER_ENGINE" == "pyxis" ]]; then export BENCH_INPUT_LEN BENCH_OUTPUT_LEN BENCH_MAX_CONCURRENCY export BENCH_REQUEST_RATE BENCH_RANDOM_RANGE_RATIO BENCH_NUM_PROMPTS_MULTIPLIER export RUN_EVAL EVAL_ONLY EVAL_CONC EVAL_FRAMEWORK EVAL_LIMIT EVAL_SUITE - export SWEBENCH_GEN_MODE SWEBENCH_USE_MODAL MODAL_TOKEN_ID MODAL_TOKEN_SECRET export IS_AGENTIC SCENARIO_TYPE FRAMEWORK PRECISION MODEL_PREFIX export RUNNER_TYPE RESULT_FILENAME SPEC_DECODING IS_MULTINODE CONFIG_FILE @@ -225,7 +220,6 @@ elif [[ "$LLMD_CONTAINER_ENGINE" == "pyxis" ]]; then PYXIS_ENV_LIST+=",BENCH_INPUT_LEN,BENCH_OUTPUT_LEN,BENCH_MAX_CONCURRENCY" PYXIS_ENV_LIST+=",BENCH_REQUEST_RATE,BENCH_RANDOM_RANGE_RATIO,BENCH_NUM_PROMPTS_MULTIPLIER" PYXIS_ENV_LIST+=",RUN_EVAL,EVAL_ONLY,EVAL_CONC,EVAL_FRAMEWORK,EVAL_LIMIT,EVAL_SUITE" - PYXIS_ENV_LIST+=",SWEBENCH_GEN_MODE,SWEBENCH_USE_MODAL,MODAL_TOKEN_ID,MODAL_TOKEN_SECRET" PYXIS_ENV_LIST+=",IS_AGENTIC,SCENARIO_TYPE,FRAMEWORK,PRECISION,MODEL_PREFIX" PYXIS_ENV_LIST+=",RUNNER_TYPE,RESULT_FILENAME,SPEC_DECODING,IS_MULTINODE,CONFIG_FILE" diff --git a/benchmarks/multi_node/llm-d/submit.sh b/benchmarks/multi_node/llm-d/submit.sh index ab1098a116..8f2f2d8e4f 100755 --- a/benchmarks/multi_node/llm-d/submit.sh +++ b/benchmarks/multi_node/llm-d/submit.sh @@ -22,7 +22,7 @@ check_env_vars \ SLURM_ACCOUNT SLURM_PARTITION TIME_LIMIT MODEL_PATH MODEL_NAME \ CONTAINER_IMAGE RUNNER_NAME BENCHMARK_LOGS_DIR GPUS_PER_NODE PREFILL_WORKERS \ DECODE_WORKERS BENCH_NUM_PROMPTS_MULTIPLIER RUN_EVAL EVAL_ONLY EVAL_FRAMEWORK \ - SWEBENCH_USE_MODAL IS_AGENTIC FRAMEWORK SPEC_DECODING IS_MULTINODE + IS_AGENTIC FRAMEWORK SPEC_DECODING IS_MULTINODE if [[ $# -ne 7 ]]; then echo "Usage: submit.sh prefill_nodes decode_nodes isl osl concurrencies request_rate random_range_ratio" >&2 exit 1 @@ -73,10 +73,6 @@ export EVAL_CONC="${EVAL_CONC:-}" export EVAL_FRAMEWORK export EVAL_LIMIT="${EVAL_LIMIT:-}" export EVAL_SUITE="${EVAL_SUITE:-}" -export SWEBENCH_GEN_MODE="${SWEBENCH_GEN_MODE:-}" -export SWEBENCH_USE_MODAL -export MODAL_TOKEN_ID="${MODAL_TOKEN_ID:-}" -export MODAL_TOKEN_SECRET="${MODAL_TOKEN_SECRET:-}" export IS_AGENTIC export SCENARIO_TYPE="${SCENARIO_TYPE:-}" export FRAMEWORK diff --git a/benchmarks/runtime_settings.sh b/benchmarks/runtime_settings.sh index a369898f8b..83b7c250ec 100644 --- a/benchmarks/runtime_settings.sh +++ b/benchmarks/runtime_settings.sh @@ -3,16 +3,6 @@ # Workflow-owned common settings, loaded before recipe-specific overrides. # Benchmark receivers validate these inputs; they do not choose missing values. export OPENAI_API_KEY='EMPTY' -export SWEBENCH_EXPECTED_INSTANCES='300' -export SWEBENCH_AGENT_STEP_LIMIT='250' -export SWEBENCH_AGENT_TIMEOUT='21600' -export SWEBENCH_AGENT_EXIT_GRACE='300' -export SWEBENCH_WATCHDOG_POLL='30' -export SWEBENCH_SANDBOX_SWEEP='1' -export SWEBENCH_SKIP_SCORE='false' -export SWEBENCH_EVAL_TIMEOUT='900' -export SWEBENCH_SCORE_TIMEOUT='7200' -export SWEBENCH_MAX_WORKERS='4' export EVAL_ENDPOINT_READY_TIMEOUT_SECONDS='1800' export EVAL_MODEL_STABILIZATION_SECONDS='30' export AIPERF_FAILED_REQUEST_THRESHOLD='0.10' @@ -32,4 +22,4 @@ export SGLANG_TORCH_PROFILER_DIR='/workspace' export VLLM_TORCH_PROFILER_DIR='/workspace' # Explicitly forward these settings across container boundaries. -export INFERENCEX_RUNTIME_ENV_VARS="OPENAI_API_KEY SWEBENCH_EXPECTED_INSTANCES SWEBENCH_AGENT_STEP_LIMIT SWEBENCH_AGENT_TIMEOUT SWEBENCH_AGENT_EXIT_GRACE SWEBENCH_WATCHDOG_POLL SWEBENCH_SANDBOX_SWEEP SWEBENCH_SKIP_SCORE SWEBENCH_EVAL_TIMEOUT SWEBENCH_SCORE_TIMEOUT SWEBENCH_MAX_WORKERS EVAL_ENDPOINT_READY_TIMEOUT_SECONDS EVAL_MODEL_STABILIZATION_SECONDS AIPERF_FAILED_REQUEST_THRESHOLD AIPERF_LIVE_FAILED_REQUEST_THRESHOLD AIPERF_TRACE_IDLE_GAP_CAP_SECONDS AIPERF_PYTHON_VERSION AIPERF_WARMUP_REQUESTS_PER_LANE AIPERF_DATASET_WEKA_LIVE_ASSISTANT_RESPONSES AGENTIC_WARMUP_GRACE_PERIOD AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS AIPERF_EXPERIMENTAL_FAST AIPERF_UNSAFE_OVERRIDE ENABLE_AGENTX_POWER REQUIRE_POWER VLLM_ENGINE_READY_TIMEOUT_S SGLANG_TORCH_PROFILER_DIR VLLM_TORCH_PROFILER_DIR" +export INFERENCEX_RUNTIME_ENV_VARS="OPENAI_API_KEY EVAL_ENDPOINT_READY_TIMEOUT_SECONDS EVAL_MODEL_STABILIZATION_SECONDS AIPERF_FAILED_REQUEST_THRESHOLD AIPERF_LIVE_FAILED_REQUEST_THRESHOLD AIPERF_TRACE_IDLE_GAP_CAP_SECONDS AIPERF_PYTHON_VERSION AIPERF_WARMUP_REQUESTS_PER_LANE AIPERF_DATASET_WEKA_LIVE_ASSISTANT_RESPONSES AGENTIC_WARMUP_GRACE_PERIOD AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS AIPERF_EXPERIMENTAL_FAST AIPERF_UNSAFE_OVERRIDE ENABLE_AGENTX_POWER REQUIRE_POWER VLLM_ENGINE_READY_TIMEOUT_S SGLANG_TORCH_PROFILER_DIR VLLM_TORCH_PROFILER_DIR" diff --git a/benchmarks/single_node/srt_eval.sh b/benchmarks/single_node/srt_eval.sh index 4fe17dafcb..33591203ed 100644 --- a/benchmarks/single_node/srt_eval.sh +++ b/benchmarks/single_node/srt_eval.sh @@ -16,11 +16,6 @@ eval_args=() if [[ "$IS_AGENTIC" != 1 ]]; then check_env_vars MAX_MODEL_LEN eval_args=(--framework lm-eval) -elif [[ "${MODEL_PREFIX:-}" == glm5.2 ]]; then - # GLM-5.2's template defaults to maximum reasoning effort without - # chat_template_kwargs, which mini-swe-agent never passes; the heavy thinking - # exhausts the shared step budget. The recipe env does not reach post-eval. - export SWEBENCH_AGENT_STEP_LIMIT=150 fi export PORT="${1##*:}" if [[ ! "$PORT" =~ ^[1-9][0-9]*$ || "$IS_MULTINODE" != false ]]; then diff --git a/docs/architecture.md b/docs/architecture.md index 35f081120f..5f51f57ab4 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -216,7 +216,7 @@ A launcher under [`runners/`](../runners/) adapts logical job metadata to one ph - choose a single-node script, a multi-node wrapper, or a checked-in external recipe. - pass the workflow environment into the runtime container or allocation. -Benchmark scripts under [`benchmarks/`](../benchmarks/) own the actual engine and client commands. Most source [`benchmarks/benchmark_lib.sh`](../benchmarks/benchmark_lib.sh), which centralizes server readiness, the serving benchmark client, GPU monitoring, lm-eval, SWE-bench, AgentX replay, and stable output helpers. +Benchmark scripts under [`benchmarks/`](../benchmarks/) own the actual engine and client commands. Most source [`benchmarks/benchmark_lib.sh`](../benchmarks/benchmark_lib.sh), which centralizes server readiness, the serving benchmark client, GPU monitoring, lm-eval, AgentX replay, and stable output helpers. The boundary is intentional. A master config remains portable and reviewable. Machine paths, scheduler details, and container mechanics stay close to the fleet that requires them. Framework flags stay close to the benchmark recipe where they can be tested against that engine. @@ -265,7 +265,7 @@ Test builders with small, independently worked examples and read-only inputs. Fo ### Eval and AgentX outputs -For eval-only jobs, throughput output is not required. The workflow instead requires at least one `results*.json`. For jobs marked to run eval, uploads may contain `meta_env.json`, `results*.json`, `sample*.jsonl`, SWE-bench predictions and reports, and trajectory files. [`infx/evals/validate_scores.py`](../infx/evals/validate_scores.py) checks produced eval scores. +For eval-only jobs, throughput output is not required. The workflow instead requires at least one `results*.json`. For jobs marked to run eval, uploads may contain `meta_env.json`, `results*.json`, `sample*.jsonl`, and trajectory files. [`infx/evals/validate_scores.py`](../infx/evals/validate_scores.py) checks produced eval scores. [`infx.results.evals`](../infx/results/evals.py) provides `extract_metrics` for loaded eval JSON and `build_rows` for collector output. Both accept explicit inputs without file I/O or input mutation. The builder applies metadata defaults and primary-score precedence, retaining failed evaluations as diagnostic rows. The CLI owns file discovery, concurrency eligibility, reporting, and artifact writes. diff --git a/docs/architecture_zh.md b/docs/architecture_zh.md index aa0d024775..4453ea6282 100644 --- a/docs/architecture_zh.md +++ b/docs/architecture_zh.md @@ -216,7 +216,7 @@ bash ./runners/launch_${RUNNER_NAME%%_*}.sh - 选择单节点脚本、多节点包装器或已签入的外部方案; - 将工作流环境传入运行时容器或分配环境。 -[`benchmarks/`](../benchmarks/) 下的基准测试脚本负责实际的引擎和客户端命令。大多数脚本会引入 [`benchmarks/benchmark_lib.sh`](../benchmarks/benchmark_lib.sh),后者集中处理服务器就绪检查、服务基准测试客户端、GPU 监控、lm-eval、SWE-bench、AgentX 重放和稳定输出辅助函数。 +[`benchmarks/`](../benchmarks/) 下的基准测试脚本负责实际的引擎和客户端命令。大多数脚本会引入 [`benchmarks/benchmark_lib.sh`](../benchmarks/benchmark_lib.sh),后者集中处理服务器就绪检查、服务基准测试客户端、GPU 监控、lm-eval、AgentX 重放和稳定输出辅助函数。 这一边界是有意设计的。主配置保持可移植且便于审查。机器路径、调度器细节和容器机制保持靠近需要它们的机群。框架标志保持靠近基准测试方案,以便针对相应引擎进行测试。 @@ -265,7 +265,7 @@ result = build_result(records, profile, server_metrics, runtime_env, ### 评测与 AgentX 输出 -对于仅评测作业,不要求吞吐量输出。工作流改为要求至少存在一个 `results*.json`。对于标记为运行评测的作业,上传内容可能包含 `meta_env.json`、`results*.json`、`sample*.jsonl`、SWE-bench 预测和报告以及轨迹文件。[`infx/evals/validate_scores.py`](../infx/evals/validate_scores.py) 会检查生成的评测分数。 +对于仅评测作业,不要求吞吐量输出。工作流改为要求至少存在一个 `results*.json`。对于标记为运行评测的作业,上传内容可能包含 `meta_env.json`、`results*.json`、`sample*.jsonl` 以及轨迹文件。[`infx/evals/validate_scores.py`](../infx/evals/validate_scores.py) 会检查生成的评测分数。 [`infx.results.evals`](../infx/results/evals.py) 提供 `extract_metrics`,用于解析已加载的评测 JSON,并提供 `build_rows`,用于构建收集器输出。两者均接收显式输入,不执行文件 I/O,也不修改输入。构建函数应用元数据默认值和主分数优先级,并将失败评测保留为诊断行。CLI 负责文件查找、并发数资格筛选、报告输出和工件写入。 diff --git a/docs/ci-procedures.md b/docs/ci-procedures.md index b3caa3802e..7f12e5c5fc 100644 --- a/docs/ci-procedures.md +++ b/docs/ci-procedures.md @@ -54,7 +54,7 @@ This page was authored from branch commit `0c28706b33d4a796b82f6f9c3594c19c46365 - The branch-local [`e2e-tests.yml`](../.github/workflows/e2e-tests.yml) requires `generate-cli-command` and hard-codes each matrix to `fail-fast: false`. The audited [`origin/main` version](https://github.com/SemiAnalysisAI/InferenceX/blob/de493d8597035e6692833de6189b567887968460/.github/workflows/e2e-tests.yml) makes that command conditionally optional and adds trusted-changelog dispatch, `fail-fast`, and power-validation inputs. - The branch-local [`run-sweep.yml`](../.github/workflows/run-sweep.yml) lacks the same-repository-head guard that `origin/main` adds before PR GPU setup. The [`trusted-external-sweep.yml` workflow](https://github.com/SemiAnalysisAI/InferenceX/blob/de493d8597035e6692833de6189b567887968460/.github/workflows/trusted-external-sweep.yml) exists on that `origin/main` snapshot but not on this branch. Do not infer external-fork secret or GPU behavior from the branch-local workflow. -- Agentic eval comments in the branch-local generator identify SWE-bench, while the audited `origin/main` generator identifies GSM8K. Inspect the target ref before describing the agentic dataset selected by `all-evals` or `evals-only`. +- Agentic eval comments in the branch-local generator may identify a different dataset than the audited `origin/main` generator. Inspect the target ref before describing the agentic dataset selected by `all-evals` or `evals-only`. A `workflow_dispatch` request uses the workflow definition from its dispatch `--ref`. The separate `inputs.ref` controls what the jobs check out. Before using inputs beyond the common example below, inspect the deployed definition: diff --git a/docs/ci-procedures_zh.md b/docs/ci-procedures_zh.md index 1378fc731a..f4e7d41e64 100644 --- a/docs/ci-procedures_zh.md +++ b/docs/ci-procedures_zh.md @@ -54,7 +54,7 @@ - 分支本地的 [`e2e-tests.yml`](../.github/workflows/e2e-tests.yml) 要求提供 `generate-cli-command`,并将各矩阵硬编码为 `fail-fast: false`。审计过的 [`origin/main` 版本](https://github.com/SemiAnalysisAI/InferenceX/blob/de493d8597035e6692833de6189b567887968460/.github/workflows/e2e-tests.yml) 仅在特定条件下要求该命令,并新增受信任 Changelog 派发、`fail-fast` 与功耗验证输入。 - 分支本地的 [`run-sweep.yml`](../.github/workflows/run-sweep.yml) 缺少 `origin/main` 在 PR GPU Setup 之前新增的“Head 仓库必须与当前仓库相同”保护。该 `origin/main` 快照存在 [`trusted-external-sweep.yml` Workflow](https://github.com/SemiAnalysisAI/InferenceX/blob/de493d8597035e6692833de6189b567887968460/.github/workflows/trusted-external-sweep.yml),本分支则没有。不要根据分支本地 Workflow 推断外部 Fork 的 Secret 或 GPU 行为。 -- 分支本地生成器的 Agentic Eval 注释指向 SWE-bench,审计过的 `origin/main` 生成器则指向 GSM8K。在描述 `all-evals` 或 `evals-only` 选择的 Agentic 数据集之前,必须检查目标 Ref。 +- 分支本地生成器的 Agentic Eval 注释可能指向与审计过的 `origin/main` 生成器不同的数据集。在描述 `all-evals` 或 `evals-only` 选择的 Agentic 数据集之前,必须检查目标 Ref。 `workflow_dispatch` 请求使用其派发 `--ref` 中的 Workflow 定义;单独的 `inputs.ref` 控制 Job Checkout 的内容。在使用下方公共示例以外的输入前,应检查已部署定义: diff --git a/docs/eval-agentx-procedures.md b/docs/eval-agentx-procedures.md index 48d480d62a..26295c6155 100644 --- a/docs/eval-agentx-procedures.md +++ b/docs/eval-agentx-procedures.md @@ -185,7 +185,7 @@ gh run download "$RUN_ID" --repo SemiAnalysisAI/InferenceX \ --pattern 'eval_*' --dir ./evals/raw ``` -Retain `meta_env.json`, `results*.json`, and `sample*.jsonl`. Agentic SWE-bench additionally uploads `agent_preds.json`, `predictions.jsonl`, `swebench_report_*.json`, and trajectory files in the single-node template. The aggregate is a navigation aid, not a substitute for raw samples and batch completeness. +Retain `meta_env.json`, `results*.json`, and `sample*.jsonl`. The aggregate is a navigation aid, not a substitute for raw samples and batch completeness. ## 7. Run AgentX: fast feedback versus canonical evidence @@ -215,19 +215,6 @@ gh workflow run e2e-tests.yml --repo SemiAnalysisAI/InferenceX --ref "$REF" \ -f agentx-fast=true ``` -Targeted AgentX SWE-bench smoke eval (first ten instances, real agentic generation): - -```bash -gh workflow run e2e-tests.yml --repo SemiAnalysisAI/InferenceX --ref "$REF" \ - -f generate-cli-command='test-config --config-keys qwen3.5-fp8-b200-sglang-agentic --conc 1 --evals-only --config-files configs/nvidia-master.yaml' \ - -f test-name='swebench-smoke-qwen35-c1' \ - -f eval-framework=swebench \ - -f eval-limit='10' \ - -f swebench-gen-mode='agentic' -``` - -For a publishable SWE-bench score, omit `eval-limit`. Do not use `single-shot`, which is only a debugging escape hatch. SWE-bench generation/scoring controls and its `0.50` full-split threshold are documented next to the implementation in [`infx/evals/EVALS.md`](../infx/evals/EVALS.md#swe-bench-lite---framework-swebench). - Treat fast results as bring-up evidence, never as a replacement for the canonical candidate. A duration below 900 seconds or `AIPERF_UNSAFE_OVERRIDE=true` adds AIPerf's `--unsafe-override` and flags the submission invalid. Use it only for smoke diagnosis ([source](../benchmarks/benchmark_lib.sh#L2266-L2268)). After a fast run is healthy, run the exact candidate canonically before claiming benchmark success. ## 8. Preserve trace and run provenance diff --git a/docs/eval-agentx-procedures_zh.md b/docs/eval-agentx-procedures_zh.md index 1e18819ff1..3ae51f9bb3 100644 --- a/docs/eval-agentx-procedures_zh.md +++ b/docs/eval-agentx-procedures_zh.md @@ -183,7 +183,7 @@ gh run download "$RUN_ID" --repo SemiAnalysisAI/InferenceX \ --pattern 'eval_*' --dir ./evals/raw ``` -保留 `meta_env.json`、`results*.json` 和 `sample*.jsonl`。Agentic SWE-bench 在单节点模板中还会上传 `agent_preds.json`、`predictions.jsonl`、`swebench_report_*.json` 和 trajectory 文件。Aggregate 是导航工具,不能替代原始样本与 batch 完整性证据。 +保留 `meta_env.json`、`results*.json` 和 `sample*.jsonl`。Aggregate 是导航工具,不能替代原始样本与 batch 完整性证据。 ## 7. 运行 AgentX:快速反馈与 canonical 证据 @@ -213,19 +213,6 @@ gh workflow run e2e-tests.yml --repo SemiAnalysisAI/InferenceX --ref "$REF" \ -f agentx-fast=true ``` -目标 AgentX SWE-bench smoke eval(前十个 instance,真实 agentic generation): - -```bash -gh workflow run e2e-tests.yml --repo SemiAnalysisAI/InferenceX --ref "$REF" \ - -f generate-cli-command='test-config --config-keys qwen3.5-fp8-b200-sglang-agentic --conc 1 --evals-only --config-files configs/nvidia-master.yaml' \ - -f test-name='swebench-smoke-qwen35-c1' \ - -f eval-framework=swebench \ - -f eval-limit='10' \ - -f swebench-gen-mode='agentic' -``` - -要得到可发布的 SWE-bench 分数,省略 `eval-limit`;不要使用 `single-shot`,它只是诊断逃生选项。SWE-bench generation/scoring 控制项以及完整 split 的 `0.50` 阈值在实现旁的 [`infx/evals/EVALS.md`](../infx/evals/EVALS.md#swe-bench-lite---framework-swebench) 中说明。 - Fast 结果只能作为 bring-up 证据,绝不能替代 canonical candidate。小于 900 秒的 duration 或 `AIPERF_UNSAFE_OVERRIDE=true` 会添加 AIPerf 的 `--unsafe-override` 并将 submission 标记为无效;只能用于 smoke 诊断([源码](../benchmarks/benchmark_lib.sh#L2266-L2268))。Fast 运行健康后,必须对完全相同的 candidate 进行 canonical 运行,才能宣称 benchmark 成功。 ## 8. 保留 trace 与运行 provenance diff --git a/docs/index.md b/docs/index.md index 211b4252cb..1f901d5860 100644 --- a/docs/index.md +++ b/docs/index.md @@ -36,7 +36,7 @@ This is the mandatory low-context router for InferenceX work. Pick the one page | [`.github/AGENT_OPERATIONS.md`](../.github/AGENT_OPERATIONS.md) | Translation terms, sweep labels, dispatch, eval selection, power, metrics, and artifacts | | [`configs/CONFIGS.md`](../configs/CONFIGS.md) | Master-config schema, search spaces, runners, and topology fields | | [`.github/workflows/README.md`](../.github/workflows/README.md) | Generator examples, workflow operation, and reuse policy | -| [`infx/evals/EVALS.md`](../infx/evals/EVALS.md) | Eval task, execution, collection, validation, and SWE-bench contracts | +| [`infx/evals/EVALS.md`](../infx/evals/EVALS.md) | Eval task, execution, collection, and validation | | [`benchmarks/multi_node/srt-slurm-recipes/RECIPES.md`](../benchmarks/multi_node/srt-slurm-recipes/RECIPES.md) | Disaggregated recipe registration and master-config coupling | | [`utils/runner_setup/RUNNER_SETUP.md`](../utils/runner_setup/RUNNER_SETUP.md) | Runner provisioning and setup | | [`MODELS.md`](MODELS.md) | Supported models, hardware coverage, and naming | diff --git a/docs/index_zh.md b/docs/index_zh.md index aafaaa7491..1bdbd341dc 100644 --- a/docs/index_zh.md +++ b/docs/index_zh.md @@ -36,7 +36,7 @@ | [`.github/AGENT_OPERATIONS.md`](../.github/AGENT_OPERATIONS.md) | 翻译术语、扫描标签、派发、Eval 选择、功耗、指标与产物 | | [`configs/CONFIGS.md`](../configs/CONFIGS.md) | 主配置 Schema、搜索空间、Runner 与拓扑字段 | | [`.github/workflows/README.md`](../.github/workflows/README.md) | 生成器示例、Workflow 操作与复用政策 | -| [`infx/evals/EVALS.md`](../infx/evals/EVALS.md) | Eval 任务、执行、收集、校验与 SWE-bench 契约 | +| [`infx/evals/EVALS.md`](../infx/evals/EVALS.md) | Eval 任务、执行、收集与校验 | | [`benchmarks/multi_node/srt-slurm-recipes/RECIPES.md`](../benchmarks/multi_node/srt-slurm-recipes/RECIPES.md) | 分离式 Recipe 注册与主配置耦合 | | [`utils/runner_setup/RUNNER_SETUP.md`](../utils/runner_setup/RUNNER_SETUP.md) | Runner 部署与初始化 | | [`MODELS_zh.md`](MODELS_zh.md) | 支持的模型、硬件覆盖与命名 | diff --git a/docs/results-and-ingestion.md b/docs/results-and-ingestion.md index a67702eda6..98842a543c 100644 --- a/docs/results-and-ingestion.md +++ b/docs/results-and-ingestion.md @@ -129,7 +129,7 @@ The native collector sets UTC and records context beside its CSV for portable re ### Per-config identity and collection -Each eval upload is named `eval__`. Its current allowed payload includes `meta_env.json`, `results*.json`, sample JSONL, predictions, SWE-bench reports, and trajectory files. The collector uses only the metadata and lm-eval result JSON for aggregate rows. +Each eval upload is named `eval__`. Its current allowed payload includes `meta_env.json`, `results*.json`, sample JSONL, predictions, and trajectory files. The collector uses only the metadata and lm-eval result JSON for aggregate rows. Collection and reuse share result reading and selection, but retain different validation policies. Collection can report completed points from a failed batch; reuse rejects failed or incomplete batches. Each phase uses its loaded JSON for selection and validation. After deduplication rewrites or removes artifacts, validation reads the resulting files afresh. diff --git a/docs/results-and-ingestion_zh.md b/docs/results-and-ingestion_zh.md index 8ceab47227..6d935ef218 100644 --- a/docs/results-and-ingestion_zh.md +++ b/docs/results-and-ingestion_zh.md @@ -129,7 +129,7 @@ PR changelog 选择具有代表性的 NVIDIA 和 AMD 覆盖,并非所有受影 ### 单配置身份和收集 -每个评测上传名为 `eval__`。当前允许的载荷包括 `meta_env.json`、`results*.json`、样本 JSONL、预测、SWE-bench 报告和轨迹文件。收集器只使用元数据和 lm-eval 结果 JSON 来生成聚合记录。 +每个评测上传名为 `eval__`。当前允许的载荷包括 `meta_env.json`、`results*.json`、样本 JSONL、预测和轨迹文件。收集器只使用元数据和 lm-eval 结果 JSON 来生成聚合记录。 收集与复用共用结果读取和选择逻辑,但保留各自的校验规则。收集可以输出失败批次中已完成的点;复用则拒绝失败或不完整的批次。每个阶段使用已读取的 JSON 完成选择和校验。去重改写或删除工件后,校验会重新读取最终文件。 diff --git a/infx/evals/EVALS.md b/infx/evals/EVALS.md index ff8919b530..76ccd48c2a 100644 --- a/infx/evals/EVALS.md +++ b/infx/evals/EVALS.md @@ -606,7 +606,7 @@ attempt cannot replace a newer failed retry. |----------|---------|-------------| | `RUN_EVAL` | `false` | Enable eval after throughput benchmark | | `EVAL_ONLY` | `false` | Skip throughput, only run evals (set by workflow) | -| `EVAL_FRAMEWORK` | Workflow: `auto`; benchmark runner: `lm-eval` | Eval runner (`lm-eval`, `swebench`, `kimi-vendor`, `minimax-vendor`, or `bfcl`). `auto` resolves from matrix metadata before reusable workflow dispatch | +| `EVAL_FRAMEWORK` | Workflow: `auto`; benchmark runner: `lm-eval` | Eval runner (`lm-eval`, `kimi-vendor`, `minimax-vendor`, or `bfcl`). `auto` resolves from matrix metadata before reusable workflow dispatch | | `EVAL_SUITE` | Matrix-selected for automatic vendor evals; otherwise basename of `EVAL_TASKS_DIR` or `gsm8k` | Provider suite selector and artifact identity. Explicit workflow overrides remain supported | | `EVAL_TASKS_DIR` | `infx/evals/gsm8k.yaml` | Path to lm-eval task YAML | | `EVAL_RESULT_DIR` | `/tmp/eval_out-*` | Output directory for eval results | @@ -641,55 +641,8 @@ Source rewrites are anchor-checked, idempotent, and atomic. (extracts `reasoning_content` when `message.content` is empty) and TRT compatibility (no `{"type": "text"}` injection for non-HF tokenizers). Copied into a temp dir as `sitecustomize.py` on `PYTHONPATH`. -- `patch_swebench_agent.py` (`_patch_swebench_agent`): mini-swe-agent/swe-rex - sandbox lifecycle cleanup, budget-exhaustion submission fallback, and the - [SWE-ReX #281](https://github.com/SWE-agent/SWE-ReX/pull/281) closed-stdin fix. -- `patch_swebench_scoring.py` (`_patch_swebench_scoring`): swebench Modal - scorer reserved-CPU reduction + sandbox termination on instance completion. - -### SWE-bench Lite (`--framework swebench`) - -SWE-bench requires applying each generated patch and running repository tests. -The dedicated framework uses mini-swe-agent for agentic generation by default, -then scores predictions with the official SWE-bench harness. It emits -`exact_match,resolved` in the existing lm-eval result shape so collection and -validation remain shared with the other evals. - -```bash -run_eval --framework swebench --port "$PORT" -append_lm_eval_summary -``` - -- Task metadata and single-shot prompt: `infx/evals/swebench_lite.yaml`. -- Scoring: `infx/evals/swebench_score.py` (diff extraction → `predictions.jsonl` → - `python -m swebench.harness.run_evaluation` → resolved-rate → results JSON). Offline - `--report` mode skips Docker for testing. -- Generation modes (`SWEBENCH_GEN_MODE`) include `agentic`, the default, which runs the - mini-swe-agent loop against the local endpoint. Each instance's shell runs in a Modal - sandbox via swe-rex, matching the real SWE-bench setting. The `single-shot` mode uses - lm-eval with one prompt per instance. It provides a roughly 10% floor baseline and is - kept only as an explicit debugging escape hatch. Agentic knobs include `SWEBENCH_AGENT_WORKERS` - (default: the config's `CONC`, else 64), `SWEBENCH_AGENT_STEP_LIMIT` (250), - `SWEBENCH_AGENT_CMD_TIMEOUT` (per command, 300s), `SWEBENCH_AGENT_TIMEOUT` (6h), - `SWEBENCH_AGENT_SANDBOX_CPU` (unset = Modal default), and `SWEBENCH_MODAL_APP_NAME` - (`infx-evals-swe`). -- Run size: an empty `EVAL_LIMIT` runs the full split of roughly 300 instances. A positive integer runs the - first N as an explicit smoke-test slice. `EVAL_LIMIT=full` (or `0`) also selects the full split. -- Scoring knobs: `SWEBENCH_TASK_NAME` (selects the YAML), `SWEBENCH_MAX_WORKERS`, - `SWEBENCH_EVAL_SANDBOX_CPU` (cores per scoring sandbox, default 2), `SWEBENCH_EVAL_TIMEOUT` - (per-instance test timeout, default 900s), `SWEBENCH_NAMESPACE` (pass `""` on arm/Mac), - `SWEBENCH_SKIP_SCORE=true` (generate-only), `SWEBENCH_USE_MODAL=true` (score on Modal remote - sandboxes instead of local Docker, as used in CI). For Modal credentials, set - `MODAL_TOKEN_ID`/`MODAL_TOKEN_SECRET` (e.g. from a GitHub secret) or provide `~/.modal.toml`. - If the file is absent, the env vars are bootstrapped into it automatically. The scoring dataset - is derived from the YAML's `dataset_path`, which keeps generation and scoring aligned. - If `SWEBENCH_DATASET` is set, it must match or the run fails fast. -- Scoring runs on Modal remote sandboxes in CI (`SWEBENCH_USE_MODAL=true`, no Docker on the GPU - nodes). Local Docker scoring needs about 120 GB of disk. The `thresholds.yaml` gate is `0.50`, - calibrated from full-split runs that scored 54%. Historical 50-instance slices scored 62–76%. ## Task files The following files are task definitions from lm-eval. More information on changes lives within the files: - `infx/evals/gsm8k.yaml` - `infx/evals/gpqa_diamond.yaml` -- `infx/evals/swebench_lite.yaml` (generation only, scored by `swebench_score.py`) diff --git a/infx/evals/patches/patch_swebench_agent.py b/infx/evals/patches/patch_swebench_agent.py deleted file mode 100644 index a25c5dddd5..0000000000 --- a/infx/evals/patches/patch_swebench_agent.py +++ /dev/null @@ -1,183 +0,0 @@ -"""Runtime fixes for mini-swe-agent 2.4.5 and swe-rex 1.4.0.""" - -import os -import sys -from pathlib import Path - - -def _patch(path: str, replacements: list[tuple[str, str, str]], label: str) -> bool: - source_path = Path(path) - source = source_path.read_text() - pending: list[tuple[str, str]] = [] - failures: list[tuple[str, int]] = [] - - for old, new, marker in replacements: - if new in source or (marker and marker in source): - continue - count = source.count(old) - if count != 1: - failures.append((old.splitlines()[0], count)) - else: - pending.append((old, new)) - - if failures: - for anchor, count in failures: - print( - f"WARN: [{label}] {source_path}: patch anchor {anchor!r} " - f"found {count} times; no changes written", - file=sys.stderr, - ) - return False - - if not pending: - print(f"[{label}] {source_path}: patch already applied") - return True - - for old, new in pending: - source = source.replace(old, new) - source_path.write_text(source) - print(f"[{label}] {source_path}: patch applied") - return True - - -def _patch_swerex_environment(path: str) -> bool: - return _patch( - path, - [ - ( - ( - ' command = action.get("command", "") if isinstance(action, dict) else action\n' - " try:" - ), - ( - ' command = action.get("command", "") if isinstance(action, dict) else action\n' - ' command = f"exec int: - import minisweagent.environments.extra.swerex_modal as mini_rex_modal - import minisweagent.run.benchmarks.swebench as mini_swebench - import swerex.deployment.modal as rex_modal - - mini_ok = _patch( - mini_swebench.__file__, - [ - ( - " agent = None\n exit_status = None", - " agent = None\n env = None\n exit_status = None", - "", - ), - ( - ( - " info = agent.run(task)\n" - ' exit_status = info.get("exit_status")\n' - ' result = info.get("submission")' - ), - ( - " info = agent.run(task)\n" - ' exit_status = info.get("exit_status")\n' - ' result = info.get("submission")\n' - " if not result and env is not None:\n" - " try:\n" - ' _fb = env.execute("git diff")\n' - ' _fb_out = (_fb.get("output") or "").strip()\n' - ' if _fb.get("returncode") == 0 and _fb_out.startswith("diff --git"):\n' - ' result = _fb_out + "\\n"\n' - ' extra_info["submission_source"] = f"fallback_after_{exit_status}"\n' - " except Exception:\n" - " pass" - ), - "", - ), - ( - ( - ' exit_status, result = type(e).__name__, ""\n' - ' extra_info = {"traceback": traceback.format_exc(), "exception_str": str(e)}' - ), - ( - ' exit_status, result = type(e).__name__, ""\n' - ' extra_info = {"traceback": traceback.format_exc(), "exception_str": str(e)}\n' - " if env is not None:\n" - " try:\n" - ' _fb = env.execute("git diff")\n' - ' _fb_out = (_fb.get("output") or "").strip()\n' - ' if _fb.get("returncode") == 0 and _fb_out.startswith("diff --git"):\n' - ' result = _fb_out + "\\n"\n' - ' extra_info["submission_source"] = f"fallback_after_{exit_status}"\n' - " except Exception:\n" - " pass" - ), - "", - ), - ( - " finally:\n if agent is not None:", - ( - " finally:\n" - ' if env is not None and callable(getattr(env, "stop", None)):\n' - " try:\n" - " env.stop()\n" - " except Exception:\n" - " pass\n" - " if agent is not None:" - ), - "", - ), - ], - "swebench-agentic", - ) - mini_env_ok = _patch_swerex_environment(mini_rex_modal.__file__) - - app_name = os.environ.get("SWEBENCH_MODAL_APP_NAME", "infx-evals-swe") - rex_ok = _patch( - rex_modal.__file__, - [ - ( - 'self._app = modal.App.lookup("swe-rex", create_if_missing=True)', - f'self._app = modal.App.lookup("{app_name}", create_if_missing=True) # isolate InferenceX sandboxes', - "# isolate InferenceX sandboxes", - ), - ( - ( - " if self._sandbox is not None:\n" - " exit_code = await self._sandbox.poll.aio()\n" - " if exit_code is not None:\n" - " await self._sandbox.terminate.aio()" - ), - ( - " if self._sandbox is not None:\n" - " try:\n" - " await self._sandbox.terminate.aio()\n" - " except Exception:\n" - " pass" - ), - "", - ), - ( - " await self._wait_until_alive(timeout=remaining_startup_timeout)", - ( - " try:\n" - " await self._wait_until_alive(timeout=remaining_startup_timeout)\n" - " except BaseException:\n" - " try:\n" - " await self._sandbox.terminate.aio()\n" - " except Exception:\n" - " pass\n" - " raise" - ), - "", - ), - ], - "swebench-agentic", - ) - return 0 if mini_ok and mini_env_ok and rex_ok else 1 - - -if __name__ == "__main__": - sys.exit(main()) diff --git a/infx/evals/patches/patch_swebench_scoring.py b/infx/evals/patches/patch_swebench_scoring.py deleted file mode 100644 index 25bf92137e..0000000000 --- a/infx/evals/patches/patch_swebench_scoring.py +++ /dev/null @@ -1,93 +0,0 @@ -"""Runtime fixes for SWE-bench 4.1.0 Modal sandbox resource handling.""" - -import math -import os -import sys -from pathlib import Path - -CPU_ANCHOR = "cpu=4," -CPU_MARKER = "reduce idle Modal CPU billing" -LIFECYCLE_ANCHOR = ( - " log_dir=log_dir,\n" - " errored=True,\n" - " )\n" - "\n" - "\n" - "def run_instances_modal(" -) -LIFECYCLE_REPLACEMENT = ( - " log_dir=log_dir,\n" - " errored=True,\n" - " )\n" - " finally: # stop billing after evaluation\n" - " try:\n" - " runner.sandbox.terminate()\n" - " except Exception:\n" - " pass\n" - "\n" - "\n" - "def run_instances_modal(" -) -LIFECYCLE_MARKER = "stop billing after evaluation" - - -def _cpu_value() -> str: - value = os.environ.get("SWEBENCH_EVAL_SANDBOX_CPU", "2") - try: - parsed = float(value) - except ValueError as error: - raise ValueError(f"SWEBENCH_EVAL_SANDBOX_CPU={value!r} must be numeric") from error - if not math.isfinite(parsed) or parsed <= 0: - raise ValueError(f"SWEBENCH_EVAL_SANDBOX_CPU={value!r} must be positive and finite") - return value - - -def patch(path: str, cpu: str) -> bool: - source_path = Path(path) - source = source_path.read_text() - cpu_applied = CPU_MARKER in source - lifecycle_applied = LIFECYCLE_MARKER in source - failures = [] - - if not cpu_applied and source.count(CPU_ANCHOR) != 1: - failures.append(f"{CPU_ANCHOR!r} found {source.count(CPU_ANCHOR)} times") - if not lifecycle_applied and source.count(LIFECYCLE_ANCHOR) != 1: - failures.append(f"lifecycle anchor found {source.count(LIFECYCLE_ANCHOR)} times") - if failures: - for failure in failures: - print( - f"WARN: [swebench] {source_path}: {failure}; no changes written", - file=sys.stderr, - ) - return False - - if not cpu_applied: - source = source.replace( - CPU_ANCHOR, - f"cpu={cpu}, # {CPU_MARKER}", - ) - if not lifecycle_applied: - source = source.replace(LIFECYCLE_ANCHOR, LIFECYCLE_REPLACEMENT) - - if cpu_applied and lifecycle_applied: - print(f"[swebench] {source_path}: patch already applied") - else: - source_path.write_text(source) - print(f"[swebench] {source_path}: patch applied with cpu={cpu}") - return True - - -def main() -> int: - try: - cpu = _cpu_value() - except ValueError as error: - print(f"WARN: {error}; leaving SWE-bench unmodified", file=sys.stderr) - return 1 - - import swebench.harness.modal_eval.run_evaluation_modal as modal_evaluation - - return 0 if patch(modal_evaluation.__file__, cpu) else 1 - - -if __name__ == "__main__": - sys.exit(main()) diff --git a/infx/evals/swebench_lite.yaml b/infx/evals/swebench_lite.yaml deleted file mode 100644 index 4d38149304..0000000000 --- a/infx/evals/swebench_lite.yaml +++ /dev/null @@ -1,39 +0,0 @@ -# Official scoring replaces lm-eval's placeholder metric. -task: swebench_lite -dataset_path: princeton-nlp/SWE-bench_Lite -output_type: generate_until -test_split: test - -doc_to_text: | - You are an expert software engineer fixing a real GitHub issue in the - repository `{{repo}}` at commit {{base_commit}}. - - - {{problem_statement}} - - - Respond with ONLY a unified diff (a git patch) that resolves the issue, using - real repository file paths. Do not include explanations. Wrap the patch in a - single fenced block exactly like: - - ```diff - diff --git a/path/to/file.py b/path/to/file.py - --- a/path/to/file.py - +++ b/path/to/file.py - @@ ... @@ - ``` -# lm-eval requires a target even though scoring ignores it. -doc_to_target: "{{patch}}" - -generation_kwargs: - until: [] - do_sample: false - temperature: 0.0 - -metric_list: - - metric: exact_match - aggregation: mean - higher_is_better: true - -metadata: - version: 0.1 diff --git a/infx/evals/swebench_score.py b/infx/evals/swebench_score.py deleted file mode 100644 index cc104bf98f..0000000000 --- a/infx/evals/swebench_score.py +++ /dev/null @@ -1,412 +0,0 @@ -"""Score SWE-bench predictions and emit the repository's lm-eval result shape.""" - -import argparse -import json -import math -import re -import subprocess -import sys -from collections.abc import Iterator -from pathlib import Path - -DEFAULT_DATASET = "princeton-nlp/SWE-bench_Lite" -DEFAULT_TASK = "swebench_lite" - -_FENCED_DIFF_RE = re.compile( - r"```(?:diff|patch)?\s*\n(?P.*?)```", - re.DOTALL | re.IGNORECASE, -) -_DIFF_GIT_RE = re.compile(r"(?:^|\n)(diff --git .*)", re.DOTALL) - -_DIFF_LINE_PREFIXES = ( - "diff ", - "index ", - "--- ", - "+++ ", - "@@", - "+", - "-", - " ", - "\\", - "old mode ", - "new mode ", - "new file mode ", - "deleted file mode ", - "rename ", - "copy ", - "similarity ", - "dissimilarity ", - "Binary files ", - "GIT binary patch", -) - - -def _trim_to_diff_body(text: str) -> str: - """Drop trailing non-diff output from a generated patch.""" - lines = text.splitlines() - out: list[str] = [] - i, n = 0, len(lines) - while i < n: - if lines[i].startswith(_DIFF_LINE_PREFIXES): - out.append(lines[i]) - i += 1 - continue - if lines[i] == "": - j = i - while j < n and lines[j] == "": - j += 1 - if j < n and lines[j].startswith(_DIFF_LINE_PREFIXES): - out.extend(lines[i:j]) - i = j - continue - break - return "\n".join(out) - - -def extract_patch(text: str) -> str: - """Extract a unified diff from a model generation.""" - if not text: - return "" - - def _finish(body: str) -> str: - body = _trim_to_diff_body(body).strip("\n") - return body + "\n" if body else "" - - for match in _FENCED_DIFF_RE.finditer(text): - body = match.group("body") - if "diff --git" in body or body.lstrip().startswith(("--- ", "+++ ")): - return _finish(body) - git_match = _DIFF_GIT_RE.search(text) - if git_match: - trimmed = _finish(git_match.group(1)) - if trimmed: - return trimmed - lone = _FENCED_DIFF_RE.search(text) - if lone: - body = lone.group("body").strip("\n") - return body + "\n" if body else "" - return text.strip("\n") + "\n" if text.strip() else "" - - -def _response_text(record: dict) -> str: - """Read response text from supported lm-eval sample schemas.""" - for key in ("filtered_resps", "resps"): - val = record.get(key) - while isinstance(val, (list, tuple)) and val: - val = val[0] - if isinstance(val, str) and val.strip(): - return val - return "" - - -def _instance_id(record: dict) -> str | None: - doc = record.get("doc") - if isinstance(doc, dict): - for key in ("instance_id", "instance", "id"): - val = doc.get(key) - if isinstance(val, str) and val: - return val - val = record.get("instance_id") - return val if isinstance(val, str) and val else None - - -def iter_samples(samples_dir: Path) -> Iterator[dict]: - """Yield JSON records from every samples_*.jsonl under ``samples_dir``.""" - files = sorted(samples_dir.rglob("samples_*.jsonl")) - if not files: - raise FileNotFoundError( - f"no samples_*.jsonl found under {samples_dir} -- did lm-eval run with --log_samples?" - ) - for path in files: - with path.open(encoding="utf-8", errors="replace") as fh: - for line in fh: - line = line.strip() - if line: - yield json.loads(line) - - -def build_predictions(samples_dir: Path, model_name: str) -> list[dict]: - """Turn lm-eval samples into swebench prediction rows (dedup by instance).""" - by_instance: dict[str, dict] = {} - skipped = 0 - for record in iter_samples(samples_dir): - instance_id = _instance_id(record) - if not instance_id: - skipped += 1 - continue - patch = extract_patch(_response_text(record)) - by_instance[instance_id] = { - "instance_id": instance_id, - "model_name_or_path": model_name, - "model_patch": patch, - } - if skipped: - print(f"WARN: skipped {skipped} sample(s) with no instance_id", file=sys.stderr) - if not by_instance: - raise ValueError("no usable predictions extracted from samples") - return list(by_instance.values()) - - -def write_predictions(predictions: list[dict], out_path: Path) -> None: - with out_path.open("w", encoding="utf-8") as fh: - for row in predictions: - fh.write(json.dumps(row) + "\n") - - -def run_harness( - predictions_path: Path, - dataset_name: str, - run_id: str, - work_dir: Path, - max_workers: int, - namespace: str | None, - modal: bool = False, - timeout: int | None = None, -) -> None: - """Invoke the official swebench harness (local Docker, or Modal sandboxes).""" - cmd = [ - sys.executable, - "-m", - "swebench.harness.run_evaluation", - "--dataset_name", - dataset_name, - "--predictions_path", - str(predictions_path), - "--run_id", - run_id, - ] - if timeout is not None: - cmd += ["--timeout", str(timeout)] - if modal: - cmd += ["--modal", "true", "--max_workers", str(max_workers)] - else: - cmd += ["--max_workers", str(max_workers)] - if namespace is not None: - cmd += ["--namespace", namespace] - print(f"[swebench] running: {' '.join(cmd)}", flush=True) - subprocess.run(cmd, cwd=str(work_dir), check=True) - - -def find_report(work_dir: Path, model_name: str, run_id: str) -> Path: - """Locate the harness report JSON, tolerant to known layout variants.""" - sanitized = model_name.replace("/", "__") - candidates = [ - work_dir / f"{sanitized}.{run_id}.json", - work_dir / f"{model_name}.{run_id}.json", - work_dir / "evaluation_results" / "results.json", - ] - for path in candidates: - if path.exists(): - return path - for path in sorted(work_dir.rglob("*.json")): - try: - data = json.loads(path.read_text()) - except (json.JSONDecodeError, OSError): - continue - if isinstance(data, dict) and ("resolved_instances" in data or "resolved_ids" in data): - return path - raise FileNotFoundError( - f"could not locate a swebench report under {work_dir} " - f"(looked for {[str(c) for c in candidates]})" - ) - - -def parse_resolved(report: dict) -> tuple[int, int]: - """Return resolved and submitted counts from a harness report. - - Sliced runs use submitted instances rather than the full dataset size. - """ - resolved: int | None = None - for key in ("resolved_instances", "resolved", "num_resolved"): - if isinstance(report.get(key), int): - resolved = report[key] - break - if resolved is None and isinstance(report.get("resolved_ids"), list): - resolved = len(report["resolved_ids"]) - - total: int | None = None - for key in ("submitted_instances", "completed_instances", "total_instances"): - val = report.get(key) - if isinstance(val, int) and val > 0: - total = val - break - if total is None: - for key in ("submitted_ids", "completed_ids"): - if isinstance(report.get(key), list) and report[key]: - total = len(report[key]) - break - - if resolved is None or total is None or total <= 0: - raise ValueError(f"could not parse resolved/total from report keys {sorted(report)}") - return resolved, total - - -def build_results_json( - task: str, - resolved: int, - total: int, - model_name: str, - lm_eval_version: str, - report: dict | None, -) -> dict: - """Publish resolved rate as the exact-match metric used by score validation.""" - rate = resolved / total - stderr = math.sqrt(rate * (1.0 - rate) / total) if total else 0.0 - return { - "lm_eval_version": lm_eval_version, - "model_name": model_name, - "results": { - task: { - "alias": task, - "exact_match,resolved": rate, - "exact_match_stderr,resolved": stderr, - } - }, - "configs": { - task: { - "metric_list": [{"metric": "exact_match"}], - "filter_list": [{"name": "resolved"}], - } - }, - "n-samples": {task: {"effective": total, "original": total}}, - "swebench": { - "resolved": resolved, - "total": total, - "resolved_rate": rate, - "report": report, - }, - } - - -def main(argv: list[str] | None = None) -> int: - parser = argparse.ArgumentParser(description="Score SWE-bench patches from lm-eval samples") - parser.add_argument( - "--samples-dir", - default=None, - help="dir containing lm-eval samples_*.jsonl (single-shot mode)", - ) - parser.add_argument( - "--predictions-file", - default=None, - help="pre-built predictions.jsonl (agentic mode) -- skips samples parsing", - ) - parser.add_argument("--out-dir", required=True, help="dir to write predictions + results JSON") - parser.add_argument( - "--model-name", required=True, help="served model name (model_name_or_path)" - ) - parser.add_argument("--dataset-name", default=DEFAULT_DATASET) - parser.add_argument("--task-name", default=DEFAULT_TASK) - parser.add_argument("--run-id", default=None, help="harness run id (default: task name)") - parser.add_argument("--max-workers", type=int, default=4) - parser.add_argument( - "--instance-timeout", - type=int, - default=None, - help="per-instance test timeout in seconds (harness default 1800)", - ) - parser.add_argument( - "--namespace", - default=None, - help="local-Docker --namespace value (pass '' on arm/Mac to build images locally)", - ) - parser.add_argument( - "--modal", - action="store_true", - help="score on Modal remote sandboxes instead of local Docker (needs modal creds)", - ) - parser.add_argument("--lm-eval-version", default="unknown") - parser.add_argument( - "--predictions-only", - action="store_true", - help="write predictions.jsonl and stop (no scoring; score elsewhere)", - ) - parser.add_argument( - "--no-run", - action="store_true", - help="skip the Docker harness; requires --report (offline/testing)", - ) - parser.add_argument( - "--report", - default=None, - help="path to a pre-computed harness report JSON (implies --no-run)", - ) - args = parser.parse_args(argv) - - out_dir = Path(args.out_dir) - out_dir.mkdir(parents=True, exist_ok=True) - run_id = args.run_id or args.task_name - - if args.predictions_file: - src = Path(args.predictions_file) - text = src.read_text(encoding="utf-8", errors="replace") - try: - blob = json.loads(text) - predictions = list(blob.values()) if isinstance(blob, dict) else blob - except json.JSONDecodeError: - predictions = [json.loads(line) for line in text.splitlines() if line.strip()] - if not predictions: - print(f"ERROR: no predictions in {src}", file=sys.stderr) - return 1 - predictions_path = out_dir / "predictions.jsonl" - write_predictions(predictions, predictions_path) - print(f"[swebench] using {len(predictions)} pre-built predictions -> {predictions_path}") - elif args.samples_dir: - predictions = build_predictions(Path(args.samples_dir), args.model_name) - predictions_path = out_dir / "predictions.jsonl" - write_predictions(predictions, predictions_path) - print(f"[swebench] wrote {len(predictions)} predictions -> {predictions_path}") - else: - print( - "ERROR: one of --samples-dir or --predictions-file is required", - file=sys.stderr, - ) - return 1 - - if args.predictions_only: - print("[swebench] predictions-only: skipping scoring (score elsewhere)") - return 0 - - if args.report: - report = json.loads(Path(args.report).read_text(encoding="utf-8", errors="replace")) - elif args.no_run: - print("ERROR: --no-run requires --report", file=sys.stderr) - return 1 - else: - run_harness( - predictions_path, - args.dataset_name, - run_id, - out_dir, - args.max_workers, - args.namespace, - modal=args.modal, - timeout=args.instance_timeout, - ) - report_path = find_report(out_dir, args.model_name, run_id) - report = json.loads(report_path.read_text(encoding="utf-8", errors="replace")) - # Artifact collection requires a stable name. - staged = out_dir / f"swebench_report_{args.task_name}.json" - if report_path.resolve() != staged.resolve(): - staged.write_text(json.dumps(report, indent=2), encoding="utf-8") - - resolved, total = parse_resolved(report) - - results = build_results_json( - args.task_name, - resolved, - total, - args.model_name, - args.lm_eval_version, - report, - ) - results_path = out_dir / f"results_{args.task_name}.json" - results_path.write_text(json.dumps(results, indent=2), encoding="utf-8") - print( - f"[swebench] {args.task_name}: resolved {resolved}/{total} " - f"= {resolved / total:.4f} -> {results_path}" - ) - return 0 - - -if __name__ == "__main__": - sys.exit(main()) diff --git a/infx/evals/thresholds.yaml b/infx/evals/thresholds.yaml index 5edab193d2..cb25739f50 100644 --- a/infx/evals/thresholds.yaml +++ b/infx/evals/thresholds.yaml @@ -25,8 +25,7 @@ "bfcl_vllm_kimi_multi_turn_miss_func": 0.0, "bfcl_vllm_kimi_multi_turn_miss_param": 0.0, "bfcl_vllm_kimi_multi_turn_long_context": 0.0, - "gpqa_diamond_cot_n_shot": 0.30, - "swebench_lite": 0.50 + "gpqa_diamond_cot_n_shot": 0.30 }, "models": { "dsr1": { diff --git a/infx/tests/evals/test_eval_patches.py b/infx/tests/evals/test_eval_patches.py index 1be32be934..940bbbfade 100644 --- a/infx/tests/evals/test_eval_patches.py +++ b/infx/tests/evals/test_eval_patches.py @@ -1,7 +1,5 @@ -import importlib.util import json import runpy -import subprocess import sys import types from pathlib import Path @@ -11,122 +9,6 @@ PATCH_DIR = Path(__file__).resolve().parents[3] / "infx/evals/patches" -def _load_patch_module(name: str): - spec = importlib.util.spec_from_file_location(name, PATCH_DIR / f"{name}.py") - assert spec is not None and spec.loader is not None - module = importlib.util.module_from_spec(spec) - spec.loader.exec_module(module) - return module - - -agent_patch = _load_patch_module("patch_swebench_agent") -scoring_patch = _load_patch_module("patch_swebench_scoring") - - -def test_agent_patch_is_atomic_and_idempotent(tmp_path): - target = tmp_path / "dependency.py" - original = "alpha\nbeta\n" - target.write_text(original) - - assert not agent_patch._patch( - str(target), - [("alpha", "patched-alpha", ""), ("missing", "patched-missing", "")], - "test", - ) - assert target.read_text() == original - - replacements = [("alpha", "patched-alpha", ""), ("beta", "patched-beta", "")] - assert agent_patch._patch(str(target), replacements, "test") - patched = target.read_text() - assert patched == "patched-alpha\npatched-beta\n" - assert agent_patch._patch(str(target), replacements, "test") - assert target.read_text() == patched - - -def test_agent_patch_closes_inherited_stdin(tmp_path): - target = tmp_path / "swerex_modal.py" - target.write_text( - """import json -import subprocess - -class Environment: - def execute(self, action): - command = action.get("command", "") if isinstance(action, dict) else action - try: - result = subprocess.run( - command, - shell=True, - timeout=0.2, - stdout=subprocess.PIPE, - text=True, - ) - except subprocess.TimeoutExpired: - return {"returncode": -1, "output": "timed out"} - return {"returncode": result.returncode, "output": result.stdout} - -environment = Environment() -commands = ["cat", "printf 'pipe-ok\\\\n' | cat"] -print(json.dumps([environment.execute({"command": command}) for command in commands])) -""" - ) - - assert agent_patch._patch_swerex_environment(str(target)) - patched = target.read_text() - assert agent_patch._patch_swerex_environment(str(target)) - assert target.read_text() == patched - - process = subprocess.Popen( - [sys.executable, str(target)], - stdin=subprocess.PIPE, - stdout=subprocess.PIPE, - stderr=subprocess.PIPE, - text=True, - ) - try: - assert process.wait(timeout=2) == 0 - finally: - if process.poll() is None: - process.kill() - process.wait() - assert process.stdin is not None - process.stdin.close() - - assert process.stdout is not None - assert process.stderr is not None - output = process.stdout.read() - errors = process.stderr.read() - process.stdout.close() - process.stderr.close() - assert not errors - assert json.loads(output) == [ - {"returncode": 0, "output": ""}, - {"returncode": 0, "output": "pipe-ok\n"}, - ] - - -def test_scoring_patch_is_atomic_and_idempotent(tmp_path): - target = tmp_path / "run_evaluation_modal.py" - target.write_text("prefix\ncpu=4,\nsuffix\n") - - assert not scoring_patch.patch(str(target), "2") - assert target.read_text() == "prefix\ncpu=4,\nsuffix\n" - - source = "cpu=4,\n" + scoring_patch.LIFECYCLE_ANCHOR - target.write_text(source) - assert scoring_patch.patch(str(target), "2") - patched = target.read_text() - assert "cpu=2," in patched - assert scoring_patch.patch(str(target), "2") - assert target.read_text() == patched - - -@pytest.mark.parametrize("value", ["0", "-1", "nan", "inf", "bad"]) -def test_scoring_patch_rejects_invalid_cpu(monkeypatch, value): - monkeypatch.setenv("SWEBENCH_EVAL_SANDBOX_CPU", value) - with pytest.raises(ValueError): - scoring_patch._cpu_value() - - def test_lm_eval_sitecustomize_hooks(monkeypatch): lm_eval = types.ModuleType("lm_eval") models = types.ModuleType("lm_eval.models") diff --git a/infx/tests/evals/test_run_eval_dispatch.py b/infx/tests/evals/test_run_eval_dispatch.py index d292e7b313..d058e9e433 100644 --- a/infx/tests/evals/test_run_eval_dispatch.py +++ b/infx/tests/evals/test_run_eval_dispatch.py @@ -4,7 +4,6 @@ import io import json import os -import signal import socket import stat import subprocess @@ -33,7 +32,6 @@ def explicit_runtime_inputs(monkeypatch: pytest.MonkeyPatch) -> None: "PORT": "8888", "CONC": "64", "VENDOR_VERIFIER_PYTHON": "python3", - "SWEBENCH_GEN_MODE": "agentic", "INFMAX_CONTAINER_WORKSPACE": str(REPO_ROOT), "AIPERF_PYTHON_VERSION": "3.11", "AIPERF_DRAIN_TIMEOUT_SECONDS": "120", @@ -43,17 +41,6 @@ def explicit_runtime_inputs(monkeypatch: pytest.MonkeyPatch) -> None: "EVAL_ENDPOINT_READY_TIMEOUT_SECONDS": "1800", "EVAL_MODEL_STABILIZATION_SECONDS": "0", "OPENAI_API_KEY": "EMPTY", - "SWEBENCH_USE_MODAL": "false", - "SWEBENCH_AGENT_STEP_LIMIT": "250", - "SWEBENCH_EXPECTED_INSTANCES": "300", - "SWEBENCH_AGENT_TIMEOUT": "21600", - "SWEBENCH_AGENT_EXIT_GRACE": "300", - "SWEBENCH_WATCHDOG_POLL": "30", - "SWEBENCH_SANDBOX_SWEEP": "1", - "SWEBENCH_SKIP_SCORE": "false", - "SWEBENCH_EVAL_TIMEOUT": "900", - "SWEBENCH_SCORE_TIMEOUT": "7200", - "SWEBENCH_MAX_WORKERS": "4", }.items(): monkeypatch.setenv(name, value) @@ -62,7 +49,6 @@ def explicit_runtime_inputs(monkeypatch: pytest.MonkeyPatch) -> None: source "$BENCHMARK_LIB" _wait_for_openai_chat_route() { echo "READY=$*"; } run_lm_eval() { echo "DISPATCH=lm-eval"; } -run_swebench_eval() { echo "DISPATCH=swebench"; } run_kimi_vendor_eval() { echo "DISPATCH=kimi-vendor"; } run_minimax_vendor_eval() { echo "DISPATCH=minimax-vendor"; } run_bfcl_eval() { echo "DISPATCH=bfcl"; } @@ -214,10 +200,6 @@ def test_environment_framework_overrides_legacy_recipe_argument() -> None: ) -def test_env_can_force_swebench_on_fixed_seqlen(): - assert "DISPATCH=swebench" in _dispatch(is_agentic="0", env_fw="swebench") - - def test_env_can_force_kimi_vendor_on_agentic_eval() -> None: assert "DISPATCH=kimi-vendor" in _dispatch( is_agentic="1", @@ -1659,7 +1641,6 @@ def test_stage_eval_artifacts_copies_eval_outputs_only(tmp_path: Path) -> None: "sample_eval.jsonl", "agent_preds.json", "predictions.jsonl", - "swebench_report_eval.json", "trace.traj.json", } for filename in expected: @@ -1885,104 +1866,6 @@ def test_env_is_true_is_case_insensitive_and_unset_safe() -> None: ] -_MODAL_CREDS_SCRIPT = r""" -source "$BENCHMARK_LIB" -_ensure_modal_credentials -echo "HOME_AFTER=$HOME" -if [ -f "$HOME/.modal.toml" ]; then - echo "TOML_EXISTS=true" - PERMS=$(stat -c '%a' "$HOME/.modal.toml" 2>/dev/null || stat -f '%A' "$HOME/.modal.toml" 2>/dev/null) - echo "TOML_PERMS=$PERMS" -fi -""" - - -def _run_modal_creds( - tmp_path: Path, *, home: str, token_id="tok-id", token_secret="tok-secret" -) -> str: - env = { - **os.environ, - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "KV_OFFLOADING": "none", - "SWEBENCH_USE_MODAL": "true", - "MODAL_TOKEN_ID": token_id, - "MODAL_TOKEN_SECRET": token_secret, - "HOME": home, - } - res = subprocess.run( - ["bash", "-c", _MODAL_CREDS_SCRIPT], - env=env, - text=True, - capture_output=True, - check=True, - ) - return res.stdout + res.stderr - - -def test_modal_creds_no_remap_when_home_writable(tmp_path): - home = str(tmp_path / "writable_home") - Path(home).mkdir() - out = _run_modal_creds(tmp_path, home=home) - assert f"HOME_AFTER={home}" in out, f"HOME should not be remapped:\n{out}" - assert "TOML_EXISTS=true" in out - toml_path = Path(home) / ".modal.toml" - assert toml_path.exists() - mode = oct(stat.S_IMODE(toml_path.stat().st_mode)) - assert mode == "0o600", f"Expected 0o600 got {mode}" - - -def test_modal_creds_remaps_home_when_not_writable_parent(tmp_path): - readonly_parent = tmp_path / "readonly_parent" - readonly_parent.mkdir(mode=0o555) - nested_home = str(readonly_parent / "nested_home") - try: - out = _run_modal_creds(tmp_path, home=nested_home) - assert "HOME_AFTER=/tmp/inferencex-modal-home" in out, ( - f"Expected HOME remap:\n{out}" - ) - assert "remapped" in out.lower() or "HOME remapped" in out - assert "TOML_EXISTS=true" in out - toml_path = Path("/tmp/inferencex-modal-home/.modal.toml") - assert toml_path.exists() - mode = oct(stat.S_IMODE(toml_path.stat().st_mode)) - assert mode == "0o600", f"Expected 0o600 got {mode}" - finally: - readonly_parent.chmod(0o755) - - -def test_modal_creds_remaps_home_when_not_writable(tmp_path): - readonly_home = tmp_path / "readonly_home" - readonly_home.mkdir(mode=0o555) - try: - out = _run_modal_creds(tmp_path, home=str(readonly_home)) - assert "HOME_AFTER=/tmp/inferencex-modal-home" in out, ( - f"Expected HOME remap:\n{out}" - ) - assert "TOML_EXISTS=true" in out - finally: - readonly_home.chmod(0o755) - - -def test_modal_creds_no_remap_when_disabled(tmp_path): - env = { - **os.environ, - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "KV_OFFLOADING": "none", - "SWEBENCH_USE_MODAL": "false", - "MODAL_TOKEN_ID": "tok", - "MODAL_TOKEN_SECRET": "sec", - "HOME": str(tmp_path), - } - res = subprocess.run( - ["bash", "-c", _MODAL_CREDS_SCRIPT], - env=env, - text=True, - capture_output=True, - check=True, - ) - out = res.stdout + res.stderr - assert "remapped" not in out.lower() - assert "TOML_EXISTS" not in out _INCLUDE_PATH_SCRIPT = r""" @@ -2038,13 +1921,13 @@ def _run_lm_eval_with_include_path( def test_include_path_injected_when_eval_include_path_set(): out = _run_lm_eval_with_include_path( eval_include_path="infx/evals", - eval_tasks_dir="swebench_lite", + eval_tasks_dir="gpqa_diamond", ) assert "--include_path infx/evals" in out, ( f"Expected '--include_path infx/evals' in output:\n{out}" ) - assert "--tasks swebench_lite" in out, ( - f"Expected '--tasks swebench_lite' in output:\n{out}" + assert "--tasks gpqa_diamond" in out, ( + f"Expected '--tasks gpqa_diamond' in output:\n{out}" ) assert ".yaml" not in out.split("--tasks")[1].split()[0], ( f"--tasks must not contain a .yaml path when include_path is set:\n{out}" @@ -2061,369 +1944,6 @@ def test_include_path_absent_when_eval_include_path_unset(): ) -def test_swebench_single_shot_registers_task_yaml(): - script = r""" -source "$BENCHMARK_LIB" -run_lm_eval() { - echo "TASK=$EVAL_TASKS_DIR" - echo "INCLUDE=$EVAL_INCLUDE_PATH" - return 9 -} -export SWEBENCH_GEN_MODE=single-shot -export EVAL_TASKS_DIR="$TASK_YAML" -export MODEL=test-model -run_swebench_eval -""" - env = { - **os.environ, - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "TASK_YAML": str(BENCHMARK_LIB.parents[1] / "infx/evals/swebench_lite.yaml"), - "KV_OFFLOADING": "none", - } - result = subprocess.run( - ["bash", "-c", script], - env=env, - text=True, - capture_output=True, - ) - assert result.returncode == 9 - assert "TASK=swebench_lite" in result.stdout - assert f"INCLUDE={BENCHMARK_LIB.parents[1] / 'infx/evals'}" in result.stdout - - -def test_modal_credentials_sanitizes_whitespace_contaminated_tokens(tmp_path): - home = tmp_path / "home" - home.mkdir() - script = r""" -source "$BENCHMARK_LIB" 2>/dev/null -export SWEBENCH_USE_MODAL=true -_ensure_modal_credentials -grep -q '^token_id = "ak-clean123"$' "$HOME/.modal.toml" || { echo ID_DIRTY; exit 1; } -grep -q '^token_secret = "as-clean456"$' "$HOME/.modal.toml" || { echo FILE_DIRTY; exit 1; } -[ "$MODAL_TOKEN_ID" = "ak-clean123" ] && [ "$MODAL_TOKEN_SECRET" = "as-clean456" ] || { echo ENV_DIRTY; exit 1; } -echo SANITIZED_OK -""" - env = { - **os.environ, - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "HOME": str(home), - "MODAL_TOKEN_ID": " \t'ak-clean123'\r\n", - "MODAL_TOKEN_SECRET": ' \t"as-clean456"\r\n', - } - res = subprocess.run( - ["bash", "-c", script], env=env, text=True, capture_output=True - ) - assert res.returncode == 0, res.stdout + res.stderr - assert "SANITIZED_OK" in res.stdout - - -# Advance the watchdog's clock by requested waits. The small real yield lets -# child processes run; their sleep, signal handling, and exit status stay real. -_AGENTIC_TEST_CLOCK = r""" -_test_now=0 -date() { - if [[ "$*" == "+%s" ]]; then printf '%s\n' "$_test_now"; else command date "$@"; fi -} -sleep() { - _test_now=$((_test_now + $1)) - command sleep 0.01 -} -""" - - -def test_agentic_generation_invokes_mini_swe_agent(tmp_path): - shim = tmp_path / "shim" - shim.mkdir() - (shim / "mini-extra").write_text( - "#!/bin/bash\n" - 'echo "MINI_ARGV: $*" >> ' + str(shim / "argv.log") + "\n" - 'out=""; prev=""\n' - 'for a in "$@"; do [ "$prev" = "-o" ] && out="$a"; prev="$a"; done\n' - 'mkdir -p "$out"\n' - 'printf \'{"i1": {"instance_id": "i1", "model_name_or_path": "m", "model_patch": "d"}}\' > "$out/preds.json"\n' - ) - (shim / "mini-extra").chmod(0o755) - default_yaml = shim / "default.yaml" - default_yaml.write_text("agent: {}\n") - (shim / "python3").write_text( - "#!/bin/bash\n" - f'if [[ "$*" == *minisweagent* ]]; then echo "This is mini-swe-agent."; echo "Check the v2 migration guide"; echo "{default_yaml}"; else exec "$TEST_PYTHON" "$@"; fi\n' - ) - (shim / "python3").chmod(0o755) - - gen_dir = tmp_path / "gen" - gen_dir.mkdir() - script = _AGENTIC_TEST_CLOCK + r""" -source "$BENCHMARK_LIB" 2>/dev/null -_install_swebench_agent_deps() { :; } -_ensure_modal_credentials() { :; } -export EVAL_LIMIT=10 MODEL_NAME=test-model SWEBENCH_SANDBOX_SWEEP=0 SWEBENCH_WATCHDOG_POLL=1 -_run_swebench_agentic_generation "$GEN_DIR" --port 8899 || exit 1 -[ -s "$GEN_DIR/agent_out/preds.json" ] || { echo NO_PREDS; exit 1; } -grep -q 'api_base: http://0.0.0.0:8899/v1' "$GEN_DIR/mini_swebench_overrides.yaml" || { echo BAD_PORT; exit 1; } -grep -q 'openai/test-model' "$GEN_DIR/mini_swebench_overrides.yaml" || { echo BAD_MODEL; exit 1; } -echo AGENTIC_GEN_OK -""" - env = { - **os.environ, - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "GEN_DIR": str(gen_dir), - "TEST_PYTHON": sys.executable, - "PATH": f"{shim}:{os.environ['PATH']}", - } - res = subprocess.run( - ["bash", "-c", script], env=env, text=True, capture_output=True - ) - assert res.returncode == 0, res.stdout + res.stderr - assert "AGENTIC_GEN_OK" in res.stdout - argv = (shim / "argv.log").read_text() - assert "--slice 0:10" in argv - assert "--environment-class swerex_modal" in argv - assert "--subset lite" in argv - - -def _agentic_shim(tmp_path, mini_body): - shim = tmp_path / "shim" - shim.mkdir() - (shim / "mini-extra").write_text("#!/bin/bash\n" + mini_body) - (shim / "mini-extra").chmod(0o755) - default_yaml = shim / "default.yaml" - default_yaml.write_text("agent: {}\n") - (shim / "python3").write_text( - "#!/bin/bash\n" - f'if [[ "$*" == *minisweagent* ]]; then echo "{default_yaml}"; else exec "$TEST_PYTHON" "$@"; fi\n' - ) - (shim / "python3").chmod(0o755) - gen_dir = tmp_path / "gen" - gen_dir.mkdir() - return shim, gen_dir - - -def _run_agentic(shim, gen_dir, extra_env=None): - script = _AGENTIC_TEST_CLOCK + r""" -source "$BENCHMARK_LIB" 2>/dev/null -_install_swebench_agent_deps() { :; } -_ensure_modal_credentials() { :; } -_run_swebench_agentic_generation "$GEN_DIR" --port 8899 -echo "GEN_RC=$?" -""" - env = { - **os.environ, - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "GEN_DIR": str(gen_dir), - "TEST_PYTHON": sys.executable, - "MODEL_NAME": "test-model", - "SWEBENCH_SANDBOX_SWEEP": "0", - "SWEBENCH_WATCHDOG_POLL": "1", - "PATH": f"{shim}:{os.environ['PATH']}", - **(extra_env or {}), - } - return subprocess.run( - ["bash", "-c", script], env=env, text=True, capture_output=True, timeout=10 - ) - - -def test_agentic_watchdog_kills_hung_mini(tmp_path): - shim, gen_dir = _agentic_shim( - tmp_path, - 'out=""; prev=""\n' - 'for a in "$@"; do [ "$prev" = "-o" ] && out="$a"; prev="$a"; done\n' - 'mkdir -p "$out"\n' - 'printf "%s\\n" "$$" > "$out/mini.pid"\n' - 'printf \'{"i1": {"instance_id": "i1", "model_patch": "d"}}\' > "$out/preds.json"\n' - "exec sleep 600 /dev/null 2>&1\n", - ) - pid_file = gen_dir / "agent_out/mini.pid" - try: - res = _run_agentic( - shim, gen_dir, {"EVAL_LIMIT": "1", "SWEBENCH_AGENT_EXIT_GRACE": "2"} - ) - assert "GEN_RC=0" in res.stdout, res.stdout + res.stderr - assert "hung after completing all instances" in res.stdout + res.stderr - with pytest.raises(ProcessLookupError): - os.kill(int(pid_file.read_text()), 0) - finally: - if pid_file.exists(): - try: - os.kill(int(pid_file.read_text()), signal.SIGKILL) - except ProcessLookupError: - pass - - -def test_agentic_salvage_partial_preds_on_failure(tmp_path): - shim, gen_dir = _agentic_shim( - tmp_path, - 'out=""; prev=""\n' - 'for a in "$@"; do [ "$prev" = "-o" ] && out="$a"; prev="$a"; done\n' - 'mkdir -p "$out"\n' - 'printf \'{"i1": {"instance_id": "i1", "model_patch": "d"}}\' > "$out/preds.json"\n' - "exit 7\n", - ) - res = _run_agentic(shim, gen_dir, {"EVAL_LIMIT": "2"}) - assert "GEN_RC=0" in res.stdout, res.stdout + res.stderr - assert "scoring the partial set" in res.stdout + res.stderr - - -def test_agentic_no_preds_still_fails(tmp_path): - shim, gen_dir = _agentic_shim(tmp_path, "exit 7\n") - res = _run_agentic(shim, gen_dir, {"EVAL_LIMIT": "2"}) - assert "GEN_RC=7" in res.stdout, res.stdout + res.stderr - - -def test_agentic_eval_limit_defaults_to_full_split(tmp_path): - shim, gen_dir = _agentic_shim( - tmp_path, - 'echo "MINI_ARGV: $*" >> ' + "ARGVLOG" + "\n" - 'out=""; prev=""\n' - 'for a in "$@"; do [ "$prev" = "-o" ] && out="$a"; prev="$a"; done\n' - 'mkdir -p "$out"\n' - 'printf \'{"i1": {"instance_id": "i1", "model_patch": "d"}}\' > "$out/preds.json"\n', - ) - body = (shim / "mini-extra").read_text().replace("ARGVLOG", str(shim / "argv.log")) - (shim / "mini-extra").write_text(body) - res = _run_agentic(shim, gen_dir) - argv = (shim / "argv.log").read_text() - assert "--slice" not in argv, argv - assert "GEN_RC=0" in res.stdout, res.stdout + res.stderr - - -_GENMODE_SCRIPT = r""" -source "$BENCHMARK_LIB" 2>/dev/null -_install_swebench_agent_deps() { :; } -_ensure_modal_credentials() { :; } -_run_swebench_agentic_generation() { - echo "GEN=agentic" - echo "SUITE=$EVAL_SUITE" - return 42 -} -run_lm_eval() { - echo "GEN=single-shot" - echo "SUITE=$EVAL_SUITE" - return 42 -} -run_swebench_eval --port 8888 -echo "RC=$?" -""" - - -def _gen_mode( - tmp_path: Path, - *, - is_agentic, - gen_mode, - eval_suite=None, -) -> str: - env = { - **os.environ, - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "KV_OFFLOADING": "none", - "IS_AGENTIC": is_agentic, - "EVAL_RESULT_DIR": str(tmp_path / "out"), - } - env.pop("SWEBENCH_GEN_MODE", None) - env.pop("SCENARIO_TYPE", None) - env.pop("EVAL_SUITE", None) - if gen_mode is not None: - env["SWEBENCH_GEN_MODE"] = gen_mode - if eval_suite is not None: - env["EVAL_SUITE"] = eval_suite - res = subprocess.run( - ["bash", "-c", _GENMODE_SCRIPT], - env=env, - text=True, - capture_output=True, - cwd=BENCHMARK_LIB.parents[1], - ) - assert "RC=42" in res.stdout, res.stdout + res.stderr - return res.stdout - - -def test_explicit_agentic_generation_mode(tmp_path): - output = _gen_mode(tmp_path, is_agentic="1", gen_mode="agentic") - assert "GEN=agentic" in output - assert "SUITE=swebench_lite" in output - - -def test_explicit_single_shot_escape_hatch(tmp_path): - output = _gen_mode(tmp_path, is_agentic="1", gen_mode="single-shot") - assert "GEN=single-shot" in output - assert "SUITE=swebench_lite" in output - - -def test_swebench_generation_modes_preserve_explicit_suite(tmp_path): - for gen_mode in ("agentic", "single-shot"): - output = _gen_mode( - tmp_path / gen_mode, - is_agentic="1", - gen_mode=gen_mode, - eval_suite="explicit_swebench", - ) - assert "SUITE=explicit_swebench" in output - - -def test_agent_sandbox_cpu_knob(tmp_path): - shim, gen_dir = _agentic_shim( - tmp_path, - 'out=""; prev=""\n' - 'for a in "$@"; do [ "$prev" = "-o" ] && out="$a"; prev="$a"; done\n' - 'mkdir -p "$out"\n' - 'printf \'{"i1": {"instance_id": "i1", "model_patch": "d"}}\' > "$out/preds.json"\n', - ) - res = _run_agentic( - shim, gen_dir, {"EVAL_LIMIT": "1", "SWEBENCH_AGENT_SANDBOX_CPU": "1"} - ) - assert "GEN_RC=0" in res.stdout, res.stdout + res.stderr - cfg = yaml.safe_load((gen_dir / "mini_swebench_overrides.yaml").read_text()) - assert cfg["environment"]["modal_sandbox_kwargs"]["cpu"] == 1 - - gen_dir2 = tmp_path / "gen2" - gen_dir2.mkdir() - res2 = _run_agentic(shim, gen_dir2, {"EVAL_LIMIT": "1"}) - assert "GEN_RC=0" in res2.stdout, res2.stdout + res2.stderr - cfg2 = yaml.safe_load((gen_dir2 / "mini_swebench_overrides.yaml").read_text()) - assert "modal_sandbox_kwargs" not in cfg2["environment"] - - -def test_eval_limit_rejects_non_positive_integer(tmp_path): - shim, gen_dir = _agentic_shim( - tmp_path, - 'out=""; prev=""\n' - 'for a in "$@"; do [ "$prev" = "-o" ] && out="$a"; prev="$a"; done\n' - 'mkdir -p "$out"\n' - 'printf \'{"i1": {"instance_id": "i1", "model_patch": "d"}}\' > "$out/preds.json"\n', - ) - for bad in ("-5", "abc", "3.5"): - gd = tmp_path / f"gen_{bad.replace('-', 'neg').replace('.', '_')}" - gd.mkdir() - res = _run_agentic(shim, gd, {"EVAL_LIMIT": bad}) - assert "GEN_RC=1" in res.stdout, ( - f"EVAL_LIMIT={bad!r} should fail: {res.stdout}{res.stderr}" - ) - assert "must be a positive integer" in res.stdout + res.stderr - - -def test_eval_limit_full_and_zero_accepted(tmp_path): - shim, gen_dir = _agentic_shim( - tmp_path, - 'echo "MINI_ARGV: $*" >> ' + "ARGVLOG" + "\n" - 'out=""; prev=""\n' - 'for a in "$@"; do [ "$prev" = "-o" ] && out="$a"; prev="$a"; done\n' - 'mkdir -p "$out"\n' - 'printf \'{"i1": {"instance_id": "i1", "model_patch": "d"}}\' > "$out/preds.json"\n', - ) - body = (shim / "mini-extra").read_text().replace("ARGVLOG", str(shim / "argv.log")) - (shim / "mini-extra").write_text(body) - for sentinel in ("full", "0"): - gd = tmp_path / f"gen_{sentinel}" - gd.mkdir() - res = _run_agentic(shim, gd, {"EVAL_LIMIT": sentinel}) - assert "GEN_RC=0" in res.stdout, ( - f"EVAL_LIMIT={sentinel!r}: {res.stdout}{res.stderr}" - ) - argv = (shim / "argv.log").read_text() - assert "--slice" not in argv - - def test_chat_route_readiness_requires_model_and_active_route(tmp_path: Path) -> None: bin_dir = tmp_path / "bin" events_path = tmp_path / "events" diff --git a/infx/tests/evals/test_swebench_eval.py b/infx/tests/evals/test_swebench_eval.py deleted file mode 100644 index 38f9562559..0000000000 --- a/infx/tests/evals/test_swebench_eval.py +++ /dev/null @@ -1,259 +0,0 @@ - -import json -import sys -from pathlib import Path - -import pytest - -import infx.evals.swebench_score as sbs - - - -def test_extract_patch_from_diff_fence(): - text = ( - "Here is the fix:\n\n```diff\n" - "diff --git a/f.py b/f.py\n--- a/f.py\n+++ b/f.py\n" - "@@ -1 +1 @@\n-old\n+new\n```\nDone." - ) - patch = sbs.extract_patch(text) - assert patch.startswith("diff --git a/f.py b/f.py") - assert patch.endswith("\n") - assert "Here is the fix" not in patch - assert "Done." not in patch - - -def test_extract_patch_bare_diff_git(): - text = "no fence\ndiff --git a/x b/x\n@@ @@\n-a\n+b\n" - patch = sbs.extract_patch(text) - assert patch.startswith("diff --git a/x b/x") - assert "no fence" not in patch - - -def test_extract_patch_bare_diff_strips_trailing_prose(): - text = ( - "diff --git a/x b/x\n--- a/x\n+++ b/x\n@@ -1 +1 @@\n-old\n+new\n" - "\nNotes:\nThis fixes #123.\n" - ) - patch = sbs.extract_patch(text) - assert patch.rstrip().endswith("+new") - assert "Notes:" not in patch - assert "This fixes" not in patch - - -def test_extract_patch_keeps_multi_file_and_interior_context(): - text = ( - "```diff\n" - "diff --git a/a b/a\n@@ -1,2 +1,2 @@\n context\n-x\n+y\n" - "diff --git a/b b/b\n@@ -1 +1 @@\n-p\n+q\n" - "```\nthanks!" - ) - patch = sbs.extract_patch(text) - assert "diff --git a/a b/a" in patch - assert "diff --git a/b b/b" in patch - assert "thanks" not in patch - - -def test_extract_patch_empty_when_no_diff(): - assert sbs.extract_patch("") == "" - assert sbs.extract_patch("just words").strip() == "just words" - - - -def _write_samples(dirpath: Path, records: list[dict]) -> None: - with (dirpath / "samples_swebench_lite_2026.jsonl").open("w") as fh: - for rec in records: - fh.write(json.dumps(rec) + "\n") - - -def test_build_predictions_extracts_instance_and_patch(tmp_path): - _write_samples(tmp_path, [ - { - "doc": {"instance_id": "repo__proj-1"}, - "filtered_resps": ["```diff\ndiff --git a/a b/a\n+x\n```"], - }, - { - "doc": {"instance_id": "repo__proj-2"}, - "resps": [["diff --git a/b b/b\n+y\n"]], - }, - ]) - preds = sbs.build_predictions(tmp_path, "my-model") - by_id = {p["instance_id"]: p for p in preds} - assert set(by_id) == {"repo__proj-1", "repo__proj-2"} - assert by_id["repo__proj-1"]["model_name_or_path"] == "my-model" - assert by_id["repo__proj-1"]["model_patch"].startswith("diff --git a/a b/a") - assert by_id["repo__proj-2"]["model_patch"].startswith("diff --git a/b b/b") - - -def test_build_predictions_raises_without_samples(tmp_path): - with pytest.raises(FileNotFoundError): - sbs.build_predictions(tmp_path, "m") - - - -def test_parse_resolved_classic_counts(): - assert sbs.parse_resolved( - {"resolved_instances": 80, "total_instances": 196} - ) == (80, 196) - - -def test_parse_resolved_prefers_submitted_over_dataset_total(): - assert sbs.parse_resolved( - {"resolved_instances": 32, "submitted_instances": 50, "total_instances": 300} - ) == (32, 50) - - -def test_parse_resolved_from_id_lists(): - report = {"resolved_ids": ["a", "b", "c"], "completed_ids": ["a", "b", "c", "d"]} - assert sbs.parse_resolved(report) == (3, 4) - - -def test_parse_resolved_raises_on_garbage(): - with pytest.raises(ValueError): - sbs.parse_resolved({"nope": 1}) - - - -def test_run_harness_instance_timeout(monkeypatch, tmp_path): - captured = {} - monkeypatch.setattr(sbs.subprocess, "run", lambda cmd, **kw: captured.setdefault("cmd", cmd)) - sbs.run_harness( - tmp_path / "p.jsonl", "ds", "rid", tmp_path, 4, None, modal=True, timeout=900, - ) - cmd = captured["cmd"] - i = cmd.index("--timeout") - assert cmd[i + 1] == "900" - - -def test_run_harness_no_timeout_by_default(monkeypatch, tmp_path): - captured = {} - monkeypatch.setattr(sbs.subprocess, "run", lambda cmd, **kw: captured.setdefault("cmd", cmd)) - sbs.run_harness(tmp_path / "p.jsonl", "ds", "rid", tmp_path, 4, None, modal=True) - assert "--timeout" not in captured["cmd"] - - -def _captured_harness_cmd(monkeypatch, tmp_path, *, modal, namespace): - captured = {} - monkeypatch.setattr(sbs.subprocess, "run", lambda cmd, **kw: captured.setdefault("cmd", cmd)) - sbs.run_harness( - tmp_path / "predictions.jsonl", "princeton-nlp/SWE-bench_Lite", "rid", - tmp_path, 8, namespace, modal=modal, - ) - return captured["cmd"] - - -def test_run_harness_modal_uses_modal_flag(monkeypatch, tmp_path): - cmd = _captured_harness_cmd(monkeypatch, tmp_path, modal=True, namespace="") - assert "--modal" in cmd - assert "--max_workers" in cmd - assert "--parallelism" not in cmd - assert "--namespace" not in cmd - - -def test_run_harness_docker_uses_max_workers_and_namespace(monkeypatch, tmp_path): - cmd = _captured_harness_cmd(monkeypatch, tmp_path, modal=False, namespace="") - assert "--max_workers" in cmd - assert "--namespace" in cmd - assert "--modal" not in cmd - - - -def test_build_results_json_is_lm_eval_shaped(): - res = sbs.build_results_json( - "swebench_lite", 49, 196, "m", "0.4.12", {"resolved_instances": 49} - ) - assert "lm_eval_version" in res - task = res["results"]["swebench_lite"] - assert task["exact_match,resolved"] == pytest.approx(0.25) - cfg = res["configs"]["swebench_lite"] - assert cfg["filter_list"] == [{"name": "resolved"}] - assert res["n-samples"]["swebench_lite"]["effective"] == 196 - - -def test_score_offline_end_to_end(tmp_path): - samples = tmp_path / "gen" - samples.mkdir() - _write_samples(samples, [ - {"doc": {"instance_id": "r__p-1"}, "filtered_resps": ["```diff\ndiff --git a/a b/a\n+x\n```"]}, - ]) - report = tmp_path / "report.json" - report.write_text(json.dumps({"resolved_instances": 1, "total_instances": 1})) - out = tmp_path / "out" - rc = sbs.main([ - "--samples-dir", str(samples), "--out-dir", str(out), - "--model-name", "m", "--report", str(report), - ]) - assert rc == 0 - assert (out / "predictions.jsonl").exists() - results = json.loads((out / "results_swebench_lite.json").read_text()) - assert results["results"]["swebench_lite"]["exact_match,resolved"] == 1.0 - - -def test_predictions_only_writes_predictions_no_results(tmp_path): - samples = tmp_path / "gen" - samples.mkdir() - _write_samples(samples, [ - {"doc": {"instance_id": "r__p-1"}, "filtered_resps": ["```diff\ndiff --git a/a b/a\n+x\n```"]}, - ]) - out = tmp_path / "out" - rc = sbs.main([ - "--samples-dir", str(samples), "--out-dir", str(out), - "--model-name", "m", "--predictions-only", - ]) - assert rc == 0 - assert (out / "predictions.jsonl").exists() - assert not (out / "results_swebench_lite.json").exists() - - - -@pytest.mark.skipif(sys.version_info < (3, 10), reason="repo modules use py3.10 syntax") -def test_results_json_flows_through_collect_and_validate(tmp_path, monkeypatch): - pytest.importorskip("tabulate") - import infx.results.collect_eval_results as cer - import infx.evals.validate_scores as vs - - art = tmp_path / "eval" - art.mkdir() - (art / "meta_env.json").write_text(json.dumps({ - "infmax_model_prefix": "dsr1", "hw": "b200", "framework": "sglang", - "precision": "fp8", "isl": 8192, "osl": 1024, - })) - res = sbs.build_results_json( - "swebench_lite", 180, 300, "dsr1", "0.4.12", None - ) - (art / "results_swebench_lite.json").write_text(json.dumps(res)) - - rows = cer.collect_eval_rows(tmp_path) - assert len(rows) == 1 - assert rows[0]["task"] == "swebench_lite" - assert rows[0]["score"] == pytest.approx(0.6) - - monkeypatch.chdir(art) - monkeypatch.setattr(sys, "argv", [ - "validate_scores.py", - "--results-glob", "results_swebench_lite.json", - ]) - assert vs.main() == 0 - - -def test_predictions_file_mode_skips_samples_and_scores(tmp_path): - preds = tmp_path / "agent_preds.jsonl" - preds.write_text(json.dumps({ - "instance_id": "r__p-1", "model_name_or_path": "m", - "model_patch": "diff --git a/a b/a\n+x\n", - }) + "\n") - report = tmp_path / "report.json" - report.write_text(json.dumps({"resolved_instances": 1, "total_instances": 1})) - out = tmp_path / "out" - rc = sbs.main([ - "--predictions-file", str(preds), "--out-dir", str(out), - "--model-name", "m", "--report", str(report), - ]) - assert rc == 0 - assert (out / "predictions.jsonl").exists() - results = json.loads((out / "results_swebench_lite.json").read_text()) - assert results["results"]["swebench_lite"]["exact_match,resolved"] == 1.0 - - -def test_no_input_source_errors(tmp_path): - rc = sbs.main(["--out-dir", str(tmp_path / "o"), "--model-name", "m"]) - assert rc == 1 diff --git a/infx/tests/matrix/test_generate_sweep_configs.py b/infx/tests/matrix/test_generate_sweep_configs.py index a23ac34788..222ef32180 100644 --- a/infx/tests/matrix/test_generate_sweep_configs.py +++ b/infx/tests/matrix/test_generate_sweep_configs.py @@ -416,9 +416,9 @@ def test_marks_agentic_entry_for_gsm8k(self): assert marked[0]["conc"] == 64 def test_marks_multinode_agentic_entry_at_highest_eligible_conc(self): - """Multi-node agentic (SWE-bench) eval selection mirrors the - fixed-seq-len multi-node policy: one eval row per parallelism - topology, at its highest eligible (>= MIN_EVAL_CONC) concurrency. + """Multi-node agentic eval selection mirrors the fixed-seq-len + multi-node policy: one eval row per parallelism topology, at its + highest eligible (>= MIN_EVAL_CONC) concurrency. Each concurrency is its own matrix entry (chunk size 1) whose exp-name embeds that concurrency, unlike fixed-seq-len multi-node diff --git a/runners/launch_h100-cr.sh b/runners/launch_h100-cr.sh index 8ea2c63fbf..9774ee2f63 100644 --- a/runners/launch_h100-cr.sh +++ b/runners/launch_h100-cr.sh @@ -16,7 +16,7 @@ docker run --rm --network=host --name=$server_name \ --runtime=nvidia --gpus="$GPU_COUNT" --ipc=host --privileged --shm-size=16g --ulimit memlock=-1 --ulimit stack=67108864 \ -v $HF_HUB_CACHE_MOUNT:$HF_HUB_CACHE \ -v $GITHUB_WORKSPACE:/workspace/ -w /workspace/ \ --e HF_TOKEN -e HF_HUB_CACHE -e MODEL -e TP -e PP_SIZE -e DCP_SIZE -e PCP_SIZE -e GPU_COUNT -e CONC -e MAX_MODEL_LEN -e ISL -e OSL -e RUN_EVAL -e EVAL_ONLY -e EVAL_FRAMEWORK -e EVAL_LIMIT -e EVAL_SUITE -e IS_AGENTIC -e SCENARIO_TYPE -e SWEBENCH_GEN_MODE -e SWEBENCH_USE_MODAL -e MODAL_TOKEN_ID -e MODAL_TOKEN_SECRET -e RUNNER_TYPE -e RESULT_FILENAME -e RANDOM_RANGE_RATIO -e PORT=$PORT \ +-e HF_TOKEN -e HF_HUB_CACHE -e MODEL -e TP -e PP_SIZE -e DCP_SIZE -e PCP_SIZE -e GPU_COUNT -e CONC -e MAX_MODEL_LEN -e ISL -e OSL -e RUN_EVAL -e EVAL_ONLY -e EVAL_FRAMEWORK -e EVAL_LIMIT -e EVAL_SUITE -e IS_AGENTIC -e SCENARIO_TYPE -e RUNNER_TYPE -e RESULT_FILENAME -e RANDOM_RANGE_RATIO -e PORT=$PORT \ -e PROFILE -e SGLANG_TORCH_PROFILER_DIR -e VLLM_TORCH_PROFILER_DIR -e VLLM_RPC_TIMEOUT \ -e PYTHONPYCACHEPREFIX=/tmp/pycache/ -e TORCH_CUDA_ARCH_LIST="9.0" -e CUDA_DEVICE_ORDER=PCI_BUS_ID -e CUDA_VISIBLE_DEVICES \ --entrypoint=/bin/bash \ diff --git a/runners/launch_mi325x-tw.sh b/runners/launch_mi325x-tw.sh index 91f0a4bf9c..3a512d6666 100644 --- a/runners/launch_mi325x-tw.sh +++ b/runners/launch_mi325x-tw.sh @@ -22,7 +22,7 @@ docker run --rm --network=host --name="$server_name" \ --security-opt seccomp=unconfined --cap-add=SYS_PTRACE \ -v "$HF_HUB_CACHE_MOUNT:$HF_HUB_CACHE" \ -v "$GITHUB_WORKSPACE:/workspace/" -w /workspace/ \ --e HF_TOKEN -e HF_HUB_CACHE -e MODEL -e TP -e PP_SIZE -e DCP_SIZE -e PCP_SIZE -e GPU_COUNT -e CONC -e MAX_MODEL_LEN -e ISL -e OSL -e RUN_EVAL -e EVAL_ONLY -e EVAL_FRAMEWORK -e EVAL_LIMIT -e EVAL_SUITE -e IS_AGENTIC -e SCENARIO_TYPE -e SWEBENCH_GEN_MODE -e SWEBENCH_USE_MODAL -e MODAL_TOKEN_ID -e MODAL_TOKEN_SECRET -e RUNNER_TYPE -e RESULT_FILENAME -e RANDOM_RANGE_RATIO -e PORT="$PORT" \ +-e HF_TOKEN -e HF_HUB_CACHE -e MODEL -e TP -e PP_SIZE -e DCP_SIZE -e PCP_SIZE -e GPU_COUNT -e CONC -e MAX_MODEL_LEN -e ISL -e OSL -e RUN_EVAL -e EVAL_ONLY -e EVAL_FRAMEWORK -e EVAL_LIMIT -e EVAL_SUITE -e IS_AGENTIC -e SCENARIO_TYPE -e RUNNER_TYPE -e RESULT_FILENAME -e RANDOM_RANGE_RATIO -e PORT="$PORT" \ -e DP_ATTENTION -e EP_SIZE -e DP_SIZE -e EVAL_MAX_MODEL_LEN -e SPEC_DECODING -e NUM_SPEC_TOKENS \ -e PROFILE -e SGLANG_TORCH_PROFILER_DIR -e VLLM_TORCH_PROFILER_DIR -e VLLM_RPC_TIMEOUT \ -e PYTHONPYCACHEPREFIX=/tmp/pycache/ -e CUDA_DEVICE_ORDER=PCI_BUS_ID \ diff --git a/runners/slurm_utils.sh b/runners/slurm_utils.sh index 50370edfeb..9dcd0e7d07 100644 --- a/runners/slurm_utils.sh +++ b/runners/slurm_utils.sh @@ -46,8 +46,7 @@ import os names = [ "EVAL_FRAMEWORK", "EVAL_CONC", "EVAL_LIMIT", "EVAL_SUITE", - "SWEBENCH_GEN_MODE", "SWEBENCH_USE_MODAL", "MODAL_TOKEN_ID", - "MODAL_TOKEN_SECRET", "IS_AGENTIC", "SCENARIO_TYPE", + "IS_AGENTIC", "SCENARIO_TYPE", "TP", "EP_SIZE", "DP_ATTENTION", "PP_SIZE", "DCP_SIZE", "PCP_SIZE", "CONC", ] print(json.dumps(names + os.environ["INFERENCEX_RUNTIME_ENV_VARS"].split()))